{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Checkout 🚀 [Ultralytics docs ](https://docs.ultralytics.com/datasets/detect/)","metadata":{}},{"cell_type":"markdown","source":"## YOLOv8 requires the label data to be provided as images in folders following specific structure and a text (.txt) file, following a specific format. \n## The folder structure should look like this:","metadata":{}},{"cell_type":"markdown","source":"\n\n<div style=\"text-align: center;\"><img src=\"https://github.com/ultralytics/docs/releases/download/0/two-persons-tie-2.avif\" width=\"500\"/></div>","metadata":{}},{"cell_type":"markdown","source":"## Each `.txt` includes the class index, coordinates of the object, all normalized to the image width and height. Each line in the text file represents an object in the corresponding image, and the format for each line. \n## As for the config `.yaml` file format, it should look something like this:\n\n```\npath: /Data/RSNA LSDC/YOLO/\ntrain: /Data/RSNA LSDC/YOLO/train/images\nval: /RSNA LSDC/YOLO/val/images\nnames:\n  0: normal\n  1: condition\n```","metadata":{}},{"cell_type":"code","source":"import os\nimport math\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport plotly.express as px\nimport plotly.graph_objects as go\n\nfrom tqdm import tqdm\nfrom PIL import Image\nfrom pathlib import Path\nfrom typing import Tuple\nfrom multiprocessing import Pool\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-09-24T04:50:12.639818Z","iopub.execute_input":"2024-09-24T04:50:12.640915Z","iopub.status.idle":"2024-09-24T04:50:12.646879Z","shell.execute_reply.started":"2024-09-24T04:50:12.640869Z","shell.execute_reply":"2024-09-24T04:50:12.645922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def parse_descriptions(image_folders_path: str, descriptions_path: str):\n    \"\"\"\n    For each scan let's create an object with the following structure:\n    \n    metadata = {\n        StudyInstanceUID: {\n            'folder_path': ... # path to the folder,\n            'SeriesInstanceUIDs': [ Array of the SeriesInstanceUIDs ],\n            'SeriesDescriptions' [ Array of the Series Descriptions ]\n        }, ...\n    }\n    \"\"\"\n    image_folders = list(filter(lambda x: x.find('.DS') == -1, os.listdir(image_folders_path)))\n    descriptions_df = pd.read_csv(descriptions_path)\n    \n    metadata = {\n        folder_name: {\n            'folder_path': os.path.join(image_folders_path, folder_name), \n            'SeriesInstanceUIDs': list(\n                filter(lambda x: x.find('.DS') == -1, os.listdir(os.path.join(image_folders_path, folder_name)))\n            )\n        } for folder_name in image_folders\n    }\n\n    # grabs the corresponding series descriptions\n    for k in tqdm(metadata, desc=\"Parsing metadata\"):\n        for s in metadata[k]['SeriesInstanceUIDs']:\n            if 'SeriesDescriptions' not in metadata[k]:\n                metadata[k]['SeriesDescriptions'] = []\n            try:\n                metadata[k]['SeriesDescriptions'].append(\n                    descriptions_df[(descriptions_df['study_id'] == int(k)) & \n                    (descriptions_df['series_id'] == int(s))]['series_description'].iloc[0])\n            except IndexError:\n                print(\"Failed on\", s, k)\n                \n    return metadata","metadata":{"execution":{"iopub.status.busy":"2024-09-24T04:34:12.837408Z","iopub.execute_input":"2024-09-24T04:34:12.838465Z","iopub.status.idle":"2024-09-24T04:34:12.850156Z","shell.execute_reply.started":"2024-09-24T04:34:12.838357Z","shell.execute_reply":"2024-09-24T04:34:12.849205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_samples_dataframe(image_folders_path: str, labels_path: str, descriptions_path: str, label_coordinates_path: str) -> pd.DataFrame:\n    metadata = parse_descriptions(image_folders_path, descriptions_path)\n    label_coordinates_df = pd.read_csv(label_coordinates_path)\n    labels_df = pd.read_csv(labels_path)\n    \n    labels_mapping = {\n        \"Normal/Mild\": 1,\n        \"Moderate\": 2,\n        \"Severe\": 3,\n    }\n    \n    data = []\n    for study_id, patient in tqdm(metadata.items(), desc=\"Creating a dataframe\", total=len(metadata)):\n        labels_per_study = labels_df[labels_df[\"study_id\"] == int(study_id)]\n        for series_folder, scan_type in zip(patient[\"SeriesInstanceUIDs\"], patient[\"SeriesDescriptions\"]):\n            # Pre-filter the label_coordinates_df for this specific study and series\n            relevant_coordinates = label_coordinates_df[\n                (label_coordinates_df[\"study_id\"] == int(study_id)) &\n                (label_coordinates_df[\"series_id\"] == int(series_folder))\n            ]\n            # Dictionary of instance_number -> relevant_coordinates\n            dicom_files = list(Path(os.path.join(patient[\"folder_path\"], series_folder)).rglob(\"*.dcm\"))\n            instance_map = {int(Path(dicom_file).stem): dicom_file for dicom_file in dicom_files}\n            \n            # Cases with coordinates\n            for _, entry in relevant_coordinates.iterrows():\n                instance_number = entry[\"instance_number\"]\n                if instance_number not in instance_map:\n                    continue\n                \n                dicom_file = instance_map[instance_number]\n                condition_column = entry[\"condition\"].lower().replace(\" \", \"_\") + \"_\" + entry[\"level\"].lower().replace(\"/\", \"_\")\n                label_row = labels_per_study[condition_column].iloc[0]\n                \n                if not isinstance(label_row, str):\n                    continue\n                \n                data.append({\n                    \"study_id\": int(study_id),\n                    \"series_id\": int(series_folder),\n                    \"instance_number\": instance_number,\n                    \"image_path\": str(dicom_file),\n                    \"scan_type\": scan_type,\n                    \"condition\": entry[\"condition\"],\n                    \"level\": entry[\"level\"],\n                    \"x\": entry[\"x\"],\n                    \"y\": entry[\"y\"],\n                    \"label\": labels_mapping[label_row],\n                })\n                \n            # Rows without coordinates (Normal cases)\n            untagged_instances = set(instance_map.keys()) - set(relevant_coordinates[\"instance_number\"].values)\n            for instance_number in untagged_instances:\n                dicom_file = instance_map[instance_number]\n                data.append({\n                    \"study_id\": int(study_id),\n                    \"series_id\": int(series_folder),\n                    \"instance_number\": instance_number,\n                    \"image_path\": str(dicom_file),\n                    \"scan_type\": scan_type,\n                    \"condition\": None,\n                    \"level\": None,\n                    \"x\": None,\n                    \"y\": None,\n                    \"label\": 0,\n                })\n\n    return pd.DataFrame(data)\n\n\ndef split_data(dataframe: pd.DataFrame, train_size: float = 0.8) -> Tuple[pd.DataFrame, pd.DataFrame]:\n    study_labels = dataframe[['study_id', 'label']].drop_duplicates()\n    \n    # Perform stratified split on study_ids based on labels\n    train_studies, val_studies = train_test_split(\n        study_labels['study_id'],\n        test_size=1 - train_size,\n        stratify=study_labels['label'],\n        random_state=42\n    )\n    \n    return (\n        dataframe[dataframe['study_id'].isin(train_studies)].reset_index(drop=True), \n        dataframe[dataframe['study_id'].isin(val_studies)].reset_index(drop=True)\n\t)\n\n\ndef save_samples_dataframe(dataframe: pd.DataFrame, save_path: str):\n    dataframe.to_csv(save_path, index=False)","metadata":{"execution":{"iopub.status.busy":"2024-09-24T04:34:12.851600Z","iopub.execute_input":"2024-09-24T04:34:12.851960Z","iopub.status.idle":"2024-09-24T04:34:12.872907Z","shell.execute_reply.started":"2024-09-24T04:34:12.851908Z","shell.execute_reply":"2024-09-24T04:34:12.871695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dicom_to_jpg(dicom_file, output_dir, sample_name=None):\n    \"\"\"\n    Converts a DICOM file to a JPG image and saves it to the specified output directory.\n\n    Parameters:\n    - dicom_file: Path to the input DICOM (.dcm) file.\n    - output_dir: Directory where the JPG image will be saved.\n    - sample_name: (Optional) Custom name for the output JPG file. If not provided, \n      the name will be derived from the DICOM filename.\n\n    Returns:\n    - image: A PIL Image object of the converted JPG image.\n    \"\"\"\n    os.makedirs(output_dir, exist_ok=True)\n    dicom_data = pydicom.dcmread(dicom_file)\n    \n    # Get pixel data and scale it to 0-255 range for PNG\n    pixel_array = dicom_data.pixel_array.astype(np.float32)\n    pixel_min = pixel_array.min()\n    pixel_max = pixel_array.max()\n    pixel_array = ((pixel_array - pixel_min) / (pixel_max - pixel_min) * 255.0).astype(np.uint8)\n    \n    if sample_name is None:\n        sample_name = os.path.basename(dicom_file).replace('.dcm', '.jpg')\n    output_path = os.path.join(output_dir, sample_name)\n    image = Image.fromarray(pixel_array)\n    image.save(output_path)\n    \n    return image\n\n\ndef create_yolo_annotation(label: int, x: float, y: float, image_width: int, image_height: int, width: float = 0.08, height: float = 0.08):\n    \"\"\"\n    Creates a YOLO annotation string for an object in an image based on its bounding box \n    center coordinates and size.\n\n    Parameters:\n    - label: Integer label for the object (0 for normal, 1 for condition).\n    - x: X-coordinate of the object’s center.\n    - y: Y-coordinate of the object’s center.\n    - image_width: Width of the image in pixels.\n    - image_height: Height of the image in pixels.\n    - width: Width of the bounding box as a fraction of the image width (default is 0.08).\n    - height: Height of the bounding box as a fraction of the image height (default is 0.08).\n\n    Returns:\n    - A string in the YOLO annotation format: \"<label> <x_center_normalized> <y_center_normalized> <width> <height>\".\n      If the x or y coordinate is NaN, an empty string is returned.\n    \"\"\"\n    if math.isnan(x) or math.isnan(y):\n        return \"\"\n        \n    return f\"{label} {x/image_width} {y/image_height} {width} {height}\"\n\n\ndef process_row(args):\n    \"\"\"\n    Processes a single row of the dataset to convert the DICOM image to JPG and generate a YOLO annotation file.\n\n    Parameters:\n    - args: A tuple containing the following:\n      - row: A pandas DataFrame row with fields 'study_id', 'series_id', 'instance_number', 'image_path', 'label', 'x', and 'y'.\n      - images_folder: Directory where the converted JPG images will be saved.\n      - labels_folder: Directory where the YOLO annotation files will be saved.\n\n    Returns:\n    - None: Saves the JPG image and corresponding YOLO annotation file to the specified folders.\n    \"\"\"\n    row, images_folder, labels_folder = args\n    sample_name = f\"{row['study_id']}_{row['series_id']}_{row['instance_number']}.jpg\"\n    \n    if os.path.exists(os.path.join(images_folder, sample_name)):\n        return\n    \n    image = dicom_to_jpg(row['image_path'], images_folder, sample_name)\n    \n    with open(os.path.join(labels_folder, sample_name.replace('.jpg', '.txt')), 'w') as f:\n        width, height = image.size\n        f.write(create_yolo_annotation(int(row['label'] > 0), row['x'], row['y'], width, height))\n\n\ndef create_yolo_dataset(dataframe: pd.DataFrame, save_folder: str):\n    \"\"\"\n    Converts a dataset of DICOM files to YOLO format by generating JPG images and corresponding YOLO annotation files.\n\n    Parameters:\n    - dataframe: A pandas DataFrame where each row contains data for a DICOM file and its annotation information.\n      Expected columns include 'study_id', 'series_id', 'instance_number', 'image_path', 'label', 'x', and 'y'.\n    - save_folder: Root folder where the dataset will be saved. Subfolders 'images' and 'labels' will be created for storing \n      the converted JPG images and annotation files respectively.\n\n    Returns:\n    - None: Parallelizes the conversion process and creates the dataset using a multiprocessing pool.\n    \"\"\"\n    images_folder = os.path.join(save_folder, \"images\")\n    labels_folder = os.path.join(save_folder, \"labels\")\n    os.makedirs(images_folder, exist_ok=True)\n    os.makedirs(labels_folder, exist_ok=True)\n    \n    args_list = [(row, images_folder, labels_folder) for index, row in dataframe.iterrows()]\n    \n    with Pool() as pool:\n        list(tqdm(pool.imap_unordered(process_row, args_list), total=len(args_list), desc=f\"Creating YOLO dataset at {save_folder}\"))\n        \n\ndef save_yaml(save_path: str):\n    \"\"\"\n    Creates a YAML configuration file for the YOLO dataset, specifying the paths for training and validation data.\n\n    Parameters:\n    - save_path: Path where the YAML configuration file will be saved.\n\n    Returns:\n    - None: Writes the YAML configuration file to the specified path.\n    \"\"\"\n    with open(save_path, 'w') as f:\n        f.write(f\"path: {yolo_dataset_folder}\\n\")\n        f.write(f\"train: {os.path.join(yolo_dataset_folder, 'train', 'images')}\\n\")\n        f.write(f\"val: {os.path.join(yolo_dataset_folder, 'val', 'images')}\\n\")\n        f.write(\"names:\\n\")\n        f.write(\"  0: normal\\n\")\n        f.write(\"  1: condition\\n\")\n        f.write(\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2024-09-24T04:34:12.961211Z","iopub.execute_input":"2024-09-24T04:34:12.961971Z","iopub.status.idle":"2024-09-24T04:34:12.983185Z","shell.execute_reply.started":"2024-09-24T04:34:12.961913Z","shell.execute_reply":"2024-09-24T04:34:12.981987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_labels_distribution(dataframe: pd.DataFrame, label_column: str = \"label\", title: str = \"Label Distribution\"):\n    labels_mapping = {\n        0: \"Normal\",\n        1: \"Mild\",\n        2: \"Moderate\",\n        3: \"Severe\",\n    }\n\n    dataframe['label_name'] = dataframe[label_column].map(labels_mapping)\n    fig = px.pie(dataframe, names='label_name', title=title, hole=0.3)\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-24T04:34:12.985587Z","iopub.execute_input":"2024-09-24T04:34:12.986097Z","iopub.status.idle":"2024-09-24T04:34:12.995067Z","shell.execute_reply.started":"2024-09-24T04:34:12.986049Z","shell.execute_reply":"2024-09-24T04:34:12.994164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_folder = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\nsamples = create_samples_dataframe(\n    os.path.join(data_folder, \"train_images\"),\n    os.path.join(data_folder, \"train.csv\"),\n    os.path.join(data_folder, \"train_series_descriptions.csv\"),\n    os.path.join(data_folder, \"train_label_coordinates.csv\")\n)\n\ntrain_samples, val_samples = split_data(samples)","metadata":{"execution":{"iopub.status.busy":"2024-09-24T04:34:12.996501Z","iopub.execute_input":"2024-09-24T04:34:12.996916Z","iopub.status.idle":"2024-09-24T04:35:48.933379Z","shell.execute_reply.started":"2024-09-24T04:34:12.996868Z","shell.execute_reply":"2024-09-24T04:35:48.932197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_labels_distribution(train_samples, title=\"Train Label Distribution\")","metadata":{"execution":{"iopub.status.busy":"2024-09-24T04:35:48.936085Z","iopub.execute_input":"2024-09-24T04:35:48.936468Z","iopub.status.idle":"2024-09-24T04:35:49.479820Z","shell.execute_reply.started":"2024-09-24T04:35:48.936429Z","shell.execute_reply":"2024-09-24T04:35:49.478540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_labels_distribution(val_samples, title=\"Validation Label Distribution\")","metadata":{"execution":{"iopub.status.busy":"2024-09-24T04:35:49.481473Z","iopub.execute_input":"2024-09-24T04:35:49.481893Z","iopub.status.idle":"2024-09-24T04:35:49.799946Z","shell.execute_reply.started":"2024-09-24T04:35:49.481850Z","shell.execute_reply":"2024-09-24T04:35:49.798951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_annotated_image(sample, box_size=0.08):\n    \"\"\"\n    Displays an image with a box annotation and label using Plotly in a Jupyter notebook.\n\n    Parameters:\n    - sample: pandas Series containing 'image_path', 'x', 'y', and 'label' fields.\n    - yolo_dataset_folder: Folder where images are stored or will be saved.\n    - box_size: Size of the box as a fraction of the image dimensions (default is 0.08).\n    \"\"\"\n    image = dicom_to_jpg(sample['image_path'], './')\n    width, height = image.size\n    image_np = np.array(image.convert('RGB'))  # Ensure the image is in RGB format\n\n    if not math.isnan(sample['x']) and not math.isnan(sample['y']):\n        x, y = sample['x'], sample['y']\n        label = sample['label']\n        \n        box_width, box_height = width * box_size, height * box_size\n        x1, y1 = int(x - box_width / 2), int(y - box_height / 2)\n        x2, y2 = int(x + box_width / 2), int(y + box_height / 2)\n\n        fig = go.Figure()\n        fig.add_trace(go.Image(z=image_np))\n        fig.add_shape(\n            type=\"rect\",\n            x0=x1, y0=y1, x1=x2, y1=y2,\n            line=dict(color=\"red\", width=3)\n        )\n        fig.add_annotation(\n            x=x1, y=y1,\n            text=str(label),\n            showarrow=False,\n            font=dict(size=12, color=\"red\"),\n            align=\"left\",\n            xanchor=\"left\", yanchor=\"bottom\"\n        )\n        fig.update_layout(\n            xaxis=dict(showgrid=False, zeroline=False, visible=False),\n            yaxis=dict(showgrid=False, zeroline=False, visible=False)\n        )\n        fig.update_yaxes(scaleanchor=\"x\", autorange=\"reversed\")\n        fig.show()\n\n\n# Call the function to display the annotated image\nplot_annotated_image(train_samples.iloc[0])","metadata":{"execution":{"iopub.status.busy":"2024-09-24T04:35:49.801394Z","iopub.execute_input":"2024-09-24T04:35:49.801705Z","iopub.status.idle":"2024-09-24T04:35:49.988830Z","shell.execute_reply.started":"2024-09-24T04:35:49.801661Z","shell.execute_reply":"2024-09-24T04:35:49.987604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yolo_dataset_folder = \"/kaggle/working/YOLO\"\nos.makedirs(yolo_dataset_folder, exist_ok=True)\n\nsave_samples_dataframe(samples, os.path.join(yolo_dataset_folder, \"all_samples.csv\"))\nsave_samples_dataframe(train_samples, os.path.join(yolo_dataset_folder, \"train_samples.csv\"))\nsave_samples_dataframe(val_samples, os.path.join(yolo_dataset_folder, \"val_samples.csv\"))\n\ncreate_yolo_dataset(train_samples, os.path.join(yolo_dataset_folder, \"train\"))\ncreate_yolo_dataset(val_samples, os.path.join(yolo_dataset_folder, \"val\"))\nsave_yaml(os.path.join(yolo_dataset_folder, \"data.yaml\"))","metadata":{"execution":{"iopub.status.busy":"2024-09-24T04:35:49.990488Z","iopub.execute_input":"2024-09-24T04:35:49.990856Z","iopub.status.idle":"2024-09-24T04:40:40.500859Z","shell.execute_reply.started":"2024-09-24T04:35:49.990817Z","shell.execute_reply":"2024-09-24T04:40:40.499588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now you cna zip the files and download them \n\nimport shutil\nfrom IPython.display import FileLink\n\n#shutil.make_archive('/kaggle/working/dataset', 'zip', '/kaggle/working')\n# Provide a link to download the zip file\nFileLink(r'working_directory_backup.zip')","metadata":{"execution":{"iopub.status.busy":"2024-09-24T04:58:26.275065Z","iopub.execute_input":"2024-09-24T04:58:26.276101Z","iopub.status.idle":"2024-09-24T04:58:26.283854Z","shell.execute_reply.started":"2024-09-24T04:58:26.276048Z","shell.execute_reply":"2024-09-24T04:58:26.282741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}