{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# siim-covid19-detection preprocessing\n\nThe intention is to create a new dataset in kaggle with all the .dcm files\nresized and converted grayscale .png files. \n\nIn addition, the csv files are processed, simplified and saved.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"# Some of the images are packed and require gdcm module.\n!pip install python-gdcm","metadata":{"execution":{"iopub.status.busy":"2021-07-08T10:03:53.184943Z","iopub.execute_input":"2021-07-08T10:03:53.185282Z","iopub.status.idle":"2021-07-08T10:04:00.422157Z","shell.execute_reply.started":"2021-07-08T10:03:53.185248Z","shell.execute_reply":"2021-07-08T10:04:00.420929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib.patches import Rectangle\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport pandas as pd\nfrom PIL import Image\nimport pydicom\nimport shutil\nfrom tqdm.auto import tqdm","metadata":{"execution":{"iopub.status.busy":"2021-07-08T10:24:04.251731Z","iopub.execute_input":"2021-07-08T10:24:04.252163Z","iopub.status.idle":"2021-07-08T10:24:04.260036Z","shell.execute_reply.started":"2021-07-08T10:24:04.252128Z","shell.execute_reply":"2021-07-08T10:24:04.258744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_shape = (768, 768)","metadata":{"execution":{"iopub.status.busy":"2021-07-08T10:04:00.93913Z","iopub.execute_input":"2021-07-08T10:04:00.939641Z","iopub.status.idle":"2021-07-08T10:04:00.943524Z","shell.execute_reply.started":"2021-07-08T10:04:00.939609Z","shell.execute_reply":"2021-07-08T10:04:00.942826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dataframe for train data.\n\nroot = \"/kaggle/input/siim-covid19-detection\"\n\n# Read the csv containing the image data\ndf_train_image = pd.read_csv(f\"{root}/train_image_level.csv\")\n\n# Handle the boxes column\ndf_train_image.loc[df_train_image[df_train_image[\"boxes\"].isna()].index, \"boxes\"] = \"[]\"\ndf_train_image[\"boxes\"] = df_train_image[\"boxes\"].apply(lambda x: eval(x))\n\n# Add a column containing the filepaths to the .dcm files.\ndef get_filepath(row):\n    identifier = row[\"id\"].split(\"_\")[0]\n    filename = identifier + \".dcm\"\n    study_identifier = row[\"StudyInstanceUID\"]\n    for path, dirs, files in os.walk(f\"{root}/train/{study_identifier}\"):\n        if filename in files:\n            return f\"{path}/{filename}\"\n    raise AssertionError(\"Could not find the file\")\n\ndf_train_image[\"filepath\"] = df_train_image.apply(get_filepath, axis=1)\n\n# Assign better names for columns and take a subset of columns\ndf_train_image[\"image_id\"] = df_train_image[\"id\"].str.replace(\"_image\", \"\")\ndf_train_image = df_train_image.rename(columns={\"StudyInstanceUID\": \"study_id\"})\ndf_train_image = df_train_image[[\"image_id\", \"study_id\", \"boxes\", \"filepath\"]]\n\n# Read the csv containing the study data\ndf_train_study = pd.read_csv(f\"{root}/train_study_level.csv\")\n\n# Create a column containing the label\ncolumn_to_label = {\n    \"Negative for Pneumonia\": \"negative\",\n    \"Typical Appearance\": \"typical\",\n    \"Indeterminate Appearance\": \"indeterminate\",\n    \"Atypical Appearance\": \"atypical\"\n}\n\ndef one_hot_to_label(row):\n    \"\"\"Given a one-hot encoding output its label\"\"\"\n    for column, label in column_to_label.items():\n        if row[column] == 1:\n            return label\n    raise AssertionError(\"Something went wrong\")\n        \ndf_train_study[\"label\"] = df_train_study.apply(one_hot_to_label, axis=1)\n\n# Create the same column as in the other data frame.\ndf_train_study[\"study_id\"] = df_train_study[\"id\"].str.replace(\"_study\", \"\")\n\n# Take a subset of columns\ndf_train_study = df_train_study[[\"study_id\", \"label\"]]\n\n# Combine the two dataframes\ndf = df_train_image.merge(\n    df_train_study,\n    how=\"outer\",\n    on=\"study_id\"\n)\n\ndf","metadata":{"execution":{"iopub.status.busy":"2021-07-08T10:05:15.496816Z","iopub.execute_input":"2021-07-08T10:05:15.497574Z","iopub.status.idle":"2021-07-08T10:05:41.463125Z","shell.execute_reply.started":"2021-07-08T10:05:15.49752Z","shell.execute_reply":"2021-07-08T10:05:41.462317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dataframe for test data.\n\ndf_test = pd.DataFrame(columns=[\"image_id\", \"study_id\", \"filepath\"])\n\ni = 0\nfor dirpath, dirs, files in os.walk(f\"{root}/test\"):\n    if not files:\n        # Ignore folders without files\n        continue\n    study_id = dirpath.split(\"/\")[-2]\n    for filename in files:\n        image_id = filename.split(\".\")[0]\n        filepath = f\"{dirpath}/{filename}\"\n        df_test.loc[i] = [image_id, study_id, filepath]\n        i += 1\n        \ndf_test","metadata":{"execution":{"iopub.status.busy":"2021-07-08T10:05:41.464397Z","iopub.execute_input":"2021-07-08T10:05:41.464824Z","iopub.status.idle":"2021-07-08T10:05:50.964067Z","shell.execute_reply.started":"2021-07-08T10:05:41.464794Z","shell.execute_reply":"2021-07-08T10:05:50.96313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_image(filepath):\n    \"\"\"Read the file to memory and process it.\n    \n    Parameters\n    ----------\n    filepath : str\n        Path the .dcm file.\n    \n    Returns\n    -------\n    PIL.Image\n    \"\"\"\n    # Read the dcm file.\n    dicom = pydicom.read_file(filepath)\n        \n    # Get the array. The array is by default uint dtype. This causes problems\n    # in inverting the colors in the next step. Thus, for convinience converting\n    # the dtype to be int which allows negative values.\n    im = dicom.pixel_array.astype(int)\n\n    # Make black 0 and white the largest number.\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        im = np.abs(im - im.max())\n        \n    # Convert the range to integers from 0 to 255\n    im = ((im - im.min()) / (im.max() - im.min()) * 255).astype(np.uint8)\n        \n    # Convert to PIL image for easier processing later on.\n    im = Image.fromarray(im)\n    \n    return im\n\n\ndef resize_image(im, target_shape):\n    \"\"\"Resize a given image.\n    \n    First rescales the image such that either width or height matches the \n    target_shape.\n    \n    Parameters\n    ----------\n    im : PIL.Image\n    target_shape : tuple of ints\n        (height, width)\n    \n    Returns\n    -------\n    PIL.Image, (x_scale, y_scale), (x_shift, y_shift)\n    \"\"\"\n    original_shape = np.array(im).shape  # Get the shape for bounding box calculations.\n\n    im.thumbnail(target_shape, Image.ANTIALIAS)\n\n    tmp_shape = np.array(im).shape  # Get the current shape for further calculations.\n    \n    # The scales for bounding box calculations\n    y_scale = tmp_shape[0] / original_shape[0]\n    x_scale = tmp_shape[1] / original_shape[1]\n    \n    # Place the resized image at the center of a larger image.\n    im_array = np.zeros(target_shape, dtype=np.uint8)\n    y0 = (target_shape[0] - tmp_shape[0]) // 2\n    y1 = int(np.floor(target_shape[0] - (target_shape[0] - tmp_shape[0]) / 2))\n    x0 = (target_shape[1] - tmp_shape[1]) // 2\n    x1 = int(np.floor(target_shape[1] - (target_shape[1] - tmp_shape[1]) / 2))\n    im_array[y0:y1, x0:x1] = np.array(im)\n    \n    im = Image.fromarray(im_array)\n    \n    return im, (x_scale, y_scale), (x0, y0)","metadata":{"execution":{"iopub.status.busy":"2021-07-08T10:59:16.086735Z","iopub.execute_input":"2021-07-08T10:59:16.087216Z","iopub.status.idle":"2021-07-08T10:59:16.100843Z","shell.execute_reply.started":"2021-07-08T10:59:16.087176Z","shell.execute_reply":"2021-07-08T10:59:16.099752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TESTING\nfig, axs = plt.subplots(2, 5, figsize=(20, 10))\n\nfor i, row in df.sample(5).reset_index().iterrows():\n    im = get_image(row[\"filepath\"])\n    axs[0, i].imshow(im, cmap=plt.cm.bone)\n    axs[0, i].set_title(row[\"label\"])\n    for box in row[\"boxes\"]:\n        rect = Rectangle((box[\"x\"], box[\"y\"]), box[\"width\"], box[\"height\"] ,linewidth=1, edgecolor='r', facecolor='none')\n        axs[0, i].add_patch(rect)\n        \n    im, (x_scale, y_scale), (x0, y0) = resize_image(im, target_shape)\n    axs[1, i].imshow(im, cmap=plt.cm.bone)\n    for box in row[\"boxes\"]:\n        x = box[\"x\"] * x_scale + x0\n        y = box[\"y\"] * y_scale + y0\n        width = box[\"width\"] * x_scale\n        height = box[\"height\"] * y_scale\n        rect = Rectangle((x, y), width, height ,linewidth=1, edgecolor='r', facecolor='none')\n        axs[1, i].add_patch(rect)","metadata":{"execution":{"iopub.status.busy":"2021-07-08T10:59:36.418437Z","iopub.execute_input":"2021-07-08T10:59:36.418873Z","iopub.status.idle":"2021-07-08T10:59:46.487174Z","shell.execute_reply.started":"2021-07-08T10:59:36.41884Z","shell.execute_reply":"2021-07-08T10:59:46.486169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the directories\ndir_data = \"/kaggle/working/data\"\ndir_train = f\"{dir_data}/train\"\nos.makedirs(dir_train, exist_ok=True)\nfor label in column_to_label.values():\n    os.makedirs(f\"{dir_train}/{label}\", exist_ok=True)\n\ndir_test = f\"{dir_data}/test\"\nos.makedirs(dir_test, exist_ok=True)\n\nprogress_bar = tqdm(range(df.shape[0] + df_test.shape[0]))\n\n# Go through all the training images\nfor index, row in df.iterrows():\n    # Read the image\n    filepath = row[\"filepath\"]\n    im = get_image(filepath)\n\n    im, (x_scale, y_scale), (x0, y0) = resize_image(im, target_shape)\n    \n    # Modify the boxes\n    for box in row[\"boxes\"]:\n        box[\"x\"] = box[\"x\"] * x_scale + x0\n        box[\"y\"] = box[\"y\"] * y_scale + y0\n        box[\"width\"] *= x_scale\n        box[\"height\"] *= y_scale\n    \n    # Save the image to new location.\n    label = row[\"label\"]\n    new_filename = os.path.basename(filepath).split(\".\")[0] + \".png\"\n    im.save(f\"{dir_train}/{label}/{new_filename}\")\n\n    progress_bar.update(1)\n\n\n# Go through all the test images.\nfor index, row in df_test.iterrows():\n    # Read the image\n    filepath = row[\"filepath\"]\n    im = get_image(filepath)\n    \n    im, (x_scale, y_scale), (x0, y0) = resize_image(im, target_shape)\n    \n    # Save the image to new location.\n    new_filename = os.path.basename(filepath).split(\".\")[0] + \".png\"\n    im.save(f\"{dir_test}/{new_filename}\")\n\n    progress_bar.update(1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the dataframes as csv\ndf.drop(\"filepath\", axis=1).to_csv(f\"{dir_data}/train.csv\")\ndf_test.drop(\"filepath\", axis=1).to_csv(f\"{dir_data}/test.csv\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Zip the images so that it is possible to create a new dataset in Kaggle.\nshutil.make_archive(\"/kaggle/working/data\", 'zip', dir_data)\n\n# Remove the data folder so that the output zip appears as output. For some\n# reasong Kaggle will just show the images.\nshutil.rmtree(dir_data)","metadata":{},"execution_count":null,"outputs":[]}]}