{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Objective\n\nIn order to normalize the images prior to processing, we need to compute the mean and variance for all the population data that we have.\n\nWe choose to include the test data in this because it gives us more information by which we can measure the stats of the population.\n\nIf in the future, there are volumes that have significantly different stats from the population that we compute, we may want to recompute the stats and use the updated stats to re-train models.","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport tifffile as tiff\nimport cv2\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-05-05T19:03:56.668829Z","iopub.execute_input":"2023-05-05T19:03:56.669739Z","iopub.status.idle":"2023-05-05T19:03:57.241647Z","shell.execute_reply.started":"2023-05-05T19:03:56.669702Z","shell.execute_reply":"2023-05-05T19:03:57.240576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_PATH = \"/kaggle/input/vesuvius-challenge-ink-detection\"","metadata":{"execution":{"iopub.status.busy":"2023-05-05T19:03:57.245874Z","iopub.execute_input":"2023-05-05T19:03:57.246201Z","iopub.status.idle":"2023-05-05T19:03:57.252492Z","shell.execute_reply.started":"2023-05-05T19:03:57.246174Z","shell.execute_reply":"2023-05-05T19:03:57.251449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_mask(volume_folder: str):\n    mask = np.array(cv2.imread(f\"{volume_folder}/mask.png\", 0)) / 255\n    mask[mask != 1] = 0\n    return mask\n\ndef load_masked_image(fname: str, mask):\n    img_array = np.array(tiff.imread(fname)) * mask\n    return img_array\n\ndef get_volume_surface_fnames(volume_folder: str):\n    fnames = []\n    for dirname, _, filenames in os.walk(f\"{volume_folder}/surface_volume\"):\n        for filename in filenames:\n            fnames.append(os.path.join(dirname, filename))\n    return fnames\n\ndef get_all_volume_surfaces():\n    folders = [\n        f\"{DATASET_PATH}/train/1\",\n        f\"{DATASET_PATH}/train/2\",\n        f\"{DATASET_PATH}/train/3\",\n        f\"{DATASET_PATH}/test/a\",\n        f\"{DATASET_PATH}/test/b\"\n    ]\n    \n    jobs = []\n    for folder in folders:\n        fnames = get_volume_surface_fnames(folder)\n        for fname in fnames:\n            jobs.append((folder, fname))\n    \n    return jobs\n\ndef get_mask_sums(jobs):\n    mask_info = {}\n    total_mask_sum = 0\n    for i in range(len(jobs)):\n        (volume_folder, surface_fname) = jobs[i]\n        if volume_folder not in mask_info:\n            mask = load_mask(volume_folder)\n            mask_sum = mask.sum()\n            mask_info[volume_folder] = {\n                \"mask_sum\": mask_sum,\n                \"mask\": mask\n            }\n                \n            total_mask_sum = total_mask_sum + mask_sum\n        else:\n            mask_sum = mask_info[volume_folder][\"mask_sum\"]\n            total_mask_sum = total_mask_sum + mask_sum\n\n    return mask_info, total_mask_sum\n\ndef compute_statistics_for_all_volumes():\n    jobs = get_all_volume_surfaces()\n    mask_info, total_mask_sum = get_mask_sums(jobs)\n\n    weighted_sums = np.zeros((len(jobs)))\n    for i in tqdm(range(len(jobs))):\n        (volume_folder, surface_fname) = jobs[i]\n        info = mask_info[volume_folder]\n        img_array = load_masked_image(surface_fname, info[\"mask\"])\n        weighted_sums[i] = img_array.mean() * (info[\"mask_sum\"] / total_mask_sum)\n        \n    mean = weighted_sums.sum()\n    print(\"mean\", mean)\n\n    weighted_variances = np.zeros((len(jobs)))\n    for i in tqdm(range(len(jobs))):\n        (volume_folder, surface_fname) = jobs[i]\n        info = mask_info[volume_folder]\n        img_array = load_masked_image(surface_fname, info[\"mask\"])\n\n        # NOTE: we use the mask to ignore pixels not in the mask\n        mean_diff = (img_array - mean) * info[\"mask\"]\n        weighted_variances[i] = ((mean_diff ** 2) * (info[\"mask_sum\"] / total_mask_sum)).mean() \n    \n    variance = np.sqrt(weighted_variances.sum())\n    print(\"variance\", variance)\n\n    return mean, variance\n","metadata":{"execution":{"iopub.status.busy":"2023-05-05T19:03:57.253889Z","iopub.execute_input":"2023-05-05T19:03:57.255683Z","iopub.status.idle":"2023-05-05T19:03:57.273152Z","shell.execute_reply.started":"2023-05-05T19:03:57.255620Z","shell.execute_reply":"2023-05-05T19:03:57.272174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"compute_statistics_for_all_volumes()","metadata":{"execution":{"iopub.status.busy":"2023-05-05T19:03:57.276030Z","iopub.execute_input":"2023-05-05T19:03:57.276906Z","iopub.status.idle":"2023-05-05T19:19:42.686109Z","shell.execute_reply.started":"2023-05-05T19:03:57.276866Z","shell.execute_reply":"2023-05-05T19:19:42.684869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can use this mean + variance to normalize data when loading during training and evaluation.","metadata":{}}]}