{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":30201,"databundleVersionId":2750748,"sourceType":"competition"}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Preparación de datos**","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"## Introducción <a class=\"anchor\"  id=\"chapter1\"></a>","metadata":{}},{"cell_type":"markdown","source":"Notebook basado en https://www.kaggle.com/code/abrachan/sartorius-unet-from-scratch-with-pytorch\nEl objetivo es: preparación de los datos.\n1. Explorar datos del dataset: EDA (Exploratory Data Analysis)","metadata":{}},{"cell_type":"markdown","source":"## Importamos librerías <a class=\"anchor\"  id=\"chapter2\"></a>","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport time\nimport random\nimport collections\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\n\nimport math\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom math import atan2, pi\n\nfrom sklearn.model_selection import KFold\n# from sklearn.model_selection import train_test_split\n\nimport cv2\nfrom PIL import Image\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n# from albumentations import HorizontalFlip, VerticalFlip, ShiftScaleRotate, Normalize, Resize, Compose, GaussNoise\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\n\nimport torchvision\nfrom torchvision.transforms import ToPILImage\nfrom torchvision.transforms import functional as F\nfrom torchvision.models.detection.faster_rcnn import FastRCNNPredictor\nfrom torchvision.models.detection.mask_rcnn import MaskRCNNPredictor\n\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Ajusta el tamaño de la letra según tu preferencia\ntamaño_de_letra = 14","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:43.23445Z","iopub.execute_input":"2024-04-20T18:00:43.234684Z","iopub.status.idle":"2024-04-20T18:00:52.089335Z","shell.execute_reply.started":"2024-04-20T18:00:43.234662Z","shell.execute_reply":"2024-04-20T18:00:52.088189Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'PyTorch version: {torch.__version__}')\nprint(f'CUDA avaliable: {torch.cuda.is_available()}')","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:52.091283Z","iopub.execute_input":"2024-04-20T18:00:52.092117Z","iopub.status.idle":"2024-04-20T18:00:52.152051Z","shell.execute_reply.started":"2024-04-20T18:00:52.092084Z","shell.execute_reply":"2024-04-20T18:00:52.150822Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Fix randomness <a class=\"anchor\"  id=\"chapter3\"></a>","metadata":{}},{"cell_type":"code","source":"# Fix randomness\n\ndef fix_all_seeds(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    # torch.backends.cudnn.deterministic = True\n    # torch.backends.cudnn.benchmark = True\n    \n    os.environ['PYTHONHASHSEED'] = str(seed)\n\n\n# Set seed\nfix_all_seeds(2023)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:52.15334Z","iopub.execute_input":"2024-04-20T18:00:52.153665Z","iopub.status.idle":"2024-04-20T18:00:52.164695Z","shell.execute_reply.started":"2024-04-20T18:00:52.153638Z","shell.execute_reply":"2024-04-20T18:00:52.163676Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Configurations <a class=\"anchor\"  id=\"chapter4\"></a>","metadata":{}},{"cell_type":"code","source":"# Directory setting\nDATA_DIR = '/kaggle/input/sartorius-cell-instance-segmentation/'\n\nTRAIN_CSV = DATA_DIR + 'train.csv'\nTRAIN_PATH = DATA_DIR + 'train/'\nTEST_PATH = DATA_DIR + 'test/'\n\nMODEL_DIR = '/kaggle/working/'\n\nORIENTATION_DIR = '/kaggle/working/'\n\nFOLDS = 3                # Kfold cross-validation, here using a low value of 3 just for demonstration.\nNUM_WORKERS = 0          # 2\nTHRESHOLD_MASK = 0.5     # Threshold for mask prediction\nBATCH_SIZE = 20\nEPOCHS = 8               # here using a low value of 8 just for demonstration.\n\nUSE_SCHEDULER = False  # Use a StepLR scheduler if True. Not tried yet.\nMOMENTUM = 0.9\nLEARNING_RATE = 0.0001\nWEIGHT_DECAY = 0.0001\n\n## Set device\n# device = torch.device('cuda' if torch.cuda.is_available() else 'cpu') \nDEVICE = torch.device('cuda') if torch.cuda.is_available() else torch.device('cpu')\nprint(f'Using {DEVICE} device')","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:52.167722Z","iopub.execute_input":"2024-04-20T18:00:52.168155Z","iopub.status.idle":"2024-04-20T18:00:52.17521Z","shell.execute_reply.started":"2024-04-20T18:00:52.168129Z","shell.execute_reply":"2024-04-20T18:00:52.174172Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Using cuda device","metadata":{}},{"cell_type":"code","source":"torch.cuda.get_device_properties(DEVICE)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:52.176448Z","iopub.execute_input":"2024-04-20T18:00:52.176835Z","iopub.status.idle":"2024-04-20T18:00:52.215931Z","shell.execute_reply.started":"2024-04-20T18:00:52.176802Z","shell.execute_reply":"2024-04-20T18:00:52.214882Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Helper functions <a class=\"anchor\"  id=\"chapter5\"></a>","metadata":{}},{"cell_type":"markdown","source":"### rle_decode()","metadata":{}},{"cell_type":"markdown","source":"Decode the RLE (annotation) of an particular cell instance in an image to its correspongding mask.","metadata":{}},{"cell_type":"code","source":"def rle_decode(rle, img_shape, color=1):\n    \"\"\"Decode the RLE (annotation) of an particular cell instance in an image to its correspongding mask.\n\n    Args:\n        rle (str): mask with run length encoding.\n        img_shape ((int, int)): (height, width) of the image, also the shape of mask np.ndarray to return.\n        color (int): brightness of the mask pixel. Default to 1.\n\n    Returns:\n        np.ndarray: 1 - mask, 0 - background.\n    \"\"\"\n    rle_list = rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (rle_list[0:][::2], rle_list[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    \n    mask = np.zeros(img_shape[0] * img_shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        mask[lo:hi] = color\n    \n    return mask.reshape(img_shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:52.217173Z","iopub.execute_input":"2024-04-20T18:00:52.217468Z","iopub.status.idle":"2024-04-20T18:00:52.225691Z","shell.execute_reply.started":"2024-04-20T18:00:52.217444Z","shell.execute_reply":"2024-04-20T18:00:52.224415Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### rle_encode()","metadata":{}},{"cell_type":"code","source":"def rle_encode(predicted_img):\n    predicted_img = (predicted_img > THRESHOLD_MASK).astype(int)\n    height, width = predicted_img.shape\n    \n    # Get the index of the masked pixel\n    pixels = predicted_img.copy()\n    pixels_list = []\n    for y in range(height):\n        for x in range(width):\n            if pixels[y][x] != 0:\n                pixels_list.append(y * width + x)\n    \n    \n    # RLE encoding\n    rle_list = []\n    start = pixels_list[0]\n    count = 1\n    for i in range(1, len(pixels_list)):\n        if pixels_list[i] == pixels_list[i-1] + 1:\n            count += 1\n        else:\n            rle_list.extend([start, count])\n            start = pixels_list[i]\n            count = 1\n    rle_list.extend([start, count])\n    \n    rle_str = [str(x) for x in rle_list]\n    \n    return ' '.join(rle_str)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:52.227088Z","iopub.execute_input":"2024-04-20T18:00:52.227471Z","iopub.status.idle":"2024-04-20T18:00:52.238126Z","shell.execute_reply.started":"2024-04-20T18:00:52.227438Z","shell.execute_reply":"2024-04-20T18:00:52.236983Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### build_image_mask","metadata":{}},{"cell_type":"markdown","source":"Decode RLEs (annotations) of all cell instance in an image into one mask image.","metadata":{}},{"cell_type":"code","source":"def build_image_mask(img, rle_list):\n    \"\"\"Decode RLEs (annotations) of all cell instance in an image into one mask image.\n    \n    Args:\n        img (np.ndarray): image with single channel or multiple channels.\n        rle_list (list of str): rles of all cell instances in an image as a list.\n    \n    Returns:\n        np.ndarray: 1 - mask, 0 - background.\n    \"\"\"\n    img_shape = img.shape\n    h = img_shape[0]\n    w = img_shape[1]\n    \n    mask = np.zeros((h, w))\n    for rle in rle_list:\n        mask += rle_decode(rle, (h, w))\n    mask = mask.clip(0,1)\n    \n    return mask","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:52.239604Z","iopub.execute_input":"2024-04-20T18:00:52.240096Z","iopub.status.idle":"2024-04-20T18:00:52.248441Z","shell.execute_reply.started":"2024-04-20T18:00:52.24006Z","shell.execute_reply":"2024-04-20T18:00:52.24761Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### transforms()","metadata":{}},{"cell_type":"markdown","source":"For segmentation tasks, the segmentation masks also need to undergo transformations and augmentations. Please also see Albumentations Documentation/Mask augmentation for segmentation. In this notebook, transforms are applied in the CellDataset() class.","metadata":{}},{"cell_type":"code","source":"def transforms(train_only=True):\n    \"\"\"converts the image, a PIL image, into a PyTorch Tensor\"\"\"\n    \n    transforms = [A.Resize(256, 256, p=1)]\n    \n    if train_only:\n        ## during training, randomly flip the training images and ground-truth for data augmentation\n        transforms.append(A.HorizontalFlip(p=0.5))\n        transforms.append(A.VerticalFlip(p=0.5))\n        transforms.append(A.Transpose(p=0.5))\n    else:\n        ## for validation phase and test phase\n        pass\n    \n    transforms.append(ToTensorV2())\n    \n    return A.Compose(transforms)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:52.249843Z","iopub.execute_input":"2024-04-20T18:00:52.25018Z","iopub.status.idle":"2024-04-20T18:00:52.257431Z","shell.execute_reply.started":"2024-04-20T18:00:52.250154Z","shell.execute_reply":"2024-04-20T18:00:52.256366Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### plot_train_val_loss()","metadata":{}},{"cell_type":"code","source":"def plot_train_val_loss(train_loss_fold_list, valid_loss_fold_list, \n                        save=False, save_dir=\"./save_images/\", save_name='segmentation_train_val_loss.png'):\n    fig = plt.figure(figsize=(10,10))\n    for i in range(FOLDS):\n        train_loss = train_loss_fold_list[i]\n        valid_loss = valid_loss_fold_list[i]\n        \n        ax = fig.add_subplot(math.ceil(np.sqrt(FOLDS)), math.ceil(np.sqrt(FOLDS)), i+1, title=f'Fold {i+1}')\n        ax.plot(range(EPOCHS), train_loss, c='orange', label='train')\n        ax.plot(range(EPOCHS), valid_loss, c='blue', label='valid')\n        ax.set_xlabel('epoch')\n        ax.set_ylabel('loss')\n        ax.legend()\n        \n    plt.tight_layout()\n    if save:\n        os.makedirs(save_dir, exist_ok=True)\n        plt.savefig(save_dir+save_name)\n    else:\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:52.262172Z","iopub.execute_input":"2024-04-20T18:00:52.262949Z","iopub.status.idle":"2024-04-20T18:00:52.271575Z","shell.execute_reply.started":"2024-04-20T18:00:52.262918Z","shell.execute_reply":"2024-04-20T18:00:52.270568Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Explorar datos del dataset: EDA (Exploratory Data Analysis) <a class=\"anchor\"  id=\"chapter6\"></a>","metadata":{}},{"cell_type":"markdown","source":"Primero vamos a hacer EDA (Exploratory Data Analysis) del train.csv para poder entender mejor el dataset de entrenamiento.","metadata":{}},{"cell_type":"code","source":"# Leemos el .csv\ndf_train = pd.read_csv(TRAIN_CSV)\npd.concat([df_train.head(), df_train.tail()])","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:52.272711Z","iopub.execute_input":"2024-04-20T18:00:52.273133Z","iopub.status.idle":"2024-04-20T18:00:52.935546Z","shell.execute_reply.started":"2024-04-20T18:00:52.273095Z","shell.execute_reply":"2024-04-20T18:00:52.934612Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Tenemos 9 columnas, formadas por el id, annotation, width, height, plate_time, sample_date, sample_id y elapsed_timedelta.","metadata":{}},{"cell_type":"code","source":"# Vemos el tamaño de la \"tabla\"\ndf_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:52.937071Z","iopub.execute_input":"2024-04-20T18:00:52.937749Z","iopub.status.idle":"2024-04-20T18:00:52.944185Z","shell.execute_reply.started":"2024-04-20T18:00:52.937712Z","shell.execute_reply":"2024-04-20T18:00:52.943123Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Vemos el tipo de las variables\ndf_train.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:52.945472Z","iopub.execute_input":"2024-04-20T18:00:52.946183Z","iopub.status.idle":"2024-04-20T18:00:52.954322Z","shell.execute_reply.started":"2024-04-20T18:00:52.946152Z","shell.execute_reply":"2024-04-20T18:00:52.953259Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Todas las variables son de tipo objeto excepto el ancho y alto de la imagen, que son de tipo entero (int64).","metadata":{}},{"cell_type":"markdown","source":"#### Cell type","metadata":{}},{"cell_type":"code","source":"# Vemos los tipos de céuluas, parece que hay más células sh5y que cort y astro\ndf_train.cell_type.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:52.955719Z","iopub.execute_input":"2024-04-20T18:00:52.956162Z","iopub.status.idle":"2024-04-20T18:00:52.97888Z","shell.execute_reply.started":"2024-04-20T18:00:52.95613Z","shell.execute_reply":"2024-04-20T18:00:52.977794Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Gráfico de barras de esta distribución\ncell_types = ['shsy5y', 'cort', 'astro']\ncounts = [52286, 10777, 10522]\n\nplt.figure(figsize=(10, 6))\n\n# Crear el gráfico de barras\nplt.bar(cell_types, counts, color=['blue', 'orange', 'green'])\n\n# Añadir etiquetas y título\n\nplt.xlabel('Tipo de célula', fontsize=tamaño_de_letra)\nplt.ylabel('Células', fontsize=tamaño_de_letra)\nplt.title('Distribución de los tipos de células', fontsize=tamaño_de_letra)\n# Ajustar el tamaño de letra de las etiquetas en los ejes x e y\nplt.xticks(fontsize=tamaño_de_letra)\nplt.yticks(fontsize=tamaño_de_letra)\nplt.show()\n\n# Mostrar el gráfico\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T19:01:26.075715Z","iopub.execute_input":"2024-04-20T19:01:26.076704Z","iopub.status.idle":"2024-04-20T19:01:26.271237Z","shell.execute_reply.started":"2024-04-20T19:01:26.076666Z","shell.execute_reply":"2024-04-20T19:01:26.270236Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Hay tres tipos de células, shsy5y que es la célula del neuroblastoma, cort que son las neuronas corticales y astro que son los astrocitos. Las células más representadas son las células del neuroblastoma, shsy5y.","metadata":{}},{"cell_type":"code","source":"# Tipos de células\ndf_train.cell_type.unique()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:53.190444Z","iopub.execute_input":"2024-04-20T18:00:53.190763Z","iopub.status.idle":"2024-04-20T18:00:53.204373Z","shell.execute_reply.started":"2024-04-20T18:00:53.190726Z","shell.execute_reply":"2024-04-20T18:00:53.203265Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Annotation (RLE)","metadata":{}},{"cell_type":"markdown","source":"La segmentacuón de los datos de entrenamiento se proporciona con Run length encoding (RLE) en la columna de anotación.\n\nRLE es una técnica de compresión sin pérdidas utilizada para representar datos que contienen secuencias largas de valores o caracteres repetidos.\n\nEl conjunto de datos actual tiene una segmentación y por tanto una célula por fila.","metadata":{}},{"cell_type":"code","source":"# Visualización de las anotciones\ndf_instances = df_train.groupby(['id']).agg({'annotation': 'count', 'cell_type': 'first'})\ndf_instances.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:53.205806Z","iopub.execute_input":"2024-04-20T18:00:53.206199Z","iopub.status.idle":"2024-04-20T18:00:53.253663Z","shell.execute_reply.started":"2024-04-20T18:00:53.206164Z","shell.execute_reply":"2024-04-20T18:00:53.25273Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tamaño tabla de anotaciones esto implica que tenemos un total de 606 imágenes\ndf_instances.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:53.255103Z","iopub.execute_input":"2024-04-20T18:00:53.255519Z","iopub.status.idle":"2024-04-20T18:00:53.262063Z","shell.execute_reply.started":"2024-04-20T18:00:53.255484Z","shell.execute_reply":"2024-04-20T18:00:53.26112Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Anotaciones por tipo de célula\ndf_instances_pentiles = df_train.groupby(['id']).agg({'annotation': 'count', 'cell_type': 'first'})\ndf_instances_pentiles = df_instances_pentiles.groupby(\"cell_type\")[['annotation']]\\\n                                             .describe(percentiles=[0.1, 0.25, 0.75, 0.8, 0.85, 0.9, 0.95, 0.99]).astype(int)\\\n                                             .T.droplevel(level=0).T.drop(['count', '50%', 'std'], axis=1)\ndf_instances_pentiles","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:53.263466Z","iopub.execute_input":"2024-04-20T18:00:53.264332Z","iopub.status.idle":"2024-04-20T18:00:53.334849Z","shell.execute_reply.started":"2024-04-20T18:00:53.264302Z","shell.execute_reply":"2024-04-20T18:00:53.333815Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Trying with different strategies\ndf_instances_pentiles['90%'].to_dict()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Como podemos observar, en media hay muchas más anotaciones del tipo celular shsy5y. ","metadata":{}},{"cell_type":"code","source":"order = ['shsy5y', 'cort', 'astro']\n\n# Calcular la desviación estándar por tipo de célula\ndf_instances_pentiles = df_train.groupby(['id']).agg({'annotation': 'count', 'cell_type': 'first'})\nstd_cell_img = df_instances_pentiles.groupby(\"cell_type\")['annotation'].std()\nmean_cell_img = df_instances_pentiles.groupby(\"cell_type\")['annotation'].mean()\n\n# Reordenar los resultados según el orden deseado\nmean_cell_img  = mean_cell_img .reindex(order)\nstd_cell_img = std_cell_img.reindex(order)\n\nplt.figure(figsize=(10, 6))\n\n# Crear el gráfico de barras\nplt.bar(cell_types, mean_cell_img,  yerr=std_cell_img, capsize=5, color=['blue', 'orange', 'green'])\n\n# Añadir etiquetas y título\nplt.xlabel('Tipo de célula', fontsize=tamaño_de_letra)\nplt.ylabel('Media de células por imagen' , fontsize=tamaño_de_letra)\nplt.title('Media y desviación típica del número de células por imagen y tipo de célula', fontsize=tamaño_de_letra)\n# Ajustar el tamaño de letra de las etiquetas en los ejes x e y\nplt.xticks(fontsize=tamaño_de_letra)\nplt.yticks(fontsize=tamaño_de_letra)\n# Mostrar el gráfico\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T19:01:12.113166Z","iopub.execute_input":"2024-04-20T19:01:12.11395Z","iopub.status.idle":"2024-04-20T19:01:12.344841Z","shell.execute_reply.started":"2024-04-20T19:01:12.113915Z","shell.execute_reply":"2024-04-20T19:01:12.343922Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Pixels Per Mask Per cell_type","metadata":{}},{"cell_type":"code","source":"# Número de píxeles por tipo de célula y máscara\ndf_train['n_pixels'] = df_train.annotation.apply(lambda x: np.sum([int(e) for e in x.split()[1:][::2]]))\n\ndf_pixels = df_train.groupby(\"cell_type\")[['n_pixels']].describe(percentiles=[0.02, 0.05, 0.1, 0.9, 0.95, 0.98])\\\n                    .astype(int).T.droplevel(level=0).T.drop(['count', '50%', 'std'], axis=1)\n\ndf_pixels","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:54.003487Z","iopub.execute_input":"2024-04-20T18:00:54.003841Z","iopub.status.idle":"2024-04-20T18:00:55.76873Z","shell.execute_reply.started":"2024-04-20T18:00:54.003806Z","shell.execute_reply":"2024-04-20T18:00:55.767703Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Tamaño en píxeles medio de cada tipo de célula","metadata":{}},{"cell_type":"code","source":"# Calcular la media por tipo de célula\nmedia_por_tipo = df_train.groupby(\"cell_type\")['n_pixels'].mean()\n\n# Calcular la desviación estándar por tipo de célula\nstd_por_tipo = df_train.groupby(\"cell_type\")['n_pixels'].std()\n\n# Definir el orden deseado de los tipos de célula\norder = ['shsy5y', 'cort', 'astro']\n\n# Reordenar los resultados según el orden deseado\nmedia_por_tipo = media_por_tipo.reindex(order)\nstd_por_tipo = std_por_tipo.reindex(order)\n\n# Crear el gráfico de barras\nplt.figure(figsize=(10, 6))\nplt.bar(media_por_tipo.index, media_por_tipo.values, yerr=std_por_tipo.values, capsize=5, color=['blue', 'orange', 'green'])\n\n# Añadir etiquetas y título\nplt.xlabel('Tipo de célula', fontsize=tamaño_de_letra)\nplt.ylabel('Tamaño medio de la célula (píxeles)', fontsize=tamaño_de_letra)\nplt.title('Tamaño medio y desviación típica de las células según el tipo de célula', fontsize=tamaño_de_letra)\n\n# Ajustar el tamaño de letra de las etiquetas en los ejes x e y\nplt.xticks(fontsize=tamaño_de_letra)\nplt.yticks(fontsize=tamaño_de_letra)\n\n# Mostrar el gráfico\nplt.xticks(rotation=45)  # Rotar las etiquetas del eje x para mayor legibilidad\nplt.tight_layout()  # Ajustar el diseño para evitar solapamientos\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:54:58.312638Z","iopub.execute_input":"2024-04-20T18:54:58.313058Z","iopub.status.idle":"2024-04-20T18:54:58.554513Z","shell.execute_reply.started":"2024-04-20T18:54:58.313027Z","shell.execute_reply":"2024-04-20T18:54:58.553684Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Hay mayor número de píxeles en las células de tipo astrocitos, las células de tipo cort y shsy5y tienen prácticamente el mismo número de píxeles.","metadata":{}},{"cell_type":"markdown","source":"#### Porcentaje de píxeles célula sobre fondo","metadata":{}},{"cell_type":"code","source":"# Obtener rutas de imágenes en el directorio de entrenamiento\nimg_paths = Path(TRAIN_PATH).glob(\"*\")\n\n# Inicializar un diccionario para almacenar el promedio de background_pixels por cell_type\navg_background_pixels = {}\n\n# Inicializar una lista para almacenar los datos\nbackground_percentages = []\n\n# Definir un diccionario para almacenar las frecuencias\nfrecuencia = {1: 0, 2: 0, 3: 0, 4: 0}\n\n# Iterar sobre las rutas de las imágenes\nfor img_path in tqdm(img_paths, total=df_train.id.unique().shape[0]):\n    # Obtener el ID de la imagen\n    img_id = img_path.stem\n    \n    # Leer la imagen en modo BGR\n    img_bgr = cv2.imread(img_path.as_posix())\n    \n    # Obtener las anotaciones (máscaras de segmentación) asociadas a la imagen\n    annotations = df_train[df_train['id'] == img_id]['annotation'].tolist()\n    \n    # Construir la máscara combinada de todas las anotaciones de la imagen\n    mask = build_image_mask(img_bgr, annotations)\n    \n    # Calcular el número de píxeles de fondo y célula de la máscara\n    total_pixels = mask.size\n    background_pixels = np.count_nonzero(mask == 0)\n    cell_pixels = total_pixels - background_pixels\n    \n    # Calcular el porcentaje de píxeles de fondo y célula de la máscara\n    percent_background = (background_pixels / total_pixels) * 100\n    percent_cell = (cell_pixels / total_pixels) * 100\n    \n    # Obtener el cell_type asociado a la imagen\n    cell_type = df_train[df_train['id'] == img_id]['cell_type'].iloc[0]\n    \n    # Porcentaje de pixeles de fondo independientemente del tipo celular\n    background_percentages.append(percent_background)\n    \n    # Porcentaje de pixeles de fondo independientemente del tipo celular\n    # Incrementar el valor en el diccionario frecuencia según el porcentaje de fondo\n    if percent_background >= 0 and percent_background < 25:\n        frecuencia[1] += 1\n    elif percent_background >= 25 and percent_background < 50:\n        frecuencia[2] += 1\n    elif percent_background >= 50 and percent_background < 75:\n        frecuencia[3] += 1\n    elif percent_background >= 75 and percent_background <= 100:\n        frecuencia[4] += 1\n    \n    # Agregar el promedio de background_pixels al diccionario por cell_type\n    if cell_type not in avg_background_pixels:\n        avg_background_pixels[cell_type] = []\n    avg_background_pixels[cell_type].append(percent_background)\n\n# Calcular los estadísticos descriptivos para cada tipo celular\nfor cell_type, pixels_list in avg_background_pixels.items():\n    df = pd.DataFrame(pixels_list, columns=['Background Pixels'])\n    descriptive_stats = df.describe()\n    print(f\"Cell Type: {cell_type}\")\n    print(descriptive_stats)\n    print()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:33:59.170071Z","iopub.execute_input":"2024-04-20T18:33:59.170885Z","iopub.status.idle":"2024-04-20T18:34:52.173125Z","shell.execute_reply.started":"2024-04-20T18:33:59.170846Z","shell.execute_reply":"2024-04-20T18:34:52.172176Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular el promedio de background_pixels por cell_type\nmean_background_pixels = {cell_type: np.mean(values) for cell_type, values in avg_background_pixels.items()}\nstd_background_pixels = {cell_type: np.std(values) for cell_type, values in avg_background_pixels.items()}\n\norder = ['shsy5y', 'cort', 'astro']\n\n# Obtener los valores de media y desviación estándar en el orden deseado\nmean_background_pixels = [mean_background_pixels[cell_type] for cell_type in order]\nstd_background_pixels = [std_background_pixels[cell_type] for cell_type in order]\n\n# Crear un gráfico de barras con la desviación estándar como error\nplt.figure(figsize=(10, 6))\nplt.bar(order, mean_background_pixels, yerr=std_background_pixels, color=['blue', 'orange', 'green'])\nplt.xlabel('Tipo de célula', fontsize=tamaño_de_letra)\nplt.ylabel('Media del porcentaje de píxeles de fondo', fontsize=tamaño_de_letra)\nplt.title('Media del porcentaje de píxeles de fondo por tipo de célula', fontsize=tamaño_de_letra)\nplt.xticks(rotation=45)\n# Ajustar el tamaño de letra de las etiquetas en los ejes x e y\nplt.xticks(fontsize=tamaño_de_letra)\nplt.yticks(fontsize=tamaño_de_letra)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T19:00:38.80155Z","iopub.execute_input":"2024-04-20T19:00:38.802295Z","iopub.status.idle":"2024-04-20T19:00:39.049124Z","shell.execute_reply.started":"2024-04-20T19:00:38.802257Z","shell.execute_reply":"2024-04-20T19:00:39.048155Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Crear bins para los porcentajes del 0 al 25%, 25 al 50%, 50 al 75% y del 75 al 100%\nbins = [0, 25, 50, 75, 100]\n\n# Crear un histograma para cada tipo celular\nplt.figure(figsize=(12, 8))\nfor cell_type, pixels_list in avg_background_pixels.items():\n    # Calcular los porcentajes\n    percentages = [np.percentile(pixels_list, bin_edge) for bin_edge in bins]\n    # Crear el histograma con los mismos bins y los porcentajes como ticks en el eje x\n    plt.hist(pixels_list, bins=bins, alpha=0.5, label=cell_type, density=True) # density = True para normalizar a 1\n    plt.xticks(bins)  # Establecer los porcentajes como etiquetas en el eje x\n# Añadir título y etiquetas\nplt.title('Histograma de píxeles de fondo por tipo celular', fontsize=tamaño_de_letra)\nplt.xlabel('Porcentaje de píxeles de fondo', fontsize=tamaño_de_letra)\nplt.ylabel('Frecuencia', fontsize=tamaño_de_letra)\n# Ajustar el tamaño de letra de las etiquetas en los ejes x e y\nplt.xticks(fontsize=tamaño_de_letra)\nplt.yticks(fontsize=tamaño_de_letra)\n# Añadir leyenda\nplt.legend()\n# Mostrar el histograma\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:28:14.87051Z","iopub.execute_input":"2024-04-20T18:28:14.871224Z","iopub.status.idle":"2024-04-20T18:28:15.224788Z","shell.execute_reply.started":"2024-04-20T18:28:14.871185Z","shell.execute_reply":"2024-04-20T18:28:15.223684Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Porcentaje de fondo independientemente del tipo celular","metadata":{}},{"cell_type":"code","source":"# Crear bins para los porcentajes del 0 al 25%, 25 al 50%, 50 al 75% y del 75 al 100%\nbins = [0, 25, 50, 75, 100]\n\n# Crear el histograma\nplt.figure(figsize=(10, 6))\nplt.hist(background_percentages, bins=bins, color='skyblue')  # Ajustar los bordes a negro\nplt.xticks(bins)  # Establecer los porcentajes como etiquetas en el eje x\nplt.xlabel('Porcentaje de píxeles de fondo', fontsize=tamaño_de_letra)\nplt.ylabel('Imágenes', fontsize=tamaño_de_letra)\nplt.title('Porcentaje de píxeles de fodo por imágen', fontsize=tamaño_de_letra)\nplt.grid(False)  # Eliminar la cuadrícula de fondo\n# Ajustar el tamaño de letra de las etiquetas en los ejes x e y\nplt.xticks(fontsize=tamaño_de_letra)\nplt.yticks(fontsize=tamaño_de_letra)\n\n# Mostrar el histograma\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:45:49.349523Z","iopub.execute_input":"2024-04-20T18:45:49.349931Z","iopub.status.idle":"2024-04-20T18:45:49.574285Z","shell.execute_reply.started":"2024-04-20T18:45:49.349902Z","shell.execute_reply":"2024-04-20T18:45:49.572764Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Image Shape and Format","metadata":{}},{"cell_type":"code","source":"df_train.width.unique()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T19:04:40.135389Z","iopub.execute_input":"2024-04-20T19:04:40.136139Z","iopub.status.idle":"2024-04-20T19:04:40.144125Z","shell.execute_reply.started":"2024-04-20T19:04:40.136105Z","shell.execute_reply":"2024-04-20T19:04:40.143135Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.height.unique()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T19:04:41.756161Z","iopub.execute_input":"2024-04-20T19:04:41.756512Z","iopub.status.idle":"2024-04-20T19:04:41.764273Z","shell.execute_reply.started":"2024-04-20T19:04:41.756483Z","shell.execute_reply":"2024-04-20T19:04:41.763138Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Plot Histogram of Pixel Values","metadata":{}},{"cell_type":"markdown","source":"Generar un histograma del valor de los píxeles puede ayudar a identificar imágenes que pueden ser outliers, como las que contienen píxeles con valores de cero.\n\nComo muestra la siguiente figura, los píxeles de las imágenes parecen bastante uniformes, lo que es característico de las imágenes de microscopía.","metadata":{}},{"cell_type":"code","source":"# Inicializar conjuntos para formas y extensiones de imágenes\nimg_shapes = set()\nimg_exts = set()\n\n# Obtener rutas de imágenes en el directorio de entrenamiento\nimg_paths = Path(TRAIN_PATH).glob(\"*\")\n\n# Inicializar variables para valor máximo y mínimo de píxeles\nmax_pixel_value = float('-inf')\nmin_pixel_value = float('inf')\n\n# Iterar sobre las rutas de las imágenes\nfor img_path in tqdm(img_paths, total=df_train.id.unique().shape[0]):\n    # Leer la imagen en modo BGR\n    img_bgr = cv2.imread(img_path.as_posix())\n    \n    # Convertir la imagen a modo RGB\n    img_rgb = img_bgr[:, :, ::-1]\n    \n    # Obtener forma y extensión de la imagen\n    img_shapes.add(img_rgb.shape)\n    img_exts.add(img_path.suffix)\n    \n    # Calcular valor máximo y mínimo de píxeles\n    max_pixel_value = max(max_pixel_value, np.max(img_rgb))\n    min_pixel_value = min(min_pixel_value, np.min(img_rgb))\n\n# Imprimir resultados\nprint(f'Image shapes are {img_shapes}.')\nprint(f'Image extensions are {img_exts}.')\nprint(f'Max pixel value: {max_pixel_value}')\nprint(f'Min pixel value: {min_pixel_value}')","metadata":{"execution":{"iopub.status.busy":"2024-04-20T19:04:44.473949Z","iopub.execute_input":"2024-04-20T19:04:44.474301Z","iopub.status.idle":"2024-04-20T19:04:53.406917Z","shell.execute_reply.started":"2024-04-20T19:04:44.47427Z","shell.execute_reply":"2024-04-20T19:04:53.40606Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_paths = Path(TRAIN_PATH).glob(\"*\")\nbar = tqdm(img_paths, total=df_train.id.unique().shape[0])     # should not be: total=len(list(img_paths))\n\nplt.figure(figsize=(8,8))\nfor img_path in bar:\n    img_bgr = cv2.imread(img_path.as_posix())    # BGR mode\n    img_rgb = img_bgr[:, :, ::-1]                # RGB mode\n    \n    hist = cv2.calcHist([img_rgb], [0], None ,[256], [0,256])\n    plt.plot(hist)\n\n# Añadir título y etiquetas\nplt.title('Valor de Píxeles por imagen', fontsize=tamaño_de_letra)\nplt.xlabel('Valor de píxeles', fontsize=tamaño_de_letra)\nplt.ylabel('Frecuencia', fontsize=tamaño_de_letra)\n# Ajustar el tamaño de letra de las etiquetas en los ejes x e y\nplt.xticks(fontsize=tamaño_de_letra)\nplt.yticks(fontsize=tamaño_de_letra)\n\n# Mostrar el gráfico\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T19:05:01.296685Z","iopub.execute_input":"2024-04-20T19:05:01.297075Z","iopub.status.idle":"2024-04-20T19:05:08.600697Z","shell.execute_reply.started":"2024-04-20T19:05:01.297042Z","shell.execute_reply":"2024-04-20T19:05:08.599835Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get unique cell types\ncell_types = df_train['cell_type'].unique()\n\n# Initialize histograms for each cell type\nhistograms = {cell_type: np.zeros(256, dtype=int) for cell_type in cell_types}\n\n# Initialize histogram array for all images\nhistogram_all_images = np.zeros(256, dtype=int)\n\n# Iterate through images\nimg_paths = list(Path(TRAIN_PATH).glob(\"*\"))\nbar = tqdm(img_paths, total=len(img_paths))\n\nfig, axs = plt.subplots(2, 2, figsize=(10, 10))\n\nfor img_path in bar:\n    # Read image\n    img_bgr = cv2.imread(img_path.as_posix())\n\n    # Convert BGR to RGB\n    img_rgb = img_bgr[:, :, ::-1]\n\n    # Flatten the image to a 1D array\n    pixels = img_rgb.flatten()\n\n    # Update histogram for each cell type\n    current_cell_type = df_train[df_train['id'] == img_path.stem]['cell_type'].values[0]\n    histograms[current_cell_type] += np.histogram(pixels, bins=256, range=[0, 256])[0]\n\n    # Update histogram for all images\n    histogram_all_images += np.histogram(pixels, bins=256, range=[0, 256])[0]\n\n# Plot histograms for each cell type\nfor i, (cell_type, histogram) in enumerate(histograms.items()):\n    axs[i//2, i%2].bar(np.arange(256), histogram)\n    axs[i//2, i%2].set_xlabel('Valor de píxel', fontsize=tamaño_de_letra)\n    axs[i//2, i%2].set_ylabel('Píxeles', fontsize=tamaño_de_letra)\n    axs[i//2, i%2].set_title(f'Histograma de {cell_type}', fontsize=tamaño_de_letra)\n\n# Plot histogram for all images\naxs[1, 1].bar(np.arange(256), histogram_all_images)\naxs[1, 1].set_xlabel('Valor de píxel', fontsize=tamaño_de_letra)\naxs[1, 1].set_ylabel('Píxeles', fontsize=tamaño_de_letra)\naxs[1, 1].set_title('Histograma del conjunto de datos', fontsize=tamaño_de_letra)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T19:07:08.082116Z","iopub.execute_input":"2024-04-20T19:07:08.082529Z","iopub.status.idle":"2024-04-20T19:07:48.618131Z","shell.execute_reply.started":"2024-04-20T19:07:08.0825Z","shell.execute_reply":"2024-04-20T19:07:48.617151Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Orientacion de las células","metadata":{}},{"cell_type":"code","source":"def getOrientation(pts):\n    \n    sz = len(pts)\n    data_pts = np.empty((sz, 2), dtype=np.float64)\n    for i in range(data_pts.shape[0]):\n        data_pts[i, 0] = pts[i, 0, 0]\n        data_pts[i, 1] = pts[i, 0, 1]\n\n    mean = np.empty((0))\n    mean, eigenvectors, eigenvalues = cv2.PCACompute2(data_pts, mean)\n\n    cntr = (int(mean[0, 0]), int(mean[0, 1]))\n\n    p1 = (cntr[0] + 0.02 * eigenvectors[0, 0] * eigenvalues[0, 0], \n          cntr[1] + 0.02 * eigenvectors[0, 1] * eigenvalues[0, 0])\n    p2 = (cntr[0] - 0.02 * eigenvectors[1, 0] * eigenvalues[1, 0], \n          cntr[1] - 0.02 * eigenvectors[1, 1] * eigenvalues[1, 0])\n\n    angle = atan2(eigenvectors[0, 1], eigenvectors[0, 0])\n \n    angle = -angle\n    \n    # Ángulo entre 0 y pi\n    if angle < 0:\n        angle += pi\n\n    return np.rad2deg(angle)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.903574Z","iopub.status.idle":"2024-04-20T18:00:57.904127Z","shell.execute_reply.started":"2024-04-20T18:00:57.903854Z","shell.execute_reply":"2024-04-20T18:00:57.903875Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def count_angles_in_ranges(angles):\n    \"\"\"\n    Count angles in specified ranges.\n\n    Args:\n    - angles: List of angles.\n\n    Returns:\n    - angle_counts: List containing counts of angles in specified ranges.\n    \"\"\"\n    angle_counts = [0, 0, 0, 0]  # [0-45, 45-90, 90-135, 135-180]\n\n    for angle in angles:\n        if 0 <= angle < 45:\n            angle_counts[0] += 1\n        elif 45 <= angle < 90:\n            angle_counts[1] += 1\n        elif 90 <= angle < 135:\n            angle_counts[2] += 1\n        elif 135 <= angle <= 180:\n            angle_counts[3] += 1\n\n    return angle_counts","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.905913Z","iopub.status.idle":"2024-04-20T18:00:57.906287Z","shell.execute_reply.started":"2024-04-20T18:00:57.906112Z","shell.execute_reply":"2024-04-20T18:00:57.906126Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load image\nimg_id = \"619f91a5c197\"\nimg = cv2.imread(TRAIN_PATH + img_id + \".png\")         # BGR mode\n\n# Get annotations associated with the image\nannotations = df_train[df_train.id == img_id].annotation.tolist()\n\n# Initialize an empty mask\nimg_mask_display = np.zeros_like(img)\n\n# Define a dictionary to store color mappings for each angle\nangle_colors = {}\n\n# Overlay angle text and colored annotations on the mask\nfor annotation in annotations:\n    # Build segmentation mask for the current annotation\n    mask = build_image_mask(img, [annotation])\n\n    # Convert 0-1 mask to uint8 mask\n    annotation_mask = (mask * 255).astype(np.uint8)\n    \n    # Find contours in the binary mask\n    contours, _ = cv2.findContours(annotation_mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n    \n    # Obtener el contorno más grande\n    cnt = max(contours, key=cv2.contourArea)\n\n    # Calcular la orientación del objeto\n    angle = getOrientation(cnt)\n    \n    # If the angle is not already mapped to a color, assign a random color\n    if angle not in angle_colors:\n        angle_colors[angle] = (random.randint(0, 255), random.randint(0, 255), random.randint(0, 255))\n    \n    # Get the color corresponding to the angle\n    color = angle_colors[angle]\n    \n    # Draw the annotation with the corresponding color on the empty mask\n    img_mask_display = cv2.drawContours(img_mask_display, contours, -1, color, thickness=cv2.FILLED)\n\n    # Get the bounding box coordinates of the contour\n    x, y, w, h = cv2.boundingRect(contours[0])\n\n    # Ensure the text fits completely inside the image\n    text_width, text_height = cv2.getTextSize(f\"{angle:.2f}\", cv2.FONT_HERSHEY_SIMPLEX, 0.75, 2)[0]\n    if x + text_width > img.shape[1]:\n        x = img.shape[1] - text_width - 5\n    if y - text_height < 0:\n        y = text_height + 5\n    \n    # Put the angle text on the image\n    cv2.putText(img_mask_display, f\"{angle:.2f}\", (x, y), cv2.FONT_HERSHEY_SIMPLEX, 0.75, color, 2)\n\n# Sort the angle-color dictionary by angle\nsorted_angle_colors = {k: angle_colors[k] for k in sorted(angle_colors)}\n\n# Create a legend for angles\nlegend_img = np.zeros((len(sorted_angle_colors) * 30, 150, 3), dtype=np.uint8)\n\n# Add text and colored rectangles to the legend image\nfor i, (angle, color) in enumerate(sorted_angle_colors.items()):\n    cv2.putText(legend_img, f\"{angle:.2f}\", (10, 30 * (i + 1) - 5), cv2.FONT_HERSHEY_SIMPLEX, 0.75, color, 2)\n    cv2.rectangle(legend_img, (100, 30 * i), (140, 30 * (i + 1)), color, thickness=cv2.FILLED)\n\n# Display the original image and the mask with angle text\nplt.figure(figsize=(15, 6))\n\n# Original Image\nplt.subplot(1, 3, 1)\nplt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\nplt.title(\"Imagen Original\", fontsize=tamaño_de_letra)\n# Ajustar el tamaño de letra de las etiquetas en los ejes x e y\nplt.xticks(fontsize=tamaño_de_letra)\nplt.yticks(fontsize=tamaño_de_letra)\nplt.axis('off')\n\n# Mask with Angle Text\nplt.subplot(1, 3, 2)\nplt.imshow(cv2.cvtColor(img_mask_display, cv2.COLOR_BGR2RGB))\nplt.title(\"Máscara con texto de ángulo\", fontsize=tamaño_de_letra)\n# Ajustar el tamaño de letra de las etiquetas en los ejes x e y\nplt.xticks(fontsize=tamaño_de_letra)\nplt.yticks(fontsize=tamaño_de_letra)\nplt.axis('off')\n\n# Legend for Angles\nplt.subplot(1, 3, 3)\nplt.imshow(cv2.cvtColor(legend_img, cv2.COLOR_BGR2RGB))\nplt.title(\"Leyenda de ángulos\", fontsize=tamaño_de_letra)\n# Ajustar el tamaño de letra de las etiquetas en los ejes x e y\nplt.xticks(fontsize=tamaño_de_letra)\nplt.yticks(fontsize=tamaño_de_letra)\nplt.axis('off')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.908357Z","iopub.status.idle":"2024-04-20T18:00:57.908704Z","shell.execute_reply.started":"2024-04-20T18:00:57.908538Z","shell.execute_reply":"2024-04-20T18:00:57.908553Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def img_orientations(img):\n    # Obtener anotaciones asociadas a la imagen\n    annotations = df_train[df_train.id == img_id].annotation.tolist()\n\n    # Inicializar una lista para almacenar todos los ángulos encontrados\n    angles = []\n\n    # Iterar sobre cada anotación\n    for annotation in annotations:\n        # Build segmentation mask for the current annotation\n        mask = build_image_mask(img, [annotation])\n\n        # Convert 0-1 mask to 0-255 mask\n        img_mask = (mask * 255).astype(np.uint8)\n\n        # Find contours in the binary mask\n        contours, _ = cv2.findContours(img_mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n        \n        # Obtener el contorno más grande\n        cnt = max(contours, key=cv2.contourArea)\n            \n        # Calcular la orientación del objeto\n        angle = getOrientation(cnt)\n\n        angles.append(angle)\n\n    # Contar el número de anotaciones y ángulos\n    num_annotations = len(annotations)\n    num_angles = len(angles)\n    \n    # Verificar si el número de anotaciones es igual al número de ángulos\n    if num_annotations != num_angles:\n        print(\"¡Error! El número de anotaciones no coincide con el número de ángulos encontrados.\")\n    \n    # Count angles in specified ranges\n    angle_counts = count_angles_in_ranges(angles)\n    \n    # Convert angle_counts to percentages\n    angle_counts_percentage = [count / num_angles for count in angle_counts]\n    \n    return angle_counts, angle_counts_percentage","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.909928Z","iopub.status.idle":"2024-04-20T18:00:57.910275Z","shell.execute_reply.started":"2024-04-20T18:00:57.910107Z","shell.execute_reply":"2024-04-20T18:00:57.910122Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_paths = Path(TRAIN_PATH).glob(\"*\")\nbar = tqdm(img_paths, total=df_train.id.unique().shape[0])\n\nall_angle = {}\n\nall_angle_percentage = {}\n\nfor img_path in bar:\n    # Obtener el tipo de célula correspondiente a esta imagen\n    img_id = img_path.stem\n    img_bgr = cv2.imread(img_path.as_posix())    # BGR mode\n    img_rgb = img_bgr[:, :, ::-1]                # RGB mode\n    \n    # Obtener las angle_counts para esta imagen\n    angle_counts, angle_counts_percentage = img_orientations(img_rgb)\n    \n    # Almacenar angle_counts en el diccionario de orientaciones\n    all_angle[img_id] = angle_counts\n    all_angle_percentage[img_id] = angle_counts_percentage\n\n# Iterar sobre las dos primeras imágenes y sus porcentajes de ángulos\nfor i, (img_id, angle_counts_percentage) in enumerate(all_angle_percentage.items()):\n    # Imprimir la información\n    print(f\"Imagen {img_id}: {angle_counts_percentage}\")\n    # Detener la iteración después de las dos primeras imágenes\n    if i == 1:\n        break","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.91145Z","iopub.status.idle":"2024-04-20T18:00:57.911827Z","shell.execute_reply.started":"2024-04-20T18:00:57.911622Z","shell.execute_reply":"2024-04-20T18:00:57.911635Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Número de imágenes que tienen ese porcentaje de células de esa orientación","metadata":{}},{"cell_type":"code","source":"# Convertir el diccionario all_angle a un DataFrame\ndf_orientations = pd.DataFrame.from_dict(all_angle_percentage, orient='index', columns=['angle_count_0', 'angle_count_1', 'angle_count_2', 'angle_count_3'])\n    \ndef add_error_bars(ax, data, mean, std, color):\n    bins, edges = np.histogram(data, bins=20)\n    bin_width = edges[1] - edges[0]\n    x_centers = edges[:-1] + bin_width / 2\n    ax.bar(x_centers, bins, width=bin_width, color=color, alpha=0.7)\n    ax.errorbar(x_centers, bins, yerr=std, fmt='o', color='black', label='Desviación estándar')\n    ax.axvline(mean, color='red', linestyle='dashed', linewidth=1, label='Media')\n    ax.legend()\n\n\n# Calcular la media y la desviación estándar de cada columna\nmean_angle_count_0 = df_orientations['angle_count_0'].mean()\nmean_angle_count_1 = df_orientations['angle_count_1'].mean()\nmean_angle_count_2 = df_orientations['angle_count_2'].mean()\nmean_angle_count_3 = df_orientations['angle_count_3'].mean()\n\nstd_angle_count_0 = df_orientations['angle_count_0'].std()\nstd_angle_count_1 = df_orientations['angle_count_1'].std()\nstd_angle_count_2 = df_orientations['angle_count_2'].std()\nstd_angle_count_3 = df_orientations['angle_count_3'].std()\n\n# Configuración de los histogramas\nfig, axs = plt.subplots(2, 2, figsize=(12, 8))\naxs = axs.ravel()\n\n# Histograma para angle_count_0\nadd_error_bars(axs[0], df_orientations['angle_count_0'], mean_angle_count_0, std_angle_count_0, 'blue')\naxs[0].set_title('Histograma orientación 0-45 grados', fontsize=tamaño_de_letra)\naxs[0].set_xlabel('Porcentaje de células', fontsize=tamaño_de_letra)\naxs[0].set_ylabel('Imágenes', fontsize=tamaño_de_letra)\n\n# Histograma para angle_count_1\nadd_error_bars(axs[1], df_orientations['angle_count_1'], mean_angle_count_1, std_angle_count_1, 'green')\naxs[1].set_title('Histograma orientación 45-90 grados', fontsize=tamaño_de_letra)\naxs[1].set_xlabel('Porcentaje de células', fontsize=tamaño_de_letra)\naxs[1].set_ylabel('Imágenes', fontsize=tamaño_de_letra)\n\n# Histograma para angle_count_2\nadd_error_bars(axs[2], df_orientations['angle_count_2'], mean_angle_count_2, std_angle_count_2, 'orange')\naxs[2].set_title('Histograma orientación 90-135 grados', fontsize=tamaño_de_letra)\naxs[2].set_xlabel('Porcentaje de células', fontsize=tamaño_de_letra)\naxs[2].set_ylabel('Imágenes', fontsize=tamaño_de_letra)\n\n# Histograma para angle_count_3\nadd_error_bars(axs[3], df_orientations['angle_count_3'], mean_angle_count_3, std_angle_count_3, 'purple')\naxs[3].set_title('Histograma orientación 135-180 grados', fontsize=tamaño_de_letra)\naxs[3].set_xlabel('Porcentaje de células', fontsize=tamaño_de_letra)\naxs[3].set_ylabel('Imágenes', fontsize=tamaño_de_letra)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.9133Z","iopub.status.idle":"2024-04-20T18:00:57.913653Z","shell.execute_reply.started":"2024-04-20T18:00:57.913482Z","shell.execute_reply":"2024-04-20T18:00:57.913497Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Distribución de anotaciones por número de ángulos","metadata":{}},{"cell_type":"code","source":"img_paths = Path(TRAIN_PATH).glob(\"*\")\nbar = tqdm(img_paths, total=df_train.id.unique().shape[0])\n\n# Obtener los ángulos y anotaciones\nangles = []  # Lista de ángulos\nannotations = []  # Lista de anotaciones\n\nfor img_path in bar:\n    # Obtener el tipo de célula correspondiente a esta imagen\n    img_id = img_path.stem\n    img_bgr = cv2.imread(img_path.as_posix())    # BGR mode\n    img_rgb = img_bgr[:, :, ::-1]                # RGB mode\n    \n    # Obtener las anotaciones para esta imagen\n    img_annotations = df_train[df_train['id'] == img_id]['annotation'].tolist()\n    \n    # Obtener los ángulos para cada anotación en esta imagen\n    for annotation in img_annotations:\n        mask = build_image_mask(img_rgb, [annotation])\n        img_mask = (mask * 255).astype(np.uint8)\n        contours, _ = cv2.findContours(img_mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n        \n        # Obtener el contorno más grande\n        cnt = max(contours, key=cv2.contourArea)\n            \n        # Calcular la orientación del objeto\n        angle = getOrientation(cnt)\n        \n        angles.append(angle)\n        annotations.append(img_id)\n\n# Count angles in specified ranges\nangle_counts = count_angles_in_ranges(angles)\n\n# Obtener los rangos de ángulos\nangle_ranges = ['0-45', '45-90', '90-135', '135-180']\n\n# Crear el histograma\nplt.figure(figsize=(10, 6))\nplt.bar(angle_ranges, angle_counts)\nplt.xlabel('Rango de ángulos', fontsize=tamaño_de_letra)\nplt.ylabel('Células', fontsize=tamaño_de_letra)\n# Ajustar el tamaño de letra de las etiquetas en los ejes x e y\nplt.xticks(fontsize=tamaño_de_letra)\nplt.yticks(fontsize=tamaño_de_letra)\nplt.title('Distribución de células por rango de ángulos', fontsize=tamaño_de_letra)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.915364Z","iopub.status.idle":"2024-04-20T18:00:57.915702Z","shell.execute_reply.started":"2024-04-20T18:00:57.915539Z","shell.execute_reply":"2024-04-20T18:00:57.915553Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Obtener máscaras de segmentación formadas solo por cada una de las orientaciones","metadata":{}},{"cell_type":"code","source":"# Definir un DataFrame para las células\ndf_45grados = pd.DataFrame(columns=df_train.columns)\ndf_90grados = pd.DataFrame(columns=df_train.columns)\ndf_135grados = pd.DataFrame(columns=df_train.columns)\ndf_180grados = pd.DataFrame(columns=df_train.columns)\n\n# Definir el rango de ángulos deseados\nangle_0 = 0\nangle_45 = 45\nangle_90 = 90\nangle_135 = 135\nangle_180 = 180\n\n# Recorrer el DataFrame df_train por fila\nfor idx, row in tqdm(df_train.iterrows(), total=len(df_train)):\n    \n    # Obtener el tipo de célula correspondiente a esta fila\n    img_id = row['id']\n    img_annotation = row['annotation']\n    img_path = os.path.join(TRAIN_PATH, img_id + \".png\")\n    \n    # Leer la imagen desde el disco\n    img = cv2.imread(img_path)\n    \n    # Build segmentation mask for the current annotation\n    mask = build_image_mask(img, [img_annotation])\n\n    # Convert 0-1 mask to 0-255 mask\n    img_mask = (mask * 255).astype(np.uint8)\n\n    # Find contours in the binary mask\n    contours, _ = cv2.findContours(img_mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n\n    # Obtener el contorno más grande\n    cnt = max(contours, key=cv2.contourArea)\n\n    # Calcular la orientación del objeto\n    angle = getOrientation(cnt)\n\n    # Verificar el ángulo y añadir la fila correspondiente a los DataFrames\n    if angle >= angle_0 and angle < angle_45:\n        df_45grados.loc[len(df_45grados)] = df_train.loc[idx]\n    elif angle >= angle_45 and angle < angle_90:\n        df_90grados.loc[len(df_90grados)] = df_train.loc[idx]\n    elif angle >= angle_90 and angle < angle_135:\n        df_135grados.loc[len(df_135grados)] = df_train.loc[idx]\n    elif angle >= angle_135 and angle <= angle_180:\n        df_180grados.loc[len(df_180grados)] = df_train.loc[idx]\n        mask180= build_image_mask(img, df_180grados[df_180grados.id == img_id].annotation.tolist())\n        np.save(os.path.join(ORIENTATION_DIR, f\"{img_id}.npy\"), mask180)\n            ","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.917048Z","iopub.status.idle":"2024-04-20T18:00:57.917416Z","shell.execute_reply.started":"2024-04-20T18:00:57.91723Z","shell.execute_reply":"2024-04-20T18:00:57.917252Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Guardar cada DataFrame en un archivo CSV\n#df_45grados.to_csv(ORIENTATION_DIR + 'df_45grados.csv', index=False)\n#df_90grados.to_csv(ORIENTATION_DIR + 'df_90grados.csv', index=False)\n#df_135grados.to_csv(ORIENTATION_DIR + 'df_135grados.csv', index=False)\n#df_180grados.to_csv(ORIENTATION_DIR + 'df_180grados.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.918982Z","iopub.status.idle":"2024-04-20T18:00:57.919342Z","shell.execute_reply.started":"2024-04-20T18:00:57.919163Z","shell.execute_reply":"2024-04-20T18:00:57.919178Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_id = '619f91a5c197'\nimg = cv2.imread(TRAIN_PATH + img_id + \".png\") \n\nmask45 = build_image_mask(img, df_45grados[df_45grados.id == img_id].annotation.tolist())\n\nmask90 = build_image_mask(img, df_90grados[df_90grados.id == img_id].annotation.tolist())\n\nmask135 = build_image_mask(img, df_135grados[df_135grados.id == img_id].annotation.tolist())\n\nmask180 = build_image_mask(img, df_180grados[df_180grados.id == img_id].annotation.tolist())\n\nplt.figure(figsize=(16, 8))\n# Original Image\nplt.subplot(2, 3, 1)\nplt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\nplt.title(\"Imagen Original\", fontsize=tamaño_de_letra)\nplt.axis('off')\n\n# Mask with Angle Text\nplt.subplot(2, 3, 2)\nplt.imshow(cv2.cvtColor(img_mask_display, cv2.COLOR_BGR2RGB))\nplt.title(\"Máscara con texto del ángulo\", fontsize=tamaño_de_letra)\nplt.axis('off')\n\n# Mask with Angle Text\nplt.subplot(2, 3, 3)\nplt.imshow(mask45, cmap='gray')\nplt.title(\"Máscara 0 - 45 grados\", fontsize=tamaño_de_letra)\nplt.axis('off')\n\n# Mask with Angle Text\nplt.subplot(2, 3, 4)\nplt.imshow(mask90, cmap='gray')\nplt.title(\"Máscara 45 - 90 grados\", fontsize=tamaño_de_letra)\nplt.axis('off')\n\n# Mask with Angle Text\nplt.subplot(2, 3, 5)\nplt.imshow(mask135, cmap='gray')\nplt.title(\"Máscara 90 - 135 grados\", fontsize=tamaño_de_letra)\nplt.axis('off')\n\n# Mask with Angle Text\nplt.subplot(2, 3, 6)\nplt.imshow(mask180, cmap='gray')\nplt.title(\"Máscara 135 - 180 grados\", fontsize=tamaño_de_letra)\nplt.axis('off')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.920504Z","iopub.status.idle":"2024-04-20T18:00:57.920871Z","shell.execute_reply.started":"2024-04-20T18:00:57.920675Z","shell.execute_reply":"2024-04-20T18:00:57.920689Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Display Some Masks","metadata":{}},{"cell_type":"markdown","source":"shsy5y","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 8))\nimg_id = \"0030fd0e6378\"\nimg = cv2.imread(TRAIN_PATH + img_id + \".png\")         # BGR mode\n# df_train[df_train.id == img_id].annotation.tolist()\nax1 = plt.subplot(121)\nax1.imshow(img)\nax1.set_title(\"Shsy5y Original Image (BGR mode)\")\n\nmask = build_image_mask(img, df_train[df_train.id == img_id].annotation.tolist())\nax2 = plt.subplot(122)\nax2.imshow(mask)\n# plt.imshow(mask, cmap=\"gray\")\n# plt.imshow(mask, cmap = plt.cm.gray)\n# plt.imshow(mask, cmap = plt.cm.gray_r)\nax2.set_title(\"Shsy5y Mask\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.922072Z","iopub.status.idle":"2024-04-20T18:00:57.922482Z","shell.execute_reply.started":"2024-04-20T18:00:57.922264Z","shell.execute_reply":"2024-04-20T18:00:57.922279Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(16, 8))\nimg_id = \"0030fd0e6378\"\nimg = cv2.imread(TRAIN_PATH + img_id + \".png\")          # BGR mode\nplt.imshow(img)\nplt.title(\"Shsy5y Original Image (BGR mode) and Mask\")\n\nmask = build_image_mask(img, df_train[df_train.id == img_id].annotation.tolist())\nplt.imshow(mask, alpha=0.2)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.924029Z","iopub.status.idle":"2024-04-20T18:00:57.924378Z","shell.execute_reply.started":"2024-04-20T18:00:57.924201Z","shell.execute_reply":"2024-04-20T18:00:57.924215Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"astro","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 8))\nimg_id = \"0140b3c8f445\"\nimg = cv2.imread(TRAIN_PATH + img_id + \".png\")         # BGR mode\n# df_train[df_train.id == img_id].annotation.tolist()\nax1 = plt.subplot(121)\nax1.imshow(img)\nax1.set_title(\"Astro Original Image (BGR mode)\")\n\nmask = build_image_mask(img, df_train[df_train.id == img_id].annotation.tolist())\nax2 = plt.subplot(122)\nax2.imshow(mask)\n# plt.imshow(mask, cmap=\"gray\")\n# plt.imshow(mask, cmap = plt.cm.gray)\n# plt.imshow(mask, cmap = plt.cm.gray_r)\nax2.set_title(\"Astro Mask\")\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.925799Z","iopub.status.idle":"2024-04-20T18:00:57.926121Z","shell.execute_reply.started":"2024-04-20T18:00:57.925962Z","shell.execute_reply":"2024-04-20T18:00:57.925975Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(16, 8))\nimg_id = \"0140b3c8f445\"\nimg = cv2.imread(TRAIN_PATH + img_id + \".png\")          # BGR mode\nplt.imshow(img)\nplt.title(\"Astro Original Image (BGR mode) and Mask\")\n\nmask = build_image_mask(img, df_train[df_train.id == img_id].annotation.tolist())\nplt.imshow(mask, alpha=0.05)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.928306Z","iopub.status.idle":"2024-04-20T18:00:57.928744Z","shell.execute_reply.started":"2024-04-20T18:00:57.928524Z","shell.execute_reply":"2024-04-20T18:00:57.928542Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"cort","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 8))\nimg_id = \"01ae5a43a2ab\"\nimg = cv2.imread(TRAIN_PATH + img_id + \".png\")         # BGR mode\n# df_train[df_train.id == img_id].annotation.tolist()\nax1 = plt.subplot(121)\nax1.imshow(img)\nax1.set_title(\"Cort Original Image (BGR mode)\")\n\nmask = build_image_mask(img, df_train[df_train.id == img_id].annotation.tolist())\nax2 = plt.subplot(122)\nax2.imshow(mask)\n# plt.imshow(mask, cmap=\"gray\")\n# plt.imshow(mask, cmap = plt.cm.gray)\n# plt.imshow(mask, cmap = plt.cm.gray_r)\nax2.set_title(\"Cort Mask\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.930315Z","iopub.status.idle":"2024-04-20T18:00:57.930631Z","shell.execute_reply.started":"2024-04-20T18:00:57.930474Z","shell.execute_reply":"2024-04-20T18:00:57.930487Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(16, 8))\nimg_id = \"01ae5a43a2ab\"\nimg = cv2.imread(TRAIN_PATH + img_id + \".png\")          # BGR mode\nplt.imshow(img)\nplt.title(\"Cort Original Image (BGR mode) and Mask\")\n\nmask = build_image_mask(img, df_train[df_train.id == img_id].annotation.tolist())\nplt.imshow(mask, alpha=0.2)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.932058Z","iopub.status.idle":"2024-04-20T18:00:57.932515Z","shell.execute_reply.started":"2024-04-20T18:00:57.93228Z","shell.execute_reply":"2024-04-20T18:00:57.9323Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Una única figura con todas las imágenes anteriores","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport cv2\n\nplt.figure(figsize=(16, 12))\n\n# Primera imagen y máscara\nimg_id1 = \"0030fd0e6378\"\nimg1 = cv2.imread(TRAIN_PATH + img_id1 + \".png\")         # BGR mode\nax1 = plt.subplot(231)\nax1.imshow(img1)\nax1.set_title(\"Imagen Original Shsy5y\", fontsize=tamaño_de_letra)\n\nmask1 = build_image_mask(img1, df_train[df_train.id == img_id1].annotation.tolist())\nax2 = plt.subplot(234)\nax2.imshow(mask1)\nax2.set_title(\"Máscara Shsy5y\", fontsize=tamaño_de_letra)\n\n# Segunda imagen y máscara\nimg_id2 = \"0140b3c8f445\"\nimg2 = cv2.imread(TRAIN_PATH + img_id2 + \".png\")         # BGR mode\nax3 = plt.subplot(232)\nax3.imshow(img2)\nax3.set_title(\"Imagen Original Astro\", fontsize=tamaño_de_letra)\n\nmask2 = build_image_mask(img2, df_train[df_train.id == img_id2].annotation.tolist())\nax4 = plt.subplot(235)\nax4.imshow(mask2)\nax4.set_title(\"Máscara Astro\", fontsize=tamaño_de_letra)\n\n# Tercera imagen y máscara\nimg_id3 = \"01ae5a43a2ab\"\nimg3 = cv2.imread(TRAIN_PATH + img_id3 + \".png\")         # BGR mode\nax5 = plt.subplot(233)\nax5.imshow(img3)\nax5.set_title(\"Imagen Original Cort\", fontsize=tamaño_de_letra)\n\nmask3 = build_image_mask(img3, df_train[df_train.id == img_id3].annotation.tolist())\nax6 = plt.subplot(236)\nax6.imshow(mask3)\nax6.set_title(\"Máscara Cort\", fontsize=tamaño_de_letra)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-20T18:00:57.934033Z","iopub.status.idle":"2024-04-20T18:00:57.934506Z","shell.execute_reply.started":"2024-04-20T18:00:57.93426Z","shell.execute_reply":"2024-04-20T18:00:57.93428Z"},"trusted":true},"outputs":[],"execution_count":null}]}