{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\nimport os\n!mkdir /kaggle/working/packages\n!cp -r /kaggle/input/pycocotools/* /kaggle/working/packages\nos.chdir(\"/kaggle/working/packages/pycocotools-2.0.6/\")\n!python setup.py install\n!pip install . --no-index --find-links /kaggle/working/packages/\nos.chdir(\"/kaggle/working\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-09T09:35:59.819561Z","iopub.execute_input":"2023-06-09T09:35:59.819939Z","iopub.status.idle":"2023-06-09T09:36:54.199382Z","shell.execute_reply.started":"2023-06-09T09:35:59.819909Z","shell.execute_reply":"2023-06-09T09:36:54.198083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport json\nfrom itertools import chain\nimport os\nimport pandas as pd\nfrom datetime import datetime\nimport shutil\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Conv2D, MaxPooling2D, UpSampling2D, concatenate, Activation\nfrom tensorflow.keras.models import load_model\nimport base64\nimport numpy as np\nfrom pycocotools import _mask as coco_mask\nimport typing as t\nimport zlib\nfrom pycocotools import _mask as coco_mask","metadata":{"execution":{"iopub.status.busy":"2023-06-09T09:36:54.203562Z","iopub.execute_input":"2023-06-09T09:36:54.204540Z","iopub.status.idle":"2023-06-09T09:37:02.481604Z","shell.execute_reply.started":"2023-06-09T09:36:54.204493Z","shell.execute_reply":"2023-06-09T09:37:02.480389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mask_generator(img_polygons):\n    mask = np.zeros((512, 512))\n    coordinates_join = []\n    for p in img_polygons:\n        coordinates = p['coordinates'][0]\n        coordinates_join.append(coordinates)\n    coordinates_vessel = list(chain.from_iterable(coordinates_join))\n    for coord in coordinates_vessel:\n        mask[coord[1], coord[0]] = 1\n    \n    mask = mask.astype(np.uint8)# * 255\n        \n    return mask\n\nfolder = '/kaggle/working/labels'\nif not os.path.exists(folder):\n    os.mkdir(folder)\n\nwith open('/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl', 'r') as f:\n    polygons = [json.loads(line) for line in f]\n\nfor i,p in enumerate(polygons):\n    img_name = p['id']\n    polygons_blood_vessel = [d for d in p['annotations'] if d['type'] == 'blood_vessel']\n    mask = mask_generator(polygons_blood_vessel)\n    cv2.imwrite(f'{folder}/{img_name}.tif', mask)","metadata":{"execution":{"iopub.status.busy":"2023-06-09T09:37:02.487130Z","iopub.execute_input":"2023-06-09T09:37:02.487831Z","iopub.status.idle":"2023-06-09T09:37:11.496630Z","shell.execute_reply.started":"2023-06-09T09:37:02.487806Z","shell.execute_reply":"2023-06-09T09:37:11.495556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n# Leer el archivo CSV\ndf = pd.read_csv('/kaggle/input/hubmap-hacking-the-human-vasculature/tile_meta.csv')\n\n# Crear la carpeta si no existe\nif not os.path.exists('/kaggle/working/train_annotation'):\n    os.makedirs('/kaggle/working/train_annotation')\n\n# Filtrar las imágenes que pertenecen al grupo 3\ngroup_1_2_images = df[(df['dataset'] == 1) | (df['dataset'] == 2)]['id']\n\n# Mover las imágenes al directorio 'train_no_annotation'\nfor image in group_1_2_images:\n    print(f'/kaggle/input/hubmap-hacking-the-human-vasculature/train/{image}.tif')\n    # Asegúrate de que la imagen exista antes de moverla\n    if os.path.isfile(f'/kaggle/input/hubmap-hacking-the-human-vasculature/train/{image}.tif'):\n        shutil.copy(f'/kaggle/input/hubmap-hacking-the-human-vasculature/train/{image}.tif', f'/kaggle/working/train_annotation/{image}.tif')\n","metadata":{"execution":{"iopub.status.busy":"2023-06-09T09:37:11.498359Z","iopub.execute_input":"2023-06-09T09:37:11.498721Z","iopub.status.idle":"2023-06-09T09:37:29.975687Z","shell.execute_reply.started":"2023-06-09T09:37:11.498687Z","shell.execute_reply":"2023-06-09T09:37:29.974528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BuildModel:\n\n    def __init__(self, img_shape, num_classes):\n\n        self.img_shape = img_shape\n        self.num_classes = num_classes\n\n    def build_model(self):\n\n        inputs = Input(shape = self.img_shape)\n\n        down0 = Conv2D(64, (3, 3), padding = 'same')(inputs)\n        down0 = Activation('relu')(down0)\n        down0_pool = MaxPooling2D((2, 2), strides = (2, 2))(down0)\n\n        down1 = Conv2D(128, (3, 3), padding = 'same')(down0_pool)\n        down1 = Activation('relu')(down1)\n        down1_pool = MaxPooling2D((2, 2), strides=(2, 2))(down1)\n\n        center = Conv2D(256, (3, 3), padding='same')(down1_pool)\n        center = Activation('relu')(center)\n\n        up1 = UpSampling2D((2,2))(center)\n        up1 = concatenate([down1, up1], axis=3)\n        up1 = Conv2D(128, (3, 3), padding='same')(up1)\n        up1 = Activation('relu')(up1)\n\n        up0 = UpSampling2D((2,2))(up1)\n\n        classify = Conv2D(self.num_classes, (1, 1), activation='sigmoid')(up0)\n\n        model = Model(inputs = inputs, outputs = classify)\n\n        return model","metadata":{"execution":{"iopub.status.busy":"2023-06-09T09:37:29.980105Z","iopub.execute_input":"2023-06-09T09:37:29.980876Z","iopub.status.idle":"2023-06-09T09:37:29.993452Z","shell.execute_reply.started":"2023-06-09T09:37:29.980847Z","shell.execute_reply":"2023-06-09T09:37:29.992409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Directorios donde se encuentran tus imágenes y etiquetas\ntrain_dir = '/kaggle/working/train_annotation'\nlabel_dir = '/kaggle/working/labels'\n\n# Lista para almacenar tus datos\ntrain_data = []\nlabel_data = []\n\n# Asegurarse de que las imágenes y las etiquetas se emparejan correctamente\nimage_filenames = sorted(os.listdir(train_dir))\nlabel_filenames = sorted(os.listdir(label_dir))\n\n\nassert image_filenames == label_filenames, \"Las imágenes y las etiquetas no se emparejan correctamente\"\n\n# Leer y procesar las imágenes y etiquetas\nfor filename in image_filenames[:600]:\n    # Leer la imagen y la etiqueta\n    img = cv2.imread(os.path.join(train_dir, filename))\n    label = cv2.imread(os.path.join(label_dir, filename), cv2.IMREAD_GRAYSCALE)  # Asegúrate de leer las etiquetas en escala de grises\n\n    # Agregar los datos a las listas\n    train_data.append(img)\n    label_data.append(label)\n\n# Convertir las listas a arrays de NumPy para que puedan ser usados por tu modelo\nX_train = np.array(train_data)\ny_train = np.array(label_data)","metadata":{"execution":{"iopub.status.busy":"2023-06-09T09:37:29.996831Z","iopub.execute_input":"2023-06-09T09:37:29.997812Z","iopub.status.idle":"2023-06-09T09:37:36.837612Z","shell.execute_reply.started":"2023-06-09T09:37:29.997775Z","shell.execute_reply":"2023-06-09T09:37:36.836511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = X_train / 255.0","metadata":{"execution":{"iopub.status.busy":"2023-06-09T09:37:36.838980Z","iopub.execute_input":"2023-06-09T09:37:36.839848Z","iopub.status.idle":"2023-06-09T09:37:38.167734Z","shell.execute_reply.started":"2023-06-09T09:37:36.839815Z","shell.execute_reply":"2023-06-09T09:37:38.166613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ahora puedes usar train_data y label_data para entrenar tu modelo\nmodel = BuildModel(img_shape = (512, 512, 3), num_classes = 1)\nmodel = model.build_model()\n\nmodel.compile(optimizer = 'adam', loss = 'binary_crossentropy', metrics = ['accuracy'])\nmodel.fit(X_train, y_train, batch_size = 4, epochs = 10, validation_split = 0.2)\n\n# Guarda el modelo con el nombre del archivo que incluye la marca de tiempo\nmodel.save('/kaggle/working/models/hubmap_model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-06-09T09:37:38.169335Z","iopub.execute_input":"2023-06-09T09:37:38.169695Z","iopub.status.idle":"2023-06-09T09:40:26.817389Z","shell.execute_reply.started":"2023-06-09T09:37:38.169660Z","shell.execute_reply":"2023-06-09T09:40:26.815699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def encode_binary_mask(mask: np.ndarray) -> t.Text:\n    \"\"\"Converts a binary mask into OID challenge encoding ascii text.\"\"\"\n\n    # check input mask --\n    if mask.dtype != np.bool:\n        raise ValueError(\n            \"encode_binary_mask expects a binary mask, received dtype == %s\" %\n            mask.dtype)\n\n    mask = np.squeeze(mask)\n    if len(mask.shape) != 2:\n        raise ValueError(\n            \"encode_binary_mask expects a 2d mask, received shape == %s\" %\n            mask.shape)\n\n    # convert input mask to expected COCO API input --\n    mask_to_encode = mask.reshape(mask.shape[0], mask.shape[1], 1)\n    mask_to_encode = mask_to_encode.astype(np.uint8)\n    mask_to_encode = np.asfortranarray(mask_to_encode)\n\n    # RLE encode mask --\n    encoded_mask = coco_mask.encode(mask_to_encode)[0][\"counts\"]\n\n    # compress and base64 encoding --\n    binary_str = zlib.compress(encoded_mask, zlib.Z_BEST_COMPRESSION)\n    base64_str = base64.b64encode(binary_str)\n    return base64_str","metadata":{"execution":{"iopub.status.busy":"2023-06-09T09:40:26.818983Z","iopub.status.idle":"2023-06-09T09:40:26.819812Z","shell.execute_reply.started":"2023-06-09T09:40:26.819521Z","shell.execute_reply":"2023-06-09T09:40:26.819559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dir = '/kaggle/input/hubmap-hacking-the-human-vasculature/test'\n\ntest_data = []\n\ntest_filenames = sorted(os.listdir(test_dir))\n\nfor filename in test_filenames:\n    test = cv2.imread(os.path.join(test_dir, filename))\n    test_data.append(test)\n\nX_test = np.array(test_data)\n\nmodel = load_model('/kaggle/working/models/hubmap_model.h5') # PONER FECHA Y HORA BIEN\n\npredictions = model.predict(X_test)\n\npredictions = (predictions > 0.5).astype(np.uint8) * 255\n\nsubmission_table = pd.DataFrame(columns = ['id', 'height', 'width', 'prediction_string'])\nfor i, img_name in enumerate(test_filenames):\n    img = predictions[i]\n    img_bool = img == 1\n    base64_encode = encode_binary_mask(img_bool)\n    prediction_string = '0 ' + '1.0 ' + base64_encode.decode('utf-8')\n    print(prediction_string)\n    new_row = {'id': img_name.replace('.tif', ''), 'height': img.shape[0], 'width': img.shape[1], 'prediction_string': prediction_string}\n    print(new_row)\n    submission_table = submission_table.append(new_row, ignore_index = True)\n    #cv2.imwrite(f'./data/predictions/{img_name}.tif', img)\n\nprint(submission_table)","metadata":{"execution":{"iopub.status.busy":"2023-06-09T09:40:26.821376Z","iopub.status.idle":"2023-06-09T09:40:26.822179Z","shell.execute_reply.started":"2023-06-09T09:40:26.821897Z","shell.execute_reply":"2023-06-09T09:40:26.821922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_table.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-09T09:40:26.823587Z","iopub.status.idle":"2023-06-09T09:40:26.824428Z","shell.execute_reply.started":"2023-06-09T09:40:26.824153Z","shell.execute_reply":"2023-06-09T09:40:26.824178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}