{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1>Table of Contents<span class=\"tocSkip\"></span></h1>\n<div class=\"toc\"><ul class=\"toc-item\"></ul></div>","metadata":{"toc":true}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom concurrent.futures import ThreadPoolExecutor\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\noutput_path = './output/'\nif not os.path.exists(output_path):\n    os.makedirs(output_path)\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2023-05-13T17:52:52.246184Z","iopub.execute_input":"2023-05-13T17:52:52.246637Z","iopub.status.idle":"2023-05-13T17:52:52.283832Z","shell.execute_reply.started":"2023-05-13T17:52:52.246595Z","shell.execute_reply":"2023-05-13T17:52:52.281974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport random\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport numpy as np\nimport glob\nfrom PIL import Image\nimport torch.utils.data as data\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nfrom tqdm import tqdm\nfrom ipywidgets import interact, fixed\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nimport torch.nn.functional as F\nfrom scipy.ndimage import morphology\nfrom tqdm.notebook import tqdm_notebook\nfrom torch.utils.data import random_split\nfrom torchvision.transforms import ToTensor\nimport torchvision.transforms as transforms\nimport gc\nfrom torch.utils.data import TensorDataset, SubsetRandomSampler\nfrom torchvision.transforms import Normalize\nimport tifffile\nimport torch.nn.functional as F\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:52:52.288165Z","iopub.execute_input":"2023-05-13T17:52:52.288458Z","iopub.status.idle":"2023-05-13T17:52:56.205152Z","shell.execute_reply.started":"2023-05-13T17:52:52.288430Z","shell.execute_reply":"2023-05-13T17:52:56.204008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostClassifier, Pool\nfrom sklearn.metrics import accuracy_score, roc_auc_score, f1_score, precision_recall_curve\nfrom sklearn.model_selection import train_test_split\nfrom joblib import Parallel, delayed","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:52:56.207304Z","iopub.execute_input":"2023-05-13T17:52:56.208245Z","iopub.status.idle":"2023-05-13T17:52:56.731142Z","shell.execute_reply.started":"2023-05-13T17:52:56.208205Z","shell.execute_reply":"2023-05-13T17:52:56.730100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = f\"cuda\" if torch.cuda.is_available() else \"cpu\"\ndevice","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:52:56.734034Z","iopub.execute_input":"2023-05-13T17:52:56.734702Z","iopub.status.idle":"2023-05-13T17:52:56.799850Z","shell.execute_reply.started":"2023-05-13T17:52:56.734659Z","shell.execute_reply":"2023-05-13T17:52:56.798783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed) # фиксируем генератор случайных чисел\n    os.environ['PYTHONHASHSEED'] = str(seed) # фиксируем заполнения хешей\n    np.random.seed(seed) # фиксируем генератор случайных чисел numpy\n    torch.manual_seed(seed) # фиксируем генератор случайных чисел pytorch\n    torch.cuda.manual_seed(seed) # фиксируем генератор случайных чисел для GPU\n    torch.backends.cudnn.deterministic = True # выбираем только детерминированные алгоритмы (для сверток)\n    torch.backends.cudnn.benchmark = False # фиксируем алгоритм вычисления сверток","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:52:56.801558Z","iopub.execute_input":"2023-05-13T17:52:56.802752Z","iopub.status.idle":"2023-05-13T17:52:56.811565Z","shell.execute_reply.started":"2023-05-13T17:52:56.802710Z","shell.execute_reply":"2023-05-13T17:52:56.810457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Задаем параметры модели\nlearning_rate = 0.0001\nbatch_size = 32 # размер батча\ncube_size = 32 # нарезка изображений на cube_size х cube_size\nepochs = 1000\nseed = 42\nstack_count = 32 # количество срезов для модели\nseed_everything(seed)\nkernel = np.array([[-1,-1,-1], [-1,9,-1], [-1,-1,-1]])\nblock_size = cube_size","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:52:56.813160Z","iopub.execute_input":"2023-05-13T17:52:56.813713Z","iopub.status.idle":"2023-05-13T17:52:56.824663Z","shell.execute_reply.started":"2023-05-13T17:52:56.813674Z","shell.execute_reply":"2023-05-13T17:52:56.823606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UNET2_3D(nn.Module):\n    def __init__(self, in_channels):\n        super().__init__()\n               \n# энкодер\n        self.conv1 = nn.Conv3d(1, 64, kernel_size=3, padding=1)\n        self.bn1 = nn.BatchNorm3d(64)\n         \n        self.conv2 = nn.Conv3d(64, 128, kernel_size=3, padding=1)\n        self.bn2 = nn.BatchNorm3d(128)\n        \n        self.conv3 = nn.Conv3d(128, 256, kernel_size=3, padding=1)\n        self.bn3 = nn.BatchNorm3d(256)\n        \n        self.conv4 = nn.Conv3d(256, 512, kernel_size=3, padding=1)\n        self.bn4 = nn.BatchNorm3d(512)\n\n        self.conv5 = nn.Conv3d(512, 1024, kernel_size=3, padding=1)\n        self.bn5 = nn.BatchNorm3d(1024)\n        \n        # декодер\n        self.upconv6 = nn.ConvTranspose3d(1024, 512, kernel_size=2, stride=2)\n        self.conv6 = nn.Conv3d(1024, 512, kernel_size=3, padding=1)\n        self.bn6 = nn.BatchNorm3d(512)\n        \n        self.upconv7 = nn.ConvTranspose3d(512, 256, kernel_size=2, stride=2)\n        self.conv7 = nn.Conv3d(512, 256, kernel_size=3, padding=1)\n        self.bn7 = nn.BatchNorm3d(256)\n        \n        self.upconv8 = nn.ConvTranspose3d(256, 128, kernel_size=2, stride=2)\n        self.conv8 = nn.Conv3d(256, 128, kernel_size=3, padding=1)\n        self.bn8 = nn.BatchNorm3d(128)\n        \n        self.upconv9 = nn.ConvTranspose3d(128, 64, kernel_size=2, stride=2)\n        self.conv9 = nn.Conv3d(128, 64, kernel_size=3, padding=1)\n        self.bn9 = nn.BatchNorm3d(64)\n        \n        self.conv10 = nn.Conv3d(64, 1, kernel_size=3, padding=1)\n        self.bn10 = nn.BatchNorm3d(1)\n               \n        self.conv11 = nn.Conv2d(in_channels, 1, kernel_size=3, padding=1)\n        \n        self.relu = nn.ReLU(inplace=True)\n        self.pool = nn.MaxPool3d(2, stride=2)\n        self.sigmoid = nn.Sigmoid()\n\n    def forward(self, x):\n            # энкодер\n            \n            x = self.relu(self.bn1(self.conv1(x)))\n            \n            x1 = x.clone() # сохраним для skip-связи\n            \n            x = self.pool(x)\n            \n            x = self.relu(self.bn2(self.conv2(x)))\n            x2 = x.clone() # сохраним для skip-связи\n           \n            x = self.pool(x)\n           \n            x = self.relu(self.bn3(self.conv3(x)))\n            \n            x3 = x.clone() # сохраним для skip-связи\n            \n            \n            x = self.pool(x)\n            x = self.relu(self.bn4(self.conv4(x)))\n            x4 = x.clone() # сохраним для skip-связи\n            \n            x = self.pool(x)\n            x = self.relu(self.bn5(self.conv5(x)))\n            \n            # декодер\n            x = self.upconv6(x)\n            x = torch.cat([x4, x], dim=1) # skip-связь\n            x = self.conv6(x)\n            \n            x = self.bn6(x)\n            x = self.relu(x)\n            \n            \n            x = self.upconv7(x)\n            \n            x = torch.cat([x3, x], dim=1) # skip-связь\n            x = self.conv7(x)\n            x = self.bn7(x)\n            x = self.relu(x)\n            \n            x = self.upconv8(x)\n            x = torch.cat([x2, x], dim=1) # skip-связь\n            x = self.conv8(x)\n            x = self.bn8(x)\n            x = self.relu(x)\n            \n            x = self.upconv9(x)\n            x = torch.cat([x1, x], dim=1) # skip-связь\n            x = self.conv9(x)\n            x = self.bn9(x)\n            x = self.relu(x)\n            \n            x = self.conv10(x)\n            x = self.bn10(x)\n            x = self.relu(x)\n            x = torch.squeeze(x, dim=1)    \n        \n       \n            x = self.conv11(x)\n            x = self.sigmoid(x)\n       \n            return x\n","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:52:56.826447Z","iopub.execute_input":"2023-05-13T17:52:56.827024Z","iopub.status.idle":"2023-05-13T17:52:56.856501Z","shell.execute_reply.started":"2023-05-13T17:52:56.826971Z","shell.execute_reply":"2023-05-13T17:52:56.855349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"state_dict = torch.load('/kaggle/input/model-2/model_Unet_3D_32.pth')\nmodel_CNN = UNET2_3D(stack_count)\nmodel_CNN.load_state_dict(state_dict)\nmodel_CNN.to(device)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:52:56.858211Z","iopub.execute_input":"2023-05-13T17:52:56.858651Z","iopub.status.idle":"2023-05-13T17:53:01.711306Z","shell.execute_reply.started":"2023-05-13T17:52:56.858609Z","shell.execute_reply":"2023-05-13T17:53:01.710107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_volume_data(path, stack_count, cube_size, model_CNN, device, batch_size):\n    \n    # Инициализация стэка объемных данных.\n    stack_shape = cv2.imread(os.path.join(path, os.listdir(path)[0])).shape\n    stack_3 = torch.zeros((stack_count, stack_shape[0], stack_shape[1])).float()\n\n    # Загрузка и обработка изображений.\n    #kernel_3 = np.array([[-1,-1,-1], [-1,9,-1], [-1,-1,-1]])\n    for idx, filename in enumerate(tqdm_notebook(os.listdir(path)[:stack_count])):\n        slice_data = cv2.imread(os.path.join(path, filename))\n        slice_data = cv2.cvtColor(slice_data, cv2.COLOR_BGR2GRAY)\n        #slice_data = cv2.filter2D(slice_data, -1, kernel)\n        #slice_data = cv2.equalizeHist(slice_data)\n        #slice_data = cv2.medianBlur(slice_data, 3)\n        slice_data = slice_data/255\n        stack_3[idx, :, :] = torch.tensor(slice_data).float()\n\n    # Обработка блоков данных на модели CNN.\n    padding = (cube_size - stack_3.shape[1] % cube_size,\n               cube_size - stack_3.shape[2] % cube_size)\n    \n    stack_shape = (1, stack_3.shape[1]+padding[0], stack_3.shape[2]+padding[1])\n    \n    stack_3 = F.pad(stack_3, (0, padding[1], 0, padding[0], 0, 0))\n    #image = F.pad(stack_3, (padding[1], 0, padding[0], 0, 0, 0))\n    blocks_stack = []\n    for i in tqdm_notebook(range(0, stack_3.shape[1], block_size)):\n        for j in range(0, stack_3.shape[2], block_size):\n            block = stack_3[:, i:i+block_size, j:j+block_size]\n            block = block.unsqueeze(0).float()\n            blocks_stack.append(block)\n    \n    del stack_3\n    gc.collect()\n    blocks_stack = torch.cat(blocks_stack).float().to(device)\n    gc.collect()\n    blocks_stack = torch.split(blocks_stack, batch_size, dim=0)\n    gc.collect()\n  \n    y_start = -cube_size\n    i = 0\n    \n    \n    stack_3 = torch.zeros(stack_shape).float()\n    gc.collect()\n    with torch.no_grad():\n        for images in tqdm_notebook(blocks_stack): \n            images = images.to(device).float()\n            images = images.unsqueeze(1)\n            outputs= model_CNN(images)\n            outputs = torch.squeeze(outputs, dim=1)\n\n            for cube in outputs.to('cpu'):\n                    x_start = (i * cube_size) % stack_shape[2]\n                    if x_start == 0:\n                        y_start +=cube_size\n                    stack_3[:, y_start:y_start+cube.shape[1], x_start:x_start+cube.shape[0]] += cube\n                    i +=1\n     \n    stack_3 = stack_3[:, :stack_shape[1]-padding[0], :stack_shape[2]-padding[1]]\n    gc.collect()\n    return stack_3","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:53:01.713060Z","iopub.execute_input":"2023-05-13T17:53:01.713767Z","iopub.status.idle":"2023-05-13T17:53:01.733308Z","shell.execute_reply.started":"2023-05-13T17:53:01.713724Z","shell.execute_reply":"2023-05-13T17:53:01.732097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Функция для кодирования маски в формат RLE\ndef rle(img):\n    flat_img = img.flatten()\n    flat_img = np.where(flat_img >= 0.4, 1, 0).astype(np.uint16)\n\n    starts = np.array((flat_img[:-1] == 0) & (flat_img[1:] == 1))\n    ends = np.array((flat_img[:-1] == 1) & (flat_img[1:] == 0))\n    starts_ix = np.where(starts)[0]  + 2\n    ends_ix = np.where(ends)[0]+ 2\n    if ends_ix.shape != starts_ix.shape:\n        lengths = ends_ix - starts_ix[1:]\n    else:\n        lengths = ends_ix - starts_ix\n    \n    gc.collect()\n    return starts_ix, lengths\n# Загружаем изображения и применяем функцию rle\npath = '/kaggle/input/vesuvius-challenge-ink-detection/test/b/surface_volume'\nbirdseye = process_volume_data(path, stack_count, cube_size, model_CNN, device, batch_size)[0]\nbirdseye_rle = rle(birdseye)\n\ndel birdseye\ntorch.cuda.empty_cache()\ngc.collect()\n\npath = '/kaggle/input/vesuvius-challenge-ink-detection/test/a/surface_volume'\ninklabels =  process_volume_data(path, stack_count, cube_size, model_CNN, device, batch_size)[0]\ninklabels_rle = rle(inklabels)\n\n\n# Записываем результаты в файл построчно\nwith open('submission.csv', 'w') as f:\n    f.write('Id,Predicted\\n')\n    f.write('a,{}\\n'.format(' '.join(map(str, sum(zip(*inklabels_rle), ())))))\n    f.write('b,{}\\n'.format(' '.join(map(str, sum(zip(*birdseye_rle), ())))))\n\nprint('Результаты записаны в файл submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-13T17:53:01.738215Z","iopub.execute_input":"2023-05-13T17:53:01.738521Z","iopub.status.idle":"2023-05-13T18:03:49.416105Z","shell.execute_reply.started":"2023-05-13T17:53:01.738492Z","shell.execute_reply":"2023-05-13T18:03:49.414828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-13T18:03:49.417928Z","iopub.execute_input":"2023-05-13T18:03:49.418356Z","iopub.status.idle":"2023-05-13T18:03:49.462709Z","shell.execute_reply.started":"2023-05-13T18:03:49.418313Z","shell.execute_reply":"2023-05-13T18:03:49.461714Z"},"trusted":true},"execution_count":null,"outputs":[]}]}