{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# RELEVANT PATCHES EXTRACTION- DATASET","metadata":{"execution":{"iopub.status.busy":"2023-05-26T12:00:09.719039Z","iopub.execute_input":"2023-05-26T12:00:09.720122Z","iopub.status.idle":"2023-05-26T12:00:24.574364Z","shell.execute_reply.started":"2023-05-26T12:00:09.720083Z","shell.execute_reply":"2023-05-26T12:00:24.572920Z"}}},{"cell_type":"markdown","source":"1. **inspired from https://www.kaggle.com/code/tanakar/2-5d-segmentaion-baseline-training**\n2. **In the above notebook the length of train images is around 14K where the mask isnt applied**\n3. **After applying the mnask the relevant patches come around 1.6K which almost reduced by 10 times**","metadata":{}},{"cell_type":"markdown","source":"# IMPORT THE LIBRARIES","metadata":{}},{"cell_type":"code","source":"pip install patchify","metadata":{"execution":{"iopub.status.busy":"2023-05-26T12:13:39.113595Z","iopub.execute_input":"2023-05-26T12:13:39.114035Z","iopub.status.idle":"2023-05-26T12:13:53.336824Z","shell.execute_reply.started":"2023-05-26T12:13:39.114004Z","shell.execute_reply":"2023-05-26T12:13:53.335501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport PIL.Image as Image\nfrom tifffile import tifffile\nfrom patchify import patchify\nfrom tqdm import tqdm\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-26T12:00:24.578726Z","iopub.execute_input":"2023-05-26T12:00:24.579150Z","iopub.status.idle":"2023-05-26T12:00:24.803100Z","shell.execute_reply.started":"2023-05-26T12:00:24.579114Z","shell.execute_reply":"2023-05-26T12:00:24.801888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_surface = lambda mode,fragment_id,i: f\"/kaggle/input/vesuvius-challenge-ink-detection/{mode}/{fragment_id}/surface_volume/{i:02}.tif\"\npath_mask = lambda mode,fragment_id : f'/kaggle/input/vesuvius-challenge-ink-detection/{mode}/{fragment_id}/mask.png'\npath_label = lambda mode,fragment_id :f'/kaggle/input/vesuvius-challenge-ink-detection/{mode}/{fragment_id}/inklabels.png'","metadata":{"execution":{"iopub.status.busy":"2023-05-26T12:00:30.523597Z","iopub.execute_input":"2023-05-26T12:00:30.524127Z","iopub.status.idle":"2023-05-26T12:00:30.530471Z","shell.execute_reply.started":"2023-05-26T12:00:30.524087Z","shell.execute_reply":"2023-05-26T12:00:30.529732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    size = 224\n    step = 224\n    valid_frag = 1\n    in_chans = 6","metadata":{"execution":{"iopub.status.busy":"2023-05-26T12:18:01.359228Z","iopub.execute_input":"2023-05-26T12:18:01.360314Z","iopub.status.idle":"2023-05-26T12:18:01.372387Z","shell.execute_reply.started":"2023-05-26T12:18:01.360257Z","shell.execute_reply":"2023-05-26T12:18:01.371135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def return_patch_details(fragment_id,size=224,mode = 'train'):\n\n    dic = {'fragment':[],'xmin':[],'ymin':[],'size':[]}\n    \n    mask = np.array(Image.open(path_mask('train',fragment_id)))\n    \n    patched_mask = patchify(mask,(size,size),step = size)\n\n    for j in tqdm(range(patched_mask.shape[0])):\n        for k in range(patched_mask.shape[1]):\n            mask_p = patched_mask[j,k,:,:]\n            if np.count_nonzero(mask_p)==size*size or mode == 'valid':    \n                dic['fragment'].append(fragment_id)\n                dic['ymin'].append(j*size)\n                dic['xmin'].append(k*size)\n                dic['size'].append(size)\n                    \n    df = pd.DataFrame(dic)\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-26T12:18:30.726330Z","iopub.execute_input":"2023-05-26T12:18:30.726738Z","iopub.status.idle":"2023-05-26T12:18:30.736386Z","shell.execute_reply.started":"2023-05-26T12:18:30.726709Z","shell.execute_reply":"2023-05-26T12:18:30.735234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def return_image_mask_patch(fragment_id,mode):\n    \n    in_chans = CFG.in_chans\n    size = CFG.size\n    valid_frag = CFG.valid_frag\n    \n    df = return_patch_details(fragment_id,size=size,mode=mode)\n    images = []\n\n    mid = 65 // 2\n    start = mid - in_chans // 2\n    end = mid + in_chans // 2\n    \n    label_ = np.array(Image.open(path_label('train',fragment_id)))\n    pad0 = 0\n    pad1 = 0\n    \n    if mode == 'valid':\n        pad0 = (size - label_.shape[0] % size)\n        pad1 = (size - label_.shape[1] % size)\n         \n    for i in tqdm(range(start,end)):\n        image = tifffile.imread(path_surface('train',fragment_id,i)).astype('float16')\n        image = np.pad(image, [(0, pad0), (0, pad1)], constant_values=0)\n        images.append(image)\n        \n    images = np.stack(images, axis=2).astype('float32')\n    label_ = np.pad(label_, [(0, pad0), (0, pad1)], constant_values=0)\n    \n    image = []\n    label = []\n    xyxy = []\n    size = df.iloc[0,3]\n    for j in range(df.shape[0]):\n        xmin = df.iloc[j,1]\n        ymin = df.iloc[j,2]\n        img = images[ymin:ymin+size,xmin:xmin+size,:]\n        lab = np.expand_dims(label_[ymin:ymin+size,xmin:xmin+size],0)\n        image.append(img)\n        label.append(lab)\n        xyxy.append([xmin,ymin,xmin+size,ymin+size])\n        \n    label = list(np.array(label).astype('float32')/255)\n    return image, label, xyxy","metadata":{"execution":{"iopub.status.busy":"2023-05-26T12:18:40.441152Z","iopub.execute_input":"2023-05-26T12:18:40.441534Z","iopub.status.idle":"2023-05-26T12:18:40.455400Z","shell.execute_reply.started":"2023-05-26T12:18:40.441506Z","shell.execute_reply":"2023-05-26T12:18:40.454527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_valid_dataset():\n    \n    in_chans = CFG.in_chans\n    size = CFG.size\n    valid_frag = CFG.valid_frag    \n    \n    train_images = []\n    train_label = []\n\n    valid_images = []\n    valid_label = []\n    \n\n    for fragment_id in range(1, 4):\n        \n        if fragment_id == valid_frag:\n            mode = 'valid'\n        else:\n            mode = 'train'\n        image, label, _ = return_image_mask_patch(fragment_id,mode = mode)\n        \n        if fragment_id == valid_frag:\n            valid_images += image\n            valid_label += label\n            valid_xyxy = _\n        else:\n            train_images += image\n            train_label += label\n\n    return train_images, train_label, valid_images, valid_label, valid_xyxy","metadata":{"execution":{"iopub.status.busy":"2023-05-26T12:04:08.351518Z","iopub.execute_input":"2023-05-26T12:04:08.351935Z","iopub.status.idle":"2023-05-26T12:04:08.360592Z","shell.execute_reply.started":"2023-05-26T12:04:08.351903Z","shell.execute_reply":"2023-05-26T12:04:08.359475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images, train_masks, valid_images, valid_masks, valid_xyxys = get_train_valid_dataset()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T12:04:25.684911Z","iopub.execute_input":"2023-05-26T12:04:25.685335Z","iopub.status.idle":"2023-05-26T12:05:15.080084Z","shell.execute_reply.started":"2023-05-26T12:04:25.685305Z","shell.execute_reply":"2023-05-26T12:05:15.078539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_xyxys = np.stack(valid_xyxys)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T12:21:03.835667Z","iopub.execute_input":"2023-05-26T12:21:03.837574Z","iopub.status.idle":"2023-05-26T12:21:03.850551Z","shell.execute_reply.started":"2023-05-26T12:21:03.837520Z","shell.execute_reply":"2023-05-26T12:21:03.849265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_images)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T12:12:14.252443Z","iopub.execute_input":"2023-05-26T12:12:14.252899Z","iopub.status.idle":"2023-05-26T12:12:14.261083Z","shell.execute_reply.started":"2023-05-26T12:12:14.252869Z","shell.execute_reply":"2023-05-26T12:12:14.259937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_images[0].shape","metadata":{"execution":{"iopub.status.busy":"2023-05-26T12:08:57.653470Z","iopub.execute_input":"2023-05-26T12:08:57.653962Z","iopub.status.idle":"2023-05-26T12:08:57.661874Z","shell.execute_reply.started":"2023-05-26T12:08:57.653930Z","shell.execute_reply":"2023-05-26T12:08:57.660780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets visualise what patches are we getting! While training!\nVisualized over Mask","metadata":{}},{"cell_type":"markdown","source":"Fragment 1. Patches of size 256*256 with no overlap","metadata":{}},{"cell_type":"code","source":"from matplotlib.patches import Rectangle\npath = path_mask('train',1)\nimg = np.array(Image.open(path))\nplt.imshow(img,cmap='gray')\ndf = return_patch_details(1,size=256,mode='train')\nfor i in range(df.shape[0]):\n    x = df.iloc[i,1]\n    y = df.iloc[i,2]\n    plt.gca().add_patch(Rectangle((x,y),256,256,\n                    edgecolor='red',\n                    facecolor='none',\n                    lw=1))\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T12:21:12.438217Z","iopub.execute_input":"2023-05-26T12:21:12.438684Z","iopub.status.idle":"2023-05-26T12:21:15.344380Z","shell.execute_reply.started":"2023-05-26T12:21:12.438652Z","shell.execute_reply":"2023-05-26T12:21:15.343196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lets store the data","metadata":{}},{"cell_type":"code","source":"import os\n\n# Create the output folder if it doesn't exist\nif not os.path.exists('output'):\n    os.makedirs('output')\n","metadata":{"execution":{"iopub.status.busy":"2023-05-26T12:27:07.960516Z","iopub.execute_input":"2023-05-26T12:27:07.961029Z","iopub.status.idle":"2023-05-26T12:27:07.967417Z","shell.execute_reply.started":"2023-05-26T12:27:07.960992Z","shell.execute_reply":"2023-05-26T12:27:07.966152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save train_images\nnp.save('output/train_images.npy', train_images)\n\n# Save train_masks\nnp.save('output/train_masks.npy', train_masks)\n\n# Save valid_images\nnp.save('output/valid_images.npy', valid_images)\n\n# Save valid_masks\nnp.save('output/valid_masks.npy', valid_masks)\n\n# Save valid_xyxys\nnp.save('output/valid_xyxys.npy', valid_xyxys)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T12:28:00.271595Z","iopub.execute_input":"2023-05-26T12:28:00.272031Z","iopub.status.idle":"2023-05-26T12:28:08.305587Z","shell.execute_reply.started":"2023-05-26T12:28:00.271998Z","shell.execute_reply":"2023-05-26T12:28:08.304382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}