{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hello fellow Kagglers,\n\nThis notebook demonstrates the preprocessing process for generating the training and validation data used from training in [this](https://www.kaggle.com/markwijkhuizen/sartorius-training-upsampling-tf-public) notebook.\n\nThe training data is split in 4 folds and saved as compressed numpy files.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom PIL import Image, ImageEnhance\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import KFold\n\nimport glob\nimport sys\nimport cv2\nimport imageio\nimport joblib\nimport math\nimport warnings\nimport os\n\ntqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:14:10.472031Z","iopub.execute_input":"2021-10-18T18:14:10.473312Z","iopub.status.idle":"2021-10-18T18:14:11.946641Z","shell.execute_reply.started":"2021-10-18T18:14:10.473180Z","shell.execute_reply":"2021-10-18T18:14:11.945560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# All training data have the same dimensions\nHEIGHT = 520\nWIDTH = 704\n\ntrain = pd.read_csv('/kaggle/input/sartorius-cell-instance-segmentation/train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:14:27.601590Z","iopub.execute_input":"2021-10-18T18:14:27.602304Z","iopub.status.idle":"2021-10-18T18:14:27.967892Z","shell.execute_reply.started":"2021-10-18T18:14:27.602263Z","shell.execute_reply":"2021-10-18T18:14:27.967103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add image file path\ndef get_file_path(image_id):\n    return f'/kaggle/input/sartorius-cell-instance-segmentation/train/{image_id}.png'\n\ntrain['file_path'] = train['id'].apply(get_file_path)","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:14:47.241715Z","iopub.execute_input":"2021-10-18T18:14:47.241994Z","iopub.status.idle":"2021-10-18T18:14:47.279996Z","shell.execute_reply.started":"2021-10-18T18:14:47.241966Z","shell.execute_reply":"2021-10-18T18:14:47.278995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add image shape\ntrain['shape'] = train[['height', 'width']].apply(tuple, axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:14:47.448230Z","iopub.execute_input":"2021-10-18T18:14:47.448533Z","iopub.status.idle":"2021-10-18T18:14:48.388728Z","shell.execute_reply.started":"2021-10-18T18:14:47.448493Z","shell.execute_reply":"2021-10-18T18:14:48.387788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train.head())","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:14:48.745102Z","iopub.execute_input":"2021-10-18T18:14:48.745372Z","iopub.status.idle":"2021-10-18T18:14:48.769658Z","shell.execute_reply.started":"2021-10-18T18:14:48.745345Z","shell.execute_reply":"2021-10-18T18:14:48.768981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train.info())","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:14:48.902347Z","iopub.execute_input":"2021-10-18T18:14:48.903308Z","iopub.status.idle":"2021-10-18T18:14:48.999420Z","shell.execute_reply.started":"2021-10-18T18:14:48.903260Z","shell.execute_reply":"2021-10-18T18:14:48.998502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analysis","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 8))\ntrain['cell_type'].value_counts().plot(kind='pie', autopct='%1.1f%%', title='Cell Type Distribution')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:15:08.072694Z","iopub.execute_input":"2021-10-18T18:15:08.072990Z","iopub.status.idle":"2021-10-18T18:15:08.198891Z","shell.execute_reply.started":"2021-10-18T18:15:08.072960Z","shell.execute_reply":"2021-10-18T18:15:08.197973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# RLE Decode","metadata":{}},{"cell_type":"code","source":"# source: https://www.kaggle.com/stainsby/fast-tested-rle\ndef rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)","metadata":{"execution":{"iopub.status.busy":"2021-10-17T18:07:23.865369Z","iopub.execute_input":"2021-10-17T18:07:23.865617Z","iopub.status.idle":"2021-10-17T18:07:23.872217Z","shell.execute_reply.started":"2021-10-17T18:07:23.865589Z","shell.execute_reply":"2021-10-17T18:07:23.871469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inspiration: https://www.kaggle.com/paulorzp/run-length-encode-and-decode\n# Modified to retrieve the full mask for a given image id\ndef rle_decode(image_id):\n    rows = train.loc[train['id'] == image_id]\n    # Shape\n    shape = train.loc[0, 'shape']\n    # Image Shape flattenned\n    mask = np.zeros(shape=shape[0]*shape[1], dtype=np.uint8)\n    \n    # Add all image masks\n    for idx, row in rows.iterrows():\n        s = row['annotation'].split()\n        starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n        starts -= 1\n        ends = starts + lengths\n        for start, end in zip(starts, ends):\n            mask[start:end] = 1\n    \n    return mask.reshape(shape)","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:16:51.683678Z","iopub.execute_input":"2021-10-18T18:16:51.683996Z","iopub.status.idle":"2021-10-18T18:16:51.693127Z","shell.execute_reply.started":"2021-10-18T18:16:51.683964Z","shell.execute_reply":"2021-10-18T18:16:51.692183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image Examples","metadata":{}},{"cell_type":"code","source":"# Shows a batch of images\ndef show_image_and_masks(rows=4, cols=4):\n    # Figure\n    fig, axes = plt.subplots(nrows=rows, ncols=cols, figsize=(cols*8, rows*6))\n    # Unique Image Ids\n    image_ids = train['id'].unique()\n    \n    for r in range(rows):\n        image_id = image_ids[r]\n        df_row = train.loc[train['id'] == image_id].head(1).squeeze()\n        # Rad Image\n        image = imageio.imread(df_row['file_path'])\n        # Enhance Contrast\n        # inspiration from: https://www.kaggle.com/dschettler8845/sartorius-segmentation-eda-efficientdet-tf\n        image = np.array(ImageEnhance.Contrast(Image.fromarray(image)).enhance(16))\n        # Plot Image\n        axes[r, 0].imshow(image)\n        axes[r, 0].set_title(f'Image {image_id}', size=18)\n        axes[r, 0].axis(False)\n        \n        # Image\n        img_norm = imageio.imread(df_row['file_path'])\n        img_norm = abs(img_norm.astype(np.float32) - 127) * 2\n        img_norm[img_norm > 255] = 255\n        axes[r, 1].imshow(img_norm)\n        axes[r, 1].set_title(f'Image {image_id} 0-255', size=18)\n        axes[r, 1].axis(False)\n        \n        # Mask\n        mask = rle_decode(image_ids[r])\n        axes[r, 2].imshow(mask)\n        axes[r, 2].set_title('Mask', size=18)\n        axes[r, 2].axis(False)\n        \n        # Image with Mask\n        axes[r, 3].imshow(cv2.cvtColor(image, cv2.COLOR_GRAY2RGB))\n        axes[r, 3].imshow((np.expand_dims(mask, axis=2) * np.array([255, 0, 0])), alpha=0.50)\n        axes[r, 3].set_title('Image and Mask', size=18)\n        axes[r, 3].axis(False)\n            \n            \n    # Adjust Vertical Space Between Subplots\n    fig.subplots_adjust(wspace=0.10)","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:17:51.887101Z","iopub.execute_input":"2021-10-18T18:17:51.887701Z","iopub.status.idle":"2021-10-18T18:17:51.902801Z","shell.execute_reply.started":"2021-10-18T18:17:51.887645Z","shell.execute_reply":"2021-10-18T18:17:51.901680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_image_and_masks(rows=8)","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:17:52.303652Z","iopub.execute_input":"2021-10-18T18:17:52.304225Z","iopub.status.idle":"2021-10-18T18:17:59.547975Z","shell.execute_reply.started":"2021-10-18T18:17:52.304180Z","shell.execute_reply":"2021-10-18T18:17:59.547166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Test KFolds","metadata":{}},{"cell_type":"code","source":"def create_fold(idxs, fold, subset):\n    # Images\n    X = np.empty(shape=[len(idxs), HEIGHT, WIDTH], dtype=np.uint8)\n    for idx, image_idx_idx in enumerate(tqdm(idxs)):\n        image_id = id_unique[image_idx_idx]\n        image = imageio.imread(image_id2file_path[image_id])\n        image = np.array(ImageEnhance.Contrast(Image.fromarray(image)).enhance(16))\n\n        X[idx] = image\n    # Save X as compressed Numpy Array\n    np.savez_compressed(f'X_fold_{fold}_{subset}.npz', v=X)\n    \n    # Labels\n    y = np.empty(shape=[len(idxs), HEIGHT, WIDTH], dtype=np.uint8)\n    for idx, image_idx_idx in enumerate(tqdm(idxs)):\n        image_id = id_unique[image_idx_idx]\n        y[idx] = rle_decode(image_id)\n    # Save y as compressed Numpy Array\n    np.savez_compressed(f'y_fold_{fold}_{subset}.npz', v=y)","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:18:03.412514Z","iopub.execute_input":"2021-10-18T18:18:03.413265Z","iopub.status.idle":"2021-10-18T18:18:03.422794Z","shell.execute_reply.started":"2021-10-18T18:18:03.413211Z","shell.execute_reply":"2021-10-18T18:18:03.421502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Unique Image Id's\nid_unique = train['id'].unique()\nprint(f'There are {len(id_unique)} unique image ids')","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:18:20.907069Z","iopub.execute_input":"2021-10-18T18:18:20.907813Z","iopub.status.idle":"2021-10-18T18:18:20.919967Z","shell.execute_reply.started":"2021-10-18T18:18:20.907771Z","shell.execute_reply":"2021-10-18T18:18:20.919018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Maps a given image id to the corresponding image file path\nimage_id2file_path = train.groupby('id')[['id', 'file_path']].head(1).set_index('id').squeeze().to_dict()","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:18:37.096513Z","iopub.execute_input":"2021-10-18T18:18:37.097308Z","iopub.status.idle":"2021-10-18T18:18:37.121083Z","shell.execute_reply.started":"2021-10-18T18:18:37.097265Z","shell.execute_reply":"2021-10-18T18:18:37.120123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make 4 folds\nN_FOLDS = 4\nkf = KFold(n_splits=N_FOLDS, shuffle=True, random_state=42)\n# Make NPY Files\nfor fold, (train_idxs, val_idxs) in enumerate(kf.split(id_unique)):\n    train_len = len(train_idxs)\n    val_len = len(val_idxs)\n    print(f'FOLD {fold} | train samples: {train_len}, val sampless: {val_len}')\n    \n    create_fold(train_idxs, fold, 'train')\n    create_fold(val_idxs, fold, 'val')","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:18:48.197805Z","iopub.execute_input":"2021-10-18T18:18:48.198501Z","iopub.status.idle":"2021-10-18T18:20:08.724048Z","shell.execute_reply.started":"2021-10-18T18:18:48.198455Z","shell.execute_reply":"2021-10-18T18:20:08.722797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check Numpy Files","metadata":{}},{"cell_type":"code","source":"# Shows a batch of images\ndef show_batch(images, rows=4, cols=2):\n    fig, axes = plt.subplots(nrows=rows, ncols=cols, figsize=(cols*8, rows*6))\n    for r in range(rows):\n        for c in range(cols):\n            axes[r, c].imshow(images[r*rows+c])\n            axes[r, c].axis(False)","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:20:11.684745Z","iopub.execute_input":"2021-10-18T18:20:11.685004Z","iopub.status.idle":"2021-10-18T18:20:11.690762Z","shell.execute_reply.started":"2021-10-18T18:20:11.684976Z","shell.execute_reply":"2021-10-18T18:20:11.689931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_fold_0 = np.load('X_fold_0_train.npz')['v']\nshow_batch(X_fold_0)","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:20:12.013587Z","iopub.execute_input":"2021-10-18T18:20:12.014184Z","iopub.status.idle":"2021-10-18T18:20:14.980124Z","shell.execute_reply.started":"2021-10-18T18:20:12.014137Z","shell.execute_reply":"2021-10-18T18:20:14.979409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_fold_0 = np.load('y_fold_0_train.npz')['v']\nshow_batch(y_fold_0)","metadata":{"execution":{"iopub.status.busy":"2021-10-18T18:20:14.981379Z","iopub.execute_input":"2021-10-18T18:20:14.981787Z","iopub.status.idle":"2021-10-18T18:20:16.818811Z","shell.execute_reply.started":"2021-10-18T18:20:14.981754Z","shell.execute_reply":"2021-10-18T18:20:16.817871Z"},"trusted":true},"execution_count":null,"outputs":[]}]}