{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport cv2\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-12T16:04:56.679861Z","iopub.execute_input":"2022-03-12T16:04:56.680182Z","iopub.status.idle":"2022-03-12T16:04:57.128764Z","shell.execute_reply.started":"2022-03-12T16:04:56.680145Z","shell.execute_reply":"2022-03-12T16:04:57.127608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/ultra-mnist/train.csv\")\ntrain_df[\"f_path\"] = \"/kaggle/input/ultra-mnist/train/\"+train_df.id+\".jpeg\"\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-03-12T16:04:33.344178Z","iopub.execute_input":"2022-03-12T16:04:33.344646Z","iopub.status.idle":"2022-03-12T16:04:33.389880Z","shell.execute_reply.started":"2022-03-12T16:04:33.344611Z","shell.execute_reply":"2022-03-12T16:04:33.388408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_img_binary(f_path, thresh=128, resize_to=1000):\n    _img = cv2.threshold(cv2.imread(f_path)[..., 0], thresh, 255, cv2.THRESH_BINARY)[-1]\n    return cv2.resize(_img, (resize_to, resize_to), interpolation=cv2.INTER_AREA)//255","metadata":{"execution":{"iopub.status.busy":"2022-03-12T16:12:47.739022Z","iopub.execute_input":"2022-03-12T16:12:47.739286Z","iopub.status.idle":"2022-03-12T16:12:47.745287Z","shell.execute_reply.started":"2022-03-12T16:12:47.739253Z","shell.execute_reply":"2022-03-12T16:12:47.744371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nplt.imshow(demo_img)","metadata":{"execution":{"iopub.status.busy":"2022-03-12T16:12:48.743941Z","iopub.execute_input":"2022-03-12T16:12:48.744530Z","iopub.status.idle":"2022-03-12T16:12:49.085140Z","shell.execute_reply.started":"2022-03-12T16:12:48.744502Z","shell.execute_reply":"2022-03-12T16:12:49.084348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-03-12T16:15:55.635545Z","iopub.execute_input":"2022-03-12T16:15:55.635853Z","iopub.status.idle":"2022-03-12T16:15:55.643933Z","shell.execute_reply.started":"2022-03-12T16:15:55.635822Z","shell.execute_reply":"2022-03-12T16:15:55.642825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keys = []\nfor i in range(5):\n    demo_img = load_img_binary(train_df.iloc[0].f_path)\n    print(len(rle_encode(demo_img).split()))\n    _keys = list(np.unique(np.array(rle_encode(demo_img).split())))\n    keys+=_keys\n    print(len(list(set(_keys))))\n    \nprint(len(list(set(keys))))\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-03-12T16:16:25.792489Z","iopub.execute_input":"2022-03-12T16:16:25.793093Z","iopub.status.idle":"2022-03-12T16:16:26.397489Z","shell.execute_reply.started":"2022-03-12T16:16:25.793005Z","shell.execute_reply":"2022-03-12T16:16:26.396828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ref.: https://www.kaggle.com/stainsby/fast-tested-rle\ndef rle_encode(img):\n    \"\"\" Generate Run Length Encoding\n    \n    Args:\n        img (np.array): \n            - 1 indicating mask\n            - 0 indicating background\n    \n    Returns: \n        run length as string formated\n    \"\"\"\n    \n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n \n# ref: https://www.kaggle.com/paulorzp/run-length-encode-and-decode\n# modified from: https://www.kaggle.com/inversion/run-length-decoding-quick-start\ndef rle_decode(mask_rle, shape, color=1):\n    \"\"\" TBD\n    \n    Args:\n        mask_rle (str): run-length as string formated (start length)\n        shape (tuple of ints): (height,width) of array to return \n    \n    Returns: \n        Mask (np.array)\n            - 1 indicating mask\n            - 0 indicating background\n\n    \"\"\"\n    # Split the string by space, then convert it into a integer array\n    s = np.array(mask_rle.split(), dtype=int)\n\n    # Every even value is the start, every odd value is the \"run\" length\n    starts = s[0::2] - 1\n    lengths = s[1::2]\n    ends = starts + lengths\n\n    # The image image is actually flattened since RLE is a 1D \"run\"\n    if len(shape)==3:\n        h, w, d = shape\n        img = np.zeros((h * w, d), dtype=np.float32)\n    else:\n        h, w = shape\n        img = np.zeros((h * w,), dtype=np.float32)\n\n    # The color here is actually just any integer you want!\n    for lo, hi in zip(starts, ends):\n        img[lo : hi] = color\n        \n    # Don't forget to change the image back to the original shape\n    return img.reshape(shape)","metadata":{"execution":{"iopub.status.busy":"2022-03-12T16:11:20.391704Z","iopub.execute_input":"2022-03-12T16:11:20.392079Z","iopub.status.idle":"2022-03-12T16:11:20.404425Z","shell.execute_reply.started":"2022-03-12T16:11:20.391959Z","shell.execute_reply":"2022-03-12T16:11:20.402980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}