{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Let's walk around the data. Here's the contents.\n\n* train.csv\n* test.csv\n* sample_submission.csv\n* train_images\n* test_images\n* train_annotations","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport cv2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-13T10:52:56.446919Z","iopub.execute_input":"2022-07-13T10:52:56.448060Z","iopub.status.idle":"2022-07-13T10:52:56.698035Z","shell.execute_reply.started":"2022-07-13T10:52:56.447950Z","shell.execute_reply":"2022-07-13T10:52:56.696852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COMP_DIR = os.path.join(\n    '..', 'input', 'hubmap-organ-segmentation')\nTRAIN_IMAGES_DIR = \\\n    os.path.join(COMP_DIR, 'train_images')\nTEST_IMAGES_DIR = \\\n    os.path.join(COMP_DIR, 'test_images')\nTRAIN_ANNOTS_DIR = \\\n    os.path.join(COMP_DIR, 'train_annotations')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:56.701219Z","iopub.execute_input":"2022-07-13T10:52:56.701760Z","iopub.status.idle":"2022-07-13T10:52:56.709166Z","shell.execute_reply.started":"2022-07-13T10:52:56.701709Z","shell.execute_reply":"2022-07-13T10:52:56.707941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# train.csv\n\n'train.csv' contains 10 columns and 351 rows.","metadata":{}},{"cell_type":"code","source":"train_csv_path = os.path.join(COMP_DIR, 'train.csv')\ntrain_df = pd.read_csv(train_csv_path)\n\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:56.710919Z","iopub.execute_input":"2022-07-13T10:52:56.711601Z","iopub.status.idle":"2022-07-13T10:52:57.094001Z","shell.execute_reply.started":"2022-07-13T10:52:56.711560Z","shell.execute_reply":"2022-07-13T10:52:57.092853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Organ\n\nThere are 5 kinds of organs. 50 ~ 100 images for each.","metadata":{}},{"cell_type":"code","source":"organ_value_counts = train_df['organ'].value_counts()\n\norgan_value_counts","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.096974Z","iopub.execute_input":"2022-07-13T10:52:57.097753Z","iopub.status.idle":"2022-07-13T10:52:57.114211Z","shell.execute_reply.started":"2022-07-13T10:52:57.097699Z","shell.execute_reply":"2022-07-13T10:52:57.113063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 5))\nplt.bar(organ_value_counts.index, organ_value_counts.values)\nplt.title('Number of images for each organs in training data')\nplt.xlabel('organ')\nplt.ylabel('count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.115780Z","iopub.execute_input":"2022-07-13T10:52:57.116347Z","iopub.status.idle":"2022-07-13T10:52:57.327832Z","shell.execute_reply.started":"2022-07-13T10:52:57.116315Z","shell.execute_reply":"2022-07-13T10:52:57.326672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data source\n\nAll training data are derived from HPA.","metadata":{}},{"cell_type":"code","source":"data_source_value_counts = train_df['data_source'].value_counts()\n\ndata_source_value_counts","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.329189Z","iopub.execute_input":"2022-07-13T10:52:57.329513Z","iopub.status.idle":"2022-07-13T10:52:57.339913Z","shell.execute_reply.started":"2022-07-13T10:52:57.329483Z","shell.execute_reply":"2022-07-13T10:52:57.338544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Image height/width\n\nMost of them are 3,000 x 3,000. All images are square.","metadata":{}},{"cell_type":"code","source":"image_dim_value_counts = train_df[['img_height', 'img_width']].value_counts()\n\nimage_dim_value_counts","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.342012Z","iopub.execute_input":"2022-07-13T10:52:57.342520Z","iopub.status.idle":"2022-07-13T10:52:57.361367Z","shell.execute_reply.started":"2022-07-13T10:52:57.342485Z","shell.execute_reply":"2022-07-13T10:52:57.360546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(train_df['img_height'] != train_df['img_width'])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.362580Z","iopub.execute_input":"2022-07-13T10:52:57.363042Z","iopub.status.idle":"2022-07-13T10:52:57.371364Z","shell.execute_reply.started":"2022-07-13T10:52:57.363000Z","shell.execute_reply":"2022-07-13T10:52:57.370114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pixel size\n\nAll pixel size is 0.4 $\\mu$m.","metadata":{}},{"cell_type":"code","source":"pixel_size_value_counts = train_df['pixel_size'].value_counts()\n\npixel_size_value_counts","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.373158Z","iopub.execute_input":"2022-07-13T10:52:57.373494Z","iopub.status.idle":"2022-07-13T10:52:57.386586Z","shell.execute_reply.started":"2022-07-13T10:52:57.373464Z","shell.execute_reply":"2022-07-13T10:52:57.385689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tissue thickness\n\nAll tissue thickness is 4 $\\mu$m.","metadata":{}},{"cell_type":"code","source":"tissue_th_value_counts = train_df['tissue_thickness'].value_counts()\n\ntissue_th_value_counts","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.390390Z","iopub.execute_input":"2022-07-13T10:52:57.391131Z","iopub.status.idle":"2022-07-13T10:52:57.400421Z","shell.execute_reply.started":"2022-07-13T10:52:57.391051Z","shell.execute_reply":"2022-07-13T10:52:57.399495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Age\n\n* Minumum: 21\n* Mean: 60\n* Maximum: 84\n* Peaks at around 60 and 80.","metadata":{}},{"cell_type":"code","source":"train_df['age'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.401886Z","iopub.execute_input":"2022-07-13T10:52:57.402393Z","iopub.status.idle":"2022-07-13T10:52:57.420130Z","shell.execute_reply.started":"2022-07-13T10:52:57.402351Z","shell.execute_reply":"2022-07-13T10:52:57.418937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 5))\nplt.hist(train_df['age'], bins=20)\nplt.title('Distribution of age in training data')\nplt.xlabel('age')\nplt.ylabel('count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.422203Z","iopub.execute_input":"2022-07-13T10:52:57.422962Z","iopub.status.idle":"2022-07-13T10:52:57.607197Z","shell.execute_reply.started":"2022-07-13T10:52:57.422913Z","shell.execute_reply":"2022-07-13T10:52:57.606152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sex\n\n2/3 is male and 1/3 is female.","metadata":{}},{"cell_type":"code","source":"sex_value_counts = train_df['sex'].value_counts()\n\nsex_value_counts","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.608710Z","iopub.execute_input":"2022-07-13T10:52:57.609251Z","iopub.status.idle":"2022-07-13T10:52:57.617851Z","shell.execute_reply.started":"2022-07-13T10:52:57.609218Z","shell.execute_reply":"2022-07-13T10:52:57.616688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(5, 5))\nplt.bar(sex_value_counts.index, sex_value_counts.values)\nplt.title('Number of males/females in training data')\nplt.xlabel('sex')\nplt.ylabel('count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.619535Z","iopub.execute_input":"2022-07-13T10:52:57.620282Z","iopub.status.idle":"2022-07-13T10:52:57.730148Z","shell.execute_reply.started":"2022-07-13T10:52:57.620248Z","shell.execute_reply":"2022-07-13T10:52:57.728953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%html\n<style>\ntable {float:left}\n</style>","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-07-13T10:52:57.731885Z","iopub.execute_input":"2022-07-13T10:52:57.732320Z","iopub.status.idle":"2022-07-13T10:52:57.742177Z","shell.execute_reply.started":"2022-07-13T10:52:57.732275Z","shell.execute_reply":"2022-07-13T10:52:57.741002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# test.csv\n\nThe test data contains just one row. 'rle', 'age', and 'sex' columns are missing.\n\n|  # | Column           | Train | Test |\n|---:|:-----------------|:-----:|:----:|\n|  1 | id               |   〇  |   〇  |\n|  2 | organ            |   〇  |   〇  |\n|  3 | data_source      |   〇  |   〇  |\n|  4 | img_height       |   〇  |   〇  |\n|  5 | img_width        |   〇  |   〇  |\n|  6 | pixel_size       |   〇  |   〇  |\n|  7 | tissue_thickness |   〇  |   〇  |\n|  8 | rle              |   〇  |  --- |\n|  9 | age              |   〇  |  --- |\n| 10 | sex              |   〇  |  --- |","metadata":{}},{"cell_type":"code","source":"test_csv_path = os.path.join(COMP_DIR, 'test.csv')\ntest_df = pd.read_csv(test_csv_path)\n\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.744070Z","iopub.execute_input":"2022-07-13T10:52:57.744820Z","iopub.status.idle":"2022-07-13T10:52:57.765417Z","shell.execute_reply.started":"2022-07-13T10:52:57.744771Z","shell.execute_reply":"2022-07-13T10:52:57.764659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# sample_submission.csv\n\nNeeds 'id' and 'rle' columns for our submissions.","metadata":{}},{"cell_type":"code","source":"sample_sub_csv_path = os.path.join(COMP_DIR, 'sample_submission.csv')\nsample_sub_df = pd.read_csv(sample_sub_csv_path)\n\nsample_sub_df","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.766582Z","iopub.execute_input":"2022-07-13T10:52:57.767441Z","iopub.status.idle":"2022-07-13T10:52:57.784985Z","shell.execute_reply.started":"2022-07-13T10:52:57.767395Z","shell.execute_reply":"2022-07-13T10:52:57.784017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# train_images\n\n* There are 351 files, the same as the train.csv row count.\n* Mean file size is 26.6 Mbytes.\n* Total file size is 9.35 Gbytes.","metadata":{}},{"cell_type":"code","source":"train_images_ls = os.listdir(TRAIN_IMAGES_DIR)\ntrain_images_df = pd.DataFrame({'file_name': train_images_ls})\n\ntrain_images_df","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.786255Z","iopub.execute_input":"2022-07-13T10:52:57.786759Z","iopub.status.idle":"2022-07-13T10:52:57.823244Z","shell.execute_reply.started":"2022-07-13T10:52:57.786726Z","shell.execute_reply":"2022-07-13T10:52:57.822090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images_df['file_size'] = train_images_df['file_name'] \\\n    .apply(lambda fn: \n           os.path.getsize(os.path.join(TRAIN_IMAGES_DIR, fn)))\n\nprint(\"Mean file size:  {0:,.0f}\".format(train_images_df['file_size'].mean()))\nprint(\"Total file size: {0:,}\".format(train_images_df['file_size'].sum()))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.824545Z","iopub.execute_input":"2022-07-13T10:52:57.825080Z","iopub.status.idle":"2022-07-13T10:52:57.837311Z","shell.execute_reply.started":"2022-07-13T10:52:57.825049Z","shell.execute_reply":"2022-07-13T10:52:57.836111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_image(image_id, image_dir):\n    image_file_name = str(image_id) + '.tiff'\n    image_path = os.path.join(image_dir, image_file_name)\n    img = Image.open(image_path)\n    return img","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.839202Z","iopub.execute_input":"2022-07-13T10:52:57.839929Z","iopub.status.idle":"2022-07-13T10:52:57.846076Z","shell.execute_reply.started":"2022-07-13T10:52:57.839881Z","shell.execute_reply":"2022-07-13T10:52:57.844890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_mask(rle, img_height, img_width):\n    # Split list of numbers by white space.\n    seg_list = rle.split()\n    # Numbers at 0th, 2nd, 4th, ... are start of a segment.\n    start_list = seg_list[0::2]\n    # Numbers at 1st, 3rd, 5th, ... are length of a segment.\n    length_list = seg_list[1::2]\n    \n    # Prepare 1D array, then fill 255 to each segments.\n    mask = np.zeros(img_height * img_width, dtype=np.uint8)\n    for start, length in zip(start_list, length_list):\n        start = int(start)\n        length = int(length)\n        mask[start:start + length] = 255\n\n    # Reshape to 2D, then swap from (x, y) to (y, x) by transposing.\n    mask = np.reshape(mask, (img_height, img_width))\n    mask = mask.T\n    mask_img = Image.fromarray(mask)\n    return mask_img","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.847350Z","iopub.execute_input":"2022-07-13T10:52:57.847903Z","iopub.status.idle":"2022-07-13T10:52:57.859715Z","shell.execute_reply.started":"2022-07-13T10:52:57.847864Z","shell.execute_reply":"2022-07-13T10:52:57.858548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def draw_image_mask(row_df):\n    row_df = row_df.reset_index()\n    image_id, organ, rle, img_height, img_width = \\\n        row_df.loc[0, ['id', 'organ', 'rle', 'img_height', 'img_width']]\n    img = read_image(image_id, TRAIN_IMAGES_DIR)\n    mask = make_mask(rle, img_height, img_width)\n    \n    plt.figure(figsize=(12, 12))\n    plt.imshow(img)\n    plt.imshow(mask, alpha=0.2)\n    title = \"Id={0}, Organ={1}, Size={2}x{3}\".format(\n        image_id, organ, img_height, img_width)\n    plt.title(title)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.861353Z","iopub.execute_input":"2022-07-13T10:52:57.861856Z","iopub.status.idle":"2022-07-13T10:52:57.874277Z","shell.execute_reply.started":"2022-07-13T10:52:57.861814Z","shell.execute_reply":"2022-07-13T10:52:57.872919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Draw sample images/masks for each organs.","metadata":{}},{"cell_type":"code","source":"for organ in train_df['organ'].unique():\n    train_organ_df = train_df[train_df['organ'] == organ]\n    sample_row = train_organ_df.sample(n=1)\n    draw_image_mask(sample_row)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:52:57.876040Z","iopub.execute_input":"2022-07-13T10:52:57.876402Z","iopub.status.idle":"2022-07-13T10:53:11.615176Z","shell.execute_reply.started":"2022-07-13T10:52:57.876370Z","shell.execute_reply":"2022-07-13T10:53:11.613936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# test_images\n\nLooks very different from train images.","metadata":{}},{"cell_type":"code","source":"test_images_ls = os.listdir(TEST_IMAGES_DIR)\ntest_images_df = pd.DataFrame({'file_name': test_images_ls})\n\ntest_images_df","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:53:11.619319Z","iopub.execute_input":"2022-07-13T10:53:11.619717Z","iopub.status.idle":"2022-07-13T10:53:11.632428Z","shell.execute_reply.started":"2022-07-13T10:53:11.619683Z","shell.execute_reply":"2022-07-13T10:53:11.631443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_id = test_df.loc[0, 'id']\ntest_image = read_image(test_id, TEST_IMAGES_DIR)\n\nplt.figure(figsize=(12, 12))\nplt.imshow(test_image)\nplt.title(test_id)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:53:11.634172Z","iopub.execute_input":"2022-07-13T10:53:11.634694Z","iopub.status.idle":"2022-07-13T10:53:12.650447Z","shell.execute_reply.started":"2022-07-13T10:53:11.634659Z","shell.execute_reply":"2022-07-13T10:53:12.649198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# train_annotations\n\n* 351 json files.\n* Contains boundary data for each images.","metadata":{}},{"cell_type":"code","source":"train_annots_ls = os.listdir(TRAIN_ANNOTS_DIR)\ntrain_annots_df = pd.DataFrame({'file_name': train_annots_ls})\n\ntrain_annots_df","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:53:12.652158Z","iopub.execute_input":"2022-07-13T10:53:12.652984Z","iopub.status.idle":"2022-07-13T10:53:12.703672Z","shell.execute_reply.started":"2022-07-13T10:53:12.652924Z","shell.execute_reply":"2022-07-13T10:53:12.702379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Each row contains a series of points for a boundary.","metadata":{}},{"cell_type":"code","source":"sample_image_id = train_df.loc[0, 'id']\nsample_json_file_name = str(sample_image_id) + '.json'\nsample_json_path = os.path.join(TRAIN_ANNOTS_DIR, sample_json_file_name)\nsample_json_df = pd.read_json(sample_json_path)\n\nsample_json_df","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:53:12.707512Z","iopub.execute_input":"2022-07-13T10:53:12.707910Z","iopub.status.idle":"2022-07-13T10:53:12.858965Z","shell.execute_reply.started":"2022-07-13T10:53:12.707876Z","shell.execute_reply":"2022-07-13T10:53:12.857847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_boundary_image(json_df, img_height, img_width):\n    bounds = np.zeros((img_height, img_width), dtype=np.uint8)\n\n    for _, row in json_df.iterrows():\n        points = np.array(row.dropna().tolist())\n        bounds = cv2.polylines(\n            bounds, [points], isClosed=True, color=1, thickness=25)\n\n    bound_image = Image.fromarray(bounds)\n    return bound_image","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:53:12.860742Z","iopub.execute_input":"2022-07-13T10:53:12.861076Z","iopub.status.idle":"2022-07-13T10:53:12.869593Z","shell.execute_reply.started":"2022-07-13T10:53:12.861045Z","shell.execute_reply":"2022-07-13T10:53:12.867872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def draw_image_boundary(row_df):\n    row_df = row_df.reset_index()\n    image_id, organ, img_height, img_width = \\\n        row_df.loc[0, ['id', 'organ', 'img_height', 'img_width']]\n    img = read_image(image_id, TRAIN_IMAGES_DIR)\n    \n    json_file_name = str(image_id) + '.json'\n    json_path = os.path.join(TRAIN_ANNOTS_DIR, json_file_name)\n    json_df = pd.read_json(json_path)\n    bound_image = make_boundary_image(json_df, img_height, img_width)\n    \n    plt.figure(figsize=(12, 12))\n    plt.imshow(img)\n    plt.imshow(bound_image, alpha=0.2)\n    title = \"Id={0}, Organ={1}, Size={2}x{3}\".format(\n        image_id, organ, img_height, img_width)\n    plt.title(title)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:53:12.873852Z","iopub.execute_input":"2022-07-13T10:53:12.874223Z","iopub.status.idle":"2022-07-13T10:53:12.882982Z","shell.execute_reply.started":"2022-07-13T10:53:12.874194Z","shell.execute_reply":"2022-07-13T10:53:12.881769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for organ in train_df['organ'].unique():\n    train_organ_df = train_df[train_df['organ'] == organ]\n    sample_row = train_organ_df.sample(n=1)\n    draw_image_boundary(sample_row)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T10:53:12.884969Z","iopub.execute_input":"2022-07-13T10:53:12.885444Z","iopub.status.idle":"2022-07-13T10:53:27.343041Z","shell.execute_reply.started":"2022-07-13T10:53:12.885398Z","shell.execute_reply":"2022-07-13T10:53:27.341844Z"},"trusted":true},"execution_count":null,"outputs":[]}]}