{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Let's segment a variety of clothing types!\n# import modules and define utils","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd\npd.set_option(\"display.max_rows\", 101)\nimport os\nprint(os.listdir(\"../input\"))\nimport cv2\nimport json\nimport matplotlib.pyplot as plt\n%matplotlib inline\nplt.rcParams[\"font.size\"] = 15\nimport seaborn as sns\nfrom collections import Counter\nfrom PIL import Image\nimport math\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T19:55:22.640054Z","iopub.execute_input":"2021-08-08T19:55:22.640339Z","iopub.status.idle":"2021-08-08T19:55:23.646017Z","shell.execute_reply.started":"2021-08-08T19:55:22.640294Z","shell.execute_reply":"2021-08-08T19:55:23.645054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_dir = \"../input/\"","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T19:55:27.465679Z","iopub.execute_input":"2021-08-08T19:55:27.466026Z","iopub.status.idle":"2021-08-08T19:55:27.470373Z","shell.execute_reply.started":"2021-08-08T19:55:27.465955Z","shell.execute_reply":"2021-08-08T19:55:27.469432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def classid2label(class_id):\n    category, *attribute = class_id.split(\"_\")\n    return category, attribute","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:06:41.798484Z","iopub.execute_input":"2021-08-08T20:06:41.799033Z","iopub.status.idle":"2021-08-08T20:06:41.804335Z","shell.execute_reply.started":"2021-08-08T20:06:41.798971Z","shell.execute_reply":"2021-08-08T20:06:41.803362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_dict(dictionary, name_dict):\n    print(\"{}{}{}{}{}\".format(\"rank\".ljust(5), \"id\".center(8), \"name\".center(40), \"amount\".rjust(10), \"ratio(%)\".rjust(10)))\n    all_num = sum(dictionary.values())\n    for i, (key, val) in enumerate(sorted(dictionary.items(), key=lambda x: -x[1])):\n        print(\"{:<5}{:^8}{:^40}{:>10}{:>10.3%}\".format(i+1, key, name_dict[key], val, val/all_num))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:06:42.346927Z","iopub.execute_input":"2021-08-08T20:06:42.34729Z","iopub.status.idle":"2021-08-08T20:06:42.35351Z","shell.execute_reply.started":"2021-08-08T20:06:42.347244Z","shell.execute_reply":"2021-08-08T20:06:42.352585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_img_with_labels(img_name, labels, category_name_dict, attribute_name_dict, ax):\n    img = np.asarray(Image.open(input_dir + \"train/\" + img_name))\n    label_interval = (img.shape[0] * 0.9) / len(labels)\n    ax.imshow(img)\n    for num, attribute_id in enumerate(labels):\n        x_pos = img.shape[1] * 1.1\n        y_pos = (img.shape[0] * 0.9) / len(labels) * (num + 2) + (img.shape[0] * 0.1)\n        if(num == 0):\n            ax.text(x_pos, y_pos-label_interval*2, \"category\", fontsize=12)\n            ax.text(x_pos, y_pos-label_interval, category_name_dict[attribute_id], fontsize=12)\n            if(len(labels) > 1):\n                ax.text(x_pos, y_pos, \"attribute\", fontsize=12)\n        else:\n            ax.text(x_pos, y_pos, attribute_name_dict[attribute_id], fontsize=12)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:09:47.552099Z","iopub.execute_input":"2021-08-08T20:09:47.552856Z","iopub.status.idle":"2021-08-08T20:09:47.56132Z","shell.execute_reply.started":"2021-08-08T20:09:47.552575Z","shell.execute_reply":"2021-08-08T20:09:47.560134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_img(img_name, ax):\n    img_df = train_df[train_df.ImageId == img_name]\n    labels = list(set(img_df[\"ClassId\"].values))\n    print_img_with_labels(img_name, labels, category_name_dict, attribute_name_dict, ax)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:09:47.931858Z","iopub.execute_input":"2021-08-08T20:09:47.932237Z","iopub.status.idle":"2021-08-08T20:09:47.937444Z","shell.execute_reply.started":"2021-08-08T20:09:47.932169Z","shell.execute_reply":"2021-08-08T20:09:47.936586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def json2df(data):\n    df = pd.DataFrame()\n    for index, el in enumerate(data):\n        for key, val in el.items():\n            df.loc[index, key] = val\n    return df","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:09:48.757677Z","iopub.execute_input":"2021-08-08T20:09:48.758218Z","iopub.status.idle":"2021-08-08T20:09:48.763111Z","shell.execute_reply.started":"2021-08-08T20:09:48.758157Z","shell.execute_reply":"2021-08-08T20:09:48.762501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# check Text Data\n* train.csv  \nTraining annotations, contains images with both segmented apparel categories and fine-grained attributes; and images with segmented apparel categories only.\n\n    * `ImageID` : the unique Id of an image\n    * `EncodedPixels` : masks in **run-length encoded format** (please refer to [evaluation page](https://www.kaggle.com/c/imaterialist-fashion-2019-FGVC6/overview/evaluation) for details).  \n        * `run-length encoded format` : In summary, '1 3 10 5' implies pixels 1,2,3,10,11,12,13,14 are to be included in the mask.  \n        (The pixels are one-indexed and numbered from top to bottom, then left to right: 1 is pixel (1,1), 2 is pixel (2,1), etc.)\n    * `ClassId` : the class id for this mask. We concatenate both category and attributes (if any) together.","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a"}},{"cell_type":"code","source":"train_df = pd.read_csv(input_dir + \"train.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-08-08T20:12:21.031596Z","iopub.execute_input":"2021-08-08T20:12:21.032004Z","iopub.status.idle":"2021-08-08T20:12:51.689184Z","shell.execute_reply.started":"2021-08-08T20:12:21.031962Z","shell.execute_reply":"2021-08-08T20:12:51.688161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(20)","metadata":{"execution":{"iopub.status.busy":"2021-08-08T20:17:07.575347Z","iopub.execute_input":"2021-08-08T20:17:07.575686Z","iopub.status.idle":"2021-08-08T20:17:07.601052Z","shell.execute_reply.started":"2021-08-08T20:17:07.57564Z","shell.execute_reply":"2021-08-08T20:17:07.600303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A image have some ClassID.  \nNext, check the label description.  \nAfter that, let's check the number of labels in each images and look some images.  ","metadata":{}},{"cell_type":"markdown","source":"* label_descriptions.json  \nA file giving the apparel categories and fine-grained attributes descriptions.","metadata":{}},{"cell_type":"code","source":"with open(input_dir + \"label_descriptions.json\") as f:\n    label_description = json.load(f)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:18:52.881878Z","iopub.execute_input":"2021-08-08T20:18:52.882163Z","iopub.status.idle":"2021-08-08T20:18:52.894953Z","shell.execute_reply.started":"2021-08-08T20:18:52.882129Z","shell.execute_reply":"2021-08-08T20:18:52.894057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"this dataset info\")\nprint(json.dumps(label_description[\"info\"], indent=2))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:18:54.524005Z","iopub.execute_input":"2021-08-08T20:18:54.524515Z","iopub.status.idle":"2021-08-08T20:18:54.529279Z","shell.execute_reply.started":"2021-08-08T20:18:54.524455Z","shell.execute_reply":"2021-08-08T20:18:54.528621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"category_df = json2df(label_description[\"categories\"])\ncategory_df[\"id\"] = category_df[\"id\"].astype(int)\ncategory_df[\"level\"] = category_df[\"level\"].astype(int)\nattribute_df = json2df(label_description[\"attributes\"])\nattribute_df[\"id\"] = attribute_df[\"id\"].astype(int)\nattribute_df[\"level\"] = attribute_df[\"level\"].astype(int)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:21:36.094705Z","iopub.execute_input":"2021-08-08T20:21:36.095085Z","iopub.status.idle":"2021-08-08T20:21:36.446272Z","shell.execute_reply.started":"2021-08-08T20:21:36.095045Z","shell.execute_reply":"2021-08-08T20:21:36.445197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Category Labels\")\ncategory_df","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:21:38.316883Z","iopub.execute_input":"2021-08-08T20:21:38.317178Z","iopub.status.idle":"2021-08-08T20:21:38.345519Z","shell.execute_reply.started":"2021-08-08T20:21:38.317132Z","shell.execute_reply":"2021-08-08T20:21:38.344344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Attribute Labels\")\nattribute_df","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:22:23.050883Z","iopub.execute_input":"2021-08-08T20:22:23.051196Z","iopub.status.idle":"2021-08-08T20:22:23.09051Z","shell.execute_reply.started":"2021-08-08T20:22:23.051132Z","shell.execute_reply":"2021-08-08T20:22:23.08941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"We have {} categories, and {} attributes.\".format(len(label_description['categories']), len(label_description['attributes'])))\nprint(\"Each label　have ID, name, supercategory, and level.\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:24:03.95746Z","iopub.execute_input":"2021-08-08T20:24:03.957783Z","iopub.status.idle":"2021-08-08T20:24:03.963186Z","shell.execute_reply.started":"2021-08-08T20:24:03.957735Z","shell.execute_reply":"2021-08-08T20:24:03.961972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have 46 categories, and 92 attributes.  \nEach label have ID, name, supercategory, and level.  \nI do not know what **level** represents.  \n\nLet's check the number of labels in each images and look some images!  ","metadata":{}},{"cell_type":"code","source":"train_df.tail(100)","metadata":{"execution":{"iopub.status.busy":"2021-08-08T20:38:53.280747Z","iopub.execute_input":"2021-08-08T20:38:53.28118Z","iopub.status.idle":"2021-08-08T20:38:53.329406Z","shell.execute_reply.started":"2021-08-08T20:38:53.28114Z","shell.execute_reply":"2021-08-08T20:38:53.328535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_label_num_df = train_df.groupby(\"ImageId\")[\"ClassId\"].count()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:25:28.951255Z","iopub.execute_input":"2021-08-08T20:25:28.951586Z","iopub.status.idle":"2021-08-08T20:25:29.071873Z","shell.execute_reply.started":"2021-08-08T20:25:28.951547Z","shell.execute_reply":"2021-08-08T20:25:29.071021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(25, 7))\nx = image_label_num_df.value_counts().index.values\ny = image_label_num_df.value_counts().values\nz = zip(x, y)\nz = sorted(z)\nx, y = zip(*z)\nindex = 0\nx_list = []\ny_list = []\nfor i in range(1, max(x)+1):\n    if(i not in x):\n        x_list.append(i)\n        y_list.append(0)\n    else:\n        x_list.append(i)\n        y_list.append(y[index])\n        index += 1\nfor i, j in zip(x_list, y_list):\n    ax.text(i-1, j, j, ha=\"center\", va=\"bottom\", fontsize=13)\nsns.barplot(x=x_list, y=y_list, ax=ax)\nax.set_xticks(list(range(0, len(x_list), 5)))\nax.set_xticklabels(list(range(1, len(x_list), 5)))\nax.set_title(\"the number of labels per image\")\nax.set_xlabel(\"the number of labels\")\nax.set_ylabel(\"amout\");","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:25:29.614706Z","iopub.execute_input":"2021-08-08T20:25:29.615035Z","iopub.status.idle":"2021-08-08T20:25:31.105773Z","shell.execute_reply.started":"2021-08-08T20:25:29.614967Z","shell.execute_reply":"2021-08-08T20:25:31.104889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most image have about 1-17 labels.  \nBut some image have too many labels about over 20.  \nMax label in a image is 74!  ","metadata":{}},{"cell_type":"code","source":"counter_category = Counter()\ncounter_attribute = Counter()\nfor class_id in train_df[\"ClassId\"]:\n    category, attribute = classid2label(class_id)\n    counter_category.update([category])\n    counter_attribute.update(attribute)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:31:42.395118Z","iopub.execute_input":"2021-08-08T20:31:42.395647Z","iopub.status.idle":"2021-08-08T20:31:45.220304Z","shell.execute_reply.started":"2021-08-08T20:31:42.395386Z","shell.execute_reply":"2021-08-08T20:31:45.219321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter_category","metadata":{"execution":{"iopub.status.busy":"2021-08-08T20:31:57.057834Z","iopub.execute_input":"2021-08-08T20:31:57.058127Z","iopub.status.idle":"2021-08-08T20:31:57.06496Z","shell.execute_reply.started":"2021-08-08T20:31:57.058081Z","shell.execute_reply":"2021-08-08T20:31:57.064179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(counter_category)","metadata":{"execution":{"iopub.status.busy":"2021-08-08T20:33:05.184188Z","iopub.execute_input":"2021-08-08T20:33:05.184508Z","iopub.status.idle":"2021-08-08T20:33:05.189896Z","shell.execute_reply.started":"2021-08-08T20:33:05.18445Z","shell.execute_reply":"2021-08-08T20:33:05.188987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter_attribute","metadata":{"execution":{"iopub.status.busy":"2021-08-08T20:33:26.396191Z","iopub.execute_input":"2021-08-08T20:33:26.396728Z","iopub.status.idle":"2021-08-08T20:33:26.406139Z","shell.execute_reply.started":"2021-08-08T20:33:26.396658Z","shell.execute_reply":"2021-08-08T20:33:26.405265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(counter_attribute)","metadata":{"execution":{"iopub.status.busy":"2021-08-08T20:34:10.279388Z","iopub.execute_input":"2021-08-08T20:34:10.279741Z","iopub.status.idle":"2021-08-08T20:34:10.285195Z","shell.execute_reply.started":"2021-08-08T20:34:10.279684Z","shell.execute_reply":"2021-08-08T20:34:10.28454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All kinds of label is in the train dataset.  ","metadata":{}},{"cell_type":"code","source":"category_name_dict = {}\nfor i in label_description[\"categories\"]:\n    category_name_dict[str(i[\"id\"])] = i[\"name\"]\nattribute_name_dict = {}\nfor i in label_description[\"attributes\"]:\n    attribute_name_dict[str(i[\"id\"])] = i[\"name\"]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:35:34.195142Z","iopub.execute_input":"2021-08-08T20:35:34.195434Z","iopub.status.idle":"2021-08-08T20:35:34.200969Z","shell.execute_reply.started":"2021-08-08T20:35:34.195395Z","shell.execute_reply":"2021-08-08T20:35:34.199928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Category label frequency\")\nprint_dict(counter_category, category_name_dict)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:35:35.28889Z","iopub.execute_input":"2021-08-08T20:35:35.289192Z","iopub.status.idle":"2021-08-08T20:35:35.300199Z","shell.execute_reply.started":"2021-08-08T20:35:35.289145Z","shell.execute_reply":"2021-08-08T20:35:35.299361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Attribute label frequency\")\nprint_dict(counter_attribute, attribute_name_dict)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:36:43.033616Z","iopub.execute_input":"2021-08-08T20:36:43.033925Z","iopub.status.idle":"2021-08-08T20:36:43.042653Z","shell.execute_reply.started":"2021-08-08T20:36:43.033883Z","shell.execute_reply":"2021-08-08T20:36:43.041715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Category of 17% in training set is `sleeve`.  \nLeast category is `leg warmer`(112/0.034%).But 112 images, not too few.  \nAttribute of 15% in training set is `symmetrical`.  \nLeast attribute is `burnout`(3/0.004%). OMG....only 3  ","metadata":{}},{"cell_type":"code","source":"train_df.ClassId.max()","metadata":{"execution":{"iopub.status.busy":"2021-08-08T20:37:34.4952Z","iopub.execute_input":"2021-08-08T20:37:34.495549Z","iopub.status.idle":"2021-08-08T20:37:34.549449Z","shell.execute_reply.started":"2021-08-08T20:37:34.495477Z","shell.execute_reply":"2021-08-08T20:37:34.548191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I have to pay enough attention to `ClassId`.  \n`ClassId` is represented　by concatenated both category and attributes (if any) together.  \nSo, we need to predict category and attribute (if any).  \nAll `ClassId` have one category and 0 or more attributes.  \n\n9_9_20_43_61_91 means `category: 9`, `attributes: 9, 20, 43, 61, and 91`  \nyou can see this information in competition [Overview/Evaluation/ClassId](https://www.kaggle.com/c/imaterialist-fashion-2019-FGVC6/overview/evaluation).  \n\nLet's check the ratio of attribute per category.  ","metadata":{}},{"cell_type":"code","source":"attribute_num_dict = {}\nnone_key = str(len(counter_attribute))\nk = list(map(str, range(len(counter_attribute) + 1)))\nv = [0] * (len(counter_attribute) + 1)\nzipped = zip(k, v)\ninit_dict = dict(zipped)\nfor class_id in train_df[\"ClassId\"].values:\n    category, attributes = classid2label(class_id)\n    if category not in attribute_num_dict.keys():\n        attribute_num_dict[category] = init_dict.copy()\n    if attributes == []:\n        attribute_num_dict[category][none_key] += 1\n        continue\n    for attribute in attributes:\n        attribute_num_dict[category][attribute] += 1","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:40:10.209564Z","iopub.execute_input":"2021-08-08T20:40:10.209906Z","iopub.status.idle":"2021-08-08T20:40:10.708441Z","shell.execute_reply.started":"2021-08-08T20:40:10.209833Z","shell.execute_reply":"2021-08-08T20:40:10.707542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(math.ceil(len(counter_category)/2), 2,\\\n                       figsize=(8*2, 6*math.ceil(len(counter_category)/2)), sharey=True)\nfor index, key in enumerate(sorted(map(int, attribute_num_dict.keys()))):\n    x = list(map(int, attribute_num_dict[str(key)].keys()))\n    total = sum(attribute_num_dict[str(key)].values())\n    y = list(map(lambda x: x / total, attribute_num_dict[str(key)].values()))\n    sns.barplot(x, y, ax=ax[index//2, index%2])\n    ax[index//2, index%2].set_title(\"category:{}({})\".format(key, category_name_dict[str(key)]))\n    ax[index//2, index%2].set_xticks(list(range(0, int(none_key), 5)))\n    ax[index//2, index%2].set_xticklabels(list(range(0, int(none_key), 5)))\nprint(\"the ratio of attribute per category(x=92 means no attribute)\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:40:19.68237Z","iopub.execute_input":"2021-08-08T20:40:19.682673Z","iopub.status.idle":"2021-08-08T20:41:17.507691Z","shell.execute_reply.started":"2021-08-08T20:40:19.682623Z","shell.execute_reply":"2021-08-08T20:41:17.506849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Wow!  \nMany category don't have any attribute.  ","metadata":{"trusted":true}},{"cell_type":"markdown","source":"# Check Image data\nLet's check the number of image!","metadata":{}},{"cell_type":"code","source":"print(\"The number of training image is {}.\".format(len(os.listdir(\"../input/train/\"))))\nprint(\"The number of test image is {}.\".format(len(os.listdir(\"../input/test/\"))))","metadata":{"execution":{"iopub.status.busy":"2021-08-08T20:49:17.933791Z","iopub.execute_input":"2021-08-08T20:49:17.934126Z","iopub.status.idle":"2021-08-08T20:49:19.211072Z","shell.execute_reply.started":"2021-08-08T20:49:17.934068Z","shell.execute_reply":"2021-08-08T20:49:19.209935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check image size!","metadata":{}},{"cell_type":"code","source":"image_shape_df = train_df.groupby(\"ImageId\")[\"Height\",\"Width\"].first()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:50:52.320247Z","iopub.execute_input":"2021-08-08T20:50:52.32062Z","iopub.status.idle":"2021-08-08T20:50:52.424135Z","shell.execute_reply.started":"2021-08-08T20:50:52.320519Z","shell.execute_reply":"2021-08-08T20:50:52.423189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(16, 5))\nax1.hist(image_shape_df.Height, bins=100)\nax1.set_title(\"Height distribution\")\nax2.hist(image_shape_df.Width, bins=100)\nax2.set_title(\"Width distribution\");","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:50:59.329803Z","iopub.execute_input":"2021-08-08T20:50:59.330114Z","iopub.status.idle":"2021-08-08T20:51:00.694637Z","shell.execute_reply.started":"2021-08-08T20:50:59.330056Z","shell.execute_reply":"2021-08-08T20:51:00.693635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_name = image_shape_df.Height.idxmin()\nheight, width = image_shape_df.loc[img_name, :]\nprint(\"Minimum height image is {},\\n(H, W) = ({}, {})\".format(img_name, height, width))\nfig, ax = plt.subplots()\nprint_img(img_name, ax)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:52:11.724195Z","iopub.execute_input":"2021-08-08T20:52:11.724493Z","iopub.status.idle":"2021-08-08T20:52:12.244289Z","shell.execute_reply.started":"2021-08-08T20:52:11.724449Z","shell.execute_reply":"2021-08-08T20:52:12.243264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_name = image_shape_df.Height.idxmax()\nheight, width = image_shape_df.loc[img_name, :]\nprint(\"Maximum height image is {},\\n(H, W) = ({}, {})\".format(img_name, height, width))\nfig, ax = plt.subplots()\nprint_img(img_name, ax)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:52:54.487395Z","iopub.execute_input":"2021-08-08T20:52:54.48786Z","iopub.status.idle":"2021-08-08T20:52:59.284469Z","shell.execute_reply.started":"2021-08-08T20:52:54.48782Z","shell.execute_reply":"2021-08-08T20:52:59.283311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_name = image_shape_df.Width.idxmin()\nheight, width = image_shape_df.loc[img_name, :]\nprint(\"Minimam width image is {},\\n(H, W) = ({}, {})\".format(img_name, height, width))\nfig, ax = plt.subplots()\nprint_img(img_name, ax)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:53:57.595587Z","iopub.execute_input":"2021-08-08T20:53:57.595926Z","iopub.status.idle":"2021-08-08T20:53:57.995874Z","shell.execute_reply.started":"2021-08-08T20:53:57.595864Z","shell.execute_reply":"2021-08-08T20:53:57.994917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_name = image_shape_df.Width.idxmax()\nheight, width = image_shape_df.loc[img_name, :]\nprint(\"Maximum width image is {},\\n(H, W) = ({}, {})\".format(img_name, height, width))\nfig, ax = plt.subplots()\nprint_img(img_name, ax)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:54:22.479428Z","iopub.execute_input":"2021-08-08T20:54:22.479904Z","iopub.status.idle":"2021-08-08T20:54:28.96594Z","shell.execute_reply.started":"2021-08-08T20:54:22.479856Z","shell.execute_reply":"2021-08-08T20:54:28.964913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, let's show segmented images.  ","metadata":{}},{"cell_type":"code","source":"pallete =  [\n    'Pastel1', 'Pastel2', 'Paired', 'Accent', 'Dark2',\n    'Set1', 'Set2', 'Set3', 'tab10', 'tab20', 'tab20b', 'tab20c']\n\n\ndef make_mask_img(segment_df):\n    category_num = len(counter_category)\n    seg_width = segment_df.at[0, \"Width\"]\n    seg_height = segment_df.at[0, \"Height\"]\n    seg_img = np.full(seg_width*seg_height, category_num-1, dtype=np.uint8)\n    for encoded_pixels, class_id in zip(segment_df[\"EncodedPixels\"].values, segment_df[\"ClassId\"].values):\n        pixel_list = list(map(int, encoded_pixels.split(\" \")))\n        for i in range(0, len(pixel_list), 2):\n            start_index = pixel_list[i] - 1\n            index_len = pixel_list[i+1] - 1\n            seg_img[start_index:start_index+index_len] =\\\n                int(int(class_id.split(\"_\")[0]) / (category_num-1) * 255)\n    seg_img = seg_img.reshape((seg_height, seg_width), order='F')\n    return seg_img\n\n\ndef train_generator(df, batch_size):\n    img_ind_num = df.groupby(\"ImageId\")[\"ClassId\"].count()\n    index = df.index.values[0]\n    trn_images = []\n    seg_images = []\n    for i, (img_name, ind_num) in enumerate(img_ind_num.items()):\n        img = cv2.imread(\"../input/train/\" + img_name)\n        segment_df = (df.loc[index:index+ind_num-1, :]).reset_index(drop=True)\n        index += ind_num\n        if segment_df[\"ImageId\"].nunique() != 1:\n            raise Exception(\"Index Range Error\")\n        seg_img = make_mask_img(segment_df)\n        \n        # HWC -> CHW\n        img = img.transpose((2, 0, 1))\n        \n        trn_images.append(img)\n        seg_images.append(seg_img)\n        if((i+1) % batch_size == 0):\n            return trn_images, seg_images","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:55:24.736939Z","iopub.execute_input":"2021-08-08T20:55:24.737412Z","iopub.status.idle":"2021-08-08T20:55:24.749004Z","shell.execute_reply.started":"2021-08-08T20:55:24.737372Z","shell.execute_reply":"2021-08-08T20:55:24.748253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cv2plt(img, isColor=True):\n    original_img = img\n    original_img = original_img.transpose(1, 2, 0)\n    original_img = cv2.cvtColor(original_img, cv2.COLOR_BGR2RGB)\n    return original_img","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T20:55:31.17772Z","iopub.execute_input":"2021-08-08T20:55:31.178234Z","iopub.status.idle":"2021-08-08T20:55:31.182247Z","shell.execute_reply.started":"2021-08-08T20:55:31.178144Z","shell.execute_reply":"2021-08-08T20:55:31.181647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(3, 2, figsize=(16, 18))\nfor i, (img, seg) in enumerate(zip(original, segmented)):\n    ax[i//2, i%2].imshow(cv2plt(img))\n  \n    ax[i//2, i%2].imshow(seg, cmap='tab20_r', alpha=0.6)\n    ax[i//2, i%2].set_title(\"Sample {}\".format(i))","metadata":{"execution":{"iopub.status.busy":"2021-08-08T21:08:15.81958Z","iopub.execute_input":"2021-08-08T21:08:15.819907Z","iopub.status.idle":"2021-08-08T21:08:21.630362Z","shell.execute_reply.started":"2021-08-08T21:08:15.819861Z","shell.execute_reply":"2021-08-08T21:08:21.629293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"original, segmented = train_generator(train_df, 12)\nfig, ax = plt.subplots(6, 2, figsize=(16, 18))\nfor i, (img, seg) in enumerate(zip(original, segmented)):\n    ax[i//2, i%2].imshow(cv2plt(img))\n    seg[seg == 45] = 255\n    ax[i//2, i%2].imshow(seg, cmap='tab20_r', alpha=0.6)\n    ax[i//2, i%2].set_title(\"Sample {}\".format(i))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-08T21:10:00.131232Z","iopub.execute_input":"2021-08-08T21:10:00.131576Z","iopub.status.idle":"2021-08-08T21:10:11.132807Z","shell.execute_reply.started":"2021-08-08T21:10:00.131495Z","shell.execute_reply":"2021-08-08T21:10:11.132031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sample Submission","metadata":{}},{"cell_type":"code","source":"sample_df = pd.read_csv(input_dir + \"sample_submission.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.head(20)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}