{"cells":[{"metadata":{},"cell_type":"markdown","source":"<a id=\"toc\"></a>\n# Table of Contents\n1. [Configure parameters](#configure_parameters)\n1. [Import modules](#import_modules)\n1. [Get annotations](#get_annotations)\n1. [Draw some charts for input dataset](#draw_some_charts_for_input_dataset)\n1. [Split dataset into training and validation sets](#split_dataset_into_training_and_validation_sets)\n1. [Draw some charts for training and validation sets](#draw_some_charts_for_training_and_validation_sets)\n1. [Visualize some images and corresponding labels](#visualize_some_images_and_corresponding_labels)\n1. [Copy images into right folders](#copy_images_into_right_folders)\n1. [Zip training and validation sets](#zip_training_and_validation_sets)\n1. [Save labels](#save_labels)"},{"metadata":{},"cell_type":"markdown","source":"<a id=\"configure_parameters\"></a>\n# Configure parameters\n[Back to Table of Contents](#toc)"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"DATASET_DIR = '../input/understanding_cloud_organization/'\nTEST_SIZE = 0.3\nRANDOM_STATE = 1024\n\nNUM_TRAIN_SAMPLES = 5 # The number of train samples used for visualization\nNUM_VAL_SAMPLES = 5 # The number of val samples used for visualization\nCOLORS = ['b', 'g', 'r', 'm'] # Color of each class","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a id=\"import_modules\"></a>\n# Import modules\n[Back to Table of Contents](#toc)"},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport os\nimport cv2\nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Polygon\nfrom matplotlib.collections import PatchCollection\n\nfrom shutil import copyfile\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm_notebook","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a id=\"get_annotations\"></a>\n# Get annotations\n[Back to Table of Contents](#toc)"},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv(os.path.join(DATASET_DIR, 'train.csv'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['Image'] = df['Image_Label'].map(lambda x: x.split('_')[0])\ndf['HavingDefection'] = df['EncodedPixels'].map(lambda x: 0 if x is np.nan else 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_col = np.array(df['Image'])\nimage_files = image_col[::4]\nall_labels = np.array(df['HavingDefection']).reshape(-1, 4)\n\nnum_img_fish = np.sum(all_labels[:, 0])\nnum_img_flower = np.sum(all_labels[:, 1])\nnum_img_gravel = np.sum(all_labels[:, 2])\nnum_img_sugar = np.sum(all_labels[:, 3])\nprint('Fish: {} images'.format(num_img_fish))\nprint('Flower: {} images'.format(num_img_flower))\nprint('Gravel: {} images'.format(num_img_gravel))\nprint('Sugar: {} images'.format(num_img_sugar))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a id=\"draw_some_charts_for_input_dataset\"></a>\n# Draw some charts for input dataset\n[Back to Table of Contents](#toc)"},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"def plot_figures(\n    sizes,\n    pie_title,\n    start_angle,\n    bar_title,\n    bar_ylabel,\n    labels=('Fish', 'Flower', 'Gravel', 'Sugar'),\n    colors=None,\n    explode=(0, 0, 0, 0.1),\n):\n    fig, axes = plt.subplots(1, 2, figsize=(18, 6))\n\n    y_pos = np.arange(len(labels))\n    barlist = axes[0].bar(y_pos, sizes, align='center')\n    axes[0].set_xticks(y_pos, labels)\n    axes[0].set_ylabel(bar_ylabel)\n    axes[0].set_title(bar_title)\n    if colors is not None:\n        for idx, item in enumerate(barlist):\n            item.set_color(colors[idx])\n\n    def autolabel(rects):\n        \"\"\"\n        Attach a text label above each bar displaying its height\n        \"\"\"\n        for rect in rects:\n            height = rect.get_height()\n            axes[0].text(\n                rect.get_x() + rect.get_width()/2., height,\n                '%d' % int(height),\n                ha='center', va='bottom', fontweight='bold'\n            )\n\n    autolabel(barlist)\n    \n    pielist = axes[1].pie(sizes, explode=explode, labels=labels, autopct='%1.1f%%', shadow=True, startangle=start_angle, counterclock=False)\n    axes[1].axis('equal')\n    axes[1].set_title(pie_title)\n    if colors is not None:\n        for idx, item in enumerate(pielist[0]):\n            item.set_color(colors[idx])\n\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('[THE WHOLE DATASET]')\n\nsum_each_class = np.sum(all_labels, axis=0)\nplot_figures(\n    sum_each_class,\n    pie_title='The percentage of each class',\n    start_angle=90,\n    bar_title='The number of images for each class',\n    bar_ylabel='Images',\n    colors=COLORS,\n    explode=(0, 0, 0, 0.1)\n)\n\nsum_each_sample = np.sum(all_labels, axis=1)\nunique, counts = np.unique(sum_each_sample, return_counts=True)\n\nplot_figures(\n    counts,\n    pie_title='The percentage of the number of classes appears in an image',\n    start_angle=100,\n    bar_title='The number of classes appears in an image',\n    bar_ylabel='Images',\n    labels=[' '.join((str(label), 'class(es)')) for label in unique],\n    explode=(0, 0.1, 0, 0)\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a id=\"split_dataset_into_training_and_validation_sets\"></a>\n# Split dataset into training and validation sets\n[Back to Table of Contents](#toc)"},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(image_files, all_labels, test_size=TEST_SIZE, random_state=RANDOM_STATE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('X_train:', X_train.shape)\nprint('y_train:', y_train.shape)\nprint('X_val:', X_val.shape)\nprint('y_val:', y_val.shape)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a id=\"draw_some_charts_for_training_and_validation_sets\"></a>\n# Draw some charts for training and validation sets\n[Back to Table of Contents](#toc)"},{"metadata":{"trusted":true},"cell_type":"code","source":"print('[TRAINING SET]')\n\nsum_each_class = np.sum(y_train, axis=0)\nplot_figures(\n    sum_each_class,\n    pie_title='The percentage of each class',\n    start_angle=90,\n    bar_title='The number of images for each class',\n    bar_ylabel='Images',\n    colors=COLORS,\n    explode=(0, 0, 0, 0.1)\n)\n\n\nsum_each_sample = np.sum(y_train, axis=1)\nunique, counts = np.unique(sum_each_sample, return_counts=True)\n\nplot_figures(\n    counts,\n    pie_title='The percentage of the number of classes appears in an image',\n    start_angle=100,\n    bar_title='The number of classes appears in an image',\n    bar_ylabel='Images',\n    labels=[' '.join((str(label), 'class(es)')) for label in unique],\n    explode=(0, 0.1, 0, 0)\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('[VALIDATION SET]')\n\nsum_each_class = np.sum(y_val, axis=0)\nplot_figures(\n    sum_each_class,\n    pie_title='The percentage of each class',\n    start_angle=90,\n    bar_title='The number of images for each class',\n    bar_ylabel='Images',\n    colors=COLORS,\n    explode=(0, 0, 0, 0.1)\n)\n\n\nsum_each_sample = np.sum(y_val, axis=1)\nunique, counts = np.unique(sum_each_sample, return_counts=True)\n\nplot_figures(\n    counts,\n    pie_title='The percentage of the number of classes appears in an image',\n    start_angle=100,\n    bar_title='The number of classes appears in an image',\n    bar_ylabel='Images',\n    labels=[' '.join((str(label), 'class(es)')) for label in unique],\n    explode=(0, 0.1, 0, 0)\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a id=\"visualize_some_images_and_corresponding_labels\"></a>\n# Visualize some images and corresponding labels\n[Back to Table of Contents](#toc)"},{"metadata":{"trusted":true},"cell_type":"code","source":"def rle2mask(mask_rle, shape=(2100, 1400)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (width,height) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def show_samples(samples):\n    for sample in samples:\n        fig, ax = plt.subplots(figsize=(15, 10))\n        img_path = os.path.join(DATASET_DIR, 'train_images', sample[0])\n        img = cv2.imread(img_path, 1)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n        # Get annotations\n        labels = df[df['Image_Label'].str.contains(sample[0])]['EncodedPixels']\n\n        patches = []\n        for idx, rle in enumerate(labels.values):\n            if rle is not np.nan:\n                mask = rle2mask(rle)\n                contours, _ = cv2.findContours(mask, cv2.RETR_TREE, cv2.CHAIN_APPROX_SIMPLE)\n                for contour in contours:\n                    poly_patch = Polygon(contour.reshape(-1, 2), closed=True, linewidth=2, edgecolor=COLORS[idx], facecolor=COLORS[idx], fill=True)\n                    patches.append(poly_patch)\n        p = PatchCollection(patches, match_original=True, cmap=matplotlib.cm.jet, alpha=0.3)\n\n        ax.imshow(img/255)\n        ax.set_title('{} - ({})'.format(sample[0], ', '.join(sample[1].astype(np.str))))\n        ax.add_collection(p)\n        ax.set_xticklabels([])\n        ax.set_yticklabels([])\n        plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_pairs = np.array(list(zip(X_train, y_train)))\ntrain_samples = train_pairs[np.random.choice(train_pairs.shape[0], NUM_TRAIN_SAMPLES, replace=False), :]\n\nshow_samples(train_samples)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_pairs = np.array(list(zip(X_val, y_val)))\nval_samples = val_pairs[np.random.choice(val_pairs.shape[0], NUM_VAL_SAMPLES, replace=False), :]\n\nshow_samples(val_samples)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a id=\"copy_images_into_right_folders\"></a>\n# Copy images into right folders\n[Back to Table of Contents](#toc)"},{"metadata":{"trusted":true},"cell_type":"code","source":"!mkdir ../train_images\n!mkdir ../val_images","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for image_file in tqdm_notebook(X_train):\n    src = os.path.join(DATASET_DIR, 'train_images', image_file)\n    dst = os.path.join('../train_images', image_file)\n    copyfile(src, dst)\n\nfor image_file in tqdm_notebook(X_val):\n    src = os.path.join(DATASET_DIR, 'train_images', image_file)\n    dst = os.path.join('../val_images', image_file)\n    copyfile(src, dst)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a id=\"zip_training_and_validation_sets\"></a>\n# Zip training and validation sets\n[Back to Table of Contents](#toc)"},{"metadata":{"trusted":true,"_kg_hide-output":true},"cell_type":"code","source":"!apt install zip","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!zip -r -m -1 -q train_images.zip ../train_images\n!zip -r -m -1 -q val_images.zip ../val_images","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a id=\"save_labels\"></a>\n# Save labels\n[Back to Table of Contents](#toc)"},{"metadata":{"trusted":true},"cell_type":"code","source":"y_train = [' '.join(y.astype(np.str)) for y in y_train]\ny_val = [' '.join(y.astype(np.str)) for y in y_val]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_set = {\n    'Image': X_train,\n    'Label': y_train\n}\n\nval_set = {\n    'Image': X_val,\n    'Label': y_val\n}\n\ntrain_df = pd.DataFrame(train_set)\nval_df = pd.DataFrame(val_set)\n\ntrain_df.to_csv('./train.csv', index=False)\nval_df.to_csv('./val.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}