{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Herbarium 2022\n### EDA Notebook\nModeling Notebooks: \n- https://www.kaggle.com/code/hanselliott/herbarium22-cnn-exploration\n- https://www.kaggle.com/code/hanselliott/herbarium22-cnn-wandb (best)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport json\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow import keras\nfrom tensorflow.keras import backend as K\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-06T22:06:48.148419Z","iopub.execute_input":"2022-07-06T22:06:48.14878Z","iopub.status.idle":"2022-07-06T22:06:55.723778Z","shell.execute_reply.started":"2022-07-06T22:06:48.14869Z","shell.execute_reply":"2022-07-06T22:06:55.722775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Competition: https://www.kaggle.com/competitions/herbarium-2022-fgvc9/overview  \nNotebook inspiration: https://www.kaggle.com/code/salaheddinelahmadi/exploration-preprocessing-baseline-cnn\n","metadata":{}},{"cell_type":"markdown","source":"## Load in metadata\nNeed the metadata to reference images","metadata":{}},{"cell_type":"code","source":"TRAIN_DIR = \"../input/herbarium-2022-fgvc9/train_images/\"\nTEST_DIR = \"../input/herbarium-2022-fgvc9/test_images/\"\n\nwith open(\"../input/herbarium-2022-fgvc9/train_metadata.json\") as json_file:\n    train_meta = json.load(json_file)\nwith open(\"../input/herbarium-2022-fgvc9/test_metadata.json\") as json_file:\n    test_meta = json.load(json_file)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:06:58.488426Z","iopub.execute_input":"2022-07-06T22:06:58.488725Z","iopub.status.idle":"2022-07-06T22:07:20.079069Z","shell.execute_reply.started":"2022-07-06T22:06:58.488694Z","shell.execute_reply":"2022-07-06T22:07:20.077646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##The JSON keys used to access data\ntrain_meta.keys()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:07:21.886056Z","iopub.execute_input":"2022-07-06T22:07:21.886348Z","iopub.status.idle":"2022-07-06T22:07:21.896445Z","shell.execute_reply.started":"2022-07-06T22:07:21.886313Z","shell.execute_reply":"2022-07-06T22:07:21.89548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" pd.DataFrame(train_meta['annotations'])[:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:07:23.821999Z","iopub.execute_input":"2022-07-06T22:07:23.822842Z","iopub.status.idle":"2022-07-06T22:07:24.92372Z","shell.execute_reply.started":"2022-07-06T22:07:23.822788Z","shell.execute_reply":"2022-07-06T22:07:24.922636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_meta['images'][:5]","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:07:27.136937Z","iopub.execute_input":"2022-07-06T22:07:27.137207Z","iopub.status.idle":"2022-07-06T22:07:27.143571Z","shell.execute_reply.started":"2022-07-06T22:07:27.137178Z","shell.execute_reply":"2022-07-06T22:07:27.142871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('the number of images in the training set : ', len(train_meta['images']))\nprint('the number of annotations in the training set :', len(train_meta['annotations']))","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:07:29.641266Z","iopub.execute_input":"2022-07-06T22:07:29.641582Z","iopub.status.idle":"2022-07-06T22:07:29.648341Z","shell.execute_reply.started":"2022-07-06T22:07:29.641548Z","shell.execute_reply":"2022-07-06T22:07:29.647397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_meta['categories'][:2]","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:07:35.51027Z","iopub.execute_input":"2022-07-06T22:07:35.510578Z","iopub.status.idle":"2022-07-06T22:07:35.51683Z","shell.execute_reply.started":"2022-07-06T22:07:35.510544Z","shell.execute_reply":"2022-07-06T22:07:35.51602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Use set to convert list of category IDs into a set of distinct IDs. Then we can determine the # of unique IDs: 15,501...\n#set([annotation[\"category_id\"] for annotation in train_meta[\"annotations\"]])\nlen(set([annotation[\"category_id\"] for annotation in train_meta[\"annotations\"]]))","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:07:51.235093Z","iopub.execute_input":"2022-07-06T22:07:51.235363Z","iopub.status.idle":"2022-07-06T22:07:51.34469Z","shell.execute_reply.started":"2022-07-06T22:07:51.235334Z","shell.execute_reply":"2022-07-06T22:07:51.343738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modify df_meta","metadata":{}},{"cell_type":"code","source":"##setting up image to category df with image paths. from the list of lists\nids = []\ncategories = []\npaths = []\n\nfor annotation, image in zip(train_meta['annotations'], train_meta['images']):\n    ids.append(image[\"image_id\"])\n    categories.append(annotation['category_id'])\n    paths.append(image[\"file_name\"])\n\ndf_meta = pd.DataFrame({\"id\":ids, \"category\":categories, \"path\":paths})\ndf_meta.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:08:18.015982Z","iopub.execute_input":"2022-07-06T22:08:18.016574Z","iopub.status.idle":"2022-07-06T22:08:18.77496Z","shell.execute_reply.started":"2022-07-06T22:08:18.016543Z","shell.execute_reply":"2022-07-06T22:08:18.77407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_meta['category'].value_counts().sort_values(ascending=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:08:26.346539Z","iopub.execute_input":"2022-07-06T22:08:26.346837Z","iopub.status.idle":"2022-07-06T22:08:26.370687Z","shell.execute_reply.started":"2022-07-06T22:08:26.346804Z","shell.execute_reply":"2022-07-06T22:08:26.369894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##view counts of each category - rather unbalanced\nsns.histplot(df_meta['category']).set(title=\"Category Frequencies\")\nplt.show()\n\nprint(\"Unique categories: \", len(df_meta['category'].unique()))","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:08:32.016433Z","iopub.execute_input":"2022-07-06T22:08:32.017274Z","iopub.status.idle":"2022-07-06T22:08:32.97951Z","shell.execute_reply.started":"2022-07-06T22:08:32.017224Z","shell.execute_reply":"2022-07-06T22:08:32.978917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##extract metadata features by category to merge with df_meta\nsci_name = {cat[\"category_id\"]:cat[\"scientificName\"] for cat in train_meta['categories']}\nfamily = {cat[\"category_id\"]:cat[\"family\"] for cat in train_meta['categories']}\ngenus = {cat[\"category_id\"]:cat[\"genus\"] for cat in train_meta['categories']}\nspecies = {cat[\"category_id\"]:cat[\"species\"] for cat in train_meta['categories']}\n\ndf_meta[\"scientific_name\"] = df_meta[\"category\"].map(sci_name)\ndf_meta[\"family\"] = df_meta[\"category\"].map(family)\ndf_meta[\"genus\"] = df_meta[\"category\"].map(genus)\ndf_meta[\"species\"] = df_meta[\"category\"].map(species)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:08:38.349304Z","iopub.execute_input":"2022-07-06T22:08:38.349584Z","iopub.status.idle":"2022-07-06T22:08:38.461459Z","shell.execute_reply.started":"2022-07-06T22:08:38.349555Z","shell.execute_reply":"2022-07-06T22:08:38.460793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_meta['categories'][0]) ##to compare\ndf_meta.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:09:23.720521Z","iopub.execute_input":"2022-07-06T22:09:23.720954Z","iopub.status.idle":"2022-07-06T22:09:23.733911Z","shell.execute_reply.started":"2022-07-06T22:09:23.72092Z","shell.execute_reply":"2022-07-06T22:09:23.732904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Image Examples","metadata":{}},{"cell_type":"code","source":"def plot_random_images(metadata, directory, n_imgs, dims=[3,4], random_seed=12):\n    \"\"\"\n    Function randomly selects paths from the train metadata and plots the corresponding image.  \n    \"\"\"\n    np.random.seed = random_seed\n    # Randomly sample n rows from metatdata\n    rndm_elems = metadata.sample(n=n_imgs)      \n    \n    # Add the img path, category, and sci name to lists\n    imgs = []\n    category = []\n    scientific_name = []\n    for path, categ, sci_name in zip(rndm_elems['path'], rndm_elems['category'], rndm_elems['scientific_name']):\n        imgs.append(cv2.imread(os.path.join(directory,path)))\n        category.append(categ)\n        scientific_name.append(sci_name)\n    # Prepare figures/axes for subplots\n    fig, axes = plt.subplots(dims[0], dims[1], figsize=(10,10))\n    axes = axes.flatten()\n    # For each image, plot image to a subplot and title with category + scientific name\n    for img, ax, c, s in zip(imgs, axes, category, scientific_name):\n        title = str(c) + \" | \" + s\n        ax.imshow(img)\n        ax.axis('off')\n        ax.title.set_text(title)\n    plt.suptitle(\"Example Images\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:13:20.768611Z","iopub.execute_input":"2022-07-06T22:13:20.768912Z","iopub.status.idle":"2022-07-06T22:13:20.778003Z","shell.execute_reply.started":"2022-07-06T22:13:20.768882Z","shell.execute_reply":"2022-07-06T22:13:20.777383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DIR = \"../input/herbarium-2022-fgvc9/train_images/\"\nplot_random_images(df_meta, TRAIN_DIR, 8, [4,2], random_seed=123)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:13:35.41701Z","iopub.execute_input":"2022-07-06T22:13:35.417303Z","iopub.status.idle":"2022-07-06T22:13:36.578057Z","shell.execute_reply.started":"2022-07-06T22:13:35.417273Z","shell.execute_reply":"2022-07-06T22:13:36.577131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_random_images(df_meta, TRAIN_DIR, 8, [4,2], random_seed=123)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:14:01.64952Z","iopub.execute_input":"2022-07-06T22:14:01.649808Z","iopub.status.idle":"2022-07-06T22:14:02.8235Z","shell.execute_reply.started":"2022-07-06T22:14:01.649779Z","shell.execute_reply":"2022-07-06T22:14:02.822546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Images to arrays","metadata":{}},{"cell_type":"code","source":"##OPEN CV (cv2): extract image example\npath = df_meta['path'][1]\nex_img = cv2.imread(os.path.join(TRAIN_DIR, path))\nprint(ex_img[0:2])\n\nplt.imshow(ex_img)\nplt.title(\"Shape: {}\".format(ex_img.shape))\nplt.show()\n##3 dimensionsal shape: 1000px tall, 666px wide, RGB","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:14:51.582452Z","iopub.execute_input":"2022-07-06T22:14:51.58275Z","iopub.status.idle":"2022-07-06T22:14:51.81702Z","shell.execute_reply.started":"2022-07-06T22:14:51.582722Z","shell.execute_reply":"2022-07-06T22:14:51.816154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def img_to_matrix(df, index):\n    \"\"\"\n    Reads the file of the image corresponding to a given index in the \"meta\" data.\n    Returns a numpy array.  \n    \"\"\"\n    label = df[\"category\"][index]\n       \n    path = df['path'][index]\n    img = cv2.imread(os.path.join(TRAIN_DIR, path))\n    return(img)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:14:57.279669Z","iopub.execute_input":"2022-07-06T22:14:57.280012Z","iopub.status.idle":"2022-07-06T22:14:57.285309Z","shell.execute_reply.started":"2022-07-06T22:14:57.279977Z","shell.execute_reply":"2022-07-06T22:14:57.284482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Ex:\nprint(\"label = \", df_meta['category'][1500])\nprint(\"path = \", df_meta['path'][1500])\nex_img = cv2.imread(os.path.join(TRAIN_DIR, df_meta['path'][1500]))\nex_img[0:2]","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:14:59.368812Z","iopub.execute_input":"2022-07-06T22:14:59.369126Z","iopub.status.idle":"2022-07-06T22:14:59.405839Z","shell.execute_reply.started":"2022-07-06T22:14:59.369092Z","shell.execute_reply":"2022-07-06T22:14:59.404958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pixel_df = []\nflat_df = []\nfor index in range(5):\n    tmp_mat = img_to_matrix(df_meta, index)\n    pixel_df.append(tmp_mat)\n    \n    flat = np.array(pixel_df).reshape(-1,3).ravel() ##convert 3d to 2d, then unravel 2d to 1d\n    flat_df.append(flat)\n\nflat_df","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:15:02.522239Z","iopub.execute_input":"2022-07-06T22:15:02.522997Z","iopub.status.idle":"2022-07-06T22:15:02.653856Z","shell.execute_reply.started":"2022-07-06T22:15:02.52294Z","shell.execute_reply":"2022-07-06T22:15:02.653032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(flat_df[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:15:05.949089Z","iopub.execute_input":"2022-07-06T22:15:05.950135Z","iopub.status.idle":"2022-07-06T22:15:05.956672Z","shell.execute_reply.started":"2022-07-06T22:15:05.950079Z","shell.execute_reply":"2022-07-06T22:15:05.955657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image Processing","metadata":{}},{"cell_type":"code","source":"ex_img_raw = ex_img\nplt.imshow(ex_img_raw)\nplt.title(\"Original Ex. Image\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:15:31.647269Z","iopub.execute_input":"2022-07-06T22:15:31.647565Z","iopub.status.idle":"2022-07-06T22:15:31.840538Z","shell.execute_reply.started":"2022-07-06T22:15:31.647535Z","shell.execute_reply":"2022-07-06T22:15:31.83969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ex_img = cv2.cvtColor(ex_img, cv2.COLOR_BGR2RGB) ##convert color scale\nplt.imshow(ex_img)\nplt.title(\"Color conversion: BGR -> RGB\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:15:34.267737Z","iopub.execute_input":"2022-07-06T22:15:34.268643Z","iopub.status.idle":"2022-07-06T22:15:34.461278Z","shell.execute_reply.started":"2022-07-06T22:15:34.268597Z","shell.execute_reply":"2022-07-06T22:15:34.460579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##resize the image\n##Image interpolation: https://docs.opencv.org/2.4/modules/imgproc/doc/geometric_transformations.html?highlight=resize#resize\nwidth = 120\nheight = 120\nres_img = cv2.resize(ex_img, (width, height), interpolation = cv2.INTER_AREA)\n\nplt.imshow(res_img)\nplt.title(\"Shape : {}\".format(res_img.shape))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:15:46.057339Z","iopub.execute_input":"2022-07-06T22:15:46.057672Z","iopub.status.idle":"2022-07-06T22:15:46.197607Z","shell.execute_reply.started":"2022-07-06T22:15:46.057638Z","shell.execute_reply":"2022-07-06T22:15:46.196649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Historgram Equalization: https://docs.opencv.org/3.4/d4/d1b/tutorial_histogram_equalization.html\n - improves the contrast in an image, in order to stretch out the intensity range ","metadata":{}},{"cell_type":"code","source":"img_yuv = cv2.cvtColor(ex_img, cv2.COLOR_RGB2YUV)\nplt.figure(figsize = (16,6))\nplt.subplot(1, 2, 1)\n\nplt.hist(img_yuv.flatten(), bins=range(256))\nplt.title('Intensity distribution')\nplt.xlabel(\"intensity\")\nplt.ylabel(\"pixels\")\n\nplt.subplot(1, 2, 2)\nplt.imshow(img_yuv)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:15:54.963993Z","iopub.execute_input":"2022-07-06T22:15:54.964334Z","iopub.status.idle":"2022-07-06T22:15:55.764608Z","shell.execute_reply.started":"2022-07-06T22:15:54.964289Z","shell.execute_reply":"2022-07-06T22:15:55.763726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(img_yuv[:, : ,0]) \nprint(\"Equalized: \")\nprint(cv2.equalizeHist(img_yuv[:, : ,0]))","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:16:00.888799Z","iopub.execute_input":"2022-07-06T22:16:00.889083Z","iopub.status.idle":"2022-07-06T22:16:00.901885Z","shell.execute_reply.started":"2022-07-06T22:16:00.889053Z","shell.execute_reply":"2022-07-06T22:16:00.900965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualize the Equalization\nimg_yuv[:, : ,0] = cv2.equalizeHist(img_yuv[:, : ,0])\nimg_equ = cv2.cvtColor(img_yuv, cv2.COLOR_YUV2RGB) ##Convert back to RGB\n\nplt.imshow(img_equ)\nplt.title(\"Post histogram equalization\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:16:02.977772Z","iopub.execute_input":"2022-07-06T22:16:02.978342Z","iopub.status.idle":"2022-07-06T22:16:03.220756Z","shell.execute_reply.started":"2022-07-06T22:16:02.978305Z","shell.execute_reply":"2022-07-06T22:16:03.220213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(img_equ.flatten(), bins=range(256))\nplt.title(\"Image historgram after equalization\")\nplt.xlabel(\"intensity\")\nplt.ylabel(\"pixels\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:16:07.796013Z","iopub.execute_input":"2022-07-06T22:16:07.796314Z","iopub.status.idle":"2022-07-06T22:16:08.401346Z","shell.execute_reply.started":"2022-07-06T22:16:07.796278Z","shell.execute_reply":"2022-07-06T22:16:08.400464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Set up augmentation for preprocessing","metadata":{}},{"cell_type":"code","source":"#from tensorflow.keras.preprocessing.image import ImageDataGenerator\n    #Generate batches of tensor image data with real-time data augmentation.\ndatagen = ImageDataGenerator(\n    rotation_range=30,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n     zoom_range=0.2,\n    shear_range=0.1,\n    horizontal_flip=True,\n    fill_mode='nearest', cval = 125)\n\nx = ex_img\nx = x.reshape((1,) + x.shape)\naug_x = datagen.flow(x)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:16:23.240549Z","iopub.execute_input":"2022-07-06T22:16:23.240826Z","iopub.status.idle":"2022-07-06T22:16:23.251482Z","shell.execute_reply.started":"2022-07-06T22:16:23.240799Z","shell.execute_reply":"2022-07-06T22:16:23.250552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aug_images = [next(aug_x)[0].astype(np.uint8) for i in range(12)]\n\nfig, axes = plt.subplots(3,4,figsize=(10,10)) ##3 img rows, 4 img cols\naxes = axes.flatten()\nfor img, ax in zip(aug_images,axes): ##zip image to its subplot\n    ax.imshow(img)\n    #ax.axis('off')\n    \nplt.suptitle(\"Data augmentation\",fontsize=24)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:16:27.418354Z","iopub.execute_input":"2022-07-06T22:16:27.419051Z","iopub.status.idle":"2022-07-06T22:16:31.201527Z","shell.execute_reply.started":"2022-07-06T22:16:27.418996Z","shell.execute_reply":"2022-07-06T22:16:31.200427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing function","metadata":{}},{"cell_type":"code","source":"##create function to preprocess images for a neural network\ndef preprocess_cnn(categories, num_categories, width, height):\n    \"\"\"\n    categories = 000, num_categories = 00 (for example)\n    \"\"\"\n    list_img = []\n    labels = []\n    for cat, num_cat in zip(categories, num_categories):\n        for ig in os.listdir(os.path.join(\"../input/herbarium-2022-fgvc9/train_images\", cat, num_cat)):\n            ##read in image\n            img = cv2.imread(os.path.join(\"../input/herbarium-2022-fgvc9/train_images\", cat, num_cat, ig))\n            ##resize\n            img = cv2.resize(img, (width, height), interpolation=cv2.INTER_LINEAR)\n            ##equalize\n            img_yuv = cv2.cvtColor(img,cv2.COLOR_RGB2YUV) ##convert to YUB\n            img_yuv[:,:,0] = cv2.equalizeHist(img_yuv[:,:,0]) ##equalize histogram\n            img_equ = cv2.cvtColor(img_yuv, cv2.COLOR_YUV2RGB) ##convert to RGB\n            \n            list_img.append(img_equ)\n            labels.append(ig.split(\"__\")[0]) ##label is the part before \"__\" (ie, 0 for 00000__001.jpg or 1 for 000100__001.jpg)\n    return list_img, labels","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:16:39.06929Z","iopub.execute_input":"2022-07-06T22:16:39.069578Z","iopub.status.idle":"2022-07-06T22:16:39.07762Z","shell.execute_reply.started":"2022-07-06T22:16:39.06955Z","shell.execute_reply":"2022-07-06T22:16:39.076655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Must Run: (add parent_folder and child_folder for retrieving images)","metadata":{}},{"cell_type":"code","source":"##split the path based on '/' into parent and child folder. \n##lambda fn is applied to each row in the column to split each path\ndf_meta['path'].apply(lambda x : x.split('/'))\n\n#add categories/num_categories equivalents to df_meta\ndf_meta['parent_folder'] = df_meta['path'].apply(lambda x : x.split('/')[0])\ndf_meta['child_folder'] = df_meta['path'].apply(lambda x : x.split('/')[1])\n\ndf_meta","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:16:42.995234Z","iopub.execute_input":"2022-07-06T22:16:42.995518Z","iopub.status.idle":"2022-07-06T22:16:46.127373Z","shell.execute_reply.started":"2022-07-06T22:16:42.995489Z","shell.execute_reply":"2022-07-06T22:16:46.126493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create a top 20 DF to use for quicker modeling","metadata":{}},{"cell_type":"code","source":"##get the indices of the top 20 most common categories\nindex_top20 = df_meta['category'].value_counts().head(20).index\n#These are the top 20:\ndf_meta['scientific_name'].value_counts().head(20)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:16:48.88525Z","iopub.execute_input":"2022-07-06T22:16:48.886188Z","iopub.status.idle":"2022-07-06T22:16:48.948368Z","shell.execute_reply.started":"2022-07-06T22:16:48.886131Z","shell.execute_reply":"2022-07-06T22:16:48.947438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##Create new df with only the top 20 categoires.\ndf_meta_top20 = df_meta[df_meta['category'].isin(index_top20)] ##subset based on top 20 indices found above\ndf_meta_top20['scientific_name'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:16:50.714484Z","iopub.execute_input":"2022-07-06T22:16:50.714762Z","iopub.status.idle":"2022-07-06T22:16:51.002328Z","shell.execute_reply.started":"2022-07-06T22:16:50.714735Z","shell.execute_reply":"2022-07-06T22:16:51.00129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"The subset of the data we will try:\")\nprint(\"Unique parent folders: \", df_meta_top20['parent_folder'].unique()) ##working with only a subset of the data now\nprint(\"Unique subfolders: \", df_meta_top20['child_folder'].unique().shape, \" corresponds to the 20 categories\") ##only 20 subfolders = to 20 unique categories","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:16:52.707264Z","iopub.execute_input":"2022-07-06T22:16:52.707582Z","iopub.status.idle":"2022-07-06T22:16:52.71537Z","shell.execute_reply.started":"2022-07-06T22:16:52.707546Z","shell.execute_reply":"2022-07-06T22:16:52.714482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing the data","metadata":{}},{"cell_type":"code","source":"top20_parent_ids = df_meta_top20['parent_folder'].unique()\ntop20_child_ids = df_meta_top20['child_folder'].unique()\n\nX, y = preprocess_cnn(categories=top20_parent_ids, ##user def\n                      num_categories=top20_child_ids,\n                      width=299, height=299)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:16:59.088787Z","iopub.execute_input":"2022-07-06T22:16:59.089085Z","iopub.status.idle":"2022-07-06T22:17:14.451483Z","shell.execute_reply.started":"2022-07-06T22:16:59.089049Z","shell.execute_reply":"2022-07-06T22:17:14.450524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##convert x and y to arrays\nX = np.array(X)\ny = np.array(y)\nprint(\"X.shape: \", X.shape, \"~ 822 imgs, each 299x299 array, with 3 dims (RGB)\") \nprint(\"y.shape: \", y.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:17:30.935539Z","iopub.execute_input":"2022-07-06T22:17:30.935846Z","iopub.status.idle":"2022-07-06T22:17:31.078057Z","shell.execute_reply.started":"2022-07-06T22:17:30.935816Z","shell.execute_reply":"2022-07-06T22:17:31.077072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"code","source":"##Prep the inputs and labels\nX = X.astype(int) ##convert pixel vals to integer\n\n# from sklearn.preprocessing import LabelEncoder\n##Encode target labels with value between 0 and n_classes-1.\n\nle = LabelEncoder()\ny = le.fit_transform(y) ##Fit label encoder and return encoded labels.\ny[0:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:18:19.722231Z","iopub.execute_input":"2022-07-06T22:18:19.72286Z","iopub.status.idle":"2022-07-06T22:18:20.45398Z","shell.execute_reply.started":"2022-07-06T22:18:19.722816Z","shell.execute_reply":"2022-07-06T22:18:20.453444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Split train and validation subsets\n# from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.25, stratify = y, random_state = 42)\nprint(X_train.shape)\nprint(X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:18:26.598316Z","iopub.execute_input":"2022-07-06T22:18:26.59922Z","iopub.status.idle":"2022-07-06T22:18:27.172934Z","shell.execute_reply.started":"2022-07-06T22:18:26.599172Z","shell.execute_reply":"2022-07-06T22:18:27.172043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n# My image preprocessing helper functions\nIncludes:\n  - Function to import image from directory, apply processing, and add to list of images\n  - Function to split a df into smaller subsets and function to apply other function in parallel across those subsets\n  - function to shuffle images\n  - saving and loading of np arrays","metadata":{}},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"def preprocess_import_deprecated(meta_data, categories=None, sub_categories=None, \n                      width=299, height=299, \n                      return_array=False, imgs_per_cat=None, \n                      directory=\"train\"):\n    \"\"\"\n    Ex: categories = 000, sub_categories = 00 (correspond to parent_folder, child_folder of an image path)\n    Function imports images from the selected categories and applies some preprocessing.\n    Produces X, y data (image, label)\n    \"\"\"\n    if directory == \"train\":\n        DIR = \"../input/herbarium-2022-fgvc9/train_images\"\n    elif directory == \"test\":\n        DIR = \"../input/herbarium-2022-fgvc9/test_images/\"\n    if categories == None:\n        categories = meta_data['parent_folder']\n        sub_categories = meta_data['child_folder']\n    else:\n        categories = categories\n        sub_categories = sub_categories\n    list_img = [] ## a list of the images\n    labels = []   ## a list of the corresponding categories\n    for cat, sub_cat in tqdm.tqdm(zip(categories, sub_categories), disable=False):\n        ## Now extract each image from the current categories/sub_categories path\n        for ig in os.listdir(os.path.join(DIR, cat, sub_cat))[0:imgs_per_cat]:\n            ##read in image\n            img = cv2.imread(os.path.join(DIR, cat, sub_cat, ig))\n            ##resize\n            img = cv2.resize(img, (width, height), interpolation=cv2.INTER_LINEAR)\n            ##equalize\n            img_yuv = cv2.cvtColor(img,cv2.COLOR_RGB2YUV) ##convert to YUB\n            img_yuv[:,:,0] = cv2.equalizeHist(img_yuv[:,:,0]) ##equalize histogram\n            img_equ = cv2.cvtColor(img_yuv, cv2.COLOR_YUV2RGB) ##convert to RGB\n            final_img = img_equ\n            img_label = ig.split(\"__\")[0]\n            list_img.append(final_img)\n            labels.append(img_label)     ##label is the part before \"__\" (eg, 2774 for 02774__001.jpg where cat=27, sub_cat=74)\n    if return_array == True:\n        list_img = np.array(list_img)\n        labels = np.array(labels)\n    return list_img, labels #X, y","metadata":{"jupyter":{"source_hidden":true}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### preprocess_import()\nImports images from metadata's path column, applies preprocessing, saves list of images and categories (X,y).","metadata":{}},{"cell_type":"code","source":"def preprocess_import(meta_data, \n              width=299, height=299, \n              #imgs_per_cat=None, \n              directory=\"train\"):\n    \"\"\"\n    Function imports images from the selected categories and applies some preprocessing.\n    Produces X, y data (image, label)\n    \"\"\"\n    if directory == \"train\":\n        DIR = \"../input/herbarium-2022-fgvc9/train_images\"\n    elif directory == \"test\":\n        DIR = \"../input/herbarium-2022-fgvc9/test_images/\"\n    list_img = [] ## a list to store the processed images\n    labels = []   ## a list to store the corresponding categories\n    for path, category in tqdm.tqdm(zip(meta_data['path'], meta_data['category'])):\n        ##read in image\n        img = cv2.imread(os.path.join(DIR, path))\n        ##resize\n        img = cv2.resize(img, (width, height), interpolation=cv2.INTER_LINEAR)\n        ##equalize\n        img_yuv = cv2.cvtColor(img,cv2.COLOR_RGB2YUV) ##convert to YUB\n        img_yuv[:,:,0] = cv2.equalizeHist(img_yuv[:,:,0]) ##equalize histogram\n        img_equ = cv2.cvtColor(img_yuv, cv2.COLOR_YUV2RGB) ##convert to RGB\n        final_img = img_equ\n        list_img.append(final_img)\n        labels.append(category)\n        \n    return list_img, labels #X, y","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:23:16.258275Z","iopub.execute_input":"2022-07-06T22:23:16.258562Z","iopub.status.idle":"2022-07-06T22:23:16.26619Z","shell.execute_reply.started":"2022-07-06T22:23:16.258529Z","shell.execute_reply":"2022-07-06T22:23:16.265348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Parallelized preprocessing function\nApplies the `preprocess_import` function in parallel across sliced chunks of metadata. Uses multiprocessing.Pool.","metadata":{}},{"cell_type":"code","source":"# Parallelize the import function\ndef train_import_args(meta_data):\n    \"\"\"\n    Function with other preprocesser arguments defined so that Pool.map knows to use the sub-df in\n    the df_list as the meta_data argument.\n    \"\"\"\n    X, y =  preprocess_import(meta_data, \n                              width=299, height=299, \n                              directory=\"train\")\n    return X, y\n\n\ndef parallel_preprocess_train(df_list, processes):\n    \"\"\"\n    Map preprocesser function to chunks of the training dataframe in parallel \n    to speed up import/preprocess time.\n    \"\"\"\n    p = Pool(processes=processes)\n    data = p.map(train_import_args, [df for df in df_list])\n    p.close()\n    return data","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:23:18.736956Z","iopub.execute_input":"2022-07-06T22:23:18.737494Z","iopub.status.idle":"2022-07-06T22:23:18.743286Z","shell.execute_reply.started":"2022-07-06T22:23:18.737459Z","shell.execute_reply":"2022-07-06T22:23:18.742701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## split_data\nsplits a df into n sub-dfs, produces a df list\n\n## concatenate_output\nconcatenates the list of parallel preprocessed images and their categories (X, y) into one object.","metadata":{}},{"cell_type":"code","source":"def split_data(n, df):\n    \"\"\"\n    Splits df into n sub-dfs and adds them to a list. Stored sequentially (in order of index)\n    \"\"\"\n    start_i = 0\n    end_i = len(df)//n\n    \n    df_list = []\n    for i in range(0, n):\n        sub_i = df[start_i:end_i].copy()\n        df_list.append(sub_i)\n        start_i = end_i\n        end_i = end_i + len(df)//n\n    \n    return df_list\n\ndef concatenate_output(results_list, splits):\n    \"\"\"\n    Concatenates the list of parallel processed results into one X list and one y list,\n    producing a final, preprocessed, and ready to go X, y list\n    \"\"\"\n    X = []\n    y = []\n    for i in range(0, splits):\n        X += results_list[i][0]\n        y += results_list[i][1]\n    X = np.array(X)\n    y = np.array(y)\n    return X, y","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:23:20.434714Z","iopub.execute_input":"2022-07-06T22:23:20.435748Z","iopub.status.idle":"2022-07-06T22:23:20.443465Z","shell.execute_reply.started":"2022-07-06T22:23:20.435697Z","shell.execute_reply":"2022-07-06T22:23:20.442472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing example","metadata":{}},{"cell_type":"code","source":"# Split data into n subsets to apply preprocessing in parallel\nsplits = 4\nmeta_df_list = split_data(splits, df_meta)\nprint(\"length of meta_df_list:\" , len(meta_df_list))\nprint(\"total length of meta_df_list contents: \", len(meta_df_list[0]) + len(meta_df_list[1]) + len(meta_df_list[2]) + len(meta_df_list[3]))\nprint(\"length of actual df: \", len(df_meta))","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:26:08.46536Z","iopub.execute_input":"2022-07-06T22:26:08.465663Z","iopub.status.idle":"2022-07-06T22:26:08.563334Z","shell.execute_reply.started":"2022-07-06T22:26:08.465633Z","shell.execute_reply":"2022-07-06T22:26:08.562282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply the preprocessing function to the training metadata to import, preprocess the images\n#data_list = parallel_preprocess_train(df_list=metadf_list, processes=splits) ##run once\n\n# Concatenates output and converts to np array \n#X, y = concatenate_output(data_list, 3)  ##run once","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:26:07.230987Z","iopub.execute_input":"2022-07-06T22:26:07.231278Z","iopub.status.idle":"2022-07-06T22:26:07.235533Z","shell.execute_reply.started":"2022-07-06T22:26:07.231249Z","shell.execute_reply":"2022-07-06T22:26:07.234584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save arrays to external file to avoid re-loading ##run once\n#with open('Xtrain.npy', 'wb') as f:\n#    np.save(file=f,arr=X)\n#with open('ytrain.npy', 'wb') as f:\n#    np.save(file=f,arr=y)\n\n# Load arrays from external files\npp = \"../input/herb22processedtrainingimgs/\"\nwith open(pp+'Xtrain.npy', 'rb') as f:\n    X = np.load(f)\nwith open(pp+'ytrain.npy', 'rb') as f:\n    y = np.load(f)\n\nprint(\"X.shape: \", X.shape)\nprint(\"y.shape: \", y.shape)\nprint(\"Number of categories: \", len(np.unique(y)))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##shuffle images\ndef shuffle_data(X, y):\n    assert len(X) == len(y)\n    p = np.random.permutation(len(X))\n    return X[p], y[p]\n\nnp.random.seed(42)\nX, y = shuffle_data(X, y)\n\n\n##Encode target labels with value between 0 and n_classes-1 (since the categories are currently an assortment of #s from all over the place)\nle = LabelEncoder() ##sklearn.preprocessing\ny = le.fit_transform(y)\n\n##One-hot encode\ny = tf.keras.utils.to_categorical(y)","metadata":{},"execution_count":null,"outputs":[]}]}