{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":33679,"databundleVersionId":3212216,"sourceType":"competition"},{"sourceId":3907121,"sourceType":"datasetVersion","datasetId":2320690}],"dockerImageVersionId":30204,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Setup","metadata":{}},{"cell_type":"code","source":"# General\nimport pandas as pd\nimport numpy as np\nimport json\nimport os\nimport random\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n# For image/data preprocessing\nimport cv2\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\n# For modeling, eval\nimport tensorflow as tf\nfrom tensorflow.keras import backend as K\nimport tensorflow_addons as tfa\n# Performance\nimport tqdm as tqdm\nimport multiprocessing\nfrom multiprocessing import Pool\n# Weights & Biases\nimport wandb\nfrom wandb.keras import WandbCallback\nfrom kaggle_secrets import UserSecretsClient","metadata":{"execution":{"iopub.status.busy":"2022-07-06T20:28:54.961872Z","iopub.execute_input":"2022-07-06T20:28:54.96257Z","iopub.status.idle":"2022-07-06T20:28:54.97268Z","shell.execute_reply.started":"2022-07-06T20:28:54.962507Z","shell.execute_reply":"2022-07-06T20:28:54.97151Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# GPU ACCELERATOR ENABLED\ndevice_name = tf.test.gpu_device_name()\nif \"GPU\" not in device_name:\n    print(\"GPU device not found\")\nelse:\n    print('Found GPU at: {}'.format(device_name))","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:27:54.899663Z","iopub.execute_input":"2022-07-06T18:27:54.900476Z","iopub.status.idle":"2022-07-06T18:27:54.911563Z","shell.execute_reply.started":"2022-07-06T18:27:54.90042Z","shell.execute_reply":"2022-07-06T18:27:54.910571Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"multiprocessing.cpu_count()  ","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:27:54.945509Z","iopub.execute_input":"2022-07-06T18:27:54.946264Z","iopub.status.idle":"2022-07-06T18:27:54.954119Z","shell.execute_reply.started":"2022-07-06T18:27:54.94623Z","shell.execute_reply":"2022-07-06T18:27:54.953119Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Exploratory Data Analysis Notebook:** https://www.kaggle.com/code/hanselliott/herbarium22-eda/","metadata":{}},{"cell_type":"markdown","source":"## Import Data","metadata":{}},{"cell_type":"code","source":"TRAIN_DIR = \"../input/herbarium-2022-fgvc9/train_images/\"\nTEST_DIR = \"../input/herbarium-2022-fgvc9/test_images/\"\n\nwith open(\"../input/herbarium-2022-fgvc9/train_metadata.json\") as json_file:\n    train_meta = json.load(json_file)\nwith open(\"../input/herbarium-2022-fgvc9/test_metadata.json\") as json_file:\n    test_meta = json.load(json_file)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:27:54.981058Z","iopub.execute_input":"2022-07-06T18:27:54.981306Z","iopub.status.idle":"2022-07-06T18:28:05.007247Z","shell.execute_reply.started":"2022-07-06T18:27:54.981284Z","shell.execute_reply":"2022-07-06T18:28:05.00628Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"../input/herbarium-2022-fgvc9/sample_submission.csv\")\nprint(\"SAMPLE SUBMISSION\")\nsample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:28:05.009183Z","iopub.execute_input":"2022-07-06T18:28:05.009576Z","iopub.status.idle":"2022-07-06T18:28:05.095353Z","shell.execute_reply.started":"2022-07-06T18:28:05.009541Z","shell.execute_reply":"2022-07-06T18:28:05.094412Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test = pd.DataFrame(test_meta)\nprint(\"TEST DATA\")\ndf_test.head() ##untouched until ready to predict","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:28:05.096712Z","iopub.execute_input":"2022-07-06T18:28:05.097309Z","iopub.status.idle":"2022-07-06T18:28:05.318171Z","shell.execute_reply.started":"2022-07-06T18:28:05.097269Z","shell.execute_reply":"2022-07-06T18:28:05.317134Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Prepare Training Meta-Data","metadata":{}},{"cell_type":"code","source":"#Create a meta-data df that can be used to call in training images\nids = []\ncategories = []\npaths = []\n\nfor annotation, image in zip(train_meta['annotations'], train_meta['images']):\n    ids.append(image[\"image_id\"])\n    categories.append(annotation['category_id'])\n    paths.append(image[\"file_name\"])\n\ndf_train = pd.DataFrame({\"id\":ids, \"category\":categories, \"path\":paths})\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:28:05.32127Z","iopub.execute_input":"2022-07-06T18:28:05.321687Z","iopub.status.idle":"2022-07-06T18:28:06.264547Z","shell.execute_reply.started":"2022-07-06T18:28:05.32165Z","shell.execute_reply":"2022-07-06T18:28:06.26362Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##extract metadata features by category to merge with df_meta\nsci_name = {cat[\"category_id\"]:cat[\"scientificName\"] for cat in train_meta['categories']}\nfamily = {cat[\"category_id\"]:cat[\"family\"] for cat in train_meta['categories']}\ngenus = {cat[\"category_id\"]:cat[\"genus\"] for cat in train_meta['categories']}\nspecies = {cat[\"category_id\"]:cat[\"species\"] for cat in train_meta['categories']}\n\ndf_train[\"scientific_name\"] = df_train[\"category\"].map(sci_name)\ndf_train[\"family\"] = df_train[\"category\"].map(family)\ndf_train[\"genus\"] = df_train[\"category\"].map(genus)\ndf_train[\"species\"] = df_train[\"category\"].map(species)\n\n##split the path based on '/' into parent and child folder. \n##lambda fn is applied to each row in the column to split each path\ndf_train['path'].apply(lambda x : x.split('/'))\n\n#add categories/sub_categories equivalents to df_meta\ndf_train['parent_folder'] = df_train['path'].apply(lambda x : x.split('/')[0])\ndf_train['child_folder'] = df_train['path'].apply(lambda x : x.split('/')[1])\n\n\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:28:06.266089Z","iopub.execute_input":"2022-07-06T18:28:06.266476Z","iopub.status.idle":"2022-07-06T18:28:09.864448Z","shell.execute_reply.started":"2022-07-06T18:28:06.266439Z","shell.execute_reply":"2022-07-06T18:28:09.863336Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Example Images","metadata":{}},{"cell_type":"code","source":"def plot_random_images(metadata, directory, n_imgs, dims=[3,4], random_seed=12):\n    \"\"\"\n    Function randomly selects paths from the train metadata and plots the corresponding image.  \n    \"\"\"\n    np.random.seed = random_seed\n    # Randomly sample n rows from metatdata\n    rndm_elems = metadata.sample(n=n_imgs)      \n    \n    # Add the img path, category, and sci name to lists\n    imgs = []\n    category = []\n    scientific_name = []\n    for path, categ, sci_name in zip(rndm_elems['path'], rndm_elems['category'], rndm_elems['scientific_name']):\n        imgs.append(cv2.imread(os.path.join(directory,path)))\n        category.append(categ)\n        scientific_name.append(sci_name)\n    # Prepare figures/axes for subplots\n    fig, axes = plt.subplots(dims[0], dims[1], figsize=(10,10))\n    axes = axes.flatten()\n    # For each image, plot image to a subplot and title with category + scientific name\n    for img, ax, c, s in zip(imgs, axes, category, scientific_name):\n        title = str(c) + \" | \" + s\n        ax.imshow(img)\n        ax.axis('off')\n        ax.title.set_text(title)\n    plt.suptitle(\"Example Images\")\n    plt.show()\n\n    \nTRAIN_DIR = \"../input/herbarium-2022-fgvc9/train_images/\"\nplot_random_images(df_train, TRAIN_DIR, 8, [4,2], random_seed=123)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:23:32.622244Z","iopub.execute_input":"2022-07-06T21:23:32.622615Z","iopub.status.idle":"2022-07-06T21:23:33.73536Z","shell.execute_reply.started":"2022-07-06T21:23:32.622583Z","shell.execute_reply":"2022-07-06T21:23:33.734335Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reduced Training Data\nI want to reduce the size of the data somewhat. There are over 1500 potential categories, which is a lot for the model to pick between.  \nOf these categories, only 174 of them have 80 image files in the training dataset - all other categories have fewer files. So I will just focus on classifying these 174 plants since we have ample training data to feed the model.    \nWhen it comes to prediction time, the model will, of course, misclassify any of the images belonging to one of the classes not in this subset. This is sort of a shortcut to getting into training so I can figure out the rest of the process and start experimentation.  ","metadata":{}},{"cell_type":"code","source":"print(\"Parent_folders\", df_train['parent_folder'].unique())\nprint(\"Child folders\", df_train['child_folder'].unique())\n\nparent_cats = str(df_train['category'].unique())\nprint(\"Categories\", parent_cats)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:28:10.231466Z","iopub.execute_input":"2022-07-06T18:28:10.232223Z","iopub.status.idle":"2022-07-06T18:28:10.361013Z","shell.execute_reply.started":"2022-07-06T18:28:10.232183Z","shell.execute_reply":"2022-07-06T18:28:10.359968Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Reduce the number of categories we train on to increase training times**","metadata":{}},{"cell_type":"code","source":"cat_val_cnt = df_train['category'].value_counts()\nprint(\"Category, Number of Images: \\n\",\n      cat_val_cnt) \nprint(\"\")\nprint(\"Number of categories which have 80 training images:\", len(cat_val_cnt[cat_val_cnt == 80]))\ncat_index = cat_val_cnt[cat_val_cnt == 80].sort_values(ascending=False).index","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:28:10.362248Z","iopub.execute_input":"2022-07-06T18:28:10.362591Z","iopub.status.idle":"2022-07-06T18:28:10.379014Z","shell.execute_reply.started":"2022-07-06T18:28:10.362547Z","shell.execute_reply":"2022-07-06T18:28:10.37796Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reduce training df to include only samples belonging to 80-image categories\nred_train_df = df_train[df_train.category.isin(cat_index)]\nred_train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:28:10.380478Z","iopub.execute_input":"2022-07-06T18:28:10.381244Z","iopub.status.idle":"2022-07-06T18:28:10.648456Z","shell.execute_reply.started":"2022-07-06T18:28:10.381205Z","shell.execute_reply":"2022-07-06T18:28:10.647419Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Now we have \", red_train_df['category'].nunique(), \"categories instead of \", df_train['category'].nunique())\nprint(\"reduced training metadata shape: \", red_train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:28:10.649852Z","iopub.execute_input":"2022-07-06T18:28:10.650645Z","iopub.status.idle":"2022-07-06T18:28:10.665463Z","shell.execute_reply.started":"2022-07-06T18:28:10.650601Z","shell.execute_reply":"2022-07-06T18:28:10.664366Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Prep metadata for flow_from_dataframe()  \nThis function requires your metadata has a column specifying the path to an image (which we have already) and a column with category labels encoded as strings - which we need to do below.  ","metadata":{}},{"cell_type":"code","source":"# Prep meta-data df for the flow_from_dataframe() fn to work properly\nred_train_df = red_train_df.reset_index()\n## Convert 'category' to string, which is required for categorical classes in flow_from_dataframe \nred_train_df['str_category'] = ''\nred_train_df[['category']] = red_train_df[['category']].astype(int)\nred_train_df[['str_category']] = red_train_df[['category']].astype(str)\nred_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:28:24.97378Z","iopub.execute_input":"2022-07-06T18:28:24.974124Z","iopub.status.idle":"2022-07-06T18:28:25.01185Z","shell.execute_reply.started":"2022-07-06T18:28:24.974093Z","shell.execute_reply":"2022-07-06T18:28:25.010826Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = red_train_df['str_category'].unique()\nlen(labels)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:29:03.49856Z","iopub.execute_input":"2022-07-06T18:29:03.498903Z","iopub.status.idle":"2022-07-06T18:29:03.507717Z","shell.execute_reply.started":"2022-07-06T18:29:03.498874Z","shell.execute_reply":"2022-07-06T18:29:03.506606Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Re-encode target labels\nUse `LabelEncoder` to rencode the categories, since they are currently an assortment of numbers (due to the fact that I am only using categories with 80 training images).  \nWe will have to reverse this encoding at the end after the model makes its predictions.  ","metadata":{}},{"cell_type":"code","source":"##Encode target labels with value between 0 and n_classes-1 (since the categories are currently an assortment of #s from all over the place)\nle = LabelEncoder() ##sklearn.preprocessing\nencoded_labels = le.fit_transform(red_train_df['str_category'])\nred_train_df['encoded_labels'] = encoded_labels\nred_train_df[['encoded_labels']] = red_train_df[['encoded_labels']].astype(str) ##convert to string for flow_from_dataframe()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:29:27.851623Z","iopub.execute_input":"2022-07-06T18:29:27.852529Z","iopub.status.idle":"2022-07-06T18:29:27.879464Z","shell.execute_reply.started":"2022-07-06T18:29:27.852488Z","shell.execute_reply":"2022-07-06T18:29:27.878385Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Setup  ","metadata":{}},{"cell_type":"markdown","source":"### Submissions are scored using a Macro F1 score","metadata":{}},{"cell_type":"code","source":"num_classes = 174 #red_train_df['category'].nunique()\nf1_macro = tfa.metrics.F1Score(num_classes=num_classes, average='macro') ##from TensorFlow Addons","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:46:47.49175Z","iopub.execute_input":"2022-07-06T21:46:47.492109Z","iopub.status.idle":"2022-07-06T21:46:47.50829Z","shell.execute_reply.started":"2022-07-06T21:46:47.492077Z","shell.execute_reply":"2022-07-06T21:46:47.507336Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Here I tell tensorflow to use \"mixed precision\" which may save some memory in the process of training.  ","metadata":{}},{"cell_type":"code","source":"# Default float type is 32\nprint(\"default float type:\", tf.keras.backend.floatx() )\n# Reduce to mixed precision: https://www.tensorflow.org/api_docs/python/tf/keras/backend/set_floatx\n#tf.keras.mixed_precision.experimental.set_policy('mixed_float16')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:29:32.65449Z","iopub.execute_input":"2022-07-06T18:29:32.655077Z","iopub.status.idle":"2022-07-06T18:29:32.66426Z","shell.execute_reply.started":"2022-07-06T18:29:32.655038Z","shell.execute_reply":"2022-07-06T18:29:32.6631Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Login to Wandb\nHere I login to Weights and Biases, create a hyperparameter dictionary, and intialize a Wandb run with `wandb.init()`.  \nThis connects my session to Wandb and starts the tracking of system performance.  \nTo login automatically (so that it works when you commit your notebook), you need to include your API key (found in your Wandb account page). You can use the kaggle_secrets module to hide this key from others.   ","metadata":{}},{"cell_type":"code","source":"wandb.login(key='fed8886fa715351293a079cd945d36b6baa126db')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:45:37.719242Z","iopub.execute_input":"2022-07-06T21:45:37.719635Z","iopub.status.idle":"2022-07-06T21:45:38.32444Z","shell.execute_reply.started":"2022-07-06T21:45:37.719604Z","shell.execute_reply":"2022-07-06T21:45:38.323457Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Param dict\ndefault=dict(\n    dropout = 0.35,\n    kernel_size=(3,3),\n    layer_1_size = 32,\n    layer_2_size = 32,\n    pool_1_size = (2,2),\n    layer_3_size=64,\n    layer_4_size=64,\n    pool_2_size=(3,3),\n    layer_5_size=1024,\n    layer_6_size=420,\n    learn_rate = 0.001,\n    beta_1 = 0.9,\n    beta_2 = 0.999,\n    epochs = 100,\n    batch_size = 64,\n    img_size = (120, 120),\n    architecture=\"CNN\",\n    infra=\"Kaggle\"\n   )\n\n# Weights & Biases Initialization\nwandb.init(anonymous='allow', project=\"herb22\", config=default)\nconfig = wandb.config","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:45:41.752224Z","iopub.execute_input":"2022-07-06T21:45:41.752596Z","iopub.status.idle":"2022-07-06T21:45:44.507238Z","shell.execute_reply.started":"2022-07-06T21:45:41.752563Z","shell.execute_reply":"2022-07-06T21:45:44.506205Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model Structure  \nHere I specify the architecture of the basic CNN with convolution layers, pooling layers, and batch normalization, followed by some dense layers and then an output layer with a node for each of the 174 classes.  \nBelow the model structure I specify the loss function as Categorical Crossentropy and the optimizer as Adam, where I choose the learning rate and beta_1, beta_2 values.  \nThe model is compiled and I choose to track accuracy as well as the F1 Macro score (which I added using TensorFlow Addons).  \nFinally I specify an ImageDataGenerator: this will randomly augment the images based on the specified parameters, ideally making the model more general to different types of images of the same category.  \nWithin the `ImageDataGenerator` I also implement a custom image preprocessing function which modifies the pixel values of the image - specifially, it converts the images from RGB to [YUV](https://en.wikipedia.org/wiki/YUV) and equalizes the luminescence (Y) component, before converting back to RGB.  ","metadata":{}},{"cell_type":"code","source":"# 2d ConvNet\nimg_shape = (120, 120, 3)\ncnnmod = tf.keras.models.Sequential([\n    tf.keras.layers.Input(shape=img_shape),\n    tf.keras.layers.Conv2D(filters=config.layer_1_size, kernel_size=config.kernel_size, padding='same', activation='relu'),\n    tf.keras.layers.Conv2D(filters=config.layer_2_size, kernel_size=config.kernel_size, padding='same', activation='relu'),\n    tf.keras.layers.MaxPool2D(pool_size=config.pool_1_size),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.Dropout(config.dropout),\n    tf.keras.layers.Conv2D(filters=config.layer_3_size, kernel_size=config.kernel_size, padding='same', activation='relu'),\n    tf.keras.layers.Conv2D(filters=config.layer_4_size, kernel_size=config.kernel_size, padding='same', activation='relu'),\n    tf.keras.layers.MaxPool2D(pool_size=config.pool_2_size),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.Dropout(config.dropout),\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dense(config.layer_5_size, activation='relu'),\n    tf.keras.layers.Dense(config.layer_6_size, activation='relu'),\n    tf.keras.layers.Dense(174, activation='softmax')\n])\n\ncnnmod.summary()\n\ncnnweights = cnnmod.get_weights()\n#fn to reset model weights to randomly initialized if want to restart training\nreset_model = lambda model, weights: model.set_weights(weights) \n# ------------------------------------------------------------------------------\n\n# Loss\nloss_fn = tf.keras.losses.CategoricalCrossentropy(from_logits=False)\n# Optimizier (adam)\noptim = tf.keras.optimizers.Adam(learning_rate=config.learn_rate,\n                                 beta_1=config.beta_1,\n                                 beta_2=config.beta_2)\n\n# Compile\ncnnmod.compile(optimizer=optim,\n               loss=loss_fn,\n               metrics=['accuracy', f1_macro])\n\n# -------------------------------------------------------------------------------\n# Data Generator (for augmentation/preprocessing in the flow of training)\n\n## Custom preprocessing function to apply to each image\ndef _adjust_image(img):\n    \"\"\"\n    Uses cv2 to apply some preprocessing to each image to improve the data.\n    \"\"\"\n    img_yuv = cv2.cvtColor(img, cv2.COLOR_BGR2YUV)  ##convert from RGB to YUV\n    img_gray = img_yuv[:,:,0].astype(np.uint8) ##convert to single channel (Y channel is the luminance component)\n    img_equ = cv2.equalizeHist(img_gray)       ##equalize histogram (note that only works for single channel unit8)\n    img_yuv[:,:,0] = img_equ                   ##add equalized channel back in\n    img_rgb = cv2.cvtColor(img_yuv, cv2.COLOR_YUV2BGR) ##convert back to RGB  \n    return img_rgb\n\n## ImageDataGenerator (train & validation)\ntrain_datagen = tf.keras.preprocessing.image.ImageDataGenerator(\n                preprocessing_function=_adjust_image,\n                rescale=1.0/255,\n                rotation_range=30,\n                width_shift_range=0.1,\n                height_shift_range=0.1,\n                shear_range=0.1,\n                zoom_range=0.2,\n                horizontal_flip=True,\n                fill_mode='reflect',\n                 #cval = 235,   ##a bright constant value for the fill mode \"constant\"\n                validation_split=0.2,\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:45:53.974978Z","iopub.execute_input":"2022-07-06T21:45:53.975354Z","iopub.status.idle":"2022-07-06T21:45:54.171131Z","shell.execute_reply.started":"2022-07-06T21:45:53.975323Z","shell.execute_reply":"2022-07-06T21:45:54.16948Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Below is a demonstration of the image preprocessing and augmentation.  \nThe cutom preprocessing function (`_adjust_image` above) aims to exaggerate the edges and form of the plant and pull it away from the underlying page.  ","metadata":{}},{"cell_type":"code","source":"# Example Image\npath = \"../input/herbarium-2022-fgvc9/train_images/021/25/02125__015.jpg\"\nex_img = cv2.imread(os.path.join(path))\nplt.imshow(ex_img.astype(np.uint8))\nplt.title(\"Example image (unprocessed)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T20:08:13.974637Z","iopub.execute_input":"2022-07-06T20:08:13.975133Z","iopub.status.idle":"2022-07-06T20:08:14.399269Z","shell.execute_reply.started":"2022-07-06T20:08:13.975091Z","shell.execute_reply":"2022-07-06T20:08:14.394668Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Image after augmentation, preprocessing is applied\nx = ex_img\nx = x.reshape((1,) + x.shape)\naug_x = train_datagen.flow(x)\naug_images = [next(aug_x)[0] for i in range(12)]\n\nfig, axes = plt.subplots(3,4,figsize=(10,10)) ##3 img rows, 4 img cols\naxes = axes.flatten()\nfor img, ax in zip(aug_images,axes): ##zip image to its subplot\n    ax.imshow(img)\n    #ax.axis('off')\n    \nplt.suptitle(\"Example image after augmentation\",fontsize=24)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T20:08:18.242418Z","iopub.execute_input":"2022-07-06T20:08:18.24312Z","iopub.status.idle":"2022-07-06T20:08:22.10125Z","shell.execute_reply.started":"2022-07-06T20:08:18.243073Z","shell.execute_reply":"2022-07-06T20:08:22.10023Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training\nTrains the model for the specified number of epochs.  \nNote the `reset_model` function restore the model's weights to the randomly initialized weights. This is useful if one wants to restart training from scratch while experimenting.  \nThe training process utilizes `flow_from_dataframe` which pulls in images from the specified path and applies the augmentation and preprocessing as specified in the `ImageDataGenerator`. In previous (naive) attempts, I tried loading in all images and preprocessing before training, which was way too memory intensive. This provides an easy to implement and effective way to reduce memory usage and still process the images.    \nFinally, note that I include `WandbCallback` in the callbacks argument which essentially connects this model to my personal Weights and Biases project dashboard. With this snippet of code, Wandb allows me to track model progress as it trains (even if I commit the notebook and exit the tab). That includes tracking loss, accuracy, and F1 macro score, for both the training and validation data.  ","metadata":{}},{"cell_type":"code","source":"reset_model(cnnmod, cnnweights) ##restore weights to random\nepochs = config.epochs\nbatch_size = config.batch_size\ntarget_size = config.img_size\nlabels = red_train_df['str_category'].unique()\n\n# TRAIN\nhistory_cnn = cnnmod.fit(train_datagen.flow_from_dataframe(dataframe=red_train_df,\n                                                                    directory=\"../input/herbarium-2022-fgvc9/train_images\",\n                                                                    x_col='path',\n                                                                    y_col='encoded_labels',\n                                                                    target_size = target_size,\n                                                                    batch_size = batch_size,\n                                                                    subset = \"training\"\n                                                                    ),\n                                   validation_data = train_datagen.flow_from_dataframe(dataframe=red_train_df,\n                                                                    directory=\"../input/herbarium-2022-fgvc9/train_images\",\n                                                                    x_col='path',\n                                                                    y_col='encoded_labels',\n                                                                    target_size = target_size,\n                                                                    batch_size = batch_size,\n                                                                    subset = \"validation\"\n                                                                    ),\n                                   epochs = epochs,\n                                   callbacks=[WandbCallback(input_type=\"images\", labels=labels)]\n                                  )","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:47:09.59408Z","iopub.execute_input":"2022-07-06T21:47:09.594536Z","iopub.status.idle":"2022-07-06T21:53:51.449134Z","shell.execute_reply.started":"2022-07-06T21:47:09.59449Z","shell.execute_reply":"2022-07-06T21:53:51.448092Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wandb.finish() ##tell wandb to stop tracking the session","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:53:54.435128Z","iopub.execute_input":"2022-07-06T21:53:54.435555Z","iopub.status.idle":"2022-07-06T21:54:00.576884Z","shell.execute_reply.started":"2022-07-06T21:53:54.435522Z","shell.execute_reply":"2022-07-06T21:54:00.576022Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Plotting the fit history\nfig, ax = plt.subplots(nrows=1, ncols=3, figsize=(20,10))\n\nax[0].plot(history_cnn.history['loss'], label='loss')\nax[0].plot(history_cnn.history['val_loss'], label='val loss')\nax[0].legend()\n\nax[1].plot(history_cnn.history['accuracy'], label='acc')\nax[1].plot(history_cnn.history['val_accuracy'], label='val acc')\nax[1].legend()\n\nax[2].plot(history_cnn.history['f1_score'], label='f1')\nax[2].plot(history_cnn.history['val_f1_score'], label='val f1')\nax[2].legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:54:06.655931Z","iopub.execute_input":"2022-07-06T21:54:06.656324Z","iopub.status.idle":"2022-07-06T21:54:07.093316Z","shell.execute_reply.started":"2022-07-06T21:54:06.656293Z","shell.execute_reply":"2022-07-06T21:54:07.092383Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Predict onto test images & submit\nWe can use `ImageDataGenerator` and `flow_from_dataframe` to process the test images in the same manner as the training images, and then make predictions directly using the generator. This is also especially useful since we can predict in batches, which avoids memory allocation issues (we cannot load in all 200,000 images at once to be predicted).    \nThen use numpy.argmax to determine which column (corresponding to the one-hot-encoded `LabelEncoder` encoded categories) has the highest predicted probability for each sample.   \nThen I reverse the label encodings to get the true category label predicted for that sample, and add this to a submission dataframe to convert to a CSV.","metadata":{}},{"cell_type":"code","source":"# Predict model onto test data\n##Test-set datagen:\ntest_datagen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255,\n                                                               preprocessing_function=_adjust_image\n).flow_from_dataframe(\n    dataframe=df_test,\n    directory=\"../input/herbarium-2022-fgvc9/test_images/\",\n    x_col='file_name',\n    y_col=None,\n    class_mode=None,\n    target_size=target_size,\n    batch_size = 128,\n    shuffle=False\n)\n\n## Predict with generator\ny_pred = cnnmod.predict(test_datagen)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T20:18:11.35821Z","iopub.execute_input":"2022-07-06T20:18:11.358629Z","iopub.status.idle":"2022-07-06T20:18:15.097201Z","shell.execute_reply.started":"2022-07-06T20:18:11.358583Z","shell.execute_reply":"2022-07-06T20:18:15.096154Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Determine the (encoded) class with highest predicted probability\ny_pred_encoded = np.argmax(y_pred, axis=1)\n\nprint(\"y_pred_encoded.shape \", y_pred_encoded.shape)\nprint(\"Model's unique class guesses (out of 174 possible): \", len(np.unique(y_pred_encoded)) )","metadata":{"execution":{"iopub.status.busy":"2022-07-06T20:18:19.534953Z","iopub.execute_input":"2022-07-06T20:18:19.535358Z","iopub.status.idle":"2022-07-06T20:18:19.544636Z","shell.execute_reply.started":"2022-07-06T20:18:19.535325Z","shell.execute_reply":"2022-07-06T20:18:19.543231Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reverse the label encodings to get true category labels\ny_pred_class = le.inverse_transform(y_pred_encoded)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T20:21:00.518927Z","iopub.execute_input":"2022-07-06T20:21:00.519745Z","iopub.status.idle":"2022-07-06T20:21:00.528725Z","shell.execute_reply.started":"2022-07-06T20:21:00.519701Z","shell.execute_reply":"2022-07-06T20:21:00.527601Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add predictions to a submission df with test-sample/image Id\nsubmission = sample_sub.copy()\nsubmission.drop(labels='Predicted',axis=1)\nsubmission['Predicted'] = y_pred_class\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T20:21:19.727896Z","iopub.execute_input":"2022-07-06T20:21:19.728472Z","iopub.status.idle":"2022-07-06T20:21:19.758551Z","shell.execute_reply.started":"2022-07-06T20:21:19.728402Z","shell.execute_reply":"2022-07-06T20:21:19.757489Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert to csv and save\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T23:52:58.939006Z","iopub.execute_input":"2022-07-05T23:52:58.939372Z","iopub.status.idle":"2022-07-05T23:52:59.240714Z","shell.execute_reply.started":"2022-07-05T23:52:58.939342Z","shell.execute_reply":"2022-07-05T23:52:59.239558Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Final notes  \nThe relatively simple CNN might not be a realistic approach to this task. The competition page shows that ResNet models have been fairly successful - these residual neural networks are very deep variants of the basic CNN, with some other tricks thrown in. This suggests that a relatively shallow CNN won't be very effective in comparison.  \nThat said, this notebook shows how to setup an image importation, preprocessing, and augmentation pipleine that can be fed right into a model. With Weights and Biases support, it's super easy to track hyperparameter tuning and results. It would be fairly simple to adjust the model (or completely change it) while still utilizing the majority of this code, which is a satisfying result itself.  ","metadata":{}}]}