{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Merging the Toxic Plant Data with the Herbarium 2022 Data\nThis notebook shows how to create a metadata datframe which merges the toxic plant classification data with the Herbarium 2022 competition's data, to increase the number of images available per class.  \nWhen we add both the Herbarium 2022 data and the toxic plant classification data as notebook Inputs, we have two different directories which we want to pull images from. Combining their metadata into one dataframe, with a row for every image and a column specifying the full path to the image, will allow for the use of operations such as tensorflow's `flow_from_dataframe`.  \n\nMake sure you add both the [Toxic Plant Data](https://www.kaggle.com/datasets/hanselliott/toxic-plant-classification) and the [Herbarium 2022 Data](https://www.kaggle.com/competitions/herbarium-2022-fgvc9/data) as notebook inputs. ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np \nimport os\nimport json\nimport random\n\nimport glob\nfrom PIL import Image\nimport tensorflow as tf\nprint(\"tensorflow version: \" + tf.__version__)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-31T02:12:14.648622Z","iopub.execute_input":"2022-08-31T02:12:14.649288Z","iopub.status.idle":"2022-08-31T02:12:14.65709Z","shell.execute_reply.started":"2022-08-31T02:12:14.649247Z","shell.execute_reply":"2022-08-31T02:12:14.655364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Random seed setting\nr_seed = 83\nrandom.seed(r_seed)\ntf.random.set_seed(r_seed)\nnp.random.seed(r_seed)","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:12:14.663586Z","iopub.execute_input":"2022-08-31T02:12:14.664341Z","iopub.status.idle":"2022-08-31T02:12:14.677741Z","shell.execute_reply.started":"2022-08-31T02:12:14.664289Z","shell.execute_reply":"2022-08-31T02:12:14.676325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prep Metadata","metadata":{}},{"cell_type":"markdown","source":"## Toxic Plants Data","metadata":{}},{"cell_type":"code","source":"toxic_meta = pd.read_csv(\"../input/toxic-plant-classification/tpc-imgs/toxic_metadata.csv\")\nnontoxic_meta = pd.read_csv(\"../input/toxic-plant-classification/tpc-imgs/nontoxic_metadata.csv\")\n\ntoxic_meta['toxicity'] = int(1)\nnontoxic_meta['toxicity'] = int(0)\ntpc_meta = pd.concat([toxic_meta, nontoxic_meta])\ntpc_meta['tox_class'] = tpc_meta['toxicity'].astype(str) + \"-\" + tpc_meta['class_id'].astype(str)\ntpc_meta","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:12:14.68044Z","iopub.execute_input":"2022-08-31T02:12:14.68127Z","iopub.status.idle":"2022-08-31T02:12:14.751642Z","shell.execute_reply.started":"2022-08-31T02:12:14.68122Z","shell.execute_reply":"2022-08-31T02:12:14.750318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Instatiate empty df\nfull_meta = pd.DataFrame()\n\n# For each class_id\nfor class_id in range(5):\n    # iterate through the \n    for tox, meta in enumerate([nontoxic_meta, toxic_meta]): #0 for nontoxic, 1 for toxic\n        # identify the path to its folder and the number of images in that folder\n        path = meta.loc[class_id, \"path\"]\n        n_imgs = len([img for img in os.listdir(path)])\n        # for each image, add a row to the empty df with the full path to the image (and other metadata)\n        for i in range(n_imgs): \n            ##ensure that the image name is correct by adding 0s where appropriate\n            if i < 10:\n                str_i = \"00\"+str(i)\n            elif 9 < i < 100:\n                str_i = \"0\"+str(i)\n            elif i > 99:\n                str_i = str(i)\n            ##create row and append to df\n            img_row = pd.DataFrame({\n                \"class_id\" : [class_id],\n                \"slang\" : meta.loc[class_id, \"slang\"],\n                \"scientific_name\" : meta.loc[class_id, \"scientific_name\"],\n                \"herbarium22_category_id\" : meta.loc[class_id, \"herbarium22_category_id\"],\n                \"path\" : path+str_i+\".jpg\",\n                \"toxicity\" : int(tox)\n            })\n            full_meta = full_meta.append(img_row)\n\n            \nfull_meta = full_meta.reset_index()\nfull_tpcmeta = full_meta.drop(labels=\"index\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:12:14.754185Z","iopub.execute_input":"2022-08-31T02:12:14.754973Z","iopub.status.idle":"2022-08-31T02:12:28.461242Z","shell.execute_reply.started":"2022-08-31T02:12:14.754927Z","shell.execute_reply":"2022-08-31T02:12:28.459946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_tpcmeta","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:12:28.466483Z","iopub.execute_input":"2022-08-31T02:12:28.467027Z","iopub.status.idle":"2022-08-31T02:12:28.486123Z","shell.execute_reply.started":"2022-08-31T02:12:28.466994Z","shell.execute_reply":"2022-08-31T02:12:28.484829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Herbarium 2022 Data","metadata":{}},{"cell_type":"code","source":"with open(\"../input/herbarium-2022-fgvc9/train_metadata.json\") as json_file:\n    train_meta = json.load(json_file)","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:12:28.489455Z","iopub.execute_input":"2022-08-31T02:12:28.490061Z","iopub.status.idle":"2022-08-31T02:12:46.498914Z","shell.execute_reply.started":"2022-08-31T02:12:28.490006Z","shell.execute_reply":"2022-08-31T02:12:46.496754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create a meta-data df that can be used to call in training images\nids = []\ncategories = []\npaths = []\n\nfor annotation, image in zip(train_meta['annotations'], train_meta['images']):\n    ids.append(image[\"image_id\"])\n    categories.append(annotation['category_id'])\n    paths.append(image[\"file_name\"])\n\nherb_meta = pd.DataFrame({\"id\":ids, \"category\":categories, \"path\":paths})\n\n##extract metadata features by category to merge with df_meta\nsci_name = {cat[\"category_id\"]:cat[\"scientificName\"] for cat in train_meta['categories']}\nherb_meta[\"scientific_name\"] = herb_meta[\"category\"].map(sci_name)\n\nherb_meta.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:19:03.344214Z","iopub.execute_input":"2022-08-31T02:19:03.344764Z","iopub.status.idle":"2022-08-31T02:19:04.424976Z","shell.execute_reply.started":"2022-08-31T02:19:03.344728Z","shell.execute_reply":"2022-08-31T02:19:04.424129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filter to just the categories we may want to use\nherb22_toxics = [c for c in toxic_meta.herbarium22_category_id]\nherb22_nontoxics = [c for c in nontoxic_meta.herbarium22_category_id]\n\nherb_meta = herb_meta[herb_meta[\"category\"].isin(herb22_toxics+herb22_nontoxics)]\nherb_meta = herb_meta.reset_index().drop(labels=\"index\",axis=1)\n\n# Make 'path column the full path'\nfull_path = [\"../input/herbarium-2022-fgvc9/train_images/\"+p for p in herb_meta['path']]\nherb_meta['path'] = full_path\n\nherb_meta","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:19:05.782039Z","iopub.execute_input":"2022-08-31T02:19:05.782462Z","iopub.status.idle":"2022-08-31T02:19:05.944388Z","shell.execute_reply.started":"2022-08-31T02:19:05.782422Z","shell.execute_reply":"2022-08-31T02:19:05.943127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# some useful dictionaries\nherb_to_tc = pd.Series(tpc_meta[\"tox_class\"].values, index=tpc_meta[\"herbarium22_category_id\"]).to_dict()\ntc_to_slang = pd.Series(tpc_meta[\"slang\"].values, index=tpc_meta[\"tox_class\"]).to_dict()\ntc_to_sciname = pd.Series(tpc_meta[\"scientific_name\"].values, index=tpc_meta[\"tox_class\"]).to_dict()","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:12:48.763714Z","iopub.execute_input":"2022-08-31T02:12:48.765016Z","iopub.status.idle":"2022-08-31T02:12:48.781522Z","shell.execute_reply.started":"2022-08-31T02:12:48.764954Z","shell.execute_reply":"2022-08-31T02:12:48.778839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"herb_meta[\"herbarium22_category_id\"] = herb_meta[\"category\"]\nherb_meta[\"tox_class\"] = herb_meta[\"herbarium22_category_id\"].map(herb_to_tc)\nherb_meta[\"slang\"] = herb_meta[\"tox_class\"].map(tc_to_slang)\nherb_meta[\"scientific_name\"] = herb_meta[\"tox_class\"].map(tc_to_sciname)\nherb_meta[['toxicity','class_id']] = herb_meta['tox_class'].str.split(\"-\", expand=True)\n# Keep only the desired columns\nherb_meta = herb_meta.loc[:, [\"class_id\", \"slang\", \"scientific_name\", \"herbarium22_category_id\", \"path\", \"toxicity\"]]","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:19:49.556195Z","iopub.execute_input":"2022-08-31T02:19:49.556656Z","iopub.status.idle":"2022-08-31T02:19:49.632651Z","shell.execute_reply.started":"2022-08-31T02:19:49.556616Z","shell.execute_reply":"2022-08-31T02:19:49.631512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"herb_meta['class_id'] = pd.to_numeric(herb_meta['class_id'])\nherb_meta['toxicity'] = pd.to_numeric(herb_meta['toxicity'])","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:22:22.825121Z","iopub.execute_input":"2022-08-31T02:22:22.825545Z","iopub.status.idle":"2022-08-31T02:22:22.83362Z","shell.execute_reply.started":"2022-08-31T02:22:22.825513Z","shell.execute_reply":"2022-08-31T02:22:22.832745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(herb_meta.shape)\nprint(full_tpcmeta.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:22:24.533686Z","iopub.execute_input":"2022-08-31T02:22:24.534103Z","iopub.status.idle":"2022-08-31T02:22:24.539604Z","shell.execute_reply.started":"2022-08-31T02:22:24.534055Z","shell.execute_reply":"2022-08-31T02:22:24.538676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Merge the TPC and Herbarium Meta Data\nNow merge the two meta-dataframes. We now have a row for every image which will allow us to flow from the two different directories. ","metadata":{}},{"cell_type":"code","source":"# Merge the metadata, drop unwanted column created by merge, reset index\nfull_meta = pd.concat([full_tpcmeta, herb_meta])\nfull_meta = full_meta.reset_index().drop(\"index\",axis=1)\nfull_meta","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:22:34.417959Z","iopub.execute_input":"2022-08-31T02:22:34.418401Z","iopub.status.idle":"2022-08-31T02:22:34.443921Z","shell.execute_reply.started":"2022-08-31T02:22:34.418364Z","shell.execute_reply":"2022-08-31T02:22:34.442763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see below that this procedure adds just about 70 images per category.","metadata":{}},{"cell_type":"code","source":"# Image counts per class\nfor tox in [0, 1]:\n    if tox == 0: print(\"NONTOXIC\")\n    else: print(\"TOXIC\")\n    for i in range(5):\n        print(f\"Class: {i} - Images: {len(full_meta[(full_meta.class_id == i) & (full_meta.toxicity==tox)])}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:22:38.260487Z","iopub.execute_input":"2022-08-31T02:22:38.261664Z","iopub.status.idle":"2022-08-31T02:22:38.277049Z","shell.execute_reply.started":"2022-08-31T02:22:38.26162Z","shell.execute_reply":"2022-08-31T02:22:38.275573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_meta.to_csv(\"full_meta_basic.csv\", index=False) ##save to csv","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:23:14.356555Z","iopub.execute_input":"2022-08-31T02:23:14.357024Z","iopub.status.idle":"2022-08-31T02:23:14.397654Z","shell.execute_reply.started":"2022-08-31T02:23:14.356982Z","shell.execute_reply":"2022-08-31T02:23:14.396155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# Further Data Manipulation","metadata":{}},{"cell_type":"markdown","source":"## Adjusting and Adding Categories\n### Collapsing classes into major species\nAdd a column which specifies whether a plant is poison oak, ivy, sumac, or nontoxic, shrinking the actual number of classes to predict and leaving more images per class for training. ","metadata":{}},{"cell_type":"code","source":"meta = full_tpcmeta ##if including herbarium photos, I would use the 'full_meta' df created above\n\n# Collapse species into 4 categories (poison oak, ivy, sumac, and nontoxic plants)\nspecieslabel_to_slang = {0:\"poison-oak\", 1:\"poison-ivy\", 2:\"poison-sumac\", 3:\"nontoxic\"}\n\nmeta[\"species_label\"] = int(0) ##Poison Oak\nmeta.loc[((meta.class_id==2) | (meta.class_id==3)) & (meta.toxicity==1), \n         \"species_label\"] = int(1) ##Poison Ivy\nmeta.loc[(meta.class_id==4) & (meta.toxicity==1), \"species_label\"] = int(2) ##Poison Sumac\nmeta.loc[(meta.toxicity==0), \"species_label\"] = int(3) ##Nontoxic\nmeta","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:26:17.833799Z","iopub.execute_input":"2022-08-31T02:26:17.834218Z","iopub.status.idle":"2022-08-31T02:26:17.862706Z","shell.execute_reply.started":"2022-08-31T02:26:17.834183Z","shell.execute_reply":"2022-08-31T02:26:17.861341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta.to_csv(\"full_tpc_meta_noherb.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:28:43.422543Z","iopub.execute_input":"2022-08-31T02:28:43.423571Z","iopub.status.idle":"2022-08-31T02:28:43.454128Z","shell.execute_reply.started":"2022-08-31T02:28:43.423525Z","shell.execute_reply":"2022-08-31T02:28:43.453004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Binary toxicity label\nWe could also train a model to simply predict if an image is of one of the toxic plants. We've already added a toxicity column which could be used as the label for such a task.","metadata":{}},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# TFRecords Dataset\nTo train neural networks on TPUs (using TensorFlow), need a TFRecords Dataset.","metadata":{}},{"cell_type":"code","source":"# Converting the values into features - helper functions\n## _int64 is used for the numeric label values\ndef _int64_feature(value):\n    return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))\n\n## _bytes is used for string/char values\ndef _bytes_feature(value):\n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:27:20.051919Z","iopub.execute_input":"2022-08-31T02:27:20.05238Z","iopub.status.idle":"2022-08-31T02:27:20.058667Z","shell.execute_reply.started":"2022-08-31T02:27:20.052337Z","shell.execute_reply":"2022-08-31T02:27:20.057281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# An example of what this process will conver the data into:\nidx=12\nimage_resize = (16, 16)\n\npath = meta.loc[idx, 'path'] ##select the filepath of this image\nspecies_label = meta.loc[idx, 'species_label'] ##select its species label\ntoxic_label = meta.loc[idx, 'toxicity']  ##select its toxicity label\nimg = Image.open(path)  ##open the image\nimg = np.array(img.resize(image_resize))  ##resize and convert to array\nfeature = { 'species_label': _int64_feature(species_label),  ##add the data to a dictionary\n           'toxic_label' : _int64_feature(toxic_label),\n          'image': _bytes_feature(img.tobytes()) }\n\nex = tf.train.Example(features=tf.train.Features(feature=feature))  ##convert into a tf.train.Example\nex","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:27:22.875361Z","iopub.execute_input":"2022-08-31T02:27:22.87574Z","iopub.status.idle":"2022-08-31T02:27:22.924259Z","shell.execute_reply.started":"2022-08-31T02:27:22.875709Z","shell.execute_reply":"2022-08-31T02:27:22.922998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First I want to split the images into training and testing, and shuffle their order\ntrain_test_split = 0.80\n## Stratify the split by class\ntrain_meta = meta.groupby(\"species_label\", group_keys=False).apply(\n    lambda x: x.sample(frac=train_test_split, random_state=r_seed)\n)\ntest_meta = meta.loc[meta.index.difference(train_meta.index), :]\n\n## Shuffle rows, reindex\ntrain_meta = train_meta.sample(frac=1).reset_index()\ntest_meta = test_meta.sample(frac=1).reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:27:36.918192Z","iopub.execute_input":"2022-08-31T02:27:36.919679Z","iopub.status.idle":"2022-08-31T02:27:36.942457Z","shell.execute_reply.started":"2022-08-31T02:27:36.919624Z","shell.execute_reply":"2022-08-31T02:27:36.941383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"train_meta images - \", len(train_meta))\nprint(\"test_meta images - \", len(test_meta))","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:27:38.943633Z","iopub.execute_input":"2022-08-31T02:27:38.94406Z","iopub.status.idle":"2022-08-31T02:27:38.949981Z","shell.execute_reply.started":"2022-08-31T02:27:38.944023Z","shell.execute_reply":"2022-08-31T02:27:38.948789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def write_to_tfrecord(tfrecord_filename, meta, image_resize):\n    \"\"\"\n    Reads in each image from the metadata, resizes, writes to tfrecord_filename - image by image\n    \"\"\"\n    # Initialize tfrecord writer\n    writer = tf.io.TFRecordWriter(tfrecord_filename)\n\n    # Iterate through each image in meta and write to the tfrecords file\n    for idx in range(len(meta)):\n        path = meta.loc[idx, 'path']\n        species_label = meta.loc[idx, 'species_label']\n        toxic_label = meta.loc[idx, 'toxicity']\n        img = Image.open(path)\n        img = np.array(img.resize((320,320)))\n        feature = { 'species_label': _int64_feature(species_label),\n                   'toxic_label' : _int64_feature(toxic_label),\n                  'image': _bytes_feature(img.tobytes()) }\n        # Create an example protocol buffer\n        example = tf.train.Example(features=tf.train.Features(feature=feature))\n        # Writing the serialized example.\n        writer.write(example.SerializeToString())\n        \n        if idx == int(len(meta)*0.25): print(\"25% done\")\n        if idx == int(len(meta)*0.5): print(\"50% done\")\n        if idx == int(len(meta)*0.75): print(\"75% done\")\n\n    writer.close()","metadata":{"execution":{"iopub.status.busy":"2022-08-31T02:27:48.686486Z","iopub.execute_input":"2022-08-31T02:27:48.687831Z","iopub.status.idle":"2022-08-31T02:27:48.700044Z","shell.execute_reply.started":"2022-08-31T02:27:48.687786Z","shell.execute_reply":"2022-08-31T02:27:48.69836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if os.path.isfile('./tpc-herb22_train.tfrecords'): ##make sure to remove if file already exists\n    os.remove('./tpc-herb22_train.tfrecords')\n    \nwrite_to_tfrecord('tpc-herb22_train.tfrecords', train_meta, (320, 320))","metadata":{"execution":{"iopub.status.busy":"2022-08-18T21:07:36.135953Z","iopub.execute_input":"2022-08-18T21:07:36.136293Z","iopub.status.idle":"2022-08-18T21:08:52.950326Z","shell.execute_reply.started":"2022-08-18T21:07:36.136268Z","shell.execute_reply":"2022-08-18T21:08:52.949472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if os.path.isfile('./tpc-herb22_test.tfrecords'): ##make sure to remove if file already exists\n    os.remove('./tpc-herb22_test.tfrecords')\n    \nwrite_to_tfrecord('tpc-herb22_test.tfrecords', test_meta, (320, 320))","metadata":{"execution":{"iopub.status.busy":"2022-08-18T21:10:47.763189Z","iopub.execute_input":"2022-08-18T21:10:47.763541Z","iopub.status.idle":"2022-08-18T21:11:04.783781Z","shell.execute_reply.started":"2022-08-18T21:10:47.763517Z","shell.execute_reply":"2022-08-18T21:11:04.782838Z"},"trusted":true},"execution_count":null,"outputs":[]}]}