{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"\n","metadata":{}},{"cell_type":"code","source":"import os\nimport json\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:18.886342Z","iopub.execute_input":"2022-04-18T12:34:18.887155Z","iopub.status.idle":"2022-04-18T12:34:19.152898Z","shell.execute_reply.started":"2022-04-18T12:34:18.887054Z","shell.execute_reply":"2022-04-18T12:34:19.152198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading and cleaning data","metadata":{}},{"cell_type":"code","source":"#loading json file for train data\ndata_path_train = \"../input/herbarium-2022-fgvc9/train_metadata.json\"\nwith open(data_path_train) as json_file:\n    meta_train = json.load(json_file)\n    \n#loading json file for test data\ndata_path_test = \"../input/herbarium-2022-fgvc9/test_metadata.json\"\nwith open(data_path_test) as json_file:\n    meta_test = json.load(json_file)    ","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:19.154594Z","iopub.execute_input":"2022-04-18T12:34:19.154839Z","iopub.status.idle":"2022-04-18T12:34:33.058315Z","shell.execute_reply.started":"2022-04-18T12:34:19.154805Z","shell.execute_reply":"2022-04-18T12:34:33.057576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Finding keys in train dictionary\nmeta_train.keys()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:33.059591Z","iopub.execute_input":"2022-04-18T12:34:33.059882Z","iopub.status.idle":"2022-04-18T12:34:33.068001Z","shell.execute_reply.started":"2022-04-18T12:34:33.059845Z","shell.execute_reply":"2022-04-18T12:34:33.067244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#creating seperate dataframes from metadata\nannotations_train =  pd.json_normalize(meta_train ['annotations'])\ncategories_train =  pd.json_normalize(meta_train ['categories'])\nimages_train =  pd.json_normalize(meta_train ['images'])\ngenera_train =  pd.json_normalize(meta_train ['genera'])\ndistance_train =  pd.json_normalize(meta_train ['distances'])\nlicenses_train =  pd.json_normalize(meta_train ['license'])\ninstitutions_train =  pd.json_normalize(meta_train ['institutions'])","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:33.070343Z","iopub.execute_input":"2022-04-18T12:34:33.070948Z","iopub.status.idle":"2022-04-18T12:34:55.716581Z","shell.execute_reply.started":"2022-04-18T12:34:33.070912Z","shell.execute_reply":"2022-04-18T12:34:55.715843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now lets check how each dataframe looks like","metadata":{}},{"cell_type":"code","source":"annotations_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:55.717926Z","iopub.execute_input":"2022-04-18T12:34:55.718167Z","iopub.status.idle":"2022-04-18T12:34:55.733589Z","shell.execute_reply.started":"2022-04-18T12:34:55.718133Z","shell.execute_reply":"2022-04-18T12:34:55.732952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categories_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:55.73532Z","iopub.execute_input":"2022-04-18T12:34:55.7358Z","iopub.status.idle":"2022-04-18T12:34:55.746576Z","shell.execute_reply.started":"2022-04-18T12:34:55.735764Z","shell.execute_reply":"2022-04-18T12:34:55.745743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:55.748082Z","iopub.execute_input":"2022-04-18T12:34:55.748386Z","iopub.status.idle":"2022-04-18T12:34:55.760388Z","shell.execute_reply.started":"2022-04-18T12:34:55.748351Z","shell.execute_reply":"2022-04-18T12:34:55.759637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"genera_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:55.761566Z","iopub.execute_input":"2022-04-18T12:34:55.761838Z","iopub.status.idle":"2022-04-18T12:34:55.769456Z","shell.execute_reply.started":"2022-04-18T12:34:55.761807Z","shell.execute_reply":"2022-04-18T12:34:55.768583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distance_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:55.770979Z","iopub.execute_input":"2022-04-18T12:34:55.771555Z","iopub.status.idle":"2022-04-18T12:34:55.782291Z","shell.execute_reply.started":"2022-04-18T12:34:55.771519Z","shell.execute_reply":"2022-04-18T12:34:55.781489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"licenses_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:55.785415Z","iopub.execute_input":"2022-04-18T12:34:55.785717Z","iopub.status.idle":"2022-04-18T12:34:55.793556Z","shell.execute_reply.started":"2022-04-18T12:34:55.785685Z","shell.execute_reply":"2022-04-18T12:34:55.792735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"institutions_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:55.794981Z","iopub.execute_input":"2022-04-18T12:34:55.795364Z","iopub.status.idle":"2022-04-18T12:34:55.807239Z","shell.execute_reply.started":"2022-04-18T12:34:55.795331Z","shell.execute_reply":"2022-04-18T12:34:55.806652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we saw there is not much usefull information in genus, licences and institutions","metadata":{}},{"cell_type":"code","source":"#Removing unused dataframe\ndel genera_train\ndel licenses_train\ndel institutions_train","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:55.808382Z","iopub.execute_input":"2022-04-18T12:34:55.808634Z","iopub.status.idle":"2022-04-18T12:34:55.812334Z","shell.execute_reply.started":"2022-04-18T12:34:55.808602Z","shell.execute_reply":"2022-04-18T12:34:55.811503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Looking at test set\ndf_test = pd.DataFrame(meta_test)\n\n#creating test data\ndf_test = df_test.drop(['license'], axis=1)\n\n\n# adding file path\ndf_test = df_test[['image_id','file_name']]\ndf_test['file_path']=\"../input/herbarium-2022-fgvc9/test_images/\"+df_test['file_name']\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:55.814391Z","iopub.execute_input":"2022-04-18T12:34:55.814935Z","iopub.status.idle":"2022-04-18T12:34:56.078092Z","shell.execute_reply.started":"2022-04-18T12:34:55.814906Z","shell.execute_reply":"2022-04-18T12:34:56.077417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Complete df\ndf_merge = pd.merge(images_train[['image_id','file_name']],annotations_train[['genus_id','category_id','image_id']] , on='image_id')\ndf_merge = pd.merge(df_merge[['genus_id','image_id','file_name','category_id']],categories_train[['category_id','scientificName','family','genus','species']] , on='category_id')\ndf_merge['file_path']=\"../input/herbarium-2022-fgvc9/train_images/\"+df_merge['file_name']\ndf_merge['name']=df_merge['genus']+' '+df_merge['species']\ndf_train = df_merge[['category_id','genus_id','image_id','family','genus','species','name','file_name','file_path']]\n\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:56.079356Z","iopub.execute_input":"2022-04-18T12:34:56.07977Z","iopub.status.idle":"2022-04-18T12:34:57.577312Z","shell.execute_reply.started":"2022-04-18T12:34:56.079733Z","shell.execute_reply":"2022-04-18T12:34:57.576505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del annotations_train\ndel categories_train\ndel images_train\n","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:57.578662Z","iopub.execute_input":"2022-04-18T12:34:57.578986Z","iopub.status.idle":"2022-04-18T12:34:57.583492Z","shell.execute_reply.started":"2022-04-18T12:34:57.578949Z","shell.execute_reply":"2022-04-18T12:34:57.582625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#removing null values\ndf_train = df_train.dropna(how = 'all')\n#cheking for missing data \ndf_train.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:57.584972Z","iopub.execute_input":"2022-04-18T12:34:57.585362Z","iopub.status.idle":"2022-04-18T12:34:58.757818Z","shell.execute_reply.started":"2022-04-18T12:34:57.585327Z","shell.execute_reply":"2022-04-18T12:34:58.756984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking for duplicates\ndf_train['file_name'].duplicated().any()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:58.759345Z","iopub.execute_input":"2022-04-18T12:34:58.759621Z","iopub.status.idle":"2022-04-18T12:34:58.893626Z","shell.execute_reply.started":"2022-04-18T12:34:58.759587Z","shell.execute_reply":"2022-04-18T12:34:58.892772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"print ('number of images in train set')\nlen(df_train['image_id'])","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:58.895575Z","iopub.execute_input":"2022-04-18T12:34:58.896035Z","iopub.status.idle":"2022-04-18T12:34:58.902987Z","shell.execute_reply.started":"2022-04-18T12:34:58.895995Z","shell.execute_reply":"2022-04-18T12:34:58.902234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print ('number of images in test set')\nlen(df_test['image_id'])","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:58.90437Z","iopub.execute_input":"2022-04-18T12:34:58.904847Z","iopub.status.idle":"2022-04-18T12:34:58.91214Z","shell.execute_reply.started":"2022-04-18T12:34:58.90481Z","shell.execute_reply":"2022-04-18T12:34:58.911288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print ('number of specific types of plants')\nlen(df_train['category_id'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:58.913484Z","iopub.execute_input":"2022-04-18T12:34:58.913932Z","iopub.status.idle":"2022-04-18T12:34:58.927284Z","shell.execute_reply.started":"2022-04-18T12:34:58.913898Z","shell.execute_reply":"2022-04-18T12:34:58.926638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print ('number of familes')\nlen(df_train['family'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:58.928473Z","iopub.execute_input":"2022-04-18T12:34:58.928864Z","iopub.status.idle":"2022-04-18T12:34:58.989572Z","shell.execute_reply.started":"2022-04-18T12:34:58.92883Z","shell.execute_reply":"2022-04-18T12:34:58.988825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print ('number of genus')\nlen(df_train['genus'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:58.990812Z","iopub.execute_input":"2022-04-18T12:34:58.991087Z","iopub.status.idle":"2022-04-18T12:34:59.051027Z","shell.execute_reply.started":"2022-04-18T12:34:58.991054Z","shell.execute_reply":"2022-04-18T12:34:59.050397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print ('number of species')\nlen(df_train['species'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:59.05223Z","iopub.execute_input":"2022-04-18T12:34:59.052663Z","iopub.status.idle":"2022-04-18T12:34:59.114621Z","shell.execute_reply.started":"2022-04-18T12:34:59.05263Z","shell.execute_reply":"2022-04-18T12:34:59.113852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We see that there are 839772 images in the dataset which can be divided into 15501 specific plants, that can be identified by category id. There are 210407 images in test dataset. The name column gives the name of the plant as species+genus. These plants can be further classified into 6932 species, 2564 genus and 272 familes. Lets create a table for this.","metadata":{}},{"cell_type":"code","source":"#finding name of families\nn = df_train['family'].unique().tolist()\nx = df_train['genus'].unique().tolist()\n\n\n#finding number of genus in each family\ng_f_n=[]\nfor i in range(len(n)):    \n    g_f_n.append(len(df_train.loc[df_train['family']==n[i],'genus' ].unique()))\n\n#finding number of species in each family\ns_f_n=[]\nfor i in range(len(n)):    \n    s_f_n.append(len(df_train.loc[df_train['family']==n[i],'species' ].unique()))\n    \n#finding number of species in each genus\ns_g_n=[]\nfor i in range(len(x)):    \n    s_g_n.append(len(df_train.loc[df_train['genus']==x[i],'species' ].unique()))  \n    \n    \n#finding number of images in each family\no_i_n=[]\nfor i in range(len(n)):\n    o_i_n.append(len(df_train.loc[df_train['family']==n[i]]))","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:34:59.115872Z","iopub.execute_input":"2022-04-18T12:34:59.116169Z","iopub.status.idle":"2022-04-18T12:40:52.281688Z","shell.execute_reply.started":"2022-04-18T12:34:59.116134Z","shell.execute_reply":"2022-04-18T12:40:52.280952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"table_order =pd.DataFrame(df_train['family'].unique(),columns =['family'])\ntable_order['Number_of_genus_in_family'] = g_f_n\ntable_order['Number_of_species_in_family'] = s_f_n\ntable_order['Number_of_images_in_family'] = o_i_n\n\ntable_order = table_order.sort_values(by=['Number_of_images_in_family'], ascending=False, ignore_index=True)\n\nprint(table_order.to_markdown())","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:41:27.346586Z","iopub.execute_input":"2022-04-18T12:41:27.347105Z","iopub.status.idle":"2022-04-18T12:41:27.457616Z","shell.execute_reply.started":"2022-04-18T12:41:27.347066Z","shell.execute_reply":"2022-04-18T12:41:27.456892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#creating table to classify by sample size\nList = [10,100,150,200,250,300,350,400,450,500,550,600,650,700,750,800,850,900,950,1000]\nsample_size = pd.DataFrame({'Sample_size':['more than 10','more than 100','more than 150','more than 200',\n                                           'more than 250','more than 300','more than 350','more than 400',\n                                           'more than 450','more than 500','more than 550','more than 600',\n                                           'more than 650','more than 700','more than 750','more than 800',\n                                           'more than 850','more than 900','more than 950','more than 1000']})\n\n#finding number of plant having particular sample size\np_n=[]\nfor i in List:\n    more= df_train['category_id'].value_counts() > i\n    p_n.append(len(more.index[more==True]))  \n\n\n#finding number of species having particular sample size\ns_n=[]\nfor i in List:\n    more= df_train['species'].value_counts() > i\n    s_n.append(len(more.index[more==True]))  \n\n#finding number of families having particular sample size\nf_n=[]\nfor i in List:\n    more= df_train['family'].value_counts() > i\n    f_n.append(len(more.index[more==True]))\n    \n#finding number of genus having particular sample size\ng_n=[]\nfor i in List:\n    more= df_train['genus'].value_counts() > i\n    g_n.append(len(more.index[more==True]))\n\n    \n\nsample_size['Number_of_plants'] = p_n\nsample_size['Number_of_families'] = f_n\nsample_size['Number_of_genus'] = g_n\nsample_size['Number_of_speciess'] = s_n\n\nprint(sample_size.to_markdown())","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:42:03.705758Z","iopub.execute_input":"2022-04-18T12:42:03.706061Z","iopub.status.idle":"2022-04-18T12:42:10.798838Z","shell.execute_reply.started":"2022-04-18T12:42:03.706029Z","shell.execute_reply":"2022-04-18T12:42:10.798147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#creating table to classify by sample size\nList = [10,20,30,40,50,60,70,80,90,100]\nsample_size = pd.DataFrame({'Sample_size':['less than 10','less than 20','less than 30','less than 40',\n                                           'less than 50','less than 60','less than 70','less than 80',\n                                           'less than 90','less than 100']})\n\n#finding number of plant having particular sample size\np_n=[]\nfor i in List:\n    less= df_train['category_id'].value_counts() < i\n    p_n.append(len(less.index[less==True]))  \n                                           \nsample_size['Number_of_images_per plant_categories'] = p_n\nprint(sample_size.to_markdown())","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:47:18.10788Z","iopub.execute_input":"2022-04-18T12:47:18.10815Z","iopub.status.idle":"2022-04-18T12:47:18.175618Z","shell.execute_reply.started":"2022-04-18T12:47:18.10812Z","shell.execute_reply":"2022-04-18T12:47:18.174858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Maximum number of samples available for a category')\nmax(df_train['category_id'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:51:36.41606Z","iopub.execute_input":"2022-04-18T12:51:36.41631Z","iopub.status.idle":"2022-04-18T12:51:36.431865Z","shell.execute_reply.started":"2022-04-18T12:51:36.416281Z","shell.execute_reply":"2022-04-18T12:51:36.431086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Minimum number of samples available for a category')\nmin(df_train['category_id'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:51:56.188548Z","iopub.execute_input":"2022-04-18T12:51:56.188954Z","iopub.status.idle":"2022-04-18T12:51:56.214555Z","shell.execute_reply.started":"2022-04-18T12:51:56.188917Z","shell.execute_reply":"2022-04-18T12:51:56.213915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Phylogenetic Distances Among Genera","metadata":{}},{"cell_type":"markdown","source":"There is also a set of pairwise phylogenetic distances among genera to test if the difference in morphological features of plant taxa well correspond to their taxonomic distances\n","metadata":{}},{"cell_type":"code","source":"distance_train.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize","metadata":{}},{"cell_type":"code","source":"#Number of samples in each famiy\nplt.figure(figsize=(25, 10))\ndf_train['family'].value_counts().plot.bar()\nplt.title(f\"value count in each family\", fontsize=10)","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:48:53.426834Z","iopub.execute_input":"2022-04-18T12:48:53.427086Z","iopub.status.idle":"2022-04-18T12:48:57.288547Z","shell.execute_reply.started":"2022-04-18T12:48:53.427058Z","shell.execute_reply":"2022-04-18T12:48:57.287879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting image by image id for single image\ndef  visualize(image_id):\n    \n    path = df_train.loc[df_train['image_id'] == image_id, 'file_path'].iloc[0]\n    family = df_train.loc[df_train['image_id'] == image_id, 'family'].iloc[0]\n    genus = df_train.loc[df_train['image_id'] == image_id, 'genus'].iloc[0]\n    species = df_train.loc[df_train['image_id'] == image_id, 'species'].iloc[0]\n    name = df_train.loc[df_train['image_id'] == image_id, 'name'].iloc[0]\n    plt.figure(figsize=(10, 10))\n    image = cv2.imread(path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(image)\n    plt.title(f\"FAMILY: {family} GENUS: {genus} SPECIES: {species}\\n NAME:{name}\\n Image_id:{image_id}\", fontsize=10)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:52:04.516823Z","iopub.execute_input":"2022-04-18T12:52:04.51727Z","iopub.status.idle":"2022-04-18T12:52:04.525114Z","shell.execute_reply.started":"2022-04-18T12:52:04.517235Z","shell.execute_reply":"2022-04-18T12:52:04.524143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize('00000__002')","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:52:05.16778Z","iopub.execute_input":"2022-04-18T12:52:05.168398Z","iopub.status.idle":"2022-04-18T12:52:06.431202Z","shell.execute_reply.started":"2022-04-18T12:52:05.168358Z","shell.execute_reply":"2022-04-18T12:52:06.430329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting image by image id for multiple image. When an image id is given, function returns 20 images \n#in same class\ndef visualize_many(image_id):\n    train_image_path = \"../input/herbarium-2022-fgvc9/train_images/\"\n    \n    category = df_train.loc[df_train['image_id'] == image_id, 'category_id'].iloc[0]\n    df = df_train.loc[df_train['category_id'] == category]\n    \n    \n                                                           \n    if  df['image_id'].count()< 20:\n        x=df['image_id'].tolist()\n        plt.figure(figsize=(18, 18))\n        for i, j in zip(x, range(20)):       \n            plt.subplot(5, 4, j + 1)\n            path = df.loc[df['image_id'] == i, 'file_name'].iloc[0]\n            image = cv2.imread(os.path.join(train_image_path, path))\n            image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n            plt.imshow(image)\n            family = df.loc[df['image_id'] == i, 'family'].iloc[0]\n            genus = df.loc[df['image_id'] == i, 'genus'].iloc[0]\n            species = df.loc[df['image_id'] == i, 'species'].iloc[0]\n            name = df.loc[df['image_id'] == i, 'name'].iloc[0]\n            imageid= df.loc[df['image_id'] == i, 'image_id'].iloc[0]\n            plt.title(f\"FAMILY: {family} GENUS: {genus} SPECIES: {species}\\n NAME:{name}\\n Image_id:{imageid}\", fontsize=10)\n            plt.axis(\"off\")\n        plt.show()\n    else:\n        x = np.random.choice(df['image_id'], 20, replace=False).tolist()\n        plt.figure(figsize=(18, 18))\n        for i, j in zip(x, range(20)):       \n            plt.subplot(5, 4, j + 1)\n            path = df.loc[df['image_id'] == i, 'file_name'].iloc[0]\n            image = cv2.imread(os.path.join(train_image_path, path))\n            image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n            plt.imshow(image)\n            family = df.loc[df['image_id'] == i, 'family'].iloc[0]\n            genus = df.loc[df['image_id'] == i, 'genus'].iloc[0]\n            species = df.loc[df['image_id'] == i, 'species'].iloc[0]\n            name = df.loc[df['image_id'] == i, 'name'].iloc[0]\n            imageid= df.loc[df['image_id'] == i, 'image_id'].iloc[0]\n            plt.title(f\"FAMILY: {family} GENUS: {genus} SPECIES: {species}\\n NAME:{name}\\n Image_id:{imageid}\", fontsize=10)\n            plt.axis(\"off\")\n    \n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:52:10.099908Z","iopub.execute_input":"2022-04-18T12:52:10.100156Z","iopub.status.idle":"2022-04-18T12:52:10.118193Z","shell.execute_reply.started":"2022-04-18T12:52:10.100127Z","shell.execute_reply":"2022-04-18T12:52:10.117484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_many('00000__003')","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:52:11.547692Z","iopub.execute_input":"2022-04-18T12:52:11.548284Z","iopub.status.idle":"2022-04-18T12:52:14.278026Z","shell.execute_reply.started":"2022-04-18T12:52:11.54825Z","shell.execute_reply":"2022-04-18T12:52:14.277448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot image by family name\ndef visualize_family(name):\n    train_image_path = \"../input/herbarium-2022-fgvc9/train_images/\"\n     \n    df = df_train.loc[df_train['family'] == name]\n    \n    x = np.random.choice(df['image_id'], 15, replace=False).tolist()\n    plt.figure(figsize=(18, 18))\n    for i, j in zip(x, range(15)):       \n            plt.subplot(3, 5, j + 1)\n            path = df.loc[df['image_id'] == i, 'file_name'].iloc[0]\n            image = cv2.imread(os.path.join(train_image_path, path))\n            image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n            plt.imshow(image)\n            family = df.loc[df['image_id'] == i, 'family'].iloc[0]\n            genus = df.loc[df['image_id'] == i, 'genus'].iloc[0]\n            species = df.loc[df['image_id'] == i, 'species'].iloc[0]\n            name = df.loc[df['image_id'] == i, 'name'].iloc[0]\n            imageid= df.loc[df['image_id'] == i, 'image_id'].iloc[0]\n            plt.title(f\"FAMILY: {family} GENUS: {genus} SPECIES: {species}\\n NAME:{name}\\n Image_id:{imageid}\", fontsize=10)\n            plt.axis(\"off\")\n            #plt.savefig('saved_figure.png')\n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:52:28.218782Z","iopub.execute_input":"2022-04-18T12:52:28.219374Z","iopub.status.idle":"2022-04-18T12:52:28.228727Z","shell.execute_reply.started":"2022-04-18T12:52:28.219336Z","shell.execute_reply":"2022-04-18T12:52:28.228008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_family('Brassicaceae')","metadata":{"execution":{"iopub.status.busy":"2022-04-18T12:52:30.478116Z","iopub.execute_input":"2022-04-18T12:52:30.478377Z","iopub.status.idle":"2022-04-18T12:52:33.340129Z","shell.execute_reply.started":"2022-04-18T12:52:30.478347Z","shell.execute_reply":"2022-04-18T12:52:33.339418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Images have different background colours, frames, and other things anlong with plant species.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}