{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport json\nimport os\nimport random\nimport matplotlib.pyplot as plt\n#from tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport tensorflow as tf\nfrom tensorflow.keras import layers \nfrom sklearn.preprocessing import LabelEncoder\nfrom tensorflow import keras\nBATCH=128","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-03T09:09:29.731443Z","iopub.execute_input":"2022-08-03T09:09:29.731839Z","iopub.status.idle":"2022-08-03T09:09:36.223202Z","shell.execute_reply.started":"2022-08-03T09:09:29.731808Z","shell.execute_reply":"2022-08-03T09:09:36.222272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DIR = \"../input/herbarium-2022-fgvc9/train_images/\"\nTEST_DIR = \"../input/herbarium-2022-fgvc9/test_images/\"\n\nwith open(\"../input/herbarium-2022-fgvc9/train_metadata.json\") as json_file:\n    train_meta = json.load(json_file)\nwith open(\"../input/herbarium-2022-fgvc9/test_metadata.json\") as json_file:\n    test_meta = json.load(json_file)\n#Create a meta-data df that can be used to call in images\nids = []\ncategories = []\npaths = []\n\nfor annotation, image in zip(train_meta['annotations'], train_meta['images']):\n    ids.append(image[\"image_id\"])\n    categories.append(annotation['category_id'])\n    paths.append(image[\"file_name\"])\n\ndf_meta = pd.DataFrame({\"id\":ids, \"category\":categories, \"path\":paths})\ndf_meta.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:09:36.225075Z","iopub.execute_input":"2022-08-03T09:09:36.225678Z","iopub.status.idle":"2022-08-03T09:09:53.404652Z","shell.execute_reply.started":"2022-08-03T09:09:36.225643Z","shell.execute_reply":"2022-08-03T09:09:53.403287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sci_name = {cat[\"category_id\"]:cat[\"scientificName\"] for cat in train_meta['categories']}\nfamily = {cat[\"category_id\"]:cat[\"family\"] for cat in train_meta['categories']}\ngenus = {cat[\"category_id\"]:cat[\"genus\"] for cat in train_meta['categories']}\nspecies = {cat[\"category_id\"]:cat[\"species\"] for cat in train_meta['categories']}\n\ndf_meta[\"scientific_name\"] = df_meta[\"category\"].map(sci_name)\ndf_meta[\"family\"] = df_meta[\"category\"].map(family)\ndf_meta[\"genus\"] = df_meta[\"category\"].map(genus)\ndf_meta[\"species\"] = df_meta[\"category\"].map(species)\npretty_df=df_meta.copy()\ndf_meta.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:09:53.406039Z","iopub.execute_input":"2022-08-03T09:09:53.406404Z","iopub.status.idle":"2022-08-03T09:09:54.27211Z","shell.execute_reply.started":"2022-08-03T09:09:53.406371Z","shell.execute_reply":"2022-08-03T09:09:54.271039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_image(index):\n    path=os.path.join(\"../input/herbarium-2022-fgvc9/train_images\",df_meta[\"path\"][index])\n    print(df_meta[\"scientific_name\"][index])\n    plt.imshow(plt.imread(path)/255)\nshow_image(20)\nLABEL=len(train_meta['categories'])\nLABEL","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:09:54.275824Z","iopub.execute_input":"2022-08-03T09:09:54.276364Z","iopub.status.idle":"2022-08-03T09:09:54.77704Z","shell.execute_reply.started":"2022-08-03T09:09:54.276281Z","shell.execute_reply":"2022-08-03T09:09:54.775757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"family_id=LabelEncoder()\ngenus_id=LabelEncoder()\nfamily_id.fit([x [\"family\"] for x in train_meta[\"categories\"]])\ngenus_id.fit([x [\"genus\"] for x in train_meta[\"categories\"]])\n#adding sublabels to make this job easier\ndata_df=df_meta.drop(columns=[\"species\",\"scientific_name\"])\ndata_df[\"family\"]=family_id.transform(data_df[\"family\"])\ndata_df[\"genus\"]=genus_id.transform(data_df[\"genus\"])\ndata_df","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:09:54.77872Z","iopub.execute_input":"2022-08-03T09:09:54.779809Z","iopub.status.idle":"2022-08-03T09:09:55.687152Z","shell.execute_reply.started":"2022-08-03T09:09:54.779771Z","shell.execute_reply":"2022-08-03T09:09:55.685895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FAM=len(family_id.classes_)\nGEN=len(genus_id.classes_)\nOLD_LABEL=LABEL\nmini=train_meta[\"categories\"]\nmini=[{\"category\":a[\"category_id\"],\"family\":family_id.transform([a[\"family\"]])[0],\"genus\":genus_id.transform([a[\"genus\"]])[0]} for a in mini] \ngenus_family={}\ncategory_genus=[]\nfor a in mini:\n    genus_family.update({a[\"genus\"]:a[\"family\"]})\n    category_genus.append([a[\"category\"],a[\"genus\"]]) \ngenus_family=[[k,v] for k,v in genus_family.items()]\ngenus_family=tf.sparse.SparseTensor(genus_family,[1 for _ in range(len(genus_family))],(GEN,FAM))\ndebug=category_genus\ntoo_big={a[0]:a[1] for a in debug if a[0]>=OLD_LABEL} \nLABEL=max([k  for k in too_big.keys()])+1\ncategory_genus=tf.sparse.SparseTensor(category_genus,[1 for _ in range(len(category_genus))],(LABEL,GEN))\nprint(genus_family)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:09:55.688602Z","iopub.execute_input":"2022-08-03T09:09:55.689138Z","iopub.status.idle":"2022-08-03T09:10:51.000129Z","shell.execute_reply.started":"2022-08-03T09:09:55.689095Z","shell.execute_reply":"2022-08-03T09:10:50.998382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pretty_df","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:10:51.001994Z","iopub.execute_input":"2022-08-03T09:10:51.002363Z","iopub.status.idle":"2022-08-03T09:10:51.027808Z","shell.execute_reply.started":"2022-08-03T09:10:51.002333Z","shell.execute_reply":"2022-08-03T09:10:51.026347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"family_df=pretty_df[\"family\"].value_counts()\nfamily_df=pd.DataFrame(family_df)\n\ngenus_df=pretty_df[\"genus\"].value_counts() \ngenus_df=pd.DataFrame(genus_df)\n\nname_df=pretty_df[\"scientific_name\"].value_counts() \nname_df=pd.DataFrame(name_df)\n\nname_df","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:10:51.029711Z","iopub.execute_input":"2022-08-03T09:10:51.030072Z","iopub.status.idle":"2022-08-03T09:10:51.197435Z","shell.execute_reply.started":"2022-08-03T09:10:51.03004Z","shell.execute_reply":"2022-08-03T09:10:51.19656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f=pd.DataFrame() \nf[\"count\"]=family_df[\"family\"]\nf[\"type\"]=\"family\" \n\ng=pd.DataFrame() \ng[\"count\"]=genus_df[\"genus\"]\ng[\"type\"]=\"genus\" \n\nn=pd.DataFrame() \nn[\"count\"]=name_df[\"scientific_name\"]\nn[\"type\"]=\"name\" \n\nbig_df=pd.concat([f,g,n]).sort_values(\"count\")\nbig_df","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:12:15.485211Z","iopub.execute_input":"2022-08-03T09:12:15.485724Z","iopub.status.idle":"2022-08-03T09:12:15.514955Z","shell.execute_reply.started":"2022-08-03T09:12:15.485684Z","shell.execute_reply":"2022-08-03T09:12:15.513665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counts=big_df.value_counts() \nx=[x[0] for x in counts.index]\ny=counts.iloc[:] \n\nplt.scatter(x,np.log(y))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:12:23.876134Z","iopub.execute_input":"2022-08-03T09:12:23.876604Z","iopub.status.idle":"2022-08-03T09:12:24.087606Z","shell.execute_reply.started":"2022-08-03T09:12:23.876569Z","shell.execute_reply":"2022-08-03T09:12:24.086285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot(p,df,name,bins=0):\n    counts=df.value_counts() \n    x=[x[0] for x in counts.index]\n    \n    if(bins>0):\n        nums, bins, edges=p.hist(x,bins)\n    \n    else: \n        bins=[]\n        y=counts.iloc[:] \n        p.scatter(x,y)\n    p.set_title(name,fontsize = 20)\n    p.set_xlabel(\"member count\",fontsize = 20)\n    p.set_ylabel(\"category count\",fontsize = 20)\n    p.legend()\n    return bins\n\nplot(plt.subplot(),f,\"fam\",10)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:12:27.833059Z","iopub.execute_input":"2022-08-03T09:12:27.833928Z","iopub.status.idle":"2022-08-03T09:12:28.04253Z","shell.execute_reply.started":"2022-08-03T09:12:27.833883Z","shell.execute_reply":"2022-08-03T09:12:28.041642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def draw(bins,data=[f,g,n]):\n    data=data.copy()\n    colecting=[]\n    figure,axs=plt.subplots(1,3,figsize=(20,7)) \n     \n    for a in axs:\n        x=data.pop()\n        label=x[\"type\"].iloc[0]\n        c=plot(a,x,label,bins.pop(0))\n        colecting.append((c,label))\n    return colecting\n\ntup=draw([19,19,19]) ","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:53:08.480015Z","iopub.execute_input":"2022-08-03T09:53:08.480428Z","iopub.status.idle":"2022-08-03T09:53:09.084651Z","shell.execute_reply.started":"2022-08-03T09:53:08.480393Z","shell.execute_reply":"2022-08-03T09:53:09.083482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INTMODE=True\ndef get_bin_df(tup):\n    def get_val(array,index):\n        if(index<len(array)):\n            if(INTMODE):\n               return int(array[index])\n            return array[index]\n        return \"Nan\" \n\n    bins=[t[0] for t in tup] \n    labels=[t[1] for t in tup]\n    maxl=max([len(b) for b in bins])\n    a=pd.DataFrame(labels).reindex(labels)\n    for i in range(maxl):\n        a[i]=[get_val(b,i) for b in bins]\n    return a \nget_bin_df(tup=draw([19,100,100]))[range(20)]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:53:12.066366Z","iopub.execute_input":"2022-08-03T09:53:12.066945Z","iopub.status.idle":"2022-08-03T09:53:13.020402Z","shell.execute_reply.started":"2022-08-03T09:53:12.066913Z","shell.execute_reply":"2022-08-03T09:53:13.019349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sample(df,size):\n    return df.sample(min(size,df.size))\n    \ndef get_samples(df,size):\n    return {i:sample(df[df[\"category\"]==i],size) for i in range(LABEL)}\n\na=get_samples(pretty_df.iloc[0:100],3) \n{k:v for k,v in a.items() if v.size>0}","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:41:28.512063Z","iopub.execute_input":"2022-08-03T09:41:28.512489Z","iopub.status.idle":"2022-08-03T09:41:38.018844Z","shell.execute_reply.started":"2022-08-03T09:41:28.512453Z","shell.execute_reply":"2022-08-03T09:41:38.017872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat([v for k,v in a.items() if v.size>0])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:47:17.585749Z","iopub.execute_input":"2022-08-03T09:47:17.586168Z","iopub.status.idle":"2022-08-03T09:47:17.736452Z","shell.execute_reply.started":"2022-08-03T09:47:17.58613Z","shell.execute_reply":"2022-08-03T09:47:17.735625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lookup=get_samples(pretty_df,3)\nsmall_df=pd.concat([v for k,v in lookup.items() if v.size>0])\nsmall_df","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:48:58.83757Z","iopub.execute_input":"2022-08-03T09:48:58.838016Z","iopub.status.idle":"2022-08-03T09:49:22.676652Z","shell.execute_reply.started":"2022-08-03T09:48:58.837983Z","shell.execute_reply":"2022-08-03T09:49:22.67557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_seperated(pretty_df):\n    family_df=pretty_df[\"family\"].value_counts()\n    family_df=pd.DataFrame(family_df)\n\n    genus_df=pretty_df[\"genus\"].value_counts() \n    genus_df=pd.DataFrame(genus_df)\n\n    name_df=pretty_df[\"scientific_name\"].value_counts() \n    name_df=pd.DataFrame(name_df)\n    \n    f=pd.DataFrame() \n    f[\"count\"]=family_df[\"family\"]\n    f[\"type\"]=\"family\" \n\n    g=pd.DataFrame() \n    g[\"count\"]=genus_df[\"genus\"]\n    g[\"type\"]=\"genus\" \n\n    n=pd.DataFrame() \n    n[\"count\"]=name_df[\"scientific_name\"]\n    n[\"type\"]=\"name\"  \n    \n    return [f,g,n]\nseprated=get_seperated(small_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:57:45.796335Z","iopub.execute_input":"2022-08-03T09:57:45.796814Z","iopub.status.idle":"2022-08-03T09:57:45.832956Z","shell.execute_reply.started":"2022-08-03T09:57:45.796778Z","shell.execute_reply":"2022-08-03T09:57:45.831904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_bin_df(tup=draw([0,5,5],data=seprated))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T09:59:27.543267Z","iopub.execute_input":"2022-08-03T09:59:27.54404Z","iopub.status.idle":"2022-08-03T09:59:28.051163Z","shell.execute_reply.started":"2022-08-03T09:59:27.543999Z","shell.execute_reply":"2022-08-03T09:59:28.050068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_size=small_df.size\npretty_size=pretty_df.size \n\nsmall_size/pretty_size","metadata":{"execution":{"iopub.status.busy":"2022-08-03T10:01:55.214222Z","iopub.execute_input":"2022-08-03T10:01:55.214611Z","iopub.status.idle":"2022-08-03T10:01:55.222651Z","shell.execute_reply.started":"2022-08-03T10:01:55.214578Z","shell.execute_reply":"2022-08-03T10:01:55.221355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}