{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# overview\nSee input data and output format ","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if filename.endswith('.jpg'):\n            break\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-27T07:50:35.563217Z","iopub.execute_input":"2021-10-27T07:50:35.563688Z","iopub.status.idle":"2021-10-27T07:55:16.624825Z","shell.execute_reply.started":"2021-10-27T07:50:35.563617Z","shell.execute_reply":"2021-10-27T07:55:16.623932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since there's a lot of images included there, we only checked non-image files and got the three above. Next, we will load the sample submission and check.","metadata":{}},{"cell_type":"code","source":"df_sample = pd.read_csv('../input/herbarium-2020-fgvc7/sample_submission.csv')\ndisplay(df_sample)","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2021-10-27T07:55:16.627928Z","iopub.execute_input":"2021-10-27T07:55:16.628787Z","iopub.status.idle":"2021-10-27T07:55:16.701967Z","shell.execute_reply.started":"2021-10-27T07:55:16.628723Z","shell.execute_reply":"2021-10-27T07:55:16.700662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Load json files","metadata":{}},{"cell_type":"code","source":"import json\nwith open(\"../input/herbarium-2020-fgvc7/nybg2020/train/metadata.json\", 'r',\n                 encoding='utf-8', errors='ignore') as f:\n    train_meta = json.load(f)\n    \nwith open(\"../input/herbarium-2020-fgvc7/nybg2020/test/metadata.json\", 'r',\n                 encoding='utf-8', errors='ignore') as f:\n    test_meta = json.load(f)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:55:16.703780Z","iopub.execute_input":"2021-10-27T07:55:16.704566Z","iopub.status.idle":"2021-10-27T07:55:23.769910Z","shell.execute_reply.started":"2021-10-27T07:55:16.704502Z","shell.execute_reply":"2021-10-27T07:55:23.768791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_meta.keys())","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:55:23.771454Z","iopub.execute_input":"2021-10-27T07:55:23.771916Z","iopub.status.idle":"2021-10-27T07:55:23.782619Z","shell.execute_reply.started":"2021-10-27T07:55:23.771813Z","shell.execute_reply":"2021-10-27T07:55:23.781004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, we will be unifying the metadata from the `*.json` files. We will first work with the `train` data.","metadata":{}},{"cell_type":"markdown","source":"First, we access the `annotations` list and convert it to a df.","metadata":{}},{"cell_type":"code","source":"train_id = pd.DataFrame(train_meta['annotations'])\ndisplay(train_id)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:55:23.788084Z","iopub.execute_input":"2021-10-27T07:55:23.788453Z","iopub.status.idle":"2021-10-27T07:55:26.045557Z","shell.execute_reply.started":"2021-10-27T07:55:23.788395Z","shell.execute_reply":"2021-10-27T07:55:26.044543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next is for `plant categories`:","metadata":{}},{"cell_type":"code","source":"train_cat = pd.DataFrame(train_meta['categories'])\ntrain_cat.columns = ['family', 'genus', 'category_id', 'category_name']\ndisplay(train_cat)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:55:26.049957Z","iopub.execute_input":"2021-10-27T07:55:26.050365Z","iopub.status.idle":"2021-10-27T07:55:26.116966Z","shell.execute_reply.started":"2021-10-27T07:55:26.050287Z","shell.execute_reply":"2021-10-27T07:55:26.115549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Followed by the `image properties`:","metadata":{}},{"cell_type":"code","source":"train_img = pd.DataFrame(train_meta['images'])\ntrain_img.columns = ['file_name', 'height', 'image_id', 'license', 'width']\ndisplay(train_img)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:55:26.118639Z","iopub.execute_input":"2021-10-27T07:55:26.119124Z","iopub.status.idle":"2021-10-27T07:55:28.622639Z","shell.execute_reply.started":"2021-10-27T07:55:26.119065Z","shell.execute_reply":"2021-10-27T07:55:28.621416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And lastly, the `region`:","metadata":{}},{"cell_type":"code","source":"train_reg = pd.DataFrame(train_meta['regions'])\ntrain_reg.columns = ['region_id', 'region_name']\ndisplay(train_reg)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:55:28.624522Z","iopub.execute_input":"2021-10-27T07:55:28.625307Z","iopub.status.idle":"2021-10-27T07:55:28.639961Z","shell.execute_reply.started":"2021-10-27T07:55:28.625014Z","shell.execute_reply":"2021-10-27T07:55:28.638515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Then, we will merge all the DataFrames and see what we got:","metadata":{}},{"cell_type":"code","source":"train_df = train_id.merge(train_cat, on='category_id', how='outer')\ntrain_df = train_df.merge(train_img, on='image_id', how='outer')\ntrain_df = train_df.merge(train_reg, on='region_id', how='outer')","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:55:28.641859Z","iopub.execute_input":"2021-10-27T07:55:28.642603Z","iopub.status.idle":"2021-10-27T07:55:30.389224Z","shell.execute_reply.started":"2021-10-27T07:55:28.642344Z","shell.execute_reply":"2021-10-27T07:55:30.388171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.info())\ndisplay(train_df)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:55:30.391042Z","iopub.execute_input":"2021-10-27T07:55:30.391512Z","iopub.status.idle":"2021-10-27T07:55:31.117120Z","shell.execute_reply.started":"2021-10-27T07:55:30.391429Z","shell.execute_reply":"2021-10-27T07:55:31.115902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looking closer, there's a line with `NaN` values there. We need to remove rows with `NaN`s so we proceed to the next line:","metadata":{}},{"cell_type":"code","source":"bools_img_path = train_df['file_name'].isna()\nkeep = [x for x in range(train_df.shape[0]) if not bools_img_path[x]]\ntrain_df = train_df.iloc[keep]","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:55:31.118869Z","iopub.execute_input":"2021-10-27T07:55:31.119520Z","iopub.status.idle":"2021-10-27T07:56:07.617720Z","shell.execute_reply.started":"2021-10-27T07:55:31.119465Z","shell.execute_reply":"2021-10-27T07:56:07.616657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After selecting the `non-NaN` items, we now reiterate on their file types. We need to save on memory, as we reached `102+ MB` for this DataFrame Only.","metadata":{}},{"cell_type":"code","source":"dtypes = ['int32', 'int32', 'int32', 'int32', 'object', 'object', 'object', 'object', 'int32', 'int32', 'int32', 'object']\nfor n, col in enumerate(train_df.columns):\n    train_df[col] = train_df[col].astype(dtypes[n])\nprint(train_df.info())\ndisplay(train_df)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:56:07.619274Z","iopub.execute_input":"2021-10-27T07:56:07.619761Z","iopub.status.idle":"2021-10-27T07:56:08.340468Z","shell.execute_reply.started":"2021-10-27T07:56:07.619678Z","shell.execute_reply":"2021-10-27T07:56:08.339302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Finally, for our `test` dataset. Since it only contains one key, `images`:","metadata":{}},{"cell_type":"code","source":"test_meta.keys()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:56:08.342041Z","iopub.execute_input":"2021-10-27T07:56:08.342575Z","iopub.status.idle":"2021-10-27T07:56:08.350526Z","shell.execute_reply.started":"2021-10-27T07:56:08.342504Z","shell.execute_reply":"2021-10-27T07:56:08.348970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.DataFrame(test_meta['images'])\ntest_df.columns = ['file_name', 'height', 'image_id', 'license', 'width']\nprint(test_df.info())\ndisplay(test_df)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:56:08.352832Z","iopub.execute_input":"2021-10-27T07:56:08.353648Z","iopub.status.idle":"2021-10-27T07:56:08.708884Z","shell.execute_reply.started":"2021-10-27T07:56:08.353513Z","shell.execute_reply":"2021-10-27T07:56:08.707534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Perfect!\n\nNow, we can go ahead and save this dataframe as a `*.csv` file for future use!","metadata":{}},{"cell_type":"code","source":"train_df.to_csv('full_train_data.csv', index=False)\ntest_df.to_csv('full_test_data.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:56:08.710615Z","iopub.execute_input":"2021-10-27T07:56:08.711183Z","iopub.status.idle":"2021-10-27T07:56:26.947360Z","shell.execute_reply.started":"2021-10-27T07:56:08.710988Z","shell.execute_reply":"2021-10-27T07:56:26.944178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Exploration\n\nWe will now start the data exploration and see what we can do with this dataset.","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:56:26.951484Z","iopub.execute_input":"2021-10-27T07:56:26.952271Z","iopub.status.idle":"2021-10-27T07:56:26.961056Z","shell.execute_reply.started":"2021-10-27T07:56:26.951891Z","shell.execute_reply":"2021-10-27T07:56:26.958105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_train = train_df['category_name'].value_counts()\nprint(len(_train),'種類')\nfig = plt.figure(figsize=(14,3))\nax1 = fig.add_subplot(1, 2, 1)\nax1.bar(_train[0:11].index,_train[0:11].values)\nplt.xticks(rotation=90)\nplt.title('top10')\nax2 = fig.add_subplot(1, 2, 2)\nax2.bar(_train[-10:-1].index,_train[-10:-1].values)\nplt.xticks(rotation=90)\nplt.title('worst10')\nfig.savefig(\"category_top10_worst_10.png\",bbox_inches=\"tight\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:56:26.963648Z","iopub.execute_input":"2021-10-27T07:56:26.965144Z","iopub.status.idle":"2021-10-27T07:56:28.777893Z","shell.execute_reply.started":"2021-10-27T07:56:26.964133Z","shell.execute_reply":"2021-10-27T07:56:28.776486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_train = train_df['family'].value_counts()\nprint(len(_train),'family')\nfig = plt.figure(figsize=(14,3))\nax1 = fig.add_subplot(1, 2, 1)\nax1.bar(_train[0:11].index,_train[0:11].values)\nplt.xticks(rotation=90)\nplt.title('top10')\nax2 = fig.add_subplot(1, 2, 2)\nax2.bar(_train[-10:-1].index,_train[-10:-1].values)\nplt.xticks(rotation=90)\nplt.title('worst10')\nfig.savefig(\"family_top10_worst_10.png\",bbox_inches=\"tight\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:56:28.779578Z","iopub.execute_input":"2021-10-27T07:56:28.780045Z","iopub.status.idle":"2021-10-27T07:56:30.055901Z","shell.execute_reply.started":"2021-10-27T07:56:28.779978Z","shell.execute_reply":"2021-10-27T07:56:30.054628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_train = train_df['genus'].value_counts()\nprint(len(_train),'genus')\nfig = plt.figure(figsize=(14,3))\nax1 = fig.add_subplot(1, 2, 1)\nax1.bar(_train[0:11].index,_train[0:11].values)\nplt.xticks(rotation=90)\nplt.title('top10')\nax2 = fig.add_subplot(1, 2, 2)\nax2.bar(_train[-10:-1].index,_train[-10:-1].values)\nplt.xticks(rotation=90)\nplt.title('worst10')\nfig.savefig(\"genus_top10_worst_10.png\",bbox_inches=\"tight\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:56:30.061274Z","iopub.execute_input":"2021-10-27T07:56:30.061919Z","iopub.status.idle":"2021-10-27T07:56:31.233362Z","shell.execute_reply.started":"2021-10-27T07:56:30.061685Z","shell.execute_reply":"2021-10-27T07:56:31.232097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Total Unique Values for each columns:\")\nprint(\"{0:10s} \\t {1:10d}\".format('train_df', len(train_df)))\nfor col in train_df.columns:\n    print(\"{0:10s} \\t {1:10d}\".format(col, len(train_df[col].unique())))","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:56:31.239292Z","iopub.execute_input":"2021-10-27T07:56:31.242516Z","iopub.status.idle":"2021-10-27T07:56:32.219813Z","shell.execute_reply.started":"2021-10-27T07:56:31.242443Z","shell.execute_reply":"2021-10-27T07:56:32.218940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here, we can see that other than the `category_id`, there's also the `family`, `genus`, `category_name`, `region_id` and `region_name` for the other probable targets. `category_id` and `category_name` are one and the same, similar to `region_id` and `region_name`.\n\nA possible approach for this kernel is to use a `CNN` to predict `family` and `genus` (we will ignore `region` for now). Then, using the `family` and `genus`, we will predict the `category_id` for the image.","metadata":{}},{"cell_type":"code","source":"family = train_df[['family', 'genus', 'category_name']].groupby(['family', 'genus']).count()\ndisplay(family.describe())","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:56:32.222325Z","iopub.execute_input":"2021-10-27T07:56:32.222756Z","iopub.status.idle":"2021-10-27T07:56:32.506086Z","shell.execute_reply.started":"2021-10-27T07:56:32.222689Z","shell.execute_reply":"2021-10-27T07:56:32.504855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"With some proper `image_data_augmentation` we can make up for the small number of samples for some images (first quartile).","metadata":{}},{"cell_type":"markdown","source":"# Model Creation\n一度にcategoryを当てに行くこともできるが、事前情報として与えられているfamilyやgenusの情報を生かして、モデルを構築するアプローチをとる。(そのまま予測する手法はほかメンバーが実施。)\n\nfamilyやgenusをcategoryの予測に生かす時、familyやgenusを予測する分類器を作って、特定の層の重みを学習させないことができる。tensorFlowであれば、レイヤーにtrainable属性が存在し、パラメータを学習させたくない層やモデルについて、trainable属性をFalseとすることで、学習をかけないでおくことができる。(ただし学習前の最後にcompile()を実行しないと属性の変更が繁栄されないので注意)\n\n参考のURLでは\n> summary()の出力結果には反映されているが、実際に設定を有効にするにはcompile()する必要があるので注意。compile()のあとでtrainableを変更した場合、再度compile()しなければならない。\n\nとしているため、このnotebookではtrainable属性の変更をしていないまま学習をかけていることになっているのではないか？\n\n\n参考\nhttps://note.nkmk.me/python-tensorflow-keras-trainable-freeze-unfreeze/\n","metadata":{}},{"cell_type":"markdown","source":"## CNNの構造の決め方について\n調べてもこういう時はこうするみたいな各事例ごとの具体例が出てくるのみ。また最近は「これが流行り」のようなモデル構造の隆盛まであるっぽい。結局どうすりゃいいかわからないから、Efficientnetに任せてしまうことにした。(AutoMLの考え方から)\n\n\n参考\nhttps://qiita.com/icoxfog417/items/5fd55fad152231d706c2","metadata":{}},{"cell_type":"markdown","source":"# Data Generator\ndata generatorを作る。\n前処理等しておくにはデータの容量が大きすぎる。バッチごとに処理をするので、dataGeneratorに任せることとする。\n\n処理時間参考\nhttps://hironsan.hatenablog.com/entry/2017/09/09/130608","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ntrain_datagen = ImageDataGenerator(featurewise_center=False,\n                                     featurewise_std_normalization=False,\n                                     rotation_range=180,\n                                     width_shift_range=0.1,\n                                     height_shift_range=0.1,\n                                     zoom_range=0.2)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:56:32.507828Z","iopub.execute_input":"2021-10-27T07:56:32.508542Z","iopub.status.idle":"2021-10-27T07:56:39.429638Z","shell.execute_reply.started":"2021-10-27T07:56:32.508465Z","shell.execute_reply":"2021-10-27T07:56:39.428443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, we will transform the `family` and `genus` to ids.","metadata":{}},{"cell_type":"code","source":"m = train_df[['file_name', 'family', 'genus', 'category_id']]\nfam = m.family.unique().tolist()\nm.family = m.family.map(lambda x: fam.index(x))\ngen = m.genus.unique().tolist()\nm.genus = m.genus.map(lambda x: gen.index(x))\ndisplay(m)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:56:39.432400Z","iopub.execute_input":"2021-10-27T07:56:39.433189Z","iopub.status.idle":"2021-10-27T07:57:14.531756Z","shell.execute_reply.started":"2021-10-27T07:56:39.433115Z","shell.execute_reply":"2021-10-27T07:57:14.530630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"os.chdir('../input')","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:57:14.534395Z","iopub.execute_input":"2021-10-27T07:57:14.535100Z","iopub.status.idle":"2021-10-27T07:57:14.541226Z","shell.execute_reply.started":"2021-10-27T07:57:14.535021Z","shell.execute_reply":"2021-10-27T07:57:14.539765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from efficientnet.efficientnet.keras import EfficientNetB3 \n#from efficientnet.efficientnet.model import EfficientNetB3\nfrom keras.models import Model\nfrom keras.layers import Dense, Dropout, Conv2D, MaxPool2D, Flatten, BatchNormalization, Input, concatenate\nfrom keras.optimizers import Adam\nfrom keras.utils import plot_model\nfrom sklearn.model_selection import train_test_split as tts","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:57:14.542955Z","iopub.execute_input":"2021-10-27T07:57:14.543710Z","iopub.status.idle":"2021-10-27T07:57:15.033291Z","shell.execute_reply.started":"2021-10-27T07:57:14.543635Z","shell.execute_reply":"2021-10-27T07:57:15.032333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fg_model(shape,lr):\n    \n    \n    actual_shape = shape\n    i = Input(actual_shape)\n    x = EfficientNetB3(weights='imagenet', include_top=False, input_shape=actual_shape, pooling='max')(i)\n    #x = Flatten()(x)\n    o1 = Dense(310, name=\"family\", activation='softmax')(x)\n    o2 = concatenate([x,o1])\n    o2 = Dense(3678, name=\"genus\", activation='softmax')(o2)\n    o3 = concatenate([x,o2])\n    o3 = Dense(32093, name=\"category_id\", activation='softmax')(o3)\n    model = Model(inputs=i,outputs=[o1,o2,o3])\n    \n    #model.layers[1].trainable = False\n    #model.get_layer('genus').trainable = False\n    \n    opt = Adam(lr=lr, amsgrad=True)\n    model.compile(optimizer=opt, loss=['sparse_categorical_crossentropy', \n                                   'sparse_categorical_crossentropy','sparse_categorical_crossentropy'],\n                 metrics=['accuracy'])\n\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:57:15.034881Z","iopub.execute_input":"2021-10-27T07:57:15.035459Z","iopub.status.idle":"2021-10-27T07:57:15.045811Z","shell.execute_reply.started":"2021-10-27T07:57:15.035393Z","shell.execute_reply":"2021-10-27T07:57:15.044641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = fg_model((300,300,3), 0.01) #Efficientnet B3 was designed for image size 300x300\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T07:57:15.047389Z","iopub.execute_input":"2021-10-27T07:57:15.048072Z","iopub.status.idle":"2021-10-27T07:57:30.425586Z","shell.execute_reply.started":"2021-10-27T07:57:15.048002Z","shell.execute_reply":"2021-10-27T07:57:30.424602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train\n\nNow, we will begin the training.","metadata":{}},{"cell_type":"code","source":"\ntrain, verif = tts(m, test_size=0.2, shuffle=True, random_state=17)\ntrain = train[:40000]\nverif = verif[:10000]\nshape = (120, 120, 3)\nepochs = 1\nbatch_size = 32\n\nmodel = fg_model(shape, 0.007)\nopt = Adam(lr=0.007, amsgrad=True)\n#Disable the last two output layers for training the Family\nmodel.get_layer('genus').trainable = True\nmodel.get_layer('family').trainable = True\nmodel.get_layer('category_id').trainable = True\n\nmodel.compile(optimizer=opt, loss=['sparse_categorical_crossentropy', \n                               'sparse_categorical_crossentropy','sparse_categorical_crossentropy'],\n             metrics=['accuracy'])\n#Train Family for 2 epochs\nmodel.fit_generator(train_datagen.flow_from_dataframe(dataframe=train,\n                                                      directory='../input/herbarium-2020-fgvc7/nybg2020/train/',\n                                                      x_col=\"file_name\",\n                                                      y_col=[\"family\", \"genus\", \"category_id\"],\n                                                      target_size=(120, 120),\n                                                      batch_size=batch_size,\n                                                      class_mode='multi_output'),\n                    validation_data=train_datagen.flow_from_dataframe(\n                        dataframe=verif,\n                        directory='../input/herbarium-2020-fgvc7/nybg2020/train/',\n                        x_col=\"file_name\",\n                        y_col=[\"family\", \"genus\", \"category_id\"],\n                        target_size=(120, 120),\n                        batch_size=batch_size,\n                        class_mode='multi_output'),\n                    epochs=epochs,\n                    steps_per_epoch=len(train)//batch_size,\n                    validation_steps=len(verif)//batch_size,\n                    verbose=1,\n                    workers=8,\n                    use_multiprocessing=True)\n\n#Reshuffle the inputs\ntrain, verif = tts(m, test_size=0.2, shuffle=True, random_state=18)\ntrain = train[:40000]\nverif = verif[:10000]\n\n#Make the Genus layer Trainable\nmodel.get_layer('genus').trainable = True\nmodel.get_layer('family').trainable = False\nmodel.get_layer('category_id').trainable = True\nmodel.compile(optimizer=opt, loss=['sparse_categorical_crossentropy', \n                               'sparse_categorical_crossentropy','sparse_categorical_crossentropy'],\n             metrics=['accuracy'])\n\n#Train Family and Genus for 2 epochs\nmodel.fit_generator(train_datagen.flow_from_dataframe(dataframe=train,\n                                                      directory='../input/herbarium-2020-fgvc7/nybg2020/train/',\n                                                      x_col=\"file_name\",\n                                                      y_col=[\"family\", \"genus\", \"category_id\"],\n                                                      target_size=(120, 120),\n                                                      batch_size=batch_size,\n                                                      class_mode='multi_output'),\n                    validation_data=train_datagen.flow_from_dataframe(\n                        dataframe=verif,\n                        directory='../input/herbarium-2020-fgvc7/nybg2020/train/',\n                        x_col=\"file_name\",\n                        y_col=[\"family\", \"genus\", \"category_id\"],\n                        target_size=(120, 120),\n                        batch_size=batch_size,\n                        class_mode='multi_output'),\n                    epochs=epochs,\n                    steps_per_epoch=len(train)//batch_size,\n                    validation_steps=len(verif)//batch_size,\n                    verbose=1,\n                    workers=8,\n                    use_multiprocessing=True)\n\n#Reshuffle the inputs\ntrain, verif = tts(m, test_size=0.2, shuffle=True, random_state=19)\ntrain = train[:40000]\nverif = verif[:10000]\n\n#Make the category_id layer Trainable\nmodel.get_layer('genus').trainable = True\nmodel.get_layer('family').trainable = True\nmodel.get_layer('category_id').trainable = True\nmodel.compile(optimizer=opt, loss=['sparse_categorical_crossentropy', \n                               'sparse_categorical_crossentropy','sparse_categorical_crossentropy'],\n             metrics=['accuracy'])\n#Train them all for 2 epochs\nmodel.fit_generator(train_datagen.flow_from_dataframe(dataframe=train,\n                                                      directory='../input/herbarium-2020-fgvc7/nybg2020/train/',\n                                                      x_col=\"file_name\",\n                                                      y_col=[\"family\", \"genus\", \"category_id\"],\n                                                      target_size=(120, 120),\n                                                      batch_size=batch_size,\n                                                      class_mode='multi_output'),\n                    validation_data=train_datagen.flow_from_dataframe(\n                        dataframe=verif,\n                        directory='../input/herbarium-2020-fgvc7/nybg2020/train/',\n                        x_col=\"file_name\",\n                        y_col=[\"family\", \"genus\", \"category_id\"],\n                        target_size=(120, 120),\n                        batch_size=batch_size,\n                        class_mode='multi_output'),\n                    epochs=epochs,\n                    steps_per_epoch=len(train)//batch_size,\n                    validation_steps=len(verif)//batch_size,\n                    verbose=1,\n                    workers=8,\n                    use_multiprocessing=True)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:01:18.842437Z","iopub.execute_input":"2021-10-27T08:01:18.842935Z","iopub.status.idle":"2021-10-27T08:53:19.479632Z","shell.execute_reply.started":"2021-10-27T08:01:18.842783Z","shell.execute_reply":"2021-10-27T08:53:19.477931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('../working/fg_model.h5')","metadata":{"execution":{"iopub.status.busy":"2021-10-27T09:10:58.210797Z","iopub.execute_input":"2021-10-27T09:10:58.211323Z","iopub.status.idle":"2021-10-27T09:11:11.780924Z","shell.execute_reply.started":"2021-10-27T09:10:58.211224Z","shell.execute_reply":"2021-10-27T09:11:11.779648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.history.history","metadata":{"execution":{"iopub.status.busy":"2021-10-27T09:21:41.753414Z","iopub.execute_input":"2021-10-27T09:21:41.754002Z","iopub.status.idle":"2021-10-27T09:21:41.766944Z","shell.execute_reply.started":"2021-10-27T09:21:41.753942Z","shell.execute_reply":"2021-10-27T09:21:41.763928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-10-27T09:22:45.511143Z","iopub.execute_input":"2021-10-27T09:22:45.511602Z","iopub.status.idle":"2021-10-27T09:22:45.537407Z","shell.execute_reply.started":"2021-10-27T09:22:45.511537Z","shell.execute_reply":"2021-10-27T09:22:45.535028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict\n\nNow, we will do our prediction. We may as well skip doing a confusion-matrix for our model because it's not even fully trained, so we go straight to our submission.\n\nSimilar to the above reason, we will be limiting the `predictions` to the first `10,000` items due to RAM limitations.","metadata":{}},{"cell_type":"code","source":"del train, verif, m, train_df, fam, gen, _train\nbatch_size = 32\n\n# generator = test_datagen.flow_from_dataframe(\n#         dataframe = test_df,#.iloc[:10000], \n#         directory = '../input/herbarium-2020-fgvc7/nybg2020/test/',\n#         x_col = 'file_name',\n#         target_size=(120, 120),\n#         batch_size=batch_size,\n#         class_mode=None,  # only data, no labels\n#         shuffle=False)\n\n# family, genus, category = model.predict_generator(generator, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-10-21T05:53:18.32024Z","iopub.execute_input":"2021-10-21T05:53:18.320759Z","iopub.status.idle":"2021-10-21T05:53:18.444854Z","shell.execute_reply.started":"2021-10-21T05:53:18.32071Z","shell.execute_reply":"2021-10-21T05:53:18.443882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ncategories = []\nfor i in range(1,len(test_df)):\n    if i % 10000 == 0:\n        test_datagen = ImageDataGenerator(featurewise_center=False,\n                                  featurewise_std_normalization=False)\n        generator = test_datagen.flow_from_dataframe(\n            dataframe = test_df.iloc[i-10000:i], \n            directory = '../input/herbarium-2020-fgvc7/nybg2020/test/',\n            x_col = 'file_name',\n            target_size=(120, 120),\n            batch_size=batch_size,\n            class_mode=None,  # only data, no labels\n            shuffle=False)\n\n        family, genus, category = model.predict_generator(generator, verbose=1,max_queue_size=10)\n        categories.append(np.argmax(category, axis=1))\n        last = i # 最後のindexを保存\n        del test_datagen, generator, family,genus\n    elif i == (len(test_df)-1):\n        test_datagen = ImageDataGenerator(featurewise_center=False,\n                                  featurewise_std_normalization=False)        \n        generator = test_datagen.flow_from_dataframe(\n            dataframe = test_df.iloc[last:i+1], \n            directory = '../input/herbarium-2020-fgvc7/nybg2020/test/',\n            x_col = 'file_name',\n            target_size=(120, 120),\n            batch_size=batch_size,\n            class_mode=None,  # only data, no labels\n            shuffle=False)\n\n        family, genus, category = model.predict_generator(generator, verbose=1,max_queue_size=10)\n        categories.append(np.argmax(category, axis=1))\n        del test_datagen, generator,family,genus\n#categories = np.concatenate(categories,axis=0)","metadata":{"execution":{"iopub.status.busy":"2021-10-21T05:53:18.44686Z","iopub.execute_input":"2021-10-21T05:53:18.447442Z","iopub.status.idle":"2021-10-21T06:27:12.338445Z","shell.execute_reply.started":"2021-10-21T05:53:18.447369Z","shell.execute_reply":"2021-10-21T06:27:12.337497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission\n\nNext, we'll save the predicted values under `predictions` into the specified format for submissions. Remember that our `predictions` is a `list` of 3-outputs, namely: `family`, `genus`, `category_id` in that order.","metadata":{}},{"cell_type":"code","source":"np.concatenate(categories[0:2])","metadata":{"execution":{"iopub.status.busy":"2021-10-21T06:38:12.245874Z","iopub.execute_input":"2021-10-21T06:38:12.246245Z","iopub.status.idle":"2021-10-21T06:38:12.252784Z","shell.execute_reply.started":"2021-10-21T06:38:12.246194Z","shell.execute_reply":"2021-10-21T06:38:12.251605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame()\nsub['Id'] = test_df.image_id\nsub['Id'] = sub['Id'].astype('int32')\nsub['Predicted'] = np.concatenate(categories)\nsub['Predicted'] = sub['Predicted'].astype('int32')\ndisplay(sub)\nsub.to_csv('../working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-21T06:38:23.689394Z","iopub.execute_input":"2021-10-21T06:38:23.689801Z","iopub.status.idle":"2021-10-21T06:38:24.534281Z","shell.execute_reply.started":"2021-10-21T06:38:23.689748Z","shell.execute_reply":"2021-10-21T06:38:24.533476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Finish\n\nThere you have it! A working model for predicting the `Category` of the plants. I hope that this kernel helped you on your journey in unraveling the mysteries of this dataset! Please upvote before forking___________3-(^_^ )","metadata":{}},{"cell_type":"code","source":"# end_time = time.time()\n# total = end_time - start_time\n# h = total//3600\n# m = (total%3600)//60\n# s = total%60\n# print(\"Total time spent: %i hours, %i minutes, and %i seconds\" %(h, m, s))","metadata":{"execution":{"iopub.status.busy":"2021-10-21T00:42:04.430395Z","iopub.status.idle":"2021-10-21T00:42:04.431227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# in_out_size = (120*120) + 3 #We will resize the image to 120*120 and we have 3 outputs\n# def xavier(shape, dtype=None):\n#     return np.random.rand(*shape)*np.sqrt(1/in_out_size)\n\n# def fg_model(shape, lr=0.001):\n#     '''Family-Genus model receives an image and outputs two integers indicating both the family and genus index.'''\n#     i = Input(shape)\n    \n#     x = Conv2D(3, (3, 3), activation='relu', padding='same', kernel_initializer=xavier)(i)\n#     x = Conv2D(3, (5, 5), activation='relu', padding='same', kernel_initializer=xavier)(x)\n#     x = MaxPool2D(pool_size=(3, 3), strides=(3,3))(x)\n#     x = BatchNormalization()(x)\n#     x = Dropout(0.5)(x)\n#     x = Conv2D(16, (5, 5), activation='relu', padding='same', kernel_initializer=xavier)(x)\n#     #x = Conv2D(16, (5, 5), activation='relu', padding='same', kernel_initializer=xavier)(x)\n#     x = MaxPool2D(pool_size=(5, 5), strides=(5,5))(x)\n#     x = BatchNormalization()(x)\n#     x = Dropout(0.5)(x)\n#     x = Flatten()(x)\n    \n#     o1 = Dense(310, activation='softmax', name='family', kernel_initializer=xavier)(x)\n    \n#     o2 = concatenate([o1, x])\n#     o2 = Dense(3678, activation='softmax', name='genus', kernel_initializer=xavier)(o2)\n    \n#     o3 = concatenate([o1, o2, x])\n#     o3 = Dense(32094, activation='softmax', name='category_id', kernel_initializer=xavier)(o3)\n    \n#     x = Model(inputs=i, outputs=[o1, o2, o3])\n    \n#     opt = Adam(lr=lr, amsgrad=True)\n#     x.compile(optimizer=opt, loss=['sparse_categorical_crossentropy', \n#                                    'sparse_categorical_crossentropy', \n#                                    'sparse_categorical_crossentropy'],\n#                  metrics=['accuracy'])\n#     return x\n\n# model = fg_model((120, 120, 3))\n# model.summary()\n# plot_model(model, to_file='full_model_plot.png', show_shapes=True, show_layer_names=True)","metadata":{"execution":{"iopub.status.busy":"2021-10-21T00:42:04.432426Z","iopub.status.idle":"2021-10-21T00:42:04.433222Z"},"trusted":true},"execution_count":null,"outputs":[]}]}