{"cells":[{"metadata":{"trusted":true,"scrolled":false},"cell_type":"code","source":"## Herbarium 2020 - FGVC7\n# read and prerpocess the json file.\nimport json\nimport codecs\nimport pandas as pd\n\n# split the dataset.\nfrom sklearn.model_selection import train_test_split\n\n# deeplearning framework.\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.layers import Dense, Input, concatenate, Flatten, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.utils import plot_model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport tensorflow as tf\n\n# other python package.\nfrom PIL import Image\nimport numpy as np\nimport os\n\n## I. Data Preprocessing define function :\n#-------------------<Data Preprocessing>----------------------------------------#\n##   Given path of json file, return the DataFrame type data \n##   ** (merge all info into 1 DataFrame) **\ndef load_data(train_path, test_path):\n    \n    def load_jsonfile(json_path):\n        with codecs.open(json_path,'r',encoding='utf-8',errors='ignore') as file:\n            meta_data = json.load(file)  ## return dictionary type data.\n        return meta_data \n    \n    train_meta = load_jsonfile(train_path)\n    test_meta = load_jsonfile(test_path)\n    \n    ## Extract from each information into corresponding DataFrame.\n    ## 1. Annotation information.\n    annotations = pd.DataFrame(train_meta['annotations'])\n    ## Debug view : \n    #print(\"\\n\\nAnnotations dataframe infomation : \\n\")\n    #print(annotations.info())\n    \n    ## 2. Categories information.\n    categories = pd.DataFrame(train_meta['categories'])\n    categories.columns = ['family', 'genus', 'category_id', 'category_name'] ## replace columns name.\n    ## Debug view :\n    #print(\"\\n\\nCategories dataframe infomation : \\n\")\n    #print(categories.info())\n    \n    ## 3. Images information.\n    images = pd.DataFrame(train_meta['images'])\n    images.columns = ['file_name', 'height', 'image_id','license','width']\n    ## Debug view :\n    #print(\"\\n\\nImages dataframe infomation : \\n\")\n    #print(images.info())\n   \n    ## 4. Regions information.\n    regions = pd.DataFrame(train_meta['regions'])\n    regions.columns = ['region_id','name']\n    ## Debug view :\n    #print(\"\\n\\nRegions dataframe infomation : \\n\")\n    #print(regions.info())\n    \n    ## Merge all dataframe to get raw training dataframe.\n    raw_train_df = annotations.merge(categories,on='category_id', how=\"left\"\n                                     ).merge(images, on=\"image_id\", how=\"outer\"\n                                            ).merge(regions, on=\"region_id\", how=\"outer\")\n    \n    raw_train_df = raw_train_df[['file_name','family','genus','category_id']]\n    ## Debug view :\n    #print(\"\\n\\nRaw dataframe of training set : \\n\")\n    #print(raw_train_df.info())\n    #print(\"\\n\\n The family, genus data type is object, which should be transformed into int type \\\n    #       via indicator (use index of list to present the corresponding content)\")\n    \n    ## Preprocess the dataframe.\n    ## 1. unique the element in the frame, \\\n    ##           and replace the 'category type' data into corresponding 'index type' data in the list.\n    name_list = raw_train_df['family'].unique().tolist()\n    raw_train_df.loc[:,'family'] = raw_train_df['family'].map(lambda x:name_list.index(x))\n    genus_list = raw_train_df['genus'].unique().tolist()\n    raw_train_df.loc[:,'genus'] = raw_train_df['genus'].map(lambda x:genus_list.index(x))\n    \n    \n    ## 2. redeclare the data type to shrink the memory usage.\n    train_df = raw_train_df.astype({'family':'int16','genus':'int16','category_id':'int16'})\n    \n    ## Extract the test dataframe information.\n    raw_test_df = pd.DataFrame(test_meta['images'])\n    raw_test_df.columns = ['file_name', 'height', 'image_id','license','width']\n    test_df = raw_test_df[['image_id','file_name']]\n    return train_df, test_df\n\ndef crop(batch_x):\n    cut1 = int(0.1*batch_x.shape[1])\n    cut2 = int(0.05*batch_x.shape[2])\n    return batch_x[:,cut1:-cut1,cut2:-cut2]\n## II. Build Classifier :\n#----------------------<self-build>------------------------#\ndef get_cnn_classifier(img_shape):\n    '''\n    def prerpocess_img(x):\n        if isinstance(x, np.ndarray):  ## Training phase, instance input.\n            return x/127.5 -1 \n        else:  ## Symbolic preprocess.\n            return tf.sub(tf.div(x, 127.5), 1)\n    '''\n    def build_classifier(img_shape):\n        model = Sequential({\n            Conv2D(filters=64, kernel_size=5, activation='relu',\\\n                   stride=2, input_shape=img_shape),\n            MaxPooling2D(2),\n            Conv2D(128, 3, activation='relu', padding='same'),\n            Conv2D(128, 3, activation='relu', padding='same'),\n            MaxPooling2D(2),\n            Conv2D(256, 3, activation='relu', padding='same'),\n            Conv2D(256, 3, activation='relu', padding='same'),\n            MaxPooling2D(2),\n            Flatten(),\n            Dense(128, activation='relu'),\n            Droupout(0.25),\n            Dense(64, activation='relu'),\n            Droupout(0.2),\n            Dense(32094, activation='softmax')      \n        })\n        return model\n    \n#-------------------<keras applications>-------------------#\n## VGG19\ndef get_VGG_based_Classifier(img_shape):\n    ## Closure function : \n    def __preprocess_vgg(x):\n        ## Take a HR image [-1, 1], convert to [0, 255], then to input for VGG network\n\n        if isinstance(x, np.ndarray):  ## Training phase, instance input.\n            return preprocess_input(x) \n        else:  ## Symbolic preprocess.\n            return Lambda(lambda x: preprocess_input(x))(x)\n    \n    def use_raw_VGG():\n        pass\n        \n    ## Build Basic VGG19 network :\n    #prepro_VGG19 = prepro_no_top_VGG()\n    #prepro_VGG19()\n    ## The modification of output prediction :  output 320000 specimens\n    ## Please take the reference of keras VGG application.\n    #VGG_classifier = Model(inputs=prepro_VGG19.input, output=)\n    \n    VGG_classifier = VGG19(weights=None, include_top=True, \\\n                      input_shape=img_shape, classes=32000)\n    VGG_classifier.compile(loss='mae',optimizer='adam')\n    return VGG_classifier\n    \n## ResNet50\nin_out_size = (120*120) + 3\ndef xavier(shape, dtype=None):\n    return np.random.rand(*shape)*np.sqrt(1/in_out_size)\ndef create_model():\n    actual_shape = (crop(np.zeros((1,img_shape[0],img_shape[1],img_shape[2]))).shape)[1:]\n    i = Input(actual_shape)\n    x = ResNet50(weights='imagenet', include_top=False, input_shape=actual_shape, pooling='max')(i)\n    x = Dropout(0.5)(x)\n    x = Flatten()(x)\n    o1 = Dense(310, activation='softmax', name='family', kernel_initializer=xavier)(x)\n    \n    o2 = concatenate([o1, x])\n    o2 = Dense(3678, activation='softmax', name='genus', kernel_initializer=xavier)(o2)\n    \n    o3 = concatenate([o1, o2, x])\n    o3 = Dense(32094, activation='softmax', name='category_id', kernel_initializer=xavier)(o3)\n    model = Model(inputs=i,outputs=[o1,o2,o3])\n    model.layers[1].trainable = False\n    model.get_layer('genus').trainable = False\n    model.get_layer('category_id').trainable = False\n    return model\n\ndef compile(model,learning_rate=0.005):\n    model.compile(optimizer=Adam(learning_rate=0.005),loss=[\"sparse_categorical_crossentropy\",\n                                     \"sparse_categorical_crossentropy\",\n                                     \"sparse_categorical_crossentropy\"],\n                                metrics=['accuracy'])\n    \n\n\n#TRAINSTEPS = (X_train.shape[0]//batchsize)+1\n#VALSTEPS = (X_dev.shape[0]//batchsize)+1\n\n## Main part :\n# self-define params : *(argparse)\nbatch_size=64\nlearning_rate=0.01\nepochs=1000\nimg_shape = (200, 200, 3) # original height = 1000, width= 682    219673 667    212347 676    190476\ntrain_path = '/kaggle/input/herbarium-2020-fgvc7/nybg2020/train/metadata.json'\ntest_path = '/kaggle/input/herbarium-2020-fgvc7/nybg2020/test/metadata.json'\n\n## I. Load & Preprocess data :\ntrain_df, test_df = load_data(train_path, test_path)\n\nnmb_cat = train_df['category_id'].max()+1\nnmb_gen = train_df['genus'].max()+1\nnmb_fam = train_df['family'].max()+1\n\ntrain_data, test_data = train_test_split(train_df, test_size=0.05, shuffle=True, random_state=13)\nprint(train_data.info())\n\n## II. Build classifier :\n#VGG_classifier = get_VGG_based_Classifier(img_shape)\nResNet_classifier = create_model()\n\n## III. Image Augumentation and Build the data flow :\n                ## Image Augumentation.\ntrain_datagen = ImageDataGenerator(horizontal_flip = True, vertical_flip = True, \\\n                                   rotation_range = 180, zoom_range=0.3, fill_mode='nearest')\n                                   \n            ## setting data flow extract from DataFrame type data.\ndata_flow = train_datagen.flow_from_dataframe(dataframe=train_data, \\\n                                  directory='../input/herbarium-2020-fgvc7/nybg2020/train/', \\\n                                  x_col=\"file_name\", y_col=[\"family\", \"genus\", \"category_id\"], \\\n                                  target_size=img_shape, batch_size=batch_size, \\\n                                  class_mode='multi_output')\n\n## IV. Training phase :\n## In the training phase, we'll first predict the family, and then use the family to predict the genus,\n##    and so on, use the predicted family & genus to predict category_id.\n#VGG_classifier.fit_generator(data_flow, epochs=epochs, verbose=2, workers=4, use_multiprocessing=False)\n#VGG_classifier.save('./classifier.h5')\nmodel = create_model()\ncompile(model,learning_rate)\n#model.summary()\nfilename=\"classifier.h5\"\nmodel.save_weights(filename)\nprint(\"Weights saved to {}\".format(filename))\nplot_model(model, show_shapes=True, show_layer_names=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true,"scrolled":false},"cell_type":"code","source":"## V. Prediction phase :\nbatch_size = 32\ntest_datagen = ImageDataGenerator(featurewise_center=False, \n                                    featurewise_std_normalization=False)\n\ngenerator = test_datagen.flow_from_dataframe(\n        dataframe = test_df.iloc[:10000], #Limiting the test to the first 10,000 items\n        directory = '../input/herbarium-2020-fgvc7/nybg2020/test/',\n        x_col = 'file_name',\n        target_size=(120, 120),\n        batch_size=batch_size,\n        class_mode=None,  # only data, no labels\n        shuffle=False)\n\nfamily, genus, category = model.predict_generator(generator, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = pd.DataFrame()\nsub['Id'] = test_df.image_id\nsub['Id'] = sub['Id'].astype('int32')\nsub['Predicted'] = np.concatenate([np.argmax(category, axis=1), 23718*np.ones((len(test_df.image_id)-len(category)))], axis=0)\nsub['Predicted'] = sub['Predicted'].astype('int32')\ndisplay(sub)\nsub.to_csv('category_submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub['Predicted'] = np.concatenate([np.argmax(family, axis=1), np.zeros((len(test_df.image_id)-len(family)))], axis=0)\nsub['Predicted'] = sub['Predicted'].astype('int32')\ndisplay(sub)\nsub.to_csv('family_submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub['Predicted'] = np.concatenate([np.argmax(genus, axis=1), np.zeros((len(test_df.image_id)-len(genus)))], axis=0)\nsub['Predicted'] = sub['Predicted'].astype('int32')\ndisplay(sub)\nsub.to_csv('genus_submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}