{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n# directories\n\n#/kaggle/input/herbarium-2022-fgvc9/train_metadata.json\n#/kaggle/input/herbarium-2022-fgvc9/sample_submission.csv\n#/kaggle/input/herbarium-2022-fgvc9/test_metadata.json\n#/kaggle/input/herbarium-2022-fgvc9/train_images/135/47/13547__018.jpg","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-10T16:57:52.585786Z","iopub.execute_input":"2022-11-10T16:57:52.586216Z","iopub.status.idle":"2022-11-10T16:57:52.616781Z","shell.execute_reply.started":"2022-11-10T16:57:52.586128Z","shell.execute_reply":"2022-11-10T16:57:52.615678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:57:52.619265Z","iopub.execute_input":"2022-11-10T16:57:52.619996Z","iopub.status.idle":"2022-11-10T16:57:52.625435Z","shell.execute_reply.started":"2022-11-10T16:57:52.619951Z","shell.execute_reply":"2022-11-10T16:57:52.624335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Load metadata ","metadata":{}},{"cell_type":"code","source":"main_dir = '/kaggle/input/herbarium-2022-fgvc9'\nmetadata_file = os.path.join(main_dir, \"{}_metadata.json\")\nimages_dir = os.path.join(main_dir, \"{}_images\")\nmetadata_file, images_dir","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:00.087314Z","iopub.execute_input":"2022-11-10T16:59:00.087747Z","iopub.status.idle":"2022-11-10T16:59:00.098149Z","shell.execute_reply.started":"2022-11-10T16:59:00.087714Z","shell.execute_reply":"2022-11-10T16:59:00.097001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_prefix = 'train'\ntest_prefix = 'test'","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:00.104487Z","iopub.execute_input":"2022-11-10T16:59:00.105236Z","iopub.status.idle":"2022-11-10T16:59:00.110323Z","shell.execute_reply.started":"2022-11-10T16:59:00.105194Z","shell.execute_reply":"2022-11-10T16:59:00.109044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load metadata\nimport json\n\ndef get_dataframe(prefix):\n    with open(metadata_file.format(prefix)) as f:\n        metadata = json.load(f)\n    \n    if (prefix == train_prefix):\n        df_ann = pd.DataFrame(metadata['annotations'])\n        df_ann = df_ann[['image_id', 'category_id']]\n        df_ann.set_index('image_id', inplace = True)\n\n        df_images = pd.DataFrame(metadata['images'])\n        df_images = df_images[['image_id', 'file_name']]\n        df_images.set_index('image_id', inplace = True)\n\n        return df_ann.join(df_images, how = 'left')\n    else:\n        df = pd.DataFrame(metadata)\n        return df","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:00.112329Z","iopub.execute_input":"2022-11-10T16:59:00.112648Z","iopub.status.idle":"2022-11-10T16:59:00.123598Z","shell.execute_reply.started":"2022-11-10T16:59:00.112596Z","shell.execute_reply":"2022-11-10T16:59:00.122467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = get_dataframe(train_prefix)\nprint(len(train_df))\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:00.131502Z","iopub.execute_input":"2022-11-10T16:59:00.132057Z","iopub.status.idle":"2022-11-10T16:59:15.992147Z","shell.execute_reply.started":"2022-11-10T16:59:00.132026Z","shell.execute_reply":"2022-11-10T16:59:15.990893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = get_dataframe(test_prefix)\nprint(len(test_df))\ntest_df.tail()","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:15.994791Z","iopub.execute_input":"2022-11-10T16:59:15.995204Z","iopub.status.idle":"2022-11-10T16:59:16.595845Z","shell.execute_reply.started":"2022-11-10T16:59:15.995164Z","shell.execute_reply":"2022-11-10T16:59:16.594699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:16.597763Z","iopub.execute_input":"2022-11-10T16:59:16.598577Z","iopub.status.idle":"2022-11-10T16:59:16.673009Z","shell.execute_reply.started":"2022-11-10T16:59:16.598533Z","shell.execute_reply":"2022-11-10T16:59:16.67159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:16.676125Z","iopub.execute_input":"2022-11-10T16:59:16.676646Z","iopub.status.idle":"2022-11-10T16:59:16.716727Z","shell.execute_reply.started":"2022-11-10T16:59:16.676574Z","shell.execute_reply":"2022-11-10T16:59:16.715233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.category_id.hist(figsize = (25, 5))","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:16.718578Z","iopub.execute_input":"2022-11-10T16:59:16.719315Z","iopub.status.idle":"2022-11-10T16:59:17.067106Z","shell.execute_reply.started":"2022-11-10T16:59:16.719273Z","shell.execute_reply":"2022-11-10T16:59:17.066059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# how many examples of each class are presented in train dataset?\nclass_count = {target: len(train_df[train_df['category_id'] == target]) for target in sorted(train_df.category_id.unique())}\nmin(class_count.values()), max(class_count.values())","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:17.068617Z","iopub.execute_input":"2022-11-10T16:59:17.069145Z","iopub.status.idle":"2022-11-10T16:59:32.37084Z","shell.execute_reply.started":"2022-11-10T16:59:17.069092Z","shell.execute_reply":"2022-11-10T16:59:32.369556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\nplt.figure(figsize = (15,5))\nsns.countplot(x = list(class_count.values()))","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:32.372726Z","iopub.execute_input":"2022-11-10T16:59:32.37314Z","iopub.status.idle":"2022-11-10T16:59:34.025741Z","shell.execute_reply.started":"2022-11-10T16:59:32.3731Z","shell.execute_reply":"2022-11-10T16:59:34.024645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a good approach to solve this task is to increase number of less presented classes by using image random transformations\n# but we would get a huuuuge dataset and it would take a lot of time to process all these images that we cannot afford in case of limited resources\n# so instead of this we take only frequently encounted class samples\n\n# let's say that the minimum number of sample of one class should be 80 or more\nthreshold = 80\nclass_count = {k:v for k, v in class_count.items() if v >= threshold}\n\nprint(len(class_count))\nplt.figure(figsize = (15,5))\nsns.countplot(x = list(class_count.values()))","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:34.02747Z","iopub.execute_input":"2022-11-10T16:59:34.027913Z","iopub.status.idle":"2022-11-10T16:59:34.234652Z","shell.execute_reply.started":"2022-11-10T16:59:34.027865Z","shell.execute_reply":"2022-11-10T16:59:34.233535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.concat([train_df[train_df['category_id'] == target] for target in class_count.keys()])\nprint(len(train_df))","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:34.236138Z","iopub.execute_input":"2022-11-10T16:59:34.237297Z","iopub.status.idle":"2022-11-10T16:59:34.429216Z","shell.execute_reply.started":"2022-11-10T16:59:34.237256Z","shell.execute_reply":"2022-11-10T16:59:34.428005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.image as mpimg\n\nimage = mpimg.imread(os.path.join(images_dir.format(train_prefix), train_df.iloc[0]['file_name']))\n\nplt.figure(figsize = (5,10))\nplt.imshow(image)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:34.432498Z","iopub.execute_input":"2022-11-10T16:59:34.432936Z","iopub.status.idle":"2022-11-10T16:59:34.840843Z","shell.execute_reply.started":"2022-11-10T16:59:34.432895Z","shell.execute_reply":"2022-11-10T16:59:34.839679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Download images","metadata":{}},{"cell_type":"code","source":"# preprocess label value\nfrom sklearn import preprocessing\n\nle = preprocessing.LabelEncoder()\ntrain_df['category_id_encoded'] = le.fit_transform(train_df['category_id'])\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:34.842066Z","iopub.execute_input":"2022-11-10T16:59:34.843284Z","iopub.status.idle":"2022-11-10T16:59:34.91203Z","shell.execute_reply.started":"2022-11-10T16:59:34.843248Z","shell.execute_reply":"2022-11-10T16:59:34.910985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# how many classes we have in the end?\nnum_of_classes = max(train_df['category_id_encoded']) + 1\nnum_of_classes","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:34.913439Z","iopub.execute_input":"2022-11-10T16:59:34.914073Z","iopub.status.idle":"2022-11-10T16:59:34.923357Z","shell.execute_reply.started":"2022-11-10T16:59:34.914031Z","shell.execute_reply":"2022-11-10T16:59:34.922186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ImageDataGenerator().flow_from_dataframe(class_mode = 'sparse') requires string labels\ntrain_df['category_id_encoded'] = train_df['category_id_encoded'].astype('str')\ntrain_df = train_df.sample(frac = 1)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-11-10T16:59:34.925129Z","iopub.execute_input":"2022-11-10T16:59:34.926585Z","iopub.status.idle":"2022-11-10T16:59:34.954492Z","shell.execute_reply.started":"2022-11-10T16:59:34.92654Z","shell.execute_reply":"2022-11-10T16:59:34.953401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nimg_size = 150\ntrain_datagen = ImageDataGenerator(rescale = 1/255.,\n                                   horizontal_flip = True,\n                                   vertical_flip = True,\n                                   zoom_range = 0.2,\n                                   rotation_range = 40,\n                                   width_shift_range = 0.2,\n                                   height_shift_range = 0.2,\n                                   shear_range = 0.2,\n                                   validation_split = 0.1)","metadata":{"execution":{"iopub.status.busy":"2022-11-10T22:26:00.263375Z","iopub.execute_input":"2022-11-10T22:26:00.263776Z","iopub.status.idle":"2022-11-10T22:26:00.271226Z","shell.execute_reply.started":"2022-11-10T22:26:00.263744Z","shell.execute_reply":"2022-11-10T22:26:00.27001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = images_dir.format(train_prefix)\nprint(train_dir)\n\ntrain_gen = train_datagen.flow_from_dataframe(dataframe = train_df,\n                                              directory = train_dir,\n                                              x_col = \"file_name\",\n                                              y_col = \"category_id_encoded\",\n                                              target_size=(img_size, img_size),\n                                              batch_size = 16,\n                                              color_mode = \"grayscale\",\n                                              class_mode = \"sparse\",\n                                              subset = \"training\")\n\nval_gen = train_datagen.flow_from_dataframe(dataframe = train_df,\n                                            directory = train_dir,\n                                            x_col = \"file_name\",\n                                            y_col = \"category_id_encoded\",\n                                            target_size = (img_size, img_size),\n                                            batch_size = 16,\n                                            color_mode = \"grayscale\",\n                                            class_mode = \"sparse\",\n                                            subset = \"validation\")","metadata":{"execution":{"iopub.status.busy":"2022-11-10T22:26:04.507407Z","iopub.execute_input":"2022-11-10T22:26:04.508214Z","iopub.status.idle":"2022-11-10T22:26:11.006143Z","shell.execute_reply.started":"2022-11-10T22:26:04.508163Z","shell.execute_reply":"2022-11-10T22:26:11.005033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Build a model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.layers import Input, Dense, Conv2D, MaxPooling2D, Flatten, Dropout, BatchNormalization, Add\nfrom tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras.optimizers import Adam\n\ninput_size = (img_size, img_size, 1) # all images are 120x120 size, 1 - \"grayscale\"\n\ninput = Input(input_size)\n\nout = Conv2D(32, 5, activation = 'relu', kernel_regularizer = tf.keras.regularizers.L2(1e-1))(input)\nout = Conv2D(32, 5, activation = 'relu', kernel_regularizer = tf.keras.regularizers.L2(1e-1))(out) \nout = MaxPooling2D()(out) \nout = BatchNormalization()(out)\nout = Dropout(0.3)(out)\n\nout = Conv2D(64, 3, activation = 'relu', kernel_regularizer = tf.keras.regularizers.L2(1e-1))(out)\nout = Conv2D(64, 3, activation = 'relu', kernel_regularizer = tf.keras.regularizers.L2(1e-1))(out) \nout = MaxPooling2D()(out) \nout = BatchNormalization()(out)\nout = Dropout(0.3)(out)\n\nout = Conv2D(64, 3, activation = 'relu', kernel_regularizer = tf.keras.regularizers.L2(1e-1))(out) \nout = Conv2D(64, 3, activation = 'relu', kernel_regularizer = tf.keras.regularizers.L2(1e-1))(out)\nout = MaxPooling2D()(out)\nout = BatchNormalization()(out)\nout = Dropout(0.3)(out)\n\nout = Flatten()(out)\nout = Dense(num_of_classes, activation = 'softmax')(out)\n\nmodel = Model(inputs = input, outputs = out)\n\nopt = Adam(learning_rate = 1e-5)\n\nmodel.compile(optimizer = opt, loss = 'sparse_categorical_crossentropy', metrics = ['accuracy'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-11-10T22:26:11.008453Z","iopub.execute_input":"2022-11-10T22:26:11.009158Z","iopub.status.idle":"2022-11-10T22:26:11.136477Z","shell.execute_reply.started":"2022-11-10T22:26:11.009116Z","shell.execute_reply":"2022-11-10T22:26:11.135409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_gen, validation_data = val_gen, epochs = 100, verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2022-11-10T22:26:11.139815Z","iopub.execute_input":"2022-11-10T22:26:11.140126Z","iopub.status.idle":"2022-11-11T00:09:29.444035Z","shell.execute_reply.started":"2022-11-10T22:26:11.140097Z","shell.execute_reply":"2022-11-11T00:09:29.442909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs = range(1, len(loss) + 1)\n\nplt.figure(figsize = (20, 6))\n\nplt.subplot(1, 2, 1)\nplt.plot(epochs, loss, 'bo', label = 'Trainig loss')\nplt.plot(epochs, val_loss, 'r', label = 'Validation loss')\nplt.legend()\n\n\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs, acc, 'bo', label = 'Trainig acc')\nplt.plot(epochs, val_acc, 'r', label = 'Validation acc')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-11T00:09:29.44734Z","iopub.execute_input":"2022-11-11T00:09:29.447848Z","iopub.status.idle":"2022-11-11T00:09:29.838172Z","shell.execute_reply.started":"2022-11-11T00:09:29.447803Z","shell.execute_reply":"2022-11-11T00:09:29.837103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Predict","metadata":{}},{"cell_type":"code","source":"test_dir = images_dir.format(test_prefix)\ntest_datagen = ImageDataGenerator(rescale = 1./255)\ntest_gen = test_datagen.flow_from_dataframe(dataframe = test_df, \n                                            directory = test_dir,\n                                            x_col = \"file_name\",\n                                            target_size = (img_size, img_size),\n                                            batch_size = 32,\n                                            color_mode = \"grayscale\",\n                                            class_mode = None,\n                                            shuffle = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = []\nfor i in range(len(test_gen)):\n    pred.extend([np.argmax(x) for x in model.predict(test_gen[i], verbose = 0)])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = le.inverse_transform(pred)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(os.path.join(main_dir, 'sample_submission.csv'))\nsub['Predicted'] = pred[:len(test_df)]\nsub.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index = False)  ","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}