{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\nimport pandas as pd\nimport numpy as np\nimport csv\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n%matplotlib inline\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-04T03:42:35.417046Z","iopub.execute_input":"2023-03-04T03:42:35.417993Z","iopub.status.idle":"2023-03-04T03:42:35.432550Z","shell.execute_reply.started":"2023-03-04T03:42:35.417946Z","shell.execute_reply":"2023-03-04T03:42:35.431395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Check data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/small-jpegs-fgvc/train_cultivar_mapping.csv')","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:42:36.600671Z","iopub.execute_input":"2023-03-04T03:42:36.601039Z","iopub.status.idle":"2023-03-04T03:42:36.691338Z","shell.execute_reply.started":"2023-03-04T03:42:36.601006Z","shell.execute_reply":"2023-03-04T03:42:36.690386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:42:37.360227Z","iopub.execute_input":"2023-03-04T03:42:37.360970Z","iopub.status.idle":"2023-03-04T03:42:37.371839Z","shell.execute_reply.started":"2023-03-04T03:42:37.360924Z","shell.execute_reply":"2023-03-04T03:42:37.369115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SIZE_TRAIN_SET = train_df.shape[0]\nLIST_LABELS = None","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:42:38.647030Z","iopub.execute_input":"2023-03-04T03:42:38.647397Z","iopub.status.idle":"2023-03-04T03:42:38.652898Z","shell.execute_reply.started":"2023-03-04T03:42:38.647365Z","shell.execute_reply":"2023-03-04T03:42:38.651531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Note:** Here, we have 22193 img to build the model. First, I split img in each label's folder. Next, I create 2 folders train, val and each has all label's folder. If i don't do step 1, i don't sure in each folder has all each label's images.","metadata":{}},{"cell_type":"code","source":"# Step 1: Split img in each label's folder\ndef order_dataset(path_to_training_folder, path_to_images, path_to_csv):\n    list_labels = set()\n    if True:\n        try:\n            with open(path_to_csv, 'r') as csvfile:\n                reader = csv.reader(csvfile, delimiter=',')\n                for i, row in enumerate(reader):\n                    if i == 0: \n                        continue\n                    img_name = row[0]\n                    if img_name == \".DS_Store\":\n                        continue\n                    label_name = row[1]\n                    list_labels.add(label_name)\n                    path_to_folder = os.path.join(path_to_training_folder, label_name)\n                    \n                    # Check folder with label is exist.\n                    if not os.path.isdir(path_to_folder):\n                        os.makedirs(path_to_folder)\n\n                    img_full_path = os.path.join(path_to_images, img_name)\n                    new_path_img = os.path.join(path_to_folder, img_name)\n                    \n                    # Check if image is exist in folder.\n                    if not os.path.exists(new_path_img):\n                        shutil.copy(img_full_path, path_to_folder)\n            return list(list_labels)\n        except:\n            print(\"Some error when run order_dataset function\")\n    \n# Step 2: Create train and val folders\ndef split_data(path_to_data, path_to_train_folder, path_to_val_folder, split_size=0.2):\n    folders = os.listdir(path_to_data)\n    for folder in folders:\n        full_path = os.path.join(path_to_data, folder)\n        img_paths = os.listdir(full_path)\n        \n        x_train, x_val = train_test_split(img_paths, test_size=split_size)\n        \n        for x in x_train:\n            path_to_folder = os.path.join(path_to_train_folder, folder)\n            \n            if not os.path.isdir(path_to_folder):\n                os.makedirs(path_to_folder)\n            \n            img_path = os.path.join(full_path, x)\n            shutil.copy(img_path, path_to_folder)\n        \n        for x in x_val:\n            path_to_folder = os.path.join(path_to_val_folder, folder)\n            \n            if not os.path.isdir(path_to_folder):\n                os.makedirs(path_to_folder)\n            \n            img_path = os.path.join(full_path, x)\n            shutil.copy(img_path, path_to_folder)","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:42:41.394502Z","iopub.execute_input":"2023-03-04T03:42:41.394990Z","iopub.status.idle":"2023-03-04T03:42:41.407177Z","shell.execute_reply.started":"2023-03-04T03:42:41.394954Z","shell.execute_reply":"2023-03-04T03:42:41.405808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_to_images = \"/kaggle/input/small-jpegs-fgvc/train\"\npath_to_csv = \"/kaggle/input/small-jpegs-fgvc/train_cultivar_mapping.csv\"\npath_to_training_folder = \"/kaggle/working/training_data\"\npath_to_train_folder = \"/kaggle/working/data/train\"\npath_to_val_folder = \"/kaggle/working/data/val\"\n\nif True:\n    # Create folder have train data sort by label\n    if not os.path.isdir(path_to_training_folder):\n        os.makedirs(path_to_training_folder)\n    LIST_LABELS = order_dataset(path_to_training_folder, path_to_images, path_to_csv)\n\nif True:\n    split_data(path_to_training_folder, path_to_train_folder, path_to_val_folder)","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:42:49.004545Z","iopub.execute_input":"2023-03-04T03:42:49.005767Z","iopub.status.idle":"2023-03-04T03:48:22.666017Z","shell.execute_reply.started":"2023-03-04T03:42:49.005726Z","shell.execute_reply":"2023-03-04T03:48:22.655920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Total labels: \", len(LIST_LABELS))","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:48:22.677218Z","iopub.execute_input":"2023-03-04T03:48:22.677633Z","iopub.status.idle":"2023-03-04T03:48:22.694965Z","shell.execute_reply.started":"2023-03-04T03:48:22.677593Z","shell.execute_reply":"2023-03-04T03:48:22.693552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Note:** We have 100 labels!","metadata":{}},{"cell_type":"code","source":"# display some images\n# because more images, so i choose images 0 in each folder.\nn_rows = 4\nn_cols = 4\npath_to_training_data = \"/kaggle/working/training_data/\"\ndef display_some_images(path_images, path_csv):\n    fig = plt.figure(figsize=(16,16))\n    for i in range(n_rows*n_cols):\n        idx_label = np.random.randint(0, len(LIST_LABELS))\n        name_label = LIST_LABELS[idx_label]\n        \n        path_folder_label = os.path.join(path_to_training_data, name_label)\n        images_with_label = os.listdir(path_folder_label)\n        \n        idx_image = np.random.randint(0, len(images_with_label))\n        img_name = images_with_label[idx_image]\n        img_path = os.path.join(path_folder_label, img_name)\n        img = mpimg.imread(img_path)\n        # Check size img\n        if i == 1:\n            print(img.shape)\n        \n        ax = fig.add_subplot(n_rows, n_cols, i+1)\n        title = img_name + \"\\n\" + name_label \n        ax.title.set_text(title)\n        ax.axis('off')\n        plt.imshow(img)\n    \n    plt.show()\n        \n    ","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:49:10.693591Z","iopub.execute_input":"2023-03-04T03:49:10.693983Z","iopub.status.idle":"2023-03-04T03:49:10.703545Z","shell.execute_reply.started":"2023-03-04T03:49:10.693948Z","shell.execute_reply":"2023-03-04T03:49:10.702498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(42)\npath_images = \"/kaggle/input/small-jpegs-fgvc/train\"\npath_csv = \"/kaggle/input/small-jpegs-fgvc/train_cultivar_mapping.csv\"\n\nif True:\n    display_some_images(path_images, path_csv)","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:49:18.914967Z","iopub.execute_input":"2023-03-04T03:49:18.915664Z","iopub.status.idle":"2023-03-04T03:49:22.910189Z","shell.execute_reply.started":"2023-03-04T03:49:18.915608Z","shell.execute_reply":"2023-03-04T03:49:22.908408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Date: 9:27PM 1/3/2023:** I have no comment about data.","metadata":{}},{"cell_type":"markdown","source":"### Processing data","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:49:28.648237Z","iopub.execute_input":"2023-03-04T03:49:28.649018Z","iopub.status.idle":"2023-03-04T03:49:28.657077Z","shell.execute_reply.started":"2023-03-04T03:49:28.648978Z","shell.execute_reply":"2023-03-04T03:49:28.655836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size=32\ndef create_generators(batch_size, image_size, train_data_path, val_data_path):\n    print(batch_size)\n    train_preprocessor = ImageDataGenerator(\n        rescale=1/255.,\n        rotation_range=10,\n        width_shift_range=0.1\n    )\n    \n    # We need origin img when classification, so don't change anything\n    test_preprocessor = ImageDataGenerator(\n        rescale=1/255.\n    )\n    \n    train_generator = train_preprocessor.flow_from_directory(\n        train_data_path,\n        class_mode='categorical',\n        target_size=(224,224),\n        color_mode='rgb',\n        shuffle=True,\n        batch_size=batch_size\n    )\n    \n    val_generator = test_preprocessor.flow_from_directory(\n        val_data_path,\n        class_mode='categorical',\n        target_size=(224,224),\n        color_mode='rgb',\n        shuffle=True,\n        batch_size=batch_size\n    )\n\n    return train_generator, val_generator","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:57:37.852755Z","iopub.execute_input":"2023-03-04T03:57:37.853445Z","iopub.status.idle":"2023-03-04T03:57:37.861255Z","shell.execute_reply.started":"2023-03-04T03:57:37.853407Z","shell.execute_reply":"2023-03-04T03:57:37.859951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_size=(224,224)\npath_to_train = \"/kaggle/working/data/train\"\npath_to_val = \"/kaggle/working/data/val\"\nif True:\n    train_generator, val_generator = create_generators(batch_size,\n                                                       image_size,\n                                                       path_to_train,\n                                                       path_to_val)","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:57:39.335639Z","iopub.execute_input":"2023-03-04T03:57:39.336075Z","iopub.status.idle":"2023-03-04T03:57:39.993077Z","shell.execute_reply.started":"2023-03-04T03:57:39.336038Z","shell.execute_reply":"2023-03-04T03:57:39.992109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Note:** Run above function, we have:\n- Train: 17716 images belonging to 100 classes.\n- Test: 4477 images belonging to 100 classes.","metadata":{}},{"cell_type":"markdown","source":"### Build model\n- Here, i choose model efficientNetB7. \n- And we save model when loss/accuracy lower/higher","metadata":{}},{"cell_type":"code","source":"if False:\n    !pip install efficientnet","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:53:10.512325Z","iopub.execute_input":"2023-03-04T03:53:10.513267Z","iopub.status.idle":"2023-03-04T03:53:10.519034Z","shell.execute_reply.started":"2023-03-04T03:53:10.513205Z","shell.execute_reply":"2023-03-04T03:53:10.517632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import efficientnet.tfkeras as efn\nfrom tensorflow.keras import Model\nfrom tensorflow.keras.layers import Input, Conv2D, MaxPool2D, MaxPooling2D,GlobalAveragePooling2D,  BatchNormalization, Dropout, Flatten, Dense","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:53:11.939625Z","iopub.execute_input":"2023-03-04T03:53:11.940272Z","iopub.status.idle":"2023-03-04T03:53:11.946515Z","shell.execute_reply.started":"2023-03-04T03:53:11.940233Z","shell.execute_reply":"2023-03-04T03:53:11.945327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if False:\n#     def efficientnet_b7(input_shape, nbr_labels):\n#         if True:\n#     #         my_input = Input(shape=(224,224,3))\n#     #         x = efn.EfficientNetB7(\n#     #                 weights='imagenet',\n#     #                 include_top=False,\n#     #                 input_shape=(224,224,3))(my_input)\n\n#     #         x = Flatten()(x)\n#     #         x = BatchNormalization()(x)\n#     #         x = Dropout(0.3)(x)\n#     #         x = Dense(100, activation='softmax')(x)\n\n#             my_input = Input(shape=(224, 224, 3))\n#             x = efn.EfficientNetB7(include_top=False, weights='imagenet')(my_input)\n#             x = Flatten()(x)\n#             x = BatchNormalization()(x)\n#             x = Dropout(0.3)(x)\n#             x = Dense(100, activation='softmax')(x)\n\n#         if False:\n#             my_input = Input(shape=(224,224,3))\n\n#             x = Conv2D(32, (3,3), activation='relu')(my_input)\n#             x = MaxPool2D()(x)\n#             x = BatchNormalization()(x)\n\n#             x = Conv2D(64, (3,3),activation='relu')(x)\n#             x = MaxPool2D()(x)\n#             x = BatchNormalization()(x)\n\n#             x = Conv2D(128, (3,3), activation='relu')(x)\n#             x = MaxPool2D()(x)\n#             x = BatchNormalization()(x)\n\n#             x = Flatten()(x)\n#             x = Dense(128, activation='relu')(x)\n#             x = Dense(nbr_labels, activation='softmax')(x)\n\n#             return Model(inputs=my_input, outputs=x)\n\n#         return Model(inputs=my_input, outputs=x) ","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:53:12.539827Z","iopub.execute_input":"2023-03-04T03:53:12.541759Z","iopub.status.idle":"2023-03-04T03:53:12.552000Z","shell.execute_reply.started":"2023-03-04T03:53:12.541711Z","shell.execute_reply":"2023-03-04T03:53:12.550613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # \n# def efficientnet_b7(input_shape, nbr_labels):\n#     my_input = Input(shape=input_shape)\n#     x = efn.EfficientNetB7(include_top=False, weights='imagenet')(my_input)\n#     x = Flatten()(x)\n#     x = BatchNormalization()(x)\n#     x = Dropout(0.3)(x)\n#     x = Dense(100, activation='softmax')(x)\n\n#     return Model(inputs=my_input, outputs=x)","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:53:13.171393Z","iopub.execute_input":"2023-03-04T03:53:13.172126Z","iopub.status.idle":"2023-03-04T03:53:13.178422Z","shell.execute_reply.started":"2023-03-04T03:53:13.172082Z","shell.execute_reply":"2023-03-04T03:53:13.177238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.optimizers.legacy import Adam\n\noptimizer = Adam(learning_rate=0.0001)","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:53:13.491266Z","iopub.execute_input":"2023-03-04T03:53:13.491612Z","iopub.status.idle":"2023-03-04T03:53:13.498234Z","shell.execute_reply.started":"2023-03-04T03:53:13.491581Z","shell.execute_reply":"2023-03-04T03:53:13.497136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_shape=(224,224,3)\nif False:\n    model = efficientnet_b7(input_shape, len(LIST_LABELS))\n    model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n    model.summary()\n    \n    \nif True:\n    model = tf.keras.Sequential([\n        efn.EfficientNetB7(\n            input_shape=(224, 224, 3),\n            weights='imagenet',\n            include_top=False\n        ),\n        MaxPooling2D(pool_size=(2,2)),\n        Flatten(),\n        BatchNormalization(),\n        Dropout(0.3),\n        Dense(100, activation='softmax')\n    ])\n    \n    model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n    model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:53:13.850139Z","iopub.execute_input":"2023-03-04T03:53:13.851740Z","iopub.status.idle":"2023-03-04T03:53:23.317853Z","shell.execute_reply.started":"2023-03-04T03:53:13.851696Z","shell.execute_reply":"2023-03-04T03:53:23.316543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping\n\npath_to_save_model = '/kaggle/working/models'\nif not os.path.isdir(path_to_save_model):\n    os.makedirs(path_to_save_model)\nckpt_saver = ModelCheckpoint(\n    path_to_save_model,\n    monitor='val_accuracy',\n    mode='max',\n    save_best_only=True,\n    save_freq='epoch',\n    verbose=1\n)\n\nearly_stop = EarlyStopping(monitor='val_accuracy', patience=10)\ncallback = tf.keras.callbacks.EarlyStopping(monitor='loss', patience=3)","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:53:23.320399Z","iopub.execute_input":"2023-03-04T03:53:23.320832Z","iopub.status.idle":"2023-03-04T03:53:23.329554Z","shell.execute_reply.started":"2023-03-04T03:53:23.320790Z","shell.execute_reply":"2023-03-04T03:53:23.328195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size=32\nepochs = 15","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:53:23.330969Z","iopub.execute_input":"2023-03-04T03:53:23.331925Z","iopub.status.idle":"2023-03-04T03:53:23.340377Z","shell.execute_reply.started":"2023-03-04T03:53:23.331896Z","shell.execute_reply":"2023-03-04T03:53:23.339002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_generator,\n            epochs=epochs,\n            batch_size=batch_size,\n            validation_data=val_generator,\n            callbacks=[ckpt_saver, early_stop])","metadata":{"execution":{"iopub.status.busy":"2023-03-04T03:57:50.482553Z","iopub.execute_input":"2023-03-04T03:57:50.482927Z","iopub.status.idle":"2023-03-04T03:58:02.961295Z","shell.execute_reply.started":"2023-03-04T03:57:50.482893Z","shell.execute_reply":"2023-03-04T03:58:02.959437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Need do tomorrow:\n- Error when use efficient -> don't enough space ram so, next time, run in computer","metadata":{"execution":{"iopub.status.busy":"2023-03-02T15:03:45.369844Z","iopub.status.idle":"2023-03-02T15:03:45.370553Z","shell.execute_reply.started":"2023-03-02T15:03:45.370248Z","shell.execute_reply":"2023-03-02T15:03:45.370284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Dense, Flatten, BatchNormalization, Dropout\nimport efficientnet.tfkeras as efn\nimport tensorflow_datasets as tfds\n\n# Load test data\ntest_data = tfds.load(\"cats_vs_dogs\", split=\"test[:10%]\", as_supervised=True)\ntest_data = test_data.map(lambda x, y: (tf.image.resize(x, (224, 224))/255.0, y))\n\n# Define model\nmy_input = Input(shape=(224, 224, 3))\nx = efn.EfficientNetB0(include_top=False, weights='imagenet')(my_input)\nx = Flatten()(x)\nx = BatchNormalization()(x)\nx = Dropout(0.3)(x)\nx = Dense(100, activation='\n","metadata":{},"execution_count":null,"outputs":[]}]}