{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q efficientnet\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport tensorflow.keras as keras\nimport PIL\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport random\nfrom tqdm import tqdm\nimport tensorflow_addons as tfa\nimport random\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom kaggle_datasets import KaggleDatasets\n\nimport efficientnet.tfkeras as efn\n\npd.set_option(\"display.max_columns\", None)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset Exploration","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/plant-pathology-2021-fgvc8/train.csv')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train))\nprint(train.columns)\nprint(train['labels'].value_counts().plot.bar())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"các loại nhãn khác nhau\n","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MultiLabelBinarizer\nlabel_split = train.labels.apply(lambda x: x.split()) #chia 1 chuỗi các nhãn thành nhiều nhãn nếu có dấu cách\nlabel_split.head()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trans_label = MultiLabelBinarizer().fit(label_split)\nlabels = pd.DataFrame(trans_label.transform(label_split), columns=trans_label.classes_)\n\nlabels.sum().plot.bar(title='Target Class Distribution');","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.sum(axis=1).value_counts().plot.bar(title='Distribution of Number of Labels per Image');","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fig1 = plt.figure(figsize=(26,10))\n\n# for i in range(1, 13):\n    \n#     rand =  random.randrange(1, 18000)\n#     sample = os.path.join('../input/plant-pathology-2021-fgvc8/train_images', train['image'][rand])\n    \n#     img = PIL.Image.open(sample)\n    \n#     ax = fig1.add_subplot(4,3,i)\n#     ax.imshow(img)\n    \n#     title = f\"{train['labels'][rand]}{img.size}\"\n#     plt.title(title)\n    \n#     fig1.tight_layout()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"size các ảnh khác nhau\n","metadata":{}},{"cell_type":"markdown","source":"# Preprocessing and Augmentation","metadata":{}},{"cell_type":"code","source":"\ntrain_df = pd.concat([train['image'], labels], axis=1)\ntrain_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"w_target = 256\nh_target = 256\nbatch_size = 32","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_data_generator = tf.keras.preprocessing.image.ImageDataGenerator(\n    horizontal_flip=True,\n    vertical_flip=True,\n    rotation_range=5,\n    zoom_range=0.1,\n    shear_range=0.05,\n    validation_split=0.03,\n    rescale=1./255)\n\ntrain_generator = image_data_generator.flow_from_dataframe(\n    dataframe=train_df,\n    directory='../input/resized-plant2021/img_sz_512',\n    x_col='image',\n    y_col=train_df.columns.tolist()[1:],\n    class_mode='raw',\n    color_mode=\"rgb\",\n    target_size=(h_target, w_target),\n    batch_size=batch_size,\n    subset='training'\n)\n\nvalid_generator = image_data_generator.flow_from_dataframe(\n    dataframe=train_df,\n    directory='../input/resized-plant2021/img_sz_512',\n    x_col='image',\n    y_col=train_df.columns.tolist()[1:],\n    class_mode='raw',\n    color_mode=\"rgb\",\n    target_size=(h_target, w_target),\n    batch_size=batch_size,\n    subset='validation'\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q efficientnet\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modelling","metadata":{}},{"cell_type":"code","source":"# inputs = tf.keras.Input(shape=(h_target, w_target, 3)\nEffnet=efn.EfficientNetB4(\n        include_top=False,\n        input_shape=(h_target,w_target, 3),\n        weights='noisy-student',\n        pooling='avg')\n# Effnet= tf.keras.applications.EfficientNetB4(weights='imagenet',include_top=False, input_shape=(h_target,w_target, 3))\nx = Effnet.output\n# for layer in VGG_16.layers:\n#     layer.trainable=False\n# x = tf.keras.layers.GlobalAveragePooling2D()(x)\n# x = tf.keras.layers.Flatten()(x)\n# x = tf.keras.layers.Dropout(0.8)(x)\noutputs = tf.keras.layers.Dense(6, activation='sigmoid')(x)\n\nmodel = tf.keras.models.Model(inputs=Effnet.input, outputs=outputs)\n\nmodel.summary()\ntf.keras.utils.plot_model(model, show_shapes=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1 = tfa.metrics.F1Score(num_classes=6, average='micro', threshold=0.5)\n\nrlp = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss',mode='min', patience=2, verbose=1, factor=0.01)\nes = tf.keras.callbacks.EarlyStopping(monitor='val_loss',mode='min', patience=5, verbose=1, restore_best_weights=True)\n\nmodel.compile(loss='binary_crossentropy', optimizer=keras.optimizers.Adam(lr=0.0001), \n              metrics= [f1])\n\nhistory = model.fit(train_generator,validation_data=valid_generator, epochs=10, callbacks=[rlp,es])\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fix, ax = plt.subplots(figsize=(20, 6))\npd.DataFrame(history.history)[['loss', 'val_loss']].plot(ax=ax, title='Model Loss Curve')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='binary_crossentropy', optimizer=keras.optimizers.Adam(lr=0.0014), \n              metrics= [f1])\nhistory = model.fit(train_generator,validation_data=valid_generator, epochs=25, callbacks=[rlp,es])\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fix, ax = plt.subplots(figsize=(20, 6))\npd.DataFrame(history.history)[['loss','val_loss']].plot(ax=ax, title='Model Loss Curve')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submissions = pd.read_csv('../input/plant-pathology-2021-fgvc8/sample_submission.csv')\nsubmissions.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_generator = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255)\n\ntest_generator = test_data_generator.flow_from_dataframe(\n    submissions,\n    directory = '../input/plant-pathology-2021-fgvc8/test_images',\n    x_col=\"image\",\n    y_col=None,\n    target_size=(h_target, w_target),\n    color_mode=\"rgb\",\n    classes=None,\n    class_mode=None,\n    shuffle=False,\n    batch_size=batch_size\n)\n\npreds = model.predict(test_generator)\nprint(preds)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"thresh = 0.5\nfor i in range(3):\n    submissions.iloc[i, 1] = ' '.join(train_df.columns[1:][preds[i] >= thresh])\n    if submissions['labels'][i] == '':\n        submissions['labels'][i] = ' '.join(train_df.columns[1:][preds[i] >= np.max(preds[i])])\n        \nsubmissions.to_csv('submission.csv', index=False)    \n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submissions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"model.save('EffnetB4.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}