{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nimport cv2\nfrom tqdm import tqdm\nimport os\nimport PIL\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\n\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization, GlobalAveragePooling2D\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\n\nfrom keras_preprocessing.image import ImageDataGenerator\nfrom keras.utils.np_utils import to_categorical\nfrom keras.preprocessing.image import img_to_array\n\nfrom sklearn.metrics import classification_report, accuracy_score\nfrom sklearn.preprocessing import MultiLabelBinarizer, LabelBinarizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/plant-pathology-2021-fgvc8/train.csv')\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['labels'].sort_values().value_counts().plot.bar()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['labels'] = df['labels'].apply(lambda s: s.split(' '))\ndf[:10]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = list(df['labels'])\nmlb = MultiLabelBinarizer()\ntrainx = pd.DataFrame(mlb.fit_transform(s), columns=mlb.classes_, index=df.index)\nprint(trainx.columns)\nprint(trainx.sum())\n\nlabels = list(trainx.sum().keys())\nprint(labels)\nlabel_counts = trainx.sum().values.tolist()\n\nfig, ax = plt.subplots(1,1, figsize=(10,6))\n\nsns.barplot(x= labels, y= label_counts, ax=ax)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen = ImageDataGenerator(\n    rescale = 1/255.0,\n    validation_split= 0.2,\n    rotation_range=5,\n    zoom_range=0.1,\n    shear_range=0.05,\n    horizontal_flip=True,\n)\nbsize = 32","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = datagen.flow_from_dataframe(\n    df,\n    directory = '../input/resized-plant2021/img_sz_256',\n    x_col = 'image',\n    y_col = 'labels',\n    subset=\"training\",\n    color_mode=\"rgb\",\n    target_size = (224,224),\n    class_mode=\"categorical\",\n    batch_size=bsize,\n    shuffle=True,\n    seed=40,\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_data = datagen.flow_from_dataframe(\n    df,\n    directory = '../input/resized-plant2021/img_sz_256',\n    x_col = 'image',\n    y_col = 'labels',\n    subset=\"validation\",\n    color_mode=\"rgb\",\n    target_size = (224,224),\n    class_mode=\"categorical\",\n    batch_size=bsize,\n    shuffle=True,\n    seed=40,\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accname = 'f1_score'\n\ndef plot_history(history): \n    fig, ax1 = plt.subplots()\n    \n    ax1.plot(history.history['loss'], 'r', label=\"training loss ({:.6f})\".format(history.history['loss'][-1]))\n    ax1.plot(history.history['val_loss'], 'r--', label=\"validation loss ({:.6f})\".format(history.history['val_loss'][-1]))\n    ax1.grid(True)\n    ax1.set_xlabel('iteration')\n    ax1.legend(loc=\"best\", fontsize=9)    \n    ax1.set_ylabel('loss', color='r')\n    ax1.tick_params('y', colors='r')\n\n    if accname in history.history:\n        ax2 = ax1.twinx()\n\n        ax2.plot(history.history[accname], 'b', label=\"training f1_score ({:.4f})\".format(history.history[accname][-1]))\n        ax2.plot(history.history['val_'+accname], 'b--', label=\"validation f1_score ({:.4f})\".format(history.history['val_'+accname][-1]))\n\n        ax2.legend(loc=\"lower right\", fontsize=9)\n        ax2.set_ylabel('acc', color='b')        \n        ax2.tick_params('y', colors='b')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weight_path = '../input/dense121-weight/densenet121_weights_tf_dim_ordering_tf_kernels_notop.h5'\nbase_model = tf.keras.applications.DenseNet121(weights=weight_path, include_top=False, pooling='avg')\nx = base_model.output\n#fully connected layer\nx = Dense(512, activation='relu')(x)\nx = Dense(256, activation='relu')(x)\n# finally, the softmax for the classifier \npredictions = Dense(6, activation='sigmoid')(x)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model(inputs=base_model.input, outputs = predictions)\nimport tensorflow_addons as tfa\nimport keras \nf1 = tfa.metrics.F1Score(num_classes=6, average='macro')\nmodel.compile(optimizer= tf.keras.optimizers.Adam(), \n              loss='binary_crossentropy', \n              metrics=[f1]\n             )\n# model.compile(optimizer=keras.optimizers.Adam(lr=0.03), \n#                  loss='binary_crossentropy', \n#                  metrics=[f1]\n#                 )\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accEarlyStop = keras.callbacks.EarlyStopping(\n    monitor=f1,     # look at the validation loss tf2.0 accuracy\n    min_delta=0.02,       # threshold to consider as no change\n    patience=5,             # stop if  epochs with no change\n    verbose=1, \n    mode='max', \n    restore_best_weights= True\n)\nlossEarlyStop = keras.callbacks.EarlyStopping(\n    monitor='val_loss',     # look at the validation loss tf2.0 accuracy\n    min_delta=0.02,       # threshold to consider as no change\n    patience=5,             # stop if  epochs with no change\n    verbose=1, \n    mode='min', \n    restore_best_weights= True\n)\nlrschedule = keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss', \n    factor=0.05, \n    patience=5, \n    verbose=1\n)\ncallbacks_list = [lrschedule]\n# callbacks_list = []","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(\n            train_data,  # data from generator\n             #steps_per_epoch=1,    # should be number of batches per epoch\n            epochs=10,\n            callbacks=callbacks_list, \n            validation_data=valid_data, \n#             validation_steps = 1,\n            verbose=True\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_history(history)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss, f1score = model.evaluate_generator(valid_data,verbose=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# _model = tf.keras.models.load_model('../input/dense121modelver2/dense121_model_ver2.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def str2max(list_str,threshold):\n    max_id =[]\n    for i in list_str:\n        if i > threshold:\n            max_id.append(1)\n        else:\n            max_id.append(0)\n    return max_id\n\ndef evaluate(data,threshold):\n    score_dic={}\n    total, right = 0., 0.\n    positive_i=0.0\n    count = 0\n    for x_true, y_true in tqdm(data):\n        count = count +1\n        a = model.predict(x_true)\n        y_pred_list=[]\n        for i in a:\n            label=str2max(i,threshold)\n            label=np.array(label)\n            y_pred_list.append(label)\n        y_pred_list=np.array(y_pred_list)\n        for i,j in zip(y_true.tolist(),y_pred_list.tolist()):\n            total+=1\n            if i==j:\n                right+=1\n        if (count == math.ceil(180/bsize)):\n            break\n    score_dic['acc']=right/total\n    score_dic['correct']=right\n    score_dic['total']=total\n    return score_dic","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(np.linspace(0,1.0,num=20))\nthresholds = np.linspace(0,1.0,num=20)\nthresholds[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc_list = []\nfor threshold in np.arange(0,1.0,0.05):\n    acc = evaluate(valid_data,threshold)['acc']\n    acc_list.append(acc)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_acc = max(acc_list)\nmax_index = acc_list.index(max_acc)\n# thresholds = np.linspace(0,1.0,num=20)\nthresholds = np.arange(0,1.0,0.05)\n\nbest_threshold = thresholds[max_index]\nprint(\"best threshold is {} with the acc {}\".format(best_threshold,max_acc))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_threshold","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.save('dense121_model_ver2.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# _model.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss, f1score = model.evaluate_generator(valid_data,verbose=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv('../input/plant-pathology-2021-fgvc8/sample_submission.csv')\nsample_sub.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for img_name in tqdm(sample_sub['image']):\n    print(img_name)\n    path = '../input/plant-pathology-2021-fgvc8/test_images/'+str(img_name)\n    with PIL.Image.open(path) as img:\n        img = img.resize((256,256))\n        img.save(f'./{img_name}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_datagen = ImageDataGenerator(\n    rescale = 1/255.0\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = test_datagen.flow_from_dataframe(\n    sample_sub,\n    directory='../input/plant-pathology-2021-fgvc8/test_images',\n    x_col='image',\n    y_col=None,\n    class_mode=None,\n    color_mode=\"rgb\",\n    target_size=(224,224),\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test_data)\nprint(preds)\npreds = preds.tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indices = []\nfor pred in preds:\n    temp = []\n    for category in pred:\n        if category>=best_threshold:\n            temp.append(pred.index(category))\n    if temp!=[]:\n        indices.append(temp)\n    else:\n        temp.append(np.argmax(pred))\n        indices.append(temp)\n    \nprint(indices)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = (train_data.class_indices)\nlabels = dict((v,k) for k,v in labels.items())\nprint(labels)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testlabels = []\n\n\nfor image in indices:\n    temp = []\n    for i in image:\n        temp.append(str(labels[i]))\n    testlabels.append(' '.join(temp))\n\nprint(testlabels)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/plant-pathology-2021-fgvc8/sample_submission.csv')\nsub['labels'] = testlabels\nsub.to_csv('submission.csv', index=False)\nsub","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}