{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport pandas as pd\nimport skimage\nimport os\nimport matplotlib.pyplot as plt\nimport imageio\nimport numpy as np\nimport skimage.io\nimport skimage.transform\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Conv2D, Flatten, MaxPool2D, Dropout, BatchNormalization,LeakyReLU, Activation, MaxPooling2D, GlobalAveragePooling2D\nfrom sklearn.metrics import classification_report\nfrom scipy.ndimage.filters import convolve\nfrom sklearn.model_selection import train_test_split\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import ResNet50\n\nfrom tensorflow.keras.applications.efficientnet import EfficientNetB4\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom sklearn.preprocessing import MultiLabelBinarizer\nimport time\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":7.947106,"end_time":"2021-12-21T18:35:42.104318","exception":false,"start_time":"2021-12-21T18:35:34.157212","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:19.488471Z","iopub.execute_input":"2021-12-22T15:13:19.488751Z","iopub.status.idle":"2021-12-22T15:13:19.497553Z","shell.execute_reply.started":"2021-12-22T15:13:19.488717Z","shell.execute_reply":"2021-12-22T15:13:19.496897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pre-final notebook\n\nIn this notebook I make big improvements compare to previous versions.\n\nBelow you can find:\n* fisrt look at data\n* some info about classes and their balance\n* blanks for data processing\n* blank model\n* correct submission sender for this competition\n\n+\n\n* data processing\n* better model\n\nFor data processing I use image augmentation.\nAlso I added some layers for models.\n","metadata":{"papermill":{"duration":0.038395,"end_time":"2021-12-21T18:35:42.181352","exception":false,"start_time":"2021-12-21T18:35:42.142957","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"For future work:\n\n* models for high leaderboard score","metadata":{"papermill":{"duration":0.038339,"end_time":"2021-12-21T18:35:42.257904","exception":false,"start_time":"2021-12-21T18:35:42.219565","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### Lets take first look on data\n\n","metadata":{"papermill":{"duration":0.037863,"end_time":"2021-12-21T18:35:42.336237","exception":false,"start_time":"2021-12-21T18:35:42.298374","status":"completed"},"tags":[]}},{"cell_type":"code","source":"data = pd.read_csv('../input/plant-pathology-2021-fgvc8/train.csv', index_col=False, dtype={'labels':'category'})\ndata.head()","metadata":{"papermill":{"duration":0.113451,"end_time":"2021-12-21T18:35:42.487672","exception":false,"start_time":"2021-12-21T18:35:42.374221","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:19.502771Z","iopub.execute_input":"2021-12-22T15:13:19.502994Z","iopub.status.idle":"2021-12-22T15:13:19.572544Z","shell.execute_reply.started":"2021-12-22T15:13:19.502964Z","shell.execute_reply":"2021-12-22T15:13:19.571864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We got 2 columns: 'image' and 'labels'\n\n'image' consists of names of images, 'labels' has linked diseases to the pictures\n\nNow lets look to the target","metadata":{"papermill":{"duration":0.037471,"end_time":"2021-12-21T18:35:42.564508","exception":false,"start_time":"2021-12-21T18:35:42.527037","status":"completed"},"tags":[]}},{"cell_type":"code","source":"img_exists = data['image'].apply(lambda f: os.path.exists('/kaggle/input/plant-pathology-2021-fgvc8/train_images/' + f))\ndata = data[img_exists]\nimg_folder = '/kaggle/input/plant-pathology-2021-fgvc8/train_images/'\n\nf_figsize = (16,5)\n\ndiseases = data['labels'].cat.categories\ndiseases","metadata":{"papermill":{"duration":15.404899,"end_time":"2021-12-21T18:35:58.006736","exception":false,"start_time":"2021-12-21T18:35:42.601837","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:19.574407Z","iopub.execute_input":"2021-12-22T15:13:19.574721Z","iopub.status.idle":"2021-12-22T15:13:31.072499Z","shell.execute_reply.started":"2021-12-22T15:13:19.574685Z","shell.execute_reply":"2021-12-22T15:13:31.071838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is 12 labels types, but they are not clear. Sometimes we got twinned label like 'rust complex'.\n\nWe want to split this labels to make it better for model, but later.\n\nNow lets use seminar code to show some pictures from dataset","metadata":{"papermill":{"duration":0.03831,"end_time":"2021-12-21T18:35:58.08295","exception":false,"start_time":"2021-12-21T18:35:58.04464","status":"completed"},"tags":[]}},{"cell_type":"code","source":"f, ax = plt.subplots(nrows=1,ncols=diseases.size - 1, figsize=f_figsize)\n\n# Draw the first found bee of given subpecies\ni=0\nfor s in diseases:\n    if s == 'healthy': continue\n    file = img_folder + data[data['labels']==s].iloc[0]['image']\n    im=imageio.imread(file)\n    ax[i].imshow(im, resample=True)\n    ax[i].set_title(s, fontsize=8)\n    i+=1\n    \nplt.suptitle(\"Plant diseases\")\nplt.tight_layout()\nplt.show()\n\n# Sample some healthy objects to have a look at\nncols = 5\nhealthy = data[data['labels'] == 'healthy'].sample(ncols)\nf, ax = plt.subplots(nrows=1,ncols=ncols, figsize=f_figsize)\n\nfor i in range(0, ncols): \n    file = img_folder + healthy.iloc[i]['image']\n    ax[i].imshow(imageio.imread(file))\n\nplt.suptitle(\"Healthy plants\")\nplt.tight_layout()\nplt.show()\n\nhealth_cats = data['labels'].cat.categories\nf, ax = plt.subplots(1, health_cats.size-1, figsize=f_figsize)\n\n# Draw the first found bee with a particulat health issue\ni=0\nfor c in health_cats:\n    if c == 'healthy': continue\n    bee = data[data['labels'] == c].sample(1).iloc[0]\n    ax[i].imshow(imageio.imread(img_folder + bee['image']))\n    ax[i].set_title(bee['labels'], fontsize=8)\n    i += 1\n    \nplt.suptitle(\"Sick Bees\")    \nplt.tight_layout()\nplt.show()","metadata":{"papermill":{"duration":25.365883,"end_time":"2021-12-21T18:36:23.487148","exception":false,"start_time":"2021-12-21T18:35:58.121265","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:31.073881Z","iopub.execute_input":"2021-12-22T15:13:31.074305Z","iopub.status.idle":"2021-12-22T15:13:53.377504Z","shell.execute_reply.started":"2021-12-22T15:13:31.074268Z","shell.execute_reply":"2021-12-22T15:13:53.376820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Of course there are imbalanced classes. But we would not do something with it.","metadata":{"papermill":{"duration":0.05373,"end_time":"2021-12-21T18:36:23.595287","exception":false,"start_time":"2021-12-21T18:36:23.541557","status":"completed"},"tags":[]}},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nplt.barh(diseases, data['labels'].value_counts())\nplt.tight_layout()\nplt.show()","metadata":{"papermill":{"duration":0.293885,"end_time":"2021-12-21T18:36:23.942333","exception":false,"start_time":"2021-12-21T18:36:23.648448","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:53.378549Z","iopub.execute_input":"2021-12-22T15:13:53.378901Z","iopub.status.idle":"2021-12-22T15:13:53.639738Z","shell.execute_reply.started":"2021-12-22T15:13:53.378866Z","shell.execute_reply":"2021-12-22T15:13:53.639093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So now lets transform classes in indicators of diseases. ","metadata":{"papermill":{"duration":0.053728,"end_time":"2021-12-21T18:36:24.051134","exception":false,"start_time":"2021-12-21T18:36:23.997406","status":"completed"},"tags":[]}},{"cell_type":"code","source":"classes = data.labels.apply(lambda x : x.split())\nmulti = MultiLabelBinarizer().fit(classes)\nlabels = pd.DataFrame(multi.transform(classes), columns = multi.classes_)\n\nlabels = pd.concat([data['image'], labels], axis=1)\nlabels.head()","metadata":{"papermill":{"duration":0.446027,"end_time":"2021-12-21T18:36:24.551798","exception":false,"start_time":"2021-12-21T18:36:24.105771","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:53.640805Z","iopub.execute_input":"2021-12-22T15:13:53.642178Z","iopub.status.idle":"2021-12-22T15:13:53.976053Z","shell.execute_reply.started":"2021-12-22T15:13:53.642138Z","shell.execute_reply":"2021-12-22T15:13:53.975247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is our final version of labels for dataset.","metadata":{"papermill":{"duration":0.055,"end_time":"2021-12-21T18:36:24.661986","exception":false,"start_time":"2021-12-21T18:36:24.606986","status":"completed"},"tags":[]}},{"cell_type":"code","source":"labels.columns[1:]","metadata":{"papermill":{"duration":0.066244,"end_time":"2021-12-21T18:36:24.784028","exception":false,"start_time":"2021-12-21T18:36:24.717784","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:53.977479Z","iopub.execute_input":"2021-12-22T15:13:53.977722Z","iopub.status.idle":"2021-12-22T15:13:53.984337Z","shell.execute_reply.started":"2021-12-22T15:13:53.977689Z","shell.execute_reply":"2021-12-22T15:13:53.983464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"Lets use some common seminar code for splitting.","metadata":{"papermill":{"duration":0.055043,"end_time":"2021-12-21T18:36:24.894711","exception":false,"start_time":"2021-12-21T18:36:24.839668","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def split(data):\n    # Split to train and test before balancing\n    train_data, test_data = train_test_split(data, test_size=0.01, random_state=24)\n\n    # Split train to train and validation datasets\n    train_data, val_data = train_test_split(train_data, test_size=0.1, random_state=24)\n\n    return(train_data, val_data, test_data)","metadata":{"papermill":{"duration":0.064938,"end_time":"2021-12-21T18:36:25.016768","exception":false,"start_time":"2021-12-21T18:36:24.95183","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:53.985788Z","iopub.execute_input":"2021-12-22T15:13:53.986070Z","iopub.status.idle":"2021-12-22T15:13:53.992992Z","shell.execute_reply.started":"2021-12-22T15:13:53.986023Z","shell.execute_reply":"2021-12-22T15:13:53.992088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_plants_bal, val_plants, test_plants = split(labels)","metadata":{"papermill":{"duration":0.069878,"end_time":"2021-12-21T18:36:25.142367","exception":false,"start_time":"2021-12-21T18:36:25.072489","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:53.996118Z","iopub.execute_input":"2021-12-22T15:13:53.996687Z","iopub.status.idle":"2021-12-22T15:13:54.009554Z","shell.execute_reply.started":"2021-12-22T15:13:53.996653Z","shell.execute_reply":"2021-12-22T15:13:54.008780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train_plants_bal), len(val_plants), len(test_plants))","metadata":{"papermill":{"duration":0.072907,"end_time":"2021-12-21T18:36:25.270135","exception":false,"start_time":"2021-12-21T18:36:25.197228","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:54.012302Z","iopub.execute_input":"2021-12-22T15:13:54.012519Z","iopub.status.idle":"2021-12-22T15:13:54.018640Z","shell.execute_reply.started":"2021-12-22T15:13:54.012489Z","shell.execute_reply":"2021-12-22T15:13:54.017778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_plants_bal[train_plants_bal[train_plants_bal.columns[1]] == 1])\n\ncount = np.zeros((len(train_plants_bal.columns) - 1))\nfor i in range(1, len(train_plants_bal.columns)):\n    count[i - 1] = len(train_plants_bal[train_plants_bal[train_plants_bal.columns[i]] == 1])\n\nprint(count)","metadata":{"papermill":{"duration":0.072941,"end_time":"2021-12-21T18:36:25.401646","exception":false,"start_time":"2021-12-21T18:36:25.328705","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:54.019936Z","iopub.execute_input":"2021-12-22T15:13:54.020359Z","iopub.status.idle":"2021-12-22T15:13:54.033318Z","shell.execute_reply.started":"2021-12-22T15:13:54.020261Z","shell.execute_reply":"2021-12-22T15:13:54.032295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nplt.barh(labels.columns[1:], count)\nplt.tight_layout()\nplt.show()","metadata":{"papermill":{"duration":0.233833,"end_time":"2021-12-21T18:36:25.692657","exception":false,"start_time":"2021-12-21T18:36:25.458824","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:54.035043Z","iopub.execute_input":"2021-12-22T15:13:54.035477Z","iopub.status.idle":"2021-12-22T15:13:54.250456Z","shell.execute_reply.started":"2021-12-22T15:13:54.035442Z","shell.execute_reply":"2021-12-22T15:13:54.249817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we can continue to data reading and models.\n\nBelow there are some constants, that we will use.","metadata":{"papermill":{"duration":0.057569,"end_time":"2021-12-21T18:36:25.811504","exception":false,"start_time":"2021-12-21T18:36:25.753935","status":"completed"},"tags":[]}},{"cell_type":"code","source":"IMAGE_WIDTH, IMAGE_HEIGHT = 128, 128\nIMAGE_SIZE = [IMAGE_WIDTH, IMAGE_HEIGHT]\nKERNEL_SIZE = 3\nIMAGE_CHANNELS = 3\nRANDOM_STATE = 1337\nN_EPOCH = 30\nBATCH_SIZE = 64\nMAX_POOL_DIM = 2\ntarget = ['complex', 'frog_eye_leaf_spot', 'healthy', 'powdery_mildew', 'rust', 'scab']","metadata":{"papermill":{"duration":0.067434,"end_time":"2021-12-21T18:36:25.936908","exception":false,"start_time":"2021-12-21T18:36:25.869474","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:54.251875Z","iopub.execute_input":"2021-12-22T15:13:54.252347Z","iopub.status.idle":"2021-12-22T15:13:54.258049Z","shell.execute_reply.started":"2021-12-22T15:13:54.252308Z","shell.execute_reply":"2021-12-22T15:13:54.257338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Also modified read_img function from seminar.\n\nIt is modified for faster reading using tensorflow","metadata":{"papermill":{"duration":0.058375,"end_time":"2021-12-21T18:36:26.054494","exception":false,"start_time":"2021-12-21T18:36:25.996119","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def read_img(file, sat_factor = 1.5, cont_factor = 1.5, img_folder='/kaggle/input/plant-pathology-2021-fgvc8/train_images/'):    \n    image = tf.io.read_file(img_folder + file)\n    image = tf.io.decode_jpeg(image, channels=IMAGE_CHANNELS)\n    image = tf.image.convert_image_dtype(image, tf.float32)\n    image = tf.image.resize(image, IMAGE_SIZE)\n    image = tf.image.adjust_saturation(image, sat_factor)\n    image = tf.image.adjust_contrast(image, cont_factor)\n    return image\n","metadata":{"papermill":{"duration":0.067803,"end_time":"2021-12-21T18:36:26.180515","exception":false,"start_time":"2021-12-21T18:36:26.112712","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:54.259524Z","iopub.execute_input":"2021-12-22T15:13:54.260050Z","iopub.status.idle":"2021-12-22T15:13:54.267686Z","shell.execute_reply.started":"2021-12-22T15:13:54.259998Z","shell.execute_reply":"2021-12-22T15:13:54.266951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Name: 8a9f2f485e20ade9.jpg\nLocation: I:/Maga/MLDM/train_images\nDimensions: 4000 x 2672\nFile Size: 963KB (986,924)\n\n\nName: 8aa78fd5c6c0cec2.jpg\nLocation: I:/Maga/MLDM/train_images\nDimensions: 4000 x 2672\nFile Size: 941KB (964,093)\n\n","metadata":{}},{"cell_type":"code","source":"img  = read_img('8aa78fd5c6c0cec2.jpg', 1, 1)\nplt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-22T15:13:54.269117Z","iopub.execute_input":"2021-12-22T15:13:54.269621Z","iopub.status.idle":"2021-12-22T15:13:56.864524Z","shell.execute_reply.started":"2021-12-22T15:13:54.269585Z","shell.execute_reply":"2021-12-22T15:13:56.863854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img  = read_img('8aa78fd5c6c0cec2.jpg')\nplt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-22T15:13:56.865548Z","iopub.execute_input":"2021-12-22T15:13:56.865919Z","iopub.status.idle":"2021-12-22T15:13:57.118506Z","shell.execute_reply.started":"2021-12-22T15:13:56.865882Z","shell.execute_reply":"2021-12-22T15:13:57.117850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.057812,"end_time":"2021-12-21T18:36:26.297356","exception":false,"start_time":"2021-12-21T18:36:26.239544","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Below we will do something with images in ImageGenerator.\n\nNow lets use ImageGenerator.\n\nFirstly, I need to zoom image, because the leaf on most pictures is in the center. And some diseases are same color as plant branches, which are near the edge usually.\n\nAlso lets do some common flips, rotations and shifts.","metadata":{"papermill":{"duration":0.057185,"end_time":"2021-12-21T18:36:26.412753","exception":false,"start_time":"2021-12-21T18:36:26.355568","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def prepare2train(train_plants, val_plants, test_plants, target):\n\n\n    print(\"Started train\")\n    train_X = np.stack(train_plants['image'].apply(read_img))\n    train_y  = train_plants[target]\n\n\n    print(\"Started val\")\n    val_X = np.stack(val_plants['image'].apply(read_img))\n    val_y = val_plants[target]\n\n\n    print(\"Started test\")\n    test_X = np.stack(test_plants['image'].apply(read_img))\n    test_y = test_plants[target]\n\n\n    generator = ImageDataGenerator(\n            featurewise_center=False,\n            samplewise_center=False,\n            featurewise_std_normalization=False,\n            samplewise_std_normalization=False,\n            rotation_range=25, \n            zoom_range = 0.15,\n            width_shift_range=0.15,\n            height_shift_range=0.15,\n            horizontal_flip=True,\n            vertical_flip=True)\n    print(\"Started generator\")\n    generator.fit(train_X)\n    return (generator, train_X, val_X, test_X, train_y, val_y, test_y)","metadata":{"papermill":{"duration":0.071015,"end_time":"2021-12-21T18:36:26.541645","exception":false,"start_time":"2021-12-21T18:36:26.47063","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T15:13:57.119660Z","iopub.execute_input":"2021-12-22T15:13:57.120131Z","iopub.status.idle":"2021-12-22T15:13:57.129009Z","shell.execute_reply.started":"2021-12-22T15:13:57.120096Z","shell.execute_reply":"2021-12-22T15:13:57.128352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So 128*128 is small. Not everything can be noticed, but there is restrictions on memory, so we will try to do best.\n\nOn the other hand, there is ImageDataGenerator with flow_from_... but it is very slow","metadata":{"papermill":{"duration":0.060206,"end_time":"2021-12-21T18:36:26.661459","exception":false,"start_time":"2021-12-21T18:36:26.601253","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Save the best model during the traning\ncheckpointer1 = keras.callbacks.ModelCheckpoint('best_model1.h5',\n                                                monitor='val_accuracy',\n                                                verbose=1,\n                                                save_best_only=True,\n                                                save_weights_only=True)\nnet = EfficientNetB4(weights=\"../input/eff-model/efficientnetb4_notop.h5\", include_top=False, input_shape=(*IMAGE_SIZE,IMAGE_CHANNELS))\n#resnet.save_weights('ResNet.h5')\nnet.trainable = True\nmodel1 = Sequential()\nmodel1.add(net)\nmodel1.add(GlobalAveragePooling2D())\nmodel1.add(Dropout(0.2))\nmodel1.add(Dense(len(target), activation='sigmoid'))\nmodel1.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-12-22T15:14:16.674361Z","iopub.execute_input":"2021-12-22T15:14:16.674612Z","iopub.status.idle":"2021-12-22T15:14:22.880545Z","shell.execute_reply.started":"2021-12-22T15:14:16.674582Z","shell.execute_reply":"2021-12-22T15:14:22.879832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1.summary()","metadata":{"execution":{"iopub.status.busy":"2021-12-22T15:14:22.881868Z","iopub.execute_input":"2021-12-22T15:14:22.882171Z","iopub.status.idle":"2021-12-22T15:14:22.915741Z","shell.execute_reply.started":"2021-12-22T15:14:22.882127Z","shell.execute_reply":"2021-12-22T15:14:22.914930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"generator, train_X, val_X, test_X, train_y, val_y, test_y = prepare2train(train_plants_bal, val_plants, test_plants, target)","metadata":{"papermill":{"duration":1891.544396,"end_time":"2021-12-21T19:07:58.264141","exception":false,"start_time":"2021-12-21T18:36:26.719745","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T10:04:57.532940Z","iopub.execute_input":"2021-12-22T10:04:57.533634Z","iopub.status.idle":"2021-12-22T10:07:41.077301Z","shell.execute_reply.started":"2021-12-22T10:04:57.533588Z","shell.execute_reply":"2021-12-22T10:07:41.075062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we have data. Now let's try model with layers like VGG16 layers. They can give good results.\n\nDue to target format we will use 'sigmoid' at last Dense layer and 'binary_crossentropy' loss","metadata":{"execution":{"iopub.execute_input":"2021-12-20T13:58:22.309089Z","iopub.status.busy":"2021-12-20T13:58:22.308852Z","iopub.status.idle":"2021-12-20T13:58:22.314173Z","shell.execute_reply":"2021-12-20T13:58:22.312882Z","shell.execute_reply.started":"2021-12-20T13:58:22.309055Z"},"papermill":{"duration":0.059354,"end_time":"2021-12-21T19:07:58.3828","exception":false,"start_time":"2021-12-21T19:07:58.323446","status":"completed"},"tags":[]}},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.709915,"end_time":"2021-12-21T19:07:59.15309","exception":false,"start_time":"2021-12-21T19:07:58.443175","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.076747,"end_time":"2021-12-21T19:07:59.289151","exception":false,"start_time":"2021-12-21T19:07:59.212404","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Remember, that we have 6 chunks, so we need to iterate over them to train model.","metadata":{"papermill":{"duration":0.059777,"end_time":"2021-12-21T19:07:59.409725","exception":false,"start_time":"2021-12-21T19:07:59.349948","status":"completed"},"tags":[]}},{"cell_type":"code","source":"training1 = model1.fit_generator(generator.flow(train_X,train_y, batch_size=BATCH_SIZE),\n                                 epochs=N_EPOCH,\n                                 validation_data=(val_X, val_y),\n                                 callbacks=[checkpointer1])\n# Get the best saved weights\nmodel1.load_weights('best_model1.h5')","metadata":{"papermill":{"duration":2526.897026,"end_time":"2021-12-21T19:50:06.366899","exception":false,"start_time":"2021-12-21T19:07:59.469873","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T10:10:11.733369Z","iopub.execute_input":"2021-12-22T10:10:11.733662Z","iopub.status.idle":"2021-12-22T10:13:01.231157Z","shell.execute_reply.started":"2021-12-22T10:10:11.733631Z","shell.execute_reply":"2021-12-22T10:13:01.229897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Also I use very usefull evaluation code from seminar. \n\nIt helps very good to check some results about model.","metadata":{"papermill":{"duration":0.809028,"end_time":"2021-12-21T19:50:07.87934","exception":false,"start_time":"2021-12-21T19:50:07.070312","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def eval_model(training, model, test_X, test_y, target):\n    \n    ## Trained model analysis and evaluation\n    f, ax = plt.subplots(2,1, figsize=(6,5))\n    ax[0].plot(training.history['loss'], label=\"Loss\")\n    ax[0].plot(training.history['val_loss'], label=\"Validation loss\")\n    ax[0].set_title('%s: loss' % target)\n    ax[0].set_xlabel('Epoch')\n    ax[0].set_ylabel('Loss')\n    ax[0].legend()\n    \n    # Accuracy\n    ax[1].plot(training1.history['accuracy'], label=\"Accuracy\")\n    ax[1].plot(training1.history['val_accuracy'], label=\"Validation accuracy\")\n    ax[1].set_title('%s: accuracy' % target)\n    ax[1].set_xlabel('Epoch')\n    ax[1].set_ylabel('Accuracy')\n    ax[1].legend()\n    plt.tight_layout()\n    plt.show()\n\n    # Accuracy by subspecies\n    test_pred = model.predict(test_X)\n    print(test_pred)\n    acc_by_subspecies = np.logical_and((test_pred > 0.5), test_y).sum()/test_y.sum()\n    acc_by_subspecies.plot(kind='bar', title='Accuracy by %s' % target)\n    plt.ylabel('Accuracy')\n    plt.show()\n\n    # Print metrics\n    print(\"Classification report\")\n    test_pred = np.argmax(test_pred, axis=1)\n    test_truth = np.argmax(test_y.values, axis=1)\n    #print(test_truth)\n    #print(test_pred)\n    print(classification_report(test_truth, test_pred, target_names=test_y.columns))\n\n    # Loss function and accuracy\n    test_res = model.evaluate(test_X, test_y.values, verbose=0)\n    print('Loss function: %s, accuracy:' % test_res[0], test_res[1])","metadata":{"papermill":{"duration":0.710045,"end_time":"2021-12-21T19:50:09.266116","exception":false,"start_time":"2021-12-21T19:50:08.556071","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T10:13:01.233444Z","iopub.execute_input":"2021-12-22T10:13:01.233804Z","iopub.status.idle":"2021-12-22T10:13:01.249700Z","shell.execute_reply.started":"2021-12-22T10:13:01.233759Z","shell.execute_reply":"2021-12-22T10:13:01.247972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#eval_model(training1, model1, test_X, test_y, ['complex', 'frog_eye_leaf_spot', 'healthy', 'powdery_mildew', 'rust', 'scab'])","metadata":{"papermill":{"duration":36.875254,"end_time":"2021-12-21T19:50:46.831399","exception":false,"start_time":"2021-12-21T19:50:09.956145","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T10:13:01.251012Z","iopub.execute_input":"2021-12-22T10:13:01.251326Z","iopub.status.idle":"2021-12-22T10:13:04.266383Z","shell.execute_reply.started":"2021-12-22T10:13:01.251283Z","shell.execute_reply":"2021-12-22T10:13:04.265185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We make simple model. Lets submit it and check results. ","metadata":{"papermill":{"duration":0.668325,"end_time":"2021-12-21T19:50:48.168133","exception":false,"start_time":"2021-12-21T19:50:47.499808","status":"completed"},"tags":[]}},{"cell_type":"code","source":"test_df = pd.read_csv('../input/plant-pathology-2021-fgvc8/sample_submission.csv')\ntest_path = \"/input/plant-pathology-2021-fgvc8/test_images\"\ntest_df","metadata":{"papermill":{"duration":0.812249,"end_time":"2021-12-21T19:50:49.64265","exception":false,"start_time":"2021-12-21T19:50:48.830401","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T10:13:10.742608Z","iopub.execute_input":"2021-12-22T10:13:10.742979Z","iopub.status.idle":"2021-12-22T10:13:10.765058Z","shell.execute_reply.started":"2021-12-22T10:13:10.742946Z","shell.execute_reply":"2021-12-22T10:13:10.764041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#kekw = read_img(test_df['image'][1], \"../input/plant-pathology-2021-fgvc8/test_images/\")","metadata":{"papermill":{"duration":0.675652,"end_time":"2021-12-21T19:50:50.980829","exception":false,"start_time":"2021-12-21T19:50:50.305177","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T10:13:12.126069Z","iopub.execute_input":"2021-12-22T10:13:12.126406Z","iopub.status.idle":"2021-12-22T10:13:12.131906Z","shell.execute_reply.started":"2021-12-22T10:13:12.126373Z","shell.execute_reply":"2021-12-22T10:13:12.130713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Submissions for this competition is allowed only from competition notebooks, so we need special saver for results.","metadata":{"execution":{"iopub.execute_input":"2021-12-20T14:17:22.135173Z","iopub.status.busy":"2021-12-20T14:17:22.134782Z","iopub.status.idle":"2021-12-20T14:17:22.140469Z","shell.execute_reply":"2021-12-20T14:17:22.139786Z","shell.execute_reply.started":"2021-12-20T14:17:22.135138Z"},"papermill":{"duration":0.660593,"end_time":"2021-12-21T19:50:52.312224","exception":false,"start_time":"2021-12-21T19:50:51.651631","status":"completed"},"tags":[]}},{"cell_type":"code","source":"submission = pd.read_csv('../input/plant-pathology-2021-fgvc8/sample_submission.csv')\nfor row in submission.index:\n\n    image = read_img(submission.loc[row,'image'],\n                    img_folder='/kaggle/input/plant-pathology-2021-fgvc8/test_images/').numpy().reshape((1, IMAGE_WIDTH, IMAGE_HEIGHT, IMAGE_CHANNELS))\n    \n    predict = model1.predict(image)[0]\n    print(predict)\n    predict = [1 if i>0.3 else 0 for i in predict]\n    if (np.sum(predict[1:]) >= 2):\n        predict[0] = 1\n    result = []\n    for i,j in enumerate(predict):\n        if j:\n            result.append(labels.columns.tolist()[i + 1])\n    result = ' '.join(result)\n    submission.loc[row,'labels'] = result\n\nsubmission.head()","metadata":{"papermill":{"duration":1.331862,"end_time":"2021-12-21T19:50:54.349124","exception":false,"start_time":"2021-12-21T19:50:53.017262","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T10:13:12.477088Z","iopub.execute_input":"2021-12-22T10:13:12.477880Z","iopub.status.idle":"2021-12-22T10:13:13.137686Z","shell.execute_reply.started":"2021-12-22T10:13:12.477845Z","shell.execute_reply":"2021-12-22T10:13:13.136599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.chdir(r'/kaggle/working')\n\nsubmission.to_csv('submission.csv', index=False)","metadata":{"papermill":{"duration":0.682225,"end_time":"2021-12-21T19:50:55.718179","exception":false,"start_time":"2021-12-21T19:50:55.035954","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-12-22T10:13:13.188693Z","iopub.execute_input":"2021-12-22T10:13:13.189573Z","iopub.status.idle":"2021-12-22T10:13:13.200193Z","shell.execute_reply.started":"2021-12-22T10:13:13.189537Z","shell.execute_reply":"2021-12-22T10:13:13.199036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{"papermill":{"duration":0.695273,"end_time":"2021-12-21T19:50:57.105881","exception":false,"start_time":"2021-12-21T19:50:56.410608","status":"completed"},"tags":[]}},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.execute_input":"2021-12-20T14:17:33.109822Z","iopub.status.busy":"2021-12-20T14:17:33.10955Z","iopub.status.idle":"2021-12-20T14:17:33.12449Z","shell.execute_reply":"2021-12-20T14:17:33.123631Z","shell.execute_reply.started":"2021-12-20T14:17:33.109787Z"},"papermill":{"duration":0.674118,"end_time":"2021-12-21T19:50:58.455021","exception":false,"start_time":"2021-12-21T19:50:57.780903","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.execute_input":"2021-12-20T14:17:34.53311Z","iopub.status.busy":"2021-12-20T14:17:34.532442Z","iopub.status.idle":"2021-12-20T14:17:34.555157Z","shell.execute_reply":"2021-12-20T14:17:34.553977Z","shell.execute_reply.started":"2021-12-20T14:17:34.533071Z"},"papermill":{"duration":0.682982,"end_time":"2021-12-21T19:50:59.808982","exception":false,"start_time":"2021-12-21T19:50:59.126","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.execute_input":"2021-12-20T13:06:16.048499Z","iopub.status.busy":"2021-12-20T13:06:16.048233Z","iopub.status.idle":"2021-12-20T13:06:16.054162Z","shell.execute_reply":"2021-12-20T13:06:16.053356Z","shell.execute_reply.started":"2021-12-20T13:06:16.048471Z"},"papermill":{"duration":0.69252,"end_time":"2021-12-21T19:51:01.197123","exception":false,"start_time":"2021-12-21T19:51:00.504603","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.execute_input":"2021-12-20T13:06:20.193997Z","iopub.status.busy":"2021-12-20T13:06:20.193739Z","iopub.status.idle":"2021-12-20T13:06:20.201087Z","shell.execute_reply":"2021-12-20T13:06:20.19941Z","shell.execute_reply.started":"2021-12-20T13:06:20.193968Z"},"papermill":{"duration":0.668382,"end_time":"2021-12-21T19:51:02.765677","exception":false,"start_time":"2021-12-21T19:51:02.097295","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.execute_input":"2021-12-20T13:06:20.988653Z","iopub.status.busy":"2021-12-20T13:06:20.9884Z","iopub.status.idle":"2021-12-20T13:06:20.994287Z","shell.execute_reply":"2021-12-20T13:06:20.993407Z","shell.execute_reply.started":"2021-12-20T13:06:20.988625Z"},"papermill":{"duration":0.687157,"end_time":"2021-12-21T19:51:04.127326","exception":false,"start_time":"2021-12-21T19:51:03.440169","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.execute_input":"2021-12-20T13:06:23.806684Z","iopub.status.busy":"2021-12-20T13:06:23.806208Z","iopub.status.idle":"2021-12-20T13:06:23.815287Z","shell.execute_reply":"2021-12-20T13:06:23.814473Z","shell.execute_reply.started":"2021-12-20T13:06:23.806645Z"},"papermill":{"duration":0.670826,"end_time":"2021-12-21T19:51:05.486718","exception":false,"start_time":"2021-12-21T19:51:04.815892","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.execute_input":"2021-12-20T13:07:01.091012Z","iopub.status.busy":"2021-12-20T13:07:01.090746Z","iopub.status.idle":"2021-12-20T13:07:01.096167Z","shell.execute_reply":"2021-12-20T13:07:01.095461Z","shell.execute_reply.started":"2021-12-20T13:07:01.090982Z"},"papermill":{"duration":0.665207,"end_time":"2021-12-21T19:51:06.824087","exception":false,"start_time":"2021-12-21T19:51:06.15888","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.671086,"end_time":"2021-12-21T19:51:08.165617","exception":false,"start_time":"2021-12-21T19:51:07.494531","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.673482,"end_time":"2021-12-21T19:51:09.596251","exception":false,"start_time":"2021-12-21T19:51:08.922769","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.739561,"end_time":"2021-12-21T19:51:11.007366","exception":false,"start_time":"2021-12-21T19:51:10.267805","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}