{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Preparing Data and Environment\n### Import modules","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.utils import to_categorical\nimport numpy as np\nfrom glob import glob\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom skimage import exposure\nimport cv2 as cv\nimport os\nimport itertools\nimport seaborn as sns\nimport os\nimport multiprocessing as mproc\nfrom keras.preprocessing import image\nfrom matplotlib.pyplot import figure\n\nfrom tqdm import tqdm\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-24T22:00:19.510843Z","iopub.execute_input":"2023-01-24T22:00:19.511290Z","iopub.status.idle":"2023-01-24T22:00:19.521504Z","shell.execute_reply.started":"2023-01-24T22:00:19.511253Z","shell.execute_reply":"2023-01-24T22:00:19.520414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### GPU availability check","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\ngpucheck = tf.config.experimental.list_physical_devices('GPU')\nfor gpu in gpucheck:\n    print(\"Name:\", gpu.name, \"  Type:\", gpu.device_type)\ntf.test.is_gpu_available()\n","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:19.525682Z","iopub.execute_input":"2023-01-24T22:00:19.526711Z","iopub.status.idle":"2023-01-24T22:00:19.543481Z","shell.execute_reply.started":"2023-01-24T22:00:19.526671Z","shell.execute_reply":"2023-01-24T22:00:19.542249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data loading and preprocessing\n\n#### File sources\n##### Images from the dataset are reduced to 256x256 to speed up the training of the model","metadata":{}},{"cell_type":"code","source":"train_dir = '../input/resized-plant2021/img_sz_256/'\ntest_dir =  '/kaggle/input/plant-pathology-2021-fgvc8/test_images/'\ndf = pd.read_csv('../input/plant-pathology-2021-fgvc8/train.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:19.545631Z","iopub.execute_input":"2023-01-24T22:00:19.546113Z","iopub.status.idle":"2023-01-24T22:00:19.570601Z","shell.execute_reply.started":"2023-01-24T22:00:19.546073Z","shell.execute_reply":"2023-01-24T22:00:19.569591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_files = glob(train_dir + '/*.jpg')\ntest_files = glob(test_dir + '/*.jpg')\nprint('# files for train',len(train_files))\nprint('# files for train',len(test_files))","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:19.572767Z","iopub.execute_input":"2023-01-24T22:00:19.573157Z","iopub.status.idle":"2023-01-24T22:00:19.634827Z","shell.execute_reply.started":"2023-01-24T22:00:19.573117Z","shell.execute_reply":"2023-01-24T22:00:19.633814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data sampling","metadata":{}},{"cell_type":"code","source":"def visualize_batch(path,image_ids, labels):\n    plt.figure(figsize=(16, 12))\n    \n    for ind, (image_id, label) in enumerate(zip(image_ids, labels)):\n        plt.subplot(3, 3, ind + 1)\n        image = cv.imread(os.path.join(path, image_id))\n        image = cv.cvtColor(image, cv.COLOR_BGR2RGB)\n\n        plt.imshow(image)\n        plt.title(f\"Класс: {label}\", fontsize=12)\n        plt.axis(\"off\")\n    plt.show()\nts = df.sample(6)\nimage_ids = ts[\"image\"].values\nlabels = ts[\"labels\"].values\nvisualize_batch(train_dir,image_ids,labels)","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:19.636227Z","iopub.execute_input":"2023-01-24T22:00:19.636923Z","iopub.status.idle":"2023-01-24T22:00:20.263335Z","shell.execute_reply.started":"2023-01-24T22:00:19.636880Z","shell.execute_reply":"2023-01-24T22:00:20.261437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Image preparation","metadata":{}},{"cell_type":"code","source":"def plot_comparison(original, filtered, title_filtered):\n  fig, (ax1, ax2) = plt.subplots(ncols=2, figsize=(8, 6), sharex=True, sharey=True)\n  ax1.imshow(original, cmap=plt.cm.gray) \n  ax1.set_title('original') \n  ax1.axis('off')\n  ax2.imshow(filtered, cmap=plt.cm.gray) \n  ax2.set_title(title_filtered) \n  ax2.axis('off')","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:20.266233Z","iopub.execute_input":"2023-01-24T22:00:20.267037Z","iopub.status.idle":"2023-01-24T22:00:20.273745Z","shell.execute_reply.started":"2023-01-24T22:00:20.266993Z","shell.execute_reply":"2023-01-24T22:00:20.272775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"leaf_img = cv.imread('../input/plant-pathology-2021-fgvc8/train_images/80230a9a3f7a9f6b.jpg')\nleaf_img = cv.cvtColor(leaf_img, cv.COLOR_BGR2RGB)\n\nplt.imshow(leaf_img)","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:20.275602Z","iopub.execute_input":"2023-01-24T22:00:20.276340Z","iopub.status.idle":"2023-01-24T22:00:22.691276Z","shell.execute_reply.started":"2023-01-24T22:00:20.276303Z","shell.execute_reply":"2023-01-24T22:00:22.690117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Contrast correction\nfrom skimage import exposure\nequalized_leaf_image = exposure.equalize_hist(leaf_img)\n\nplot_comparison(leaf_img, equalized_leaf_image, 'Histogram equalization')","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:22.693173Z","iopub.execute_input":"2023-01-24T22:00:22.693578Z","iopub.status.idle":"2023-01-24T22:00:28.908596Z","shell.execute_reply.started":"2023-01-24T22:00:22.693540Z","shell.execute_reply":"2023-01-24T22:00:28.907674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adapthits_leag_image = exposure.equalize_adapthist(leaf_img)\n\nplot_comparison(leaf_img, adapthits_leag_image, 'Adaptive Histogram equalization')","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:28.910241Z","iopub.execute_input":"2023-01-24T22:00:28.910909Z","iopub.status.idle":"2023-01-24T22:00:40.982629Z","shell.execute_reply.started":"2023-01-24T22:00:28.910865Z","shell.execute_reply":"2023-01-24T22:00:40.981606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Histograms","metadata":{}},{"cell_type":"code","source":"red = leaf_img[:, :, 0]\ngreen = leaf_img[:, 0, :]\nblue = leaf_img[0, :, :]\n\n\nplt.hist(red.ravel(), color = '#CC0000', bins=30) \nplt.title('Red_Histogram')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:40.984130Z","iopub.execute_input":"2023-01-24T22:00:40.985997Z","iopub.status.idle":"2023-01-24T22:00:41.377850Z","shell.execute_reply.started":"2023-01-24T22:00:40.985955Z","shell.execute_reply":"2023-01-24T22:00:41.376684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(green.ravel(), color = 'green', bins=30) \nplt.title('Green_Histogram')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:41.381773Z","iopub.execute_input":"2023-01-24T22:00:41.382080Z","iopub.status.idle":"2023-01-24T22:00:41.630714Z","shell.execute_reply.started":"2023-01-24T22:00:41.382050Z","shell.execute_reply":"2023-01-24T22:00:41.629689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(blue.ravel(), bins=30) \nplt.title('Blue_Histogram')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:41.632172Z","iopub.execute_input":"2023-01-24T22:00:41.632893Z","iopub.status.idle":"2023-01-24T22:00:41.861659Z","shell.execute_reply.started":"2023-01-24T22:00:41.632852Z","shell.execute_reply":"2023-01-24T22:00:41.860705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Class labels","metadata":{}},{"cell_type":"code","source":"all_labels = list(itertools.chain(*[lbs.split(\" \") for lbs in df['labels']]))\nfigure(figsize=(16, 6), dpi=80)\n\nax = sns.countplot(x=sorted(all_labels), orient='y')\nax.grid()","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:41.863350Z","iopub.execute_input":"2023-01-24T22:00:41.863818Z","iopub.status.idle":"2023-01-24T22:00:42.364881Z","shell.execute_reply.started":"2023-01-24T22:00:41.863756Z","shell.execute_reply":"2023-01-24T22:00:42.363676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_labels = list(itertools.chain([lbs for lbs in df['labels']]))\nfigure(figsize=(12, 6), dpi=80)\n\nax = sns.countplot(y=sorted(all_labels), orient='x')\nax.grid()","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:42.366395Z","iopub.execute_input":"2023-01-24T22:00:42.368372Z","iopub.status.idle":"2023-01-24T22:00:42.677749Z","shell.execute_reply.started":"2023-01-24T22:00:42.368330Z","shell.execute_reply":"2023-01-24T22:00:42.676740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.head(10))\nprint(df.labels.value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:42.679173Z","iopub.execute_input":"2023-01-24T22:00:42.680093Z","iopub.status.idle":"2023-01-24T22:00:42.693364Z","shell.execute_reply.started":"2023-01-24T22:00:42.680049Z","shell.execute_reply":"2023-01-24T22:00:42.692221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Labels contains spaces between words. ","metadata":{}},{"cell_type":"code","source":"# converting labels to binary attributes\ntrain = df.copy()\ntrain['labels'] = df['labels'].apply(lambda string: string.split(' '))\n\ns = list(train['labels'])\nmlb = MultiLabelBinarizer()\ntrainx = pd.DataFrame(mlb.fit_transform(s), columns=mlb.classes_, index=train.index)\ntrainx['image'] = train['image']\n\nprint(trainx.head(10))\nprint(trainx.columns)","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:42.695200Z","iopub.execute_input":"2023-01-24T22:00:42.695867Z","iopub.status.idle":"2023-01-24T22:00:42.732594Z","shell.execute_reply.started":"2023-01-24T22:00:42.695826Z","shell.execute_reply":"2023-01-24T22:00:42.731460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model setup","metadata":{}},{"cell_type":"code","source":"# training parameters\nTARGET_SIZE = 128\nBATCH_SIZE = 64\nEPOCHS = 50\nDATA_LIMIT = 12000\ntrainx = trainx[:DATA_LIMIT]","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:42.734316Z","iopub.execute_input":"2023-01-24T22:00:42.734697Z","iopub.status.idle":"2023-01-24T22:00:42.739780Z","shell.execute_reply.started":"2023-01-24T22:00:42.734660Z","shell.execute_reply":"2023-01-24T22:00:42.738662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data loading for Model training","metadata":{}},{"cell_type":"code","source":"# Load images resized to TARGET_SIZE into np.array\ntrain_image = []\nfor i in tqdm(range(trainx.shape[0])):\n    \n    img = image.load_img(train_dir+ trainx['image'][i],target_size=(TARGET_SIZE,TARGET_SIZE,3))\n    img = image.img_to_array(img)\n    img = img/255\n    train_image.append(img)\n\nX = np.array(train_image)","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:00:42.741575Z","iopub.execute_input":"2023-01-24T22:00:42.741958Z","iopub.status.idle":"2023-01-24T22:01:46.407934Z","shell.execute_reply.started":"2023-01-24T22:00:42.741920Z","shell.execute_reply":"2023-01-24T22:01:46.406868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check that we have the right sizes images in the X array\nplt.imshow(X[8])","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:01:46.409550Z","iopub.execute_input":"2023-01-24T22:01:46.410207Z","iopub.status.idle":"2023-01-24T22:01:46.657228Z","shell.execute_reply.started":"2023-01-24T22:01:46.410161Z","shell.execute_reply":"2023-01-24T22:01:46.656203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transform the binary attributes of classes into an array Y and check its dimension\ny = np.array(trainx.drop(['image'],axis=1))\ny.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:01:46.658749Z","iopub.execute_input":"2023-01-24T22:01:46.659343Z","iopub.status.idle":"2023-01-24T22:01:46.668770Z","shell.execute_reply.started":"2023-01-24T22:01:46.659300Z","shell.execute_reply":"2023-01-24T22:01:46.667691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# splitting data into training and validation sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.1)","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:01:46.670531Z","iopub.execute_input":"2023-01-24T22:01:46.671075Z","iopub.status.idle":"2023-01-24T22:01:47.334961Z","shell.execute_reply.started":"2023-01-24T22:01:46.670993Z","shell.execute_reply":"2023-01-24T22:01:47.333889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model initialization","metadata":{}},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten\nfrom keras.layers import Conv2D, MaxPooling2D\n\nmodel = Sequential()\nmodel.add(Conv2D(filters=16, kernel_size=(5, 5), activation=\"relu\", input_shape=(TARGET_SIZE,TARGET_SIZE,3)))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\nmodel.add(Conv2D(filters=32, kernel_size=(5, 5), activation='relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\nmodel.add(Conv2D(filters=64, kernel_size=(5, 5), activation=\"relu\"))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\nmodel.add(Conv2D(filters=64, kernel_size=(5, 5), activation='relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(6, activation='sigmoid'))\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:01:47.336488Z","iopub.execute_input":"2023-01-24T22:01:47.336897Z","iopub.status.idle":"2023-01-24T22:01:47.843622Z","shell.execute_reply.started":"2023-01-24T22:01:47.336858Z","shell.execute_reply":"2023-01-24T22:01:47.841804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Training","metadata":{}},{"cell_type":"code","source":"model.fit(X_train, y_train, epochs=EPOCHS, validation_data=(X_test, y_test), batch_size=BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:01:47.844961Z","iopub.execute_input":"2023-01-24T22:01:47.845354Z","iopub.status.idle":"2023-01-24T22:04:41.815712Z","shell.execute_reply.started":"2023-01-24T22:01:47.845313Z","shell.execute_reply":"2023-01-24T22:04:41.813792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model validation on test data\n\nimg = image.load_img(train_dir+ trainx['image'][9],target_size=(TARGET_SIZE,TARGET_SIZE,3))\nimg = image.img_to_array(img)\nimg = img/255\nclasses = np.array(trainx.columns[:-1])\nproba = model.predict(img.reshape(1,TARGET_SIZE,TARGET_SIZE,3))\ntop_3 = np.argsort(proba[0])[:-4:-1]\nfor i in range(3):\n    print(\"{}\".format(classes[top_3[i]])+\" ({:.3})\".format(proba[0][top_3[i]]))\n\nplt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:04:41.820089Z","iopub.execute_input":"2023-01-24T22:04:41.820405Z","iopub.status.idle":"2023-01-24T22:04:42.320917Z","shell.execute_reply.started":"2023-01-24T22:04:41.820374Z","shell.execute_reply":"2023-01-24T22:04:42.319916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting the occuring as the class predicted by model.predict() will be valid\nname = {0: 'complex',\n        1: 'frog_eye_leaf_spot',\n        2: 'healthy',\n        3: 'powdery_mildew',\n        4: 'rust',\n        5: 'scab'}\n\nthreshold = {0: 0.3,\n             1: 0.3,\n             2: 0.4,\n             3: 0.3,\n             4: 0.3,\n            5:0.3}","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:04:42.322263Z","iopub.execute_input":"2023-01-24T22:04:42.322930Z","iopub.status.idle":"2023-01-24T22:04:42.330056Z","shell.execute_reply.started":"2023-01-24T22:04:42.322890Z","shell.execute_reply":"2023-01-24T22:04:42.329007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The class labels are concatenated into a string and written to the final submission.csv dataframe\nim_labels = []\n\nfor z in test_files:\n    img = image.load_img(z,target_size=(TARGET_SIZE,TARGET_SIZE,3))\n    img = image.img_to_array(img)\n    img = img/255\n    proba = list(model.predict(img.reshape(1,TARGET_SIZE,TARGET_SIZE,3))[0])\n    filename = z.split('/')[-1]\n    p = []\n    for i in range(len(proba)):\n        if proba[i] > threshold[i]:\n            p.append(name[i])\n            \n    im_labels.append({'image': filename, 'labels': ' '.join(p)})\n    \ndf = pd.DataFrame(im_labels)\ndf.to_csv('submission.csv', index=False)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-24T22:04:42.334885Z","iopub.execute_input":"2023-01-24T22:04:42.335840Z","iopub.status.idle":"2023-01-24T22:04:43.063966Z","shell.execute_reply.started":"2023-01-24T22:04:42.335799Z","shell.execute_reply":"2023-01-24T22:04:43.062899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}