{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport random\nfrom tensorflow.keras.utils import load_img\nimport matplotlib.pyplot as plt \nimport glob as gb\nfrom kaggle_datasets import KaggleDatasets\n!pip install -q efficientnet\nimport efficientnet.tfkeras as efn\nimport tensorflow as tf\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.applications import EfficientNetB7\nfrom tensorflow.keras.applications import EfficientNetB4\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, TensorBoard, ModelCheckpoint\nfrom tensorflow.keras.utils import plot_model\nfrom IPython.display import SVG, Image\nimport cv2\nfrom sklearn.preprocessing import MultiLabelBinarizer\nimport os\nimport hashlib\n\nfrom PIL import Image\n!pip install albumentations\nimport albumentations as A","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-20T13:28:12.600787Z","iopub.execute_input":"2023-05-20T13:28:12.601615Z","iopub.status.idle":"2023-05-20T13:28:26.574642Z","shell.execute_reply.started":"2023-05-20T13:28:12.601578Z","shell.execute_reply":"2023-05-20T13:28:26.566625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GPU usage\nMaking sure that *GPU* is used when using *GPU* accelator turned on as it seems not to work sometimes\n\n> Notebook was runed using: GPU P100","metadata":{}},{"cell_type":"code","source":"gpus = tf.config.list_physical_devices('GPU'); print(gpus)\nif len(gpus)==1: strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\nelse: strategy = tf.distribute.MirroredStrategy()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.575517Z","iopub.status.idle":"2023-05-20T13:28:26.575882Z","shell.execute_reply.started":"2023-05-20T13:28:26.575713Z","shell.execute_reply":"2023-05-20T13:28:26.575733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using auto mixed precision as it seems to improve speed of *GPU*","metadata":{}},{"cell_type":"code","source":"tf.config.optimizer.set_experimental_options({\"auto_mixed_precision\": True})\nprint('Mixed precision enabled')","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.578212Z","iopub.status.idle":"2023-05-20T13:28:26.578702Z","shell.execute_reply.started":"2023-05-20T13:28:26.578446Z","shell.execute_reply":"2023-05-20T13:28:26.578468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Defining load_image functions and paths","metadata":{}},{"cell_type":"code","source":"# # GCS_DS_PATH = KaggleDatasets().get_gcs_path()\n# GCS_DS_PATH = KaggleDatasets().get_gcs_path(\"resized-plant2021\")\n# #to verify your dir\n# !gsutil ls $GCS_DS_PATH","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.580402Z","iopub.status.idle":"2023-05-20T13:28:26.580889Z","shell.execute_reply.started":"2023-05-20T13:28:26.580640Z","shell.execute_reply":"2023-05-20T13:28:26.580679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image(filename):\n    image = load_img(\"../input/plant-pathology-2021-fgvc8/train_images/\"+filename)\n    plt.imshow(image) \n\ndef load_random_image(filenames):\n    sample = random.choice(filenames)\n    image = load_img(\"../input/plant-pathology-2021-fgvc8/train_images/\"+sample)\n    plt.imshow(image)     \n\ndef load_augmented_random_image(filenames):\n    sample = random.choice(filenames)\n    print(sample)\n    image = load_img(\"/kaggle/working/\"+sample)\n    plt.imshow(image)         \n    \ndef load_image_for_augmentation(image_path):\n    image = cv2.imread(image_path)\n    if image.shape[-1] == 1:\n        image = cv2.cvtColor(image, cv2.COLOR_GRAY2RGB)\n    \n    image = np.array(image)\n    return image    \n\n# adding path we will use in this nothebook \ndef format_path_gcs(st):\n    return GCS_DS_PATH + '/train_images/' + st\ndef format_resized_image_path_gcs(st):\n    return GCS_DS_PATH + '/img_sz_384/' + st\ndef format_tpu_path(st):\n    return '/kaggle/input/resized-plant2021' + '/img_sz_256/' + st","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.582671Z","iopub.status.idle":"2023-05-20T13:28:26.583127Z","shell.execute_reply.started":"2023-05-20T13:28:26.582891Z","shell.execute_reply":"2023-05-20T13:28:26.582914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining mem usage reduction function to try to make dataframes lighter\n\ndef reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.585078Z","iopub.status.idle":"2023-05-20T13:28:26.585542Z","shell.execute_reply.started":"2023-05-20T13:28:26.585299Z","shell.execute_reply":"2023-05-20T13:28:26.585322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading train dataframe and defining image paths","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/plant-pathology-2021-fgvc8/train.csv\")\n# test_df = pd.read_csv(\"../input/plant-pathology-2021-fgvc8/test.csv\")\n\nIMAGE_PATH = \"../input/plant-pathology-2021-fgvc8/test-images/\"\n\ntrain_df = reduce_mem_usage(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.587203Z","iopub.status.idle":"2023-05-20T13:28:26.587709Z","shell.execute_reply.started":"2023-05-20T13:28:26.587425Z","shell.execute_reply":"2023-05-20T13:28:26.587446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since ***plant-pathology-2021-fgvc8*** has very high resolution (almost 4k resolution) we use resized images found on kaggle to make image reading a lot faster","metadata":{}},{"cell_type":"markdown","source":"We use resized image path of ***384 x 384*** as we will be using **EfficientNetB4** for training model and this good size for it ","metadata":{}},{"cell_type":"code","source":"RESIZED_IMAGE_PATH = \"../input/resized-plant2021/img_sz_256/\"","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.589284Z","iopub.status.idle":"2023-05-20T13:28:26.589619Z","shell.execute_reply.started":"2023-05-20T13:28:26.589452Z","shell.execute_reply":"2023-05-20T13:28:26.589468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explaining ***plant-pathology*** and ***plant-pathology-2021-fgvc8*** dataset","metadata":{}},{"cell_type":"markdown","source":"Firstly we analyze dataframe and it labels distribution","metadata":{}},{"cell_type":"code","source":"train_df.labels.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.591463Z","iopub.status.idle":"2023-05-20T13:28:26.592352Z","shell.execute_reply.started":"2023-05-20T13:28:26.592173Z","shell.execute_reply":"2023-05-20T13:28:26.592192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the count of each label\nlabel_counts = train_df['labels'].value_counts()\n\n# Define a colormap for assigning colors to labels\ncolormap = plt.cm.get_cmap('tab10')\n\n# Get the number of unique labels\nnum_labels = len(label_counts)\n\n# Generate an array of colors based on the colormap\ncolors = [colormap(i) for i in range(num_labels)]\n\n# Plot the label distribution with colored bars\nplt.figure(figsize=(10, 6))\nplt.bar(label_counts.index, label_counts.values, color=colors)\nplt.xlabel('Labels')\nplt.ylabel('Count')\nplt.title('Label Distribution')\nplt.xticks(rotation=45)\n\n# Add count labels above the bars\nfor i, count in enumerate(label_counts):\n    plt.text(i, count + 50, str(count), ha='center')\n\n# Add a line below the total count\ntotal_count = label_counts.sum()\nplt.axhline(total_count, color='black', linestyle='--', alpha=0.5)\n\n# Add total count of all labels above the title with increased font size and bold style\nplt.text(0.5, 1.08, f'Total Count: {total_count}', transform=plt.gca().transAxes, ha='center', fontsize=12, fontweight='bold')\n\n# Calculate the maximum count for setting the y-axis limit\nmax_count = max(label_counts)\ny_limit = max_count + max_count * 0.1  # Add some padding\n\nplt.ylim(top=y_limit)  # Set the y-axis limit to accommodate the line\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.593493Z","iopub.status.idle":"2023-05-20T13:28:26.593921Z","shell.execute_reply.started":"2023-05-20T13:28:26.593689Z","shell.execute_reply":"2023-05-20T13:28:26.593711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We try to detect duplicates in dataframe and delete them.","metadata":{}},{"cell_type":"code","source":"initial_length = len(train_df)\n\n# create dictionary to store hashes and paths\nhashes = {}\n\nduplicates = []\noriginals = []\n\n# loop over rows in the dataframe\nfor index, row in train_df.iterrows():\n    # get the filename of the image\n    filename = row['image']\n    \n    # compute the hash of the image\n    with open(os.path.join(RESIZED_IMAGE_PATH, filename), 'rb') as f:\n        hash = hashlib.md5(f.read()).hexdigest()\n    \n    # check if hash already exists in dictionary\n    if hash in hashes:\n        duplicates.append(filename)\n        \n        originals.append(hashes[hash])\n        # delete duplicate row from dataframe\n        train_df.drop(index, inplace=True)\n    else:\n        # add hash and path to dictionary\n        hashes[hash] = filename\n\n# print number of duplicates found\nprint(f\"Number of duplicates found: {initial_length-len(hashes)}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.595764Z","iopub.status.idle":"2023-05-20T13:28:26.596218Z","shell.execute_reply.started":"2023-05-20T13:28:26.595984Z","shell.execute_reply":"2023-05-20T13:28:26.596007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading original and duplicated images found","metadata":{}},{"cell_type":"code","source":"# List of image filenames\nimage_filenames = [ originals[0], duplicates[0],\n                    originals[1], duplicates[1],\n                    originals[2], duplicates[2]]\n\n# Create a figure with a 3x2 grid of subplots\nfig, axes = plt.subplots(3, 2, figsize=(10, 8))\n\n# Iterate over the image filenames and corresponding subplots\nfor i, (image_filename, ax) in enumerate(zip(image_filenames, axes.flatten())):\n    # Load the image\n    image = plt.imread(format_tpu_path(image_filename))\n    \n    # Show the image in the subplot\n    ax.imshow(image)\n    ax.axis('off')\n\n    # Set subplot title\n    if i % 2 == 0:\n        title = 'Original'\n    else:\n        title = 'Duplicate'\n    ax.set_title(title)\n# Adjust the layout of subplots to avoid overlapping\nplt.tight_layout()\n\n# Display the figure\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.598028Z","iopub.status.idle":"2023-05-20T13:28:26.598476Z","shell.execute_reply.started":"2023-05-20T13:28:26.598246Z","shell.execute_reply":"2023-05-20T13:28:26.598268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we can compare how two grafs look like: before deleting duplicates and after","metadata":{}},{"cell_type":"code","source":"train_df.labels.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.600157Z","iopub.status.idle":"2023-05-20T13:28:26.600615Z","shell.execute_reply.started":"2023-05-20T13:28:26.600378Z","shell.execute_reply":"2023-05-20T13:28:26.600400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the count of each label\nlabel_counts = train_df['labels'].value_counts()\n\n# Define a colormap for assigning colors to labels\ncolormap = plt.cm.get_cmap('tab10')\n\n# Get the number of unique labels\nnum_labels = len(label_counts)\n\n# Generate an array of colors based on the colormap\ncolors = [colormap(i) for i in range(num_labels)]\n\n# Plot the label distribution with colored bars\nplt.figure(figsize=(10, 6))\nplt.bar(label_counts.index, label_counts.values, color=colors)\nplt.xlabel('Labels')\nplt.ylabel('Count')\nplt.title('Label Distribution')\nplt.xticks(rotation=45)\n\n# Add count labels above the bars\nfor i, count in enumerate(label_counts):\n    plt.text(i, count + 50, str(count), ha='center')\n\n# Add a line below the total count\ntotal_count = label_counts.sum()\nplt.axhline(total_count, color='black', linestyle='--', alpha=0.5)\n\n# Add total count of all labels above the title with increased font size and bold style\nplt.text(0.5, 1.08, f'Total Count: {total_count}', transform=plt.gca().transAxes, ha='center', fontsize=12, fontweight='bold')\n\n# Calculate the maximum count for setting the y-axis limit\nmax_count = max(label_counts)\ny_limit = max_count + max_count * 0.1  # Add some padding\n\nplt.ylim(top=y_limit)  # Set the y-axis limit to accommodate the line\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.603239Z","iopub.status.idle":"2023-05-20T13:28:26.604414Z","shell.execute_reply.started":"2023-05-20T13:28:26.604136Z","shell.execute_reply":"2023-05-20T13:28:26.604163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Performing **One Hot Encoding** on labels. <br>\n**One Hot Encoding** converts categorical data into numeric values.","metadata":{}},{"cell_type":"markdown","source":"As we see from two graphs above that there is 12 labels but we can also notice that this classification task requires multilabel approach. \n\nAs we approach multilable classification we need to find unique labels in dataframe labels column and separate them to their own column.\n\nAfter we have different columns we need to convert categorical labels to numerical values to be able to use *'binary_crossentropy'* loss function","metadata":{}},{"cell_type":"code","source":"# Extract the labels column from the dataframe and store it in a list\nlabels = train_df['labels'].tolist()\n\n# Create a set of unique labels\nunique_labels = set()\n\n# Iterate through each label in the list of labels and add it to the set of unique labels\nfor label in labels:\n    unique_labels.update(label.split())\n\n# Print the number of unique labels\nprint(unique_labels, \"suma:\", len(unique_labels))","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.605638Z","iopub.status.idle":"2023-05-20T13:28:26.606422Z","shell.execute_reply.started":"2023-05-20T13:28:26.606180Z","shell.execute_reply":"2023-05-20T13:28:26.606202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.607896Z","iopub.status.idle":"2023-05-20T13:28:26.608922Z","shell.execute_reply.started":"2023-05-20T13:28:26.608680Z","shell.execute_reply":"2023-05-20T13:28:26.608704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select the six most common unique labels\ncommon_labels = [label[0] for label in pd.Series(labels).str.split(expand=True).stack().value_counts()[:6].items()]\n# Create a MultiLabelBinarizer object with the six common labels\nmlb = MultiLabelBinarizer(classes=common_labels)\n\n# Transform the labels into a binary matrix with one-hot encoding for the six common labels\nlabel_matrix = mlb.fit_transform(train_df['labels'].str.split())\n\n# Create a new dataframe with the one-hot encoded labels\nlabel_df = pd.DataFrame(label_matrix, columns=common_labels)\n\n# fixes train_df row number error\ntrain_df.reset_index(drop=True, inplace=True)\nlabel_df.reset_index(drop=True, inplace=True)\n\n# Concatenate the new label dataframe with the original dataframe\nnew_df = pd.concat([train_df, label_df], axis=1)\n\n# Drop the original labels column from the new dataframe\ntrain_df = new_df.drop('labels', axis=1)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.610114Z","iopub.status.idle":"2023-05-20T13:28:26.610911Z","shell.execute_reply.started":"2023-05-20T13:28:26.610669Z","shell.execute_reply":"2023-05-20T13:28:26.610693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the sum of each label column\nlabel_counts = train_df.iloc[:, 1:].sum()\n\n# Define a colormap for assigning colors to labels\ncolormap = plt.cm.get_cmap('tab10')\n\n# Get the number of unique labels\nnum_labels = len(label_counts)\n\n# Generate an array of colors based on the colormap\ncolors = [colormap(i) for i in np.linspace(0, 1, num_labels)]\n\n# Plot the label distribution with colored bars\nplt.figure(figsize=(10, 6))\nplt.bar(label_counts.index, label_counts.values, color=colors)\nplt.xlabel('Labels')\nplt.ylabel('Count')\nplt.title('Label Distribution')\nplt.xticks(rotation=45)\n\n# Add count labels above the bars\nfor i, count in enumerate(label_counts):\n    plt.text(i, count + 50, str(count), ha='center')\n\n# Add a line below the total count\ntotal_count = label_counts.sum()\nplt.axhline(total_count, color='black', linestyle='--', alpha=0.5)\n\n# Add total count of all labels above the title with increased font size and bold style\nplt.text(0.5, 1.08, f'Total Count: {total_count}', transform=plt.gca().transAxes, ha='center', fontsize=12, fontweight='bold')\n\n# Calculate the maximum count for setting the y-axis limit\nmax_count = max(label_counts)\ny_limit = max_count + max_count * 0.1  # Add some padding\n\nplt.ylim(top=y_limit)  # Set the y-axis limit to accommodate the line\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.612357Z","iopub.status.idle":"2023-05-20T13:28:26.613356Z","shell.execute_reply.started":"2023-05-20T13:28:26.613122Z","shell.execute_reply":"2023-05-20T13:28:26.613146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"34# Albumentations","metadata":{}},{"cell_type":"code","source":"frog_eye_leaf_spot = list(train_df[train_df[\"frog_eye_leaf_spot\"]==1].image)\nrust = list(train_df[train_df[\"rust\"]==1].image)\nscab = list(train_df[train_df[\"scab\"]==1].image)\ncomplex = list(train_df[train_df[\"complex\"]==1].image)\nhealthy = list(train_df[train_df[\"healthy\"]==1].image)\npowdery_mildew = list(train_df[train_df[\"powdery_mildew\"]==1].image)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.614509Z","iopub.status.idle":"2023-05-20T13:28:26.615497Z","shell.execute_reply.started":"2023-05-20T13:28:26.615263Z","shell.execute_reply":"2023-05-20T13:28:26.615285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading sample images from every class and saving them to output folder for later use","metadata":{}},{"cell_type":"code","source":"load_random_image(frog_eye_leaf_spot)\nplt.savefig('frog_eye_leaf_spot.png')","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.616650Z","iopub.status.idle":"2023-05-20T13:28:26.617635Z","shell.execute_reply.started":"2023-05-20T13:28:26.617398Z","shell.execute_reply":"2023-05-20T13:28:26.617421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_random_image(rust)\nplt.savefig('rust.png')","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.618804Z","iopub.status.idle":"2023-05-20T13:28:26.619813Z","shell.execute_reply.started":"2023-05-20T13:28:26.619561Z","shell.execute_reply":"2023-05-20T13:28:26.619585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_random_image(scab)\nplt.savefig('scab.png')","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.620963Z","iopub.status.idle":"2023-05-20T13:28:26.621962Z","shell.execute_reply.started":"2023-05-20T13:28:26.621725Z","shell.execute_reply":"2023-05-20T13:28:26.621748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_random_image(complex)\nplt.savefig('complex.png')","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.623123Z","iopub.status.idle":"2023-05-20T13:28:26.624017Z","shell.execute_reply.started":"2023-05-20T13:28:26.623734Z","shell.execute_reply":"2023-05-20T13:28:26.623760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_random_image(healthy)\nplt.savefig('healthy.png')","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.625431Z","iopub.status.idle":"2023-05-20T13:28:26.626414Z","shell.execute_reply.started":"2023-05-20T13:28:26.626180Z","shell.execute_reply":"2023-05-20T13:28:26.626203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_random_image(powdery_mildew)\nplt.savefig('powdery_mildew.png')","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.627551Z","iopub.status.idle":"2023-05-20T13:28:26.628333Z","shell.execute_reply.started":"2023-05-20T13:28:26.628096Z","shell.execute_reply":"2023-05-20T13:28:26.628119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training and validation split","metadata":{}},{"cell_type":"code","source":"#===============================================================\n#===============================================================\n# X = train_df.image.apply(format_resized_image_path_gcs).values\nX = train_df.image.apply(format_tpu_path).values\ny = np.float32(train_df.loc[:, 'healthy':'scab'].values)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.629861Z","iopub.status.idle":"2023-05-20T13:28:26.630914Z","shell.execute_reply.started":"2023-05-20T13:28:26.630674Z","shell.execute_reply":"2023-05-20T13:28:26.630700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = train_df\n\n# Get the hot encoded label columns\nlabel_cols = [col for col in df.columns if not col.startswith('image')]\n\n# Split the dataframe into features (X) and labels (y)\n# X = df.drop(label_cols, axis=1)\ny = df[label_cols]\n\n# Determine the size of the validation set\nvalidation_size = 0.2\n\n# Split the data into training and validation sets while ensuring that the label distribution is balanced in both sets\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=validation_size, stratify=y, random_state=25)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.632360Z","iopub.status.idle":"2023-05-20T13:28:26.633554Z","shell.execute_reply.started":"2023-05-20T13:28:26.633317Z","shell.execute_reply":"2023-05-20T13:28:26.633339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(y_train)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.634897Z","iopub.status.idle":"2023-05-20T13:28:26.635900Z","shell.execute_reply.started":"2023-05-20T13:28:26.635666Z","shell.execute_reply":"2023-05-20T13:28:26.635690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the sum of each label column\nlabel_counts = y_train.iloc[:, 0:].sum()\n\n# Define a colormap for assigning colors to labels\ncolormap = plt.cm.get_cmap('tab10')\n\n# Get the number of unique labels\nnum_labels = len(label_counts)\n\n# Generate an array of colors based on the colormap\ncolors = [colormap(i) for i in np.linspace(0, 1, num_labels)]\n\n# Plot the label distribution with colored bars\nplt.figure(figsize=(10, 6))\nplt.bar(label_counts.index, label_counts.values, color=colors)\nplt.xlabel('Labels')\nplt.ylabel('Count')\nplt.title('Label Distribution')\nplt.xticks(rotation=45)\n\n# Add count labels above the bars\nfor i, count in enumerate(label_counts):\n    plt.text(i, count + 50, str(count), ha='center')\n\n# Add a line below the total count\ntotal_count = label_counts.sum()\nplt.axhline(total_count, color='black', linestyle='--', alpha=0.5)\n\n# Add total count of all labels above the title with increased font size and bold style\nplt.text(0.5, 1.08, f'Total Count: {total_count}', transform=plt.gca().transAxes, ha='center', fontsize=12, fontweight='bold')\n\n# Calculate the maximum count for setting the y-axis limit\nmax_count = max(label_counts)\ny_limit = max_count + max_count * 0.1  # Add some padding\n\nplt.ylim(top=y_limit)  # Set the y-axis limit to accommodate the line\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.637066Z","iopub.status.idle":"2023-05-20T13:28:26.638062Z","shell.execute_reply.started":"2023-05-20T13:28:26.637810Z","shell.execute_reply":"2023-05-20T13:28:26.637834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"arr_df = pd.DataFrame(X_train, columns=['image'])\n\n# Reset the index of arr_df\narr_df = arr_df.reset_index(drop=True)\ny_train = y_train.reset_index(drop=True)\n\ntrain_df = pd.concat([arr_df, y_train], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.639199Z","iopub.status.idle":"2023-05-20T13:28:26.640194Z","shell.execute_reply.started":"2023-05-20T13:28:26.639942Z","shell.execute_reply":"2023-05-20T13:28:26.639964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cutout_image(image):\n    image_array = np.array(image)\n    augmented_array = A.CoarseDropout(max_holes=8, max_height=12, max_width=12, always_apply=True)(image=image_array)['image']\n\n    return Image.fromarray(augmented_array)\n\ndef elastic_transform_image(image):\n    image_array = np.array(image)\n    augmented_array = A.ElasticTransform(alpha=120, sigma=120 * 0.05, alpha_affine=120 * 0.03, always_apply=True)(image=image_array)['image']\n\n    return Image.fromarray(augmented_array)\n\ndef gaussian_noise_image(image):\n    image_array = np.array(image)\n    augmented_array = A.GaussNoise(var_limit=(15.0, 60.0), always_apply=True)(image=image_array)['image']\n\n    return Image.fromarray(augmented_array)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.641341Z","iopub.status.idle":"2023-05-20T13:28:26.641938Z","shell.execute_reply.started":"2023-05-20T13:28:26.641702Z","shell.execute_reply":"2023-05-20T13:28:26.641724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.643446Z","iopub.status.idle":"2023-05-20T13:28:26.644370Z","shell.execute_reply.started":"2023-05-20T13:28:26.644090Z","shell.execute_reply":"2023-05-20T13:28:26.644117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_labels","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.645766Z","iopub.status.idle":"2023-05-20T13:28:26.646720Z","shell.execute_reply.started":"2023-05-20T13:28:26.646461Z","shell.execute_reply":"2023-05-20T13:28:26.646485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = train_df\n\nlabels = unique_labels\n\n# Get the counts of each label\nlabel_counts = df[list(unique_labels)].sum()\n\n# Calculate the number of rows to be added for each label\ntotal_rows_added = int(len(df) * 0.03)\nrows_added_per_label = (total_rows_added * label_counts / label_counts.sum()).astype(int)\n\n# Find the maximum value\nmax_value = rows_added_per_label.max()\n\n# Make the smallest value the largest\nrows_added_per_label = max_value - rows_added_per_label + max_value\n\nprint(rows_added_per_label)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.647922Z","iopub.status.idle":"2023-05-20T13:28:26.648944Z","shell.execute_reply.started":"2023-05-20T13:28:26.648710Z","shell.execute_reply":"2023-05-20T13:28:26.648733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a new DataFrame for the augmented images\ncutout_df = pd.DataFrame(columns=df.columns)\nelastic_transform_df = pd.DataFrame(columns=df.columns)\ngaus_df = pd.DataFrame(columns=df.columns)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.650088Z","iopub.status.idle":"2023-05-20T13:28:26.650678Z","shell.execute_reply.started":"2023-05-20T13:28:26.650426Z","shell.execute_reply":"2023-05-20T13:28:26.650448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a new directory to save augmented images\ncutout_dir = 'cutout_images'\nos.makedirs(cutout_dir, exist_ok=True)\n\n# Create a new directory to save augmented images\nelastic_dir = 'elastic_images'\nos.makedirs(elastic_dir, exist_ok=True)\n\n# Create a new directory to save augmented images\ngaus_dir = 'gaus_images'\nos.makedirs(gaus_dir, exist_ok=True)\n\n# Track the selected indices\nselected_indices_set = set()\n\n# Augment the data for each label\nfor label in labels:\n    # Get the indices of rows with the current label\n    label_indices = df[df[label] == 1].index\n    # Convert label indices to a NumPy array\n    label_indices = np.array(label_indices)\n\n    # Shuffle the label indices randomly\n    np.random.shuffle(label_indices)\n    \n    # Get the number of rows to be added for the current label\n    rows_to_add = rows_added_per_label[label]\n\n    # Select the indices for augmentation, excluding the ones already selected\n    selected_indices = np.random.choice(\n    np.setdiff1d(label_indices, list(selected_indices_set)),\n    size=min(rows_to_add, len(label_indices)),\n    replace=False)\n\n    # Update the set of selected indices\n    selected_indices_set.update(selected_indices)\n\n    # Augment the selected rows and update the image paths in the dataframe\n    for index in selected_indices:\n        image_path = df.at[index, 'image']\n        \n        image = load_image_for_augmentation(image_path)\n        \n        # Apply cutout\n        augmented_image = cutout_image(image)\n        \n        # Save the augmented image\n        augmented_image_path = os.path.join(cutout_dir, f'cutout_{label}_{index}.png')\n        # Save the augmented image\n        augmented_image.save(augmented_image_path)\n\n        new_row = df.iloc[index].copy()\n        new_row['image'] = augmented_image_path\n\n        # save to cutout dataframe\n        cutout_df.loc[len(cutout_df)] = new_row.values\n        \n        \n        # Apply elastic transformations\n        augmented_image = elastic_transform_image(image)\n        \n        # Save the augmented image\n        augmented_image_path = os.path.join(elastic_dir, f'elastic_{label}_{index}.png')\n        augmented_image.save(augmented_image_path)\n\n        new_row = df.iloc[index].copy()\n        new_row['image'] = augmented_image_path\n        \n        # save to elastic dataframe\n        elastic_transform_df.loc[len(elastic_transform_df)] = new_row.values\n        \n        \n        # Apply gaussian noise\n        augmented_image = gaussian_noise_image(image)\n        \n        # Save the augmented image\n        augmented_image_path = os.path.join(gaus_dir, f'gaus_{label}_{index}.png')\n        augmented_image.save(augmented_image_path)\n\n        new_row = df.iloc[index].copy()\n        new_row['image'] = augmented_image_path\n        \n        # save to gaus dataframe\n        gaus_df.loc[len(gaus_df)] = new_row.values\n\n# Save the cutout DataFrame to a new CSV file\naugmented_dataframe_path = 'cutout_df.csv'\ncutout_df.to_csv(augmented_dataframe_path, index=False)\n\n# Save the elastic DataFrame to a new CSV file\naugmented_dataframe_path = 'elastic_transform_df.csv'\nelastic_transform_df.to_csv(augmented_dataframe_path, index=False)\n\n\n# Save the elastic DataFrame to a new CSV file\naugmented_dataframe_path = 'gaus_df.csv'\ngaus_df.to_csv(augmented_dataframe_path, index=False)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-05-20T13:28:26.652301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cutout_df = pd.read_csv(\"/kaggle/working/cutout_df.csv\")\n# elastic_transform_df =  pd.read_csv(\"/kaggle/working/elastic_transform_df.csv\")\n# gaus_df =  pd.read_csv(\"/kaggle/working/gaus_df.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cutout_frog_eye_leaf_spot = list(cutout_df[cutout_df[\"frog_eye_leaf_spot\"]==1].image)\ncutout_rust = list(cutout_df[cutout_df[\"rust\"]==1].image)\ncutout_scab = list(cutout_df[cutout_df[\"scab\"]==1].image)\ncutout_complex = list(cutout_df[cutout_df[\"complex\"]==1].image)\ncutout_healthy = list(cutout_df[cutout_df[\"healthy\"]==1].image)\ncutout_powdery_mildew = list(cutout_df[cutout_df[\"powdery_mildew\"]==1].image)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# List of image filenames and corresponding titles\nimage_filenames = [cutout_frog_eye_leaf_spot[0], cutout_rust[0],\n                   cutout_scab[0], cutout_complex[0],\n                   cutout_healthy[0], cutout_powdery_mildew[0]]\nimage_titles = unique_labels\n\n# Create a figure with a 2x3 grid of subplots\nfig, axes = plt.subplots(2, 3, figsize=(10, 8))\n\n# Iterate over the image filenames, titles, and corresponding subplots\nfor i, (image_filename, image_title, ax) in enumerate(zip(image_filenames, image_titles, axes.flatten())):\n    # Load the image\n    image = plt.imread(image_filename)\n    \n    # Show the image in the subplot\n    ax.imshow(image)\n    ax.set_title(image_title)\n    ax.axis('off')\n\n# Adjust the layout of subplots to avoid overlapping\nplt.tight_layout()\n\n# Display the figure\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"elastic_frog_eye_leaf_spot = list(elastic_transform_df[elastic_transform_df[\"frog_eye_leaf_spot\"]==1].image)\nelastic_rust = list(elastic_transform_df[elastic_transform_df[\"rust\"]==1].image)\nelastic_scab = list(elastic_transform_df[elastic_transform_df[\"scab\"]==1].image)\nelastic_complex = list(elastic_transform_df[elastic_transform_df[\"complex\"]==1].image)\nelastic_healthy = list(elastic_transform_df[elastic_transform_df[\"healthy\"]==1].image)\nelastic_powdery_mildew = list(elastic_transform_df[elastic_transform_df[\"powdery_mildew\"]==1].image)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# List of image filenames and corresponding titles\nimage_filenames = [elastic_frog_eye_leaf_spot[0], elastic_rust[0],\n                   elastic_scab[0], elastic_complex[0],\n                   elastic_healthy[0], elastic_powdery_mildew[0]]\nimage_titles = unique_labels\n\n# Create a figure with a 2x3 grid of subplots\nfig, axes = plt.subplots(2, 3, figsize=(10, 8))\n\n# Iterate over the image filenames, titles, and corresponding subplots\nfor i, (image_filename, image_title, ax) in enumerate(zip(image_filenames, image_titles, axes.flatten())):\n    # Load the image\n    image = plt.imread(image_filename)\n    \n    # Show the image in the subplot\n    ax.imshow(image)\n    ax.set_title(image_title)\n    ax.axis('off')\n\n# Adjust the layout of subplots to avoid overlapping\nplt.tight_layout()\n\n# Display the figure\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gaus_frog_eye_leaf_spot = list(gaus_df[gaus_df[\"frog_eye_leaf_spot\"]==1].image)\ngaus_rust = list(gaus_df[gaus_df[\"rust\"]==1].image)\ngaus_scab = list(gaus_df[gaus_df[\"scab\"]==1].image)\ngaus_complex = list(gaus_df[gaus_df[\"complex\"]==1].image)\ngaus_healthy = list(gaus_df[gaus_df[\"healthy\"]==1].image)\ngaus_powdery_mildew = list(gaus_df[gaus_df[\"powdery_mildew\"]==1].image)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# List of image filenames and corresponding titles\nimage_filenames = [gaus_frog_eye_leaf_spot[0], gaus_rust[0],\n                   gaus_scab[0], gaus_complex[0],\n                   gaus_healthy[0], gaus_powdery_mildew[0]]\nimage_titles = unique_labels\n\n# Create a figure with a 2x3 grid of subplots\nfig, axes = plt.subplots(2, 3, figsize=(10, 8))\n\n# Iterate over the image filenames, titles, and corresponding subplots\nfor i, (image_filename, image_title, ax) in enumerate(zip(image_filenames, image_titles, axes.flatten())):\n    # Load the image\n    image = plt.imread(image_filename)\n    \n    # Show the image in the subplot\n    ax.imshow(image)\n    ax.set_title(image_title)\n    ax.axis('off')\n\n# Adjust the layout of subplots to avoid overlapping\nplt.tight_layout()\n\n# Display the figure\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Label distribution after cutout and elastic transformation augmentation","metadata":{}},{"cell_type":"code","source":"# Reset the index \ntrain_df.reset_index(drop=True, inplace=True)\ncutout_df.reset_index(drop=True, inplace=True)\nelastic_transform_df.reset_index(drop=True, inplace=True)\n\n# Concatenate the two dataframes vertically\ncombined_df = pd.concat([train_df, cutout_df, elastic_transform_df, gaus_df], axis=0)\n\n# Reset the index of the combined dataframe\ncombined_df.reset_index(drop=True, inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Original dataframe length:\", len(train_df),\n      \"\\nCutout dataframe length:\", len(cutout_df),\n      \"\\nElastic transform dataframe length:\", len(elastic_transform_df),\n      \"\\nGaussian noise dataframe length:\", len(gaus_df))\n\nprint(\"------------------------------------------------------------\")\nprint(\"Combined length:\", len(combined_df))\nprint(\"____________________________________________________________\")\nprint(\"Added length:\", len(combined_df) - len(train_df))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = combined_df[['image']]\ny_train = combined_df.drop('image', axis=1)\ny_train = y_train.to_numpy()\ny_train = y_train.astype(np.float32)\n\n# Convert DataFrame column to NumPy array and change shape\nX_train = X_train['image'].values.squeeze()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the sum of each label column\nlabel_counts = combined_df.iloc[:, 1:].sum()\n\n# Define a colormap for assigning colors to labels\ncolormap = plt.cm.get_cmap('tab10')\n\n# Get the number of unique labels\nnum_labels = len(label_counts)\n\n# Generate an array of colors based on the colormap\ncolors = [colormap(i) for i in np.linspace(0, 1, num_labels)]\n\n# Plot the label distribution with colored bars\nplt.figure(figsize=(10, 6))\nplt.bar(label_counts.index, label_counts.values, color=colors)\nplt.xlabel('Labels')\nplt.ylabel('Count')\nplt.title('Label Distribution')\nplt.xticks(rotation=45)\n\n# Add count labels above the bars\nfor i, count in enumerate(label_counts):\n    plt.text(i, count + 50, str(count), ha='center')\n\n# Add a line below the total count\ntotal_count = label_counts.sum()\nplt.axhline(total_count, color='black', linestyle='--', alpha=0.5)\n\n# Add total count of all labels above the title with increased font size and bold style\nplt.text(0.5, 1.08, f'Total Count: {total_count}', transform=plt.gca().transAxes, ha='center', fontsize=12, fontweight='bold')\n\n# Calculate the maximum count for setting the y-axis limit\nmax_count = max(label_counts)\ny_limit = max_count + max_count * 0.1  # Add some padding\n\nplt.ylim(top=y_limit)  # Set the y-axis limit to accommodate the line\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Shape of training and validation after augmentation ","metadata":{}},{"cell_type":"code","source":"print('Shape of X_train : ',X_train.shape)\nprint('Shape of y_train : ',y_train.shape)\nprint('===============================================')\nprint('Shape of X_val : ',X_val.shape)\nprint('Shape of y_val : ',y_val.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nAUTO = tf.data.experimental.AUTOTUNE\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Set the batch size and image size","metadata":{}},{"cell_type":"code","source":"if(strategy.num_replicas_in_sync != 1 ):\n    print('TPU used')\n    BATCH_SIZE = 48 * strategy.num_replicas_in_sync\nelse:\n    BATCH_SIZE = 12\n    \nprint(\"Batch size:\", BATCH_SIZE)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEPS_PER_EPOCH = y_train.shape[0] // BATCH_SIZE\nVALIDATION_STEPS = y_val.shape[0] // BATCH_SIZE\n\nprint (\"Steps per epoch: \", STEPS_PER_EPOCH, \"\\nValidation steps: \", VALIDATION_STEPS)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_height = 380\nimage_width = 380","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image(filename, label=None, image_size=(image_height, image_width)):\n    bits = tf.io.read_file(filename)\n    image = tf.image.decode_jpeg(bits, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.image.resize(image, image_size)\n    \n    if label is None:\n        return image\n    else:\n        return image, label\n\ndef data_augment(image, label=None):\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_flip_up_down(image)\n    image = tf.image.random_brightness(image, max_delta=0.3)\n    image = tf.image.random_hue(image, max_delta=0.3)\n    \n    if label is None:\n        return image\n    else:\n        return image, label","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(AUTO)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset.from_tensor_slices((X_train, y_train))\n    .map(decode_image, num_parallel_calls = AUTO)\n    .map(data_augment, num_parallel_calls = AUTO)\n    .repeat()\n    .shuffle(256)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((X_val, y_val))\n    .map(decode_image, num_parallel_calls = AUTO)\n    .cache()\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Training** **model**","metadata":{}},{"cell_type":"markdown","source":"# Callback","metadata":{}},{"cell_type":"code","source":"tensorboard = TensorBoard(log_dir = 'logs')\n\ncheckpoint = ModelCheckpoint(\"effnet.h5\", monitor=\"val_accuracy\",save_best_only=True,\n                             mode=\"auto\", verbose=1)\nreduce_lr = ReduceLROnPlateau(monitor = 'val_accuracy', factor = 0.1, patience = 2, min_delta = 0.001,\n                              mode='auto', verbose=1, min_lr=1e-9)\n\n# Choosing to monitor val_accuracy since it is classification task and accuracy is more important\nearly_stop=EarlyStopping(monitor='val_accuracy', restore_best_weights= True,\n                             patience=7, verbose=1)\n\nfilename='history.csv'\nhistory_logger=tf.keras.callbacks.CSVLogger(filename, separator=\",\", append=True)\n\n\ncallback_options = [tensorboard, checkpoint, reduce_lr, early_stop, history_logger]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transfer learning\nIn this notebook for training we are using **Transfer Learning** concept <br>\n\nI will be using **EfficientNetB4** model with weights from **imagenet**. <br>\nAnd I will using **include_top=False** option since it allows me to add my own output layer. <br>\nAnd also i will freeze base model layers using ***effnet.trainable = False***<br>\n**EfficientNetB4** model is used because *image size is *380x380*\n\nThe original image sizes used for every version of EfficientNet are:\n\n    EfficientNetB0 - (224, 224, 3)\n    EfficientNetB1 - (240, 240, 3)\n    EfficientNetB2 - (260, 260, 3)\n    EfficientNetB3 - (300, 300, 3)\n    EfficientNetB4 - (380, 380, 3)\n    EfficientNetB5 - (456, 456, 3)\n    EfficientNetB6 - (528, 528, 3)\n    EfficientNetB7 - (600, 600, 3)","metadata":{}},{"cell_type":"code","source":"# effnet = Sequential([efn.EfficientNetB4(input_shape=(image_height, image_width,3), weights='noisy-student', include_top=False)])\n\neffnet = Sequential([efn.EfficientNetB4(input_shape=(image_height, image_width,3), weights='noisy-student', include_top=False)])\n# effnet.trainable = False","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = effnet.output\nmodel = tf.keras.layers.GlobalAveragePooling2D()(model)\nmodel = tf.keras.layers.Dense(256,activation='relu')(model)\nmodel = tf.keras.layers.Dense(128,activation='relu')(model)\nmodel = tf.keras.layers.BatchNormalization()(model)\nmodel = tf.keras.layers.Dropout(rate=0.2)(model)\nmodel = tf.keras.layers.Dense(64,activation='relu')(model)\nmodel = tf.keras.layers.BatchNormalization()(model)\nmodel = tf.keras.layers.Dropout(rate=0.2)(model)\nmodel = tf.keras.layers.Dense(6,activation='sigmoid')(model)\nmodel = tf.keras.models.Model(inputs=effnet.input, outputs = model)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = effnet.output\n# model = tf.keras.layers.Conv2D(32, (5, 5), activation='relu', padding='same')(model)\n# model = tf.keras.layers.MaxPooling2D((2, 2))(model)\n# model = tf.keras.layers.Conv2D(64, (3, 3), activation='relu', padding='same')(model)\n# model = tf.keras.layers.MaxPooling2D((2, 2))(model)\n# model = tf.keras.layers.Conv2D(128, (5, 5), activation='relu', padding='same')(model)\n# model = tf.keras.layers.MaxPooling2D((2, 2))(model)\n# model = tf.keras.layers.Conv2D(256, (3, 3), activation='relu', padding='same')(model)\n# model = tf.keras.layers.MaxPooling2D((2, 2))(model)\n# model = tf.keras.layers.GlobalAveragePooling2D()(model)\n# model = tf.keras.layers.Dense(1024,activation='relu')(model)\n# model = tf.keras.layers.Dense(512,activation='relu')(model)\n# model = tf.keras.layers.Dense(256,activation='relu', kernel_regularizer='l2')(model)\n# model = tf.keras.layers.Dense(128,activation='relu')(model)\n# model = tf.keras.layers.BatchNormalization()(model)\n# model = tf.keras.layers.Dropout(rate=0.2)(model)\n# model = tf.keras.layers.Dense(64,activation='relu')(model)\n# model = tf.keras.layers.Dense(6,activation='sigmoid')(model)\n# model = tf.keras.models.Model(inputs=effnet.input, outputs = model)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Image\n\nmodel.summary()\nplot_model(model, to_file='model_Eff_B7.png', show_shapes=True, show_layer_names=True)\nImage('model_Eff_B7.png',width=400, height=200)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\n\n# Count the number of occurrences for each label\nlabel_counts = np.sum(y_train, axis=0)\n\n# Calculate the class weights based on label frequencies\ntotal_samples = y_train.shape[0]\nclass_weights = total_samples / (len(label_counts) * label_counts)\n\nclass_weights_dict = dict(zip(range(len(label_counts)), class_weights))\n\nprint(class_weights_dict)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the custom loss function\ndef weighted_loss(y_true, y_pred, class_weights):\n    weights = tf.reduce_sum(class_weights * y_true, axis=1)\n    loss = tf.keras.losses.BinaryCrossentropy()(y_true, y_pred)\n    weighted_loss = tf.reduce_mean(loss * weights)\n    return weighted_loss","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.compile(loss='binary_crossentropy',optimizer = 'Adam', metrics= ['accuracy'], loss_weights=class_weights_dict)\n\nmodel.compile(loss='binary_crossentropy',optimizer = 'Adam', metrics= ['accuracy'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_dataset,\n                    epochs=30,\n                    callbacks = callback_options,\n                    class_weight=class_weights_dict,\n                    steps_per_epoch=STEPS_PER_EPOCH,\n                    validation_data=valid_dataset,\n                    validation_steps=VALIDATION_STEPS)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\n\n# Save the training history as a CSV file\nhistory_df.to_csv('training_history.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.716207Z","iopub.status.idle":"2023-05-20T13:28:26.716898Z","shell.execute_reply.started":"2023-05-20T13:28:26.716645Z","shell.execute_reply":"2023-05-20T13:28:26.716685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure()\nfig,(ax1, ax2)=plt.subplots(1,2,figsize=(19,7))\nax1.plot(history.history['loss'])\nax1.plot(history.history['val_loss'])\nax1.legend(['training','validation'])\nax1.set_title('loss')\nax1.set_xlabel('epoch')\n\nax2.plot(history.history['accuracy'])\nax2.plot(history.history['val_accuracy'])\nax2.legend(['training','validation'])\nax2.set_title('Acurracy')\nax2.set_xlabel('epoch')","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.719114Z","iopub.status.idle":"2023-05-20T13:28:26.719826Z","shell.execute_reply.started":"2023-05-20T13:28:26.719564Z","shell.execute_reply":"2023-05-20T13:28:26.719588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Access the best epoch and corresponding accuracy value\nbest_epoch = early_stop.stopped_epoch + 1\nbest_accuracy = max(history.history['val_accuracy'])\n\n# Print the best epoch and accuracy value\nprint(f\"Best Epoch: {best_epoch}\")\nprint(f\"Best Accuracy: {best_accuracy}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.721060Z","iopub.status.idle":"2023-05-20T13:28:26.721824Z","shell.execute_reply.started":"2023-05-20T13:28:26.721564Z","shell.execute_reply":"2023-05-20T13:28:26.721588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction","metadata":{}},{"cell_type":"code","source":"SUB_PATH = \"../input/plant-pathology-2021-fgvc8/sample_submission.csv\"\nsub_df = pd.read_csv(SUB_PATH)\n\ndef format_test_path(st):\n    return '/kaggle/input/plant-pathology-2021-fgvc8/test_images/' + st\n\ndef format_own_path(st):\n    return '/kaggle/input/own-pictures' + st\n\ntest_paths = sub_df.image.apply(format_test_path).values","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.723053Z","iopub.status.idle":"2023-05-20T13:28:26.723816Z","shell.execute_reply.started":"2023-05-20T13:28:26.723479Z","shell.execute_reply":"2023-05-20T13:28:26.723546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transform the labels into a binary matrix with one-hot encoding for the six common labels\nlabel_matrix = mlb.fit_transform(sub_df['labels'].str.split())\n\n# Create a new dataframe with the one-hot encoded labels\nlabel_df = pd.DataFrame(label_matrix, columns=common_labels)\n\n# fixes train_df row number error\nsub_df.reset_index(drop=True, inplace=True)\nlabel_df.reset_index(drop=True, inplace=True)\n\n# Concatenate the new label dataframe with the original dataframe\nnew_df = pd.concat([sub_df, label_df], axis=1)\n\n# Drop the original labels column from the new dataframe\nsub_df = new_df.drop('labels', axis=1)\nsub_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.725231Z","iopub.status.idle":"2023-05-20T13:28:26.725930Z","shell.execute_reply.started":"2023-05-20T13:28:26.725690Z","shell.execute_reply":"2023-05-20T13:28:26.725713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(test_paths)\n    .map(decode_image)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.727133Z","iopub.status.idle":"2023-05-20T13:28:26.727829Z","shell.execute_reply.started":"2023-05-20T13:28:26.727570Z","shell.execute_reply":"2023-05-20T13:28:26.727593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probs_efn = model.predict(test_dataset, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.729043Z","iopub.status.idle":"2023-05-20T13:28:26.729744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.loc[:, 'scab':] = probs_efn\nsub_df.to_csv('prediction.csv', index=False)\nsub_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.730939Z","iopub.status.idle":"2023-05-20T13:28:26.731618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making predictions with my own pictures","metadata":{}},{"cell_type":"code","source":"# Recreate the exact same model, including weights and optimizer.\nmodel = tf.keras.models.load_model('/kaggle/input/model-from-46-version/effnet.h5')\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.732841Z","iopub.status.idle":"2023-05-20T13:28:26.733750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_dir = \"/kaggle/input/own-pictures-new\"  # Directory path containing the images\n\nimage_paths = []  # Array to store the image paths\nimage_names = []\n\n# Iterate over each file in the directory\nfor file in os.listdir(image_dir):\n    # Check if the file is an image file\n    image_path = os.path.join(image_dir, file)\n        \n    # Append the image path to the array\n    image_paths.append(image_path)\n    image_names.append(file)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.734962Z","iopub.status.idle":"2023-05-20T13:28:26.735684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" BATCH_SIZE = 80 * 4\n\nown_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(image_paths)\n    .map(decode_image)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.736896Z","iopub.status.idle":"2023-05-20T13:28:26.737619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probs_own = model.predict(own_dataset, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.738841Z","iopub.status.idle":"2023-05-20T13:28:26.739518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probs_own","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.740792Z","iopub.status.idle":"2023-05-20T13:28:26.741455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = pd.DataFrame({'Column1': image_names})\ndf2 = pd.DataFrame(probs_own, columns=common_labels)\n\ndf = pd.concat([df1, df2], axis=1)\n\ndf","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.742693Z","iopub.status.idle":"2023-05-20T13:28:26.743416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Recreate the exact same model, including weights and optimizer.\nmodel_26_88acc = tf.keras.models.load_model('/kaggle/input/model-from-46-version/version_26_88acc.h5')\nmodel_26_88acc.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.744880Z","iopub.status.idle":"2023-05-20T13:28:26.745565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probs_own = model_26_88acc.predict(own_dataset, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.746802Z","iopub.status.idle":"2023-05-20T13:28:26.747469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probs_own","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.748697Z","iopub.status.idle":"2023-05-20T13:28:26.749360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = pd.DataFrame({'Column1': image_names})\ndf2 = pd.DataFrame(probs_own, columns=common_labels)\n\ndf = pd.concat([df1, df2], axis=1)\n\ndf.iloc[:1]","metadata":{"execution":{"iopub.status.busy":"2023-05-20T13:28:26.750586Z","iopub.status.idle":"2023-05-20T13:28:26.751270Z"},"trusted":true},"execution_count":null,"outputs":[]}]}