{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.9","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"datasetVersion","sourceId":6523161,"datasetId":3771148,"databundleVersionId":6605744}],"dockerImageVersionId":30061,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nDATASET_PATH = \"/kaggle/input/five-crop-diseases-dataset\"\n\nprint(\"Main folders (crops):\")\nprint(os.listdir(DATASET_PATH))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:08:24.989488Z","iopub.execute_input":"2025-12-16T18:08:24.989749Z","iopub.status.idle":"2025-12-16T18:08:24.998122Z","shell.execute_reply.started":"2025-12-16T18:08:24.989687Z","shell.execute_reply":"2025-12-16T18:08:24.997242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nDATASET_PATH = \"/kaggle/input/five-crop-diseases-dataset/Crop Diseases Dataset\"\n\nprint(\"Folders inside dataset:\")\nprint(os.listdir(DATASET_PATH))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:08:33.415248Z","iopub.execute_input":"2025-12-16T18:08:33.415540Z","iopub.status.idle":"2025-12-16T18:08:33.425373Z","shell.execute_reply.started":"2025-12-16T18:08:33.415515Z","shell.execute_reply":"2025-12-16T18:08:33.424579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nDATASET_PATH = \"/kaggle/input/five-crop-diseases-dataset/Crop Diseases Dataset/Crop Diseases\"\n\nprint(\"Final folders:\")\nprint(os.listdir(DATASET_PATH))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:08:38.954540Z","iopub.execute_input":"2025-12-16T18:08:38.954816Z","iopub.status.idle":"2025-12-16T18:08:38.963719Z","shell.execute_reply.started":"2025-12-16T18:08:38.954793Z","shell.execute_reply":"2025-12-16T18:08:38.962869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nDATASET_PATH = \"/kaggle/input/five-crop-diseases-dataset/Crop Diseases Dataset/Crop Diseases/Crop___Disease\"\n\nprint(\"Actual class folders:\")\nprint(os.listdir(DATASET_PATH))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:08:43.573958Z","iopub.execute_input":"2025-12-16T18:08:43.574294Z","iopub.status.idle":"2025-12-16T18:08:43.582389Z","shell.execute_reply.started":"2025-12-16T18:08:43.574267Z","shell.execute_reply":"2025-12-16T18:08:43.581589Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Libraries and Configurations","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow.keras.layers as L\nimport tensorflow_addons as tfa\nimport glob, random, os, warnings\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, classification_report\nimport seaborn as sns\n\nprint('TensorFlow Version ' + tf.__version__)\n\ndef seed_everything(seed = 0):\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    os.environ['TF_DETERMINISTIC_OPS'] = '1'\n\nseed_everything()\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:08:50.835386Z","iopub.execute_input":"2025-12-16T18:08:50.835679Z","iopub.status.idle":"2025-12-16T18:09:01.002146Z","shell.execute_reply.started":"2025-12-16T18:08:50.835656Z","shell.execute_reply":"2025-12-16T18:09:01.001406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nDATASET_PATH = \"/kaggle/input/five-crop-diseases-dataset/Crop Diseases Dataset/Crop Diseases/Crop___Disease\"\n\nIMAGE_SIZE = 224\nBATCH_SIZE = 16\n\ndatagen = ImageDataGenerator(\n    rescale=1./255,\n    validation_split=0.2\n)\n\ntrain_generator = datagen.flow_from_directory(\n    DATASET_PATH,\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode=\"categorical\",\n    subset=\"training\"\n)\n\nval_generator = datagen.flow_from_directory(\n    DATASET_PATH,\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode=\"categorical\",\n    subset=\"validation\"\n)\n\nNUM_CLASSES = train_generator.num_classes\nprint(\" Number of classes detected:\", NUM_CLASSES)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:09:35.087419Z","iopub.execute_input":"2025-12-16T18:09:35.087709Z","iopub.status.idle":"2025-12-16T18:09:40.740619Z","shell.execute_reply.started":"2025-12-16T18:09:35.087685Z","shell.execute_reply":"2025-12-16T18:09:40.739830Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Augmentations","metadata":{}},{"cell_type":"code","source":"def data_augment(image):\n    p_spatial = tf.random.uniform([], 0, 1.0, dtype = tf.float32)\n    p_rotate = tf.random.uniform([], 0, 1.0, dtype = tf.float32)\n \n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_flip_up_down(image)\n    \n    if p_spatial > .75:\n        image = tf.image.transpose(image)\n        \n    # Rotates\n    if p_rotate > .75:\n        image = tf.image.rot90(image, k = 3) # rotate 270º\n    elif p_rotate > .5:\n        image = tf.image.rot90(image, k = 2) # rotate 180º\n    elif p_rotate > .25:\n        image = tf.image.rot90(image, k = 1) # rotate 90º\n        \n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:10:06.521003Z","iopub.execute_input":"2025-12-16T18:10:06.521314Z","iopub.status.idle":"2025-12-16T18:10:06.527362Z","shell.execute_reply.started":"2025-12-16T18:10:06.521290Z","shell.execute_reply":"2025-12-16T18:10:06.526244Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Sample Images Visualization","metadata":{}},{"cell_type":"code","source":"images, labels = next(train_generator)\n\nplt.figure(figsize=(10,10))\nfor i in range(9):\n    plt.subplot(3,3,i+1)\n    plt.imshow(images[i])\n    plt.axis(\"off\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:10:13.694337Z","iopub.execute_input":"2025-12-16T18:10:13.694644Z","iopub.status.idle":"2025-12-16T18:10:14.684974Z","shell.execute_reply.started":"2025-12-16T18:10:13.694621Z","shell.execute_reply":"2025-12-16T18:10:14.684295Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Hyperparameters","metadata":{}},{"cell_type":"code","source":"learning_rate = 0.0001\nweight_decay = 0.0001\nnum_epochs = 10\n\npatch_size = 7\nnum_patches = (IMAGE_SIZE // patch_size) ** 2\nprojection_dim = 64\nnum_heads = 4\ntransformer_units = [\n    projection_dim * 2,\n    projection_dim,\n]\ntransformer_layers = 8\nmlp_head_units = [56, 28]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:10:29.796959Z","iopub.execute_input":"2025-12-16T18:10:29.797302Z","iopub.status.idle":"2025-12-16T18:10:29.801456Z","shell.execute_reply.started":"2025-12-16T18:10:29.797272Z","shell.execute_reply":"2025-12-16T18:10:29.800782Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Building the Model and it's Components","metadata":{}},{"cell_type":"markdown","source":"## 1. Multilayer Perceptron (MLP)","metadata":{}},{"cell_type":"code","source":"def mlp(x, hidden_units, dropout_rate):\n    for units in hidden_units:\n        x = L.Dense(units, activation = tf.nn.gelu)(x)\n        x = L.Dropout(dropout_rate)(x)\n    return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:10:53.270382Z","iopub.execute_input":"2025-12-16T18:10:53.270663Z","iopub.status.idle":"2025-12-16T18:10:53.274910Z","shell.execute_reply.started":"2025-12-16T18:10:53.270639Z","shell.execute_reply":"2025-12-16T18:10:53.274134Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Patch Creation Layer","metadata":{}},{"cell_type":"code","source":"class Patches(L.Layer):\n    def __init__(self, patch_size):\n        super(Patches, self).__init__()\n        self.patch_size = patch_size\n\n    def call(self, images):\n        batch_size = tf.shape(images)[0]\n        patches = tf.image.extract_patches(\n            images = images,\n            sizes = [1, self.patch_size, self.patch_size, 1],\n            strides = [1, self.patch_size, self.patch_size, 1],\n            rates = [1, 1, 1, 1],\n            padding = 'VALID',\n        )\n        patch_dims = patches.shape[-1]\n        patches = tf.reshape(patches, [batch_size, -1, patch_dims])\n        return patches","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:11:01.976202Z","iopub.execute_input":"2025-12-16T18:11:01.976520Z","iopub.status.idle":"2025-12-16T18:11:01.982278Z","shell.execute_reply.started":"2025-12-16T18:11:01.976490Z","shell.execute_reply":"2025-12-16T18:11:01.981258Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Sample Image Patches Visualization","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(4, 4))\n\nimages, labels = next(train_generator)\nimage = images[0]\n\nplt.imshow(image)\nplt.axis('off')\n\nresized_image = tf.image.resize(\n    tf.convert_to_tensor([image]), size=(IMAGE_SIZE, IMAGE_SIZE)\n)\n\npatches = Patches(patch_size)(resized_image)\nprint(f'Image size: {IMAGE_SIZE} X {IMAGE_SIZE}')\nprint(f'Patch size: {patch_size} X {patch_size}')\nprint(f'Patches per image: {patches.shape[1]}')\nprint(f'Elements per patch: {patches.shape[-1]}')\n\nn = int(np.sqrt(patches.shape[1]))\nplt.figure(figsize=(4, 4))\n\nfor i, patch in enumerate(patches[0]):\n    ax = plt.subplot(n, n, i + 1)\n    patch_img = tf.reshape(patch, (patch_size, patch_size, 3))\n    plt.imshow(patch_img.numpy())\n    plt.axis('off')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:11:10.065472Z","iopub.execute_input":"2025-12-16T18:11:10.065759Z","iopub.status.idle":"2025-12-16T18:11:50.211900Z","shell.execute_reply.started":"2025-12-16T18:11:10.065735Z","shell.execute_reply":"2025-12-16T18:11:50.210956Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Patch Encoding Layer\nThe `PatchEncoder` layer will linearly transform a patch by projecting it into a vector of size `projection_dim`. In addition, it adds a learnable position embedding to the projected vector.","metadata":{}},{"cell_type":"code","source":"class PatchEncoder(L.Layer):\n    def __init__(self, num_patches, projection_dim):\n        super(PatchEncoder, self).__init__()\n        self.num_patches = num_patches\n        self.projection = L.Dense(units = projection_dim)\n        self.position_embedding = L.Embedding(\n            input_dim = num_patches, output_dim = projection_dim\n        )\n\n    def call(self, patch):\n        positions = tf.range(start = 0, limit = self.num_patches, delta = 1)\n        encoded = self.projection(patch) + self.position_embedding(positions)\n        return encoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:12:02.011565Z","iopub.execute_input":"2025-12-16T18:12:02.011853Z","iopub.status.idle":"2025-12-16T18:12:02.017249Z","shell.execute_reply.started":"2025-12-16T18:12:02.011829Z","shell.execute_reply":"2025-12-16T18:12:02.016375Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Build the ViT model\nThe ViT model consists of multiple Transformer blocks, which use the `MultiHeadAttention` layer as a self-attention mechanism applied to the sequence of patches. The Transformer blocks produce a `[batch_size, num_patches, projection_dim]` tensor, which is processed via an classifier head with softmax to produce the final class probabilities output.\n\nUnlike the technique described in the paper, which prepends a learnable embedding to the sequence of encoded patches to serve as the image representation, all the outputs of the final Transformer block are reshaped with `Flatten()` and used as the image representation input to the classifier head. Note that the `GlobalAveragePooling1D` layer could also be used instead to aggregate the outputs of the Transformer block, especially when the number of patches and the projection dimensions are large.","metadata":{}},{"cell_type":"code","source":"def vision_transformer():\n    inputs = L.Input(shape=(IMAGE_SIZE, IMAGE_SIZE, 3))\n    \n    # Create patches.\n    patches = Patches(patch_size)(inputs)\n    \n    # Encode patches.\n    encoded_patches = PatchEncoder(num_patches, projection_dim)(patches)\n\n    # Create multiple layers of the Transformer block.\n    for _ in range(transformer_layers):\n        x1 = L.LayerNormalization(epsilon=1e-6)(encoded_patches)\n        \n        attention_output = L.MultiHeadAttention(\n            num_heads=num_heads, key_dim=projection_dim, dropout=0.1\n        )(x1, x1)\n        \n        x2 = L.Add()([attention_output, encoded_patches])\n        x3 = L.LayerNormalization(epsilon=1e-6)(x2)\n        x3 = mlp(x3, hidden_units=transformer_units, dropout_rate=0.1)\n        encoded_patches = L.Add()([x3, x2])\n\n    representation = L.LayerNormalization(epsilon=1e-6)(encoded_patches)\n    representation = L.Flatten()(representation)\n    representation = L.Dropout(0.5)(representation)\n\n    features = mlp(representation, hidden_units=mlp_head_units, dropout_rate=0.5)\n\n    # Correct output layer for your dataset\n    logits = L.Dense(NUM_CLASSES, activation=\"softmax\")(features)\n\n    model = tf.keras.Model(inputs=inputs, outputs=logits)\n    \n    return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:12:08.289716Z","iopub.execute_input":"2025-12-16T18:12:08.290029Z","iopub.status.idle":"2025-12-16T18:12:08.296594Z","shell.execute_reply.started":"2025-12-16T18:12:08.289986Z","shell.execute_reply":"2025-12-16T18:12:08.295791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"decay_steps = train_generator.samples // train_generator.batch_size\ninitial_learning_rate = learning_rate\n\nlr_decayed_fn = tf.keras.experimental.CosineDecay(\n    initial_learning_rate,\n    decay_steps\n)\n\nlr_scheduler = tf.keras.callbacks.LearningRateScheduler(lr_decayed_fn)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:12:15.688192Z","iopub.execute_input":"2025-12-16T18:12:15.688524Z","iopub.status.idle":"2025-12-16T18:12:15.692398Z","shell.execute_reply.started":"2025-12-16T18:12:15.688493Z","shell.execute_reply":"2025-12-16T18:12:15.691593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer = tf.keras.optimizers.Adam(learning_rate=learning_rate)\n\nmodel = vision_transformer()\n\nmodel.compile(\n    optimizer=optimizer,\n    loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.1),\n    metrics=[\"accuracy\"]\n)\n\nSTEP_SIZE_TRAIN = train_generator.samples // train_generator.batch_size\nSTEP_SIZE_VALID = val_generator.samples // val_generator.batch_size\n\nearlystopping = tf.keras.callbacks.EarlyStopping(\n    monitor=\"val_accuracy\",\n    min_delta=1e-4,\n    patience=5,\n    mode=\"max\",\n    restore_best_weights=True,\n    verbose=1\n)\n\ncheckpointer = tf.keras.callbacks.ModelCheckpoint(\n    filepath=\"./model.hdf5\",\n    monitor=\"val_accuracy\",\n    verbose=1,\n    save_best_only=True,\n    save_weights_only=True,\n    mode=\"max\"\n)\n\ncallbacks = [earlystopping, lr_scheduler, checkpointer]\n\nhistory = model.fit(\n    train_generator,\n    steps_per_epoch=STEP_SIZE_TRAIN,\n    validation_data=val_generator,\n    validation_steps=STEP_SIZE_VALID,\n    epochs=num_epochs,\n    callbacks=callbacks\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:12:26.613555Z","iopub.execute_input":"2025-12-16T18:12:26.613868Z","iopub.status.idle":"2025-12-16T19:08:09.973298Z","shell.execute_reply.started":"2025-12-16T18:12:26.613836Z","shell.execute_reply":"2025-12-16T19:08:09.972480Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Results","metadata":{}},{"cell_type":"code","source":"print(\"Training results\")\ntrain_loss, train_acc = model.evaluate(train_generator)\n\nprint(\"Validation results\")\nval_loss, val_acc = model.evaluate(val_generator)\n\nprint(f\"Training Accuracy: {train_acc * 100:.2f}%\")\nprint(f\"Validation Accuracy: {val_acc * 100:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T19:08:58.642122Z","iopub.execute_input":"2025-12-16T19:08:58.642429Z","iopub.status.idle":"2025-12-16T19:13:45.107864Z","shell.execute_reply.started":"2025-12-16T19:08:58.642405Z","shell.execute_reply":"2025-12-16T19:13:45.107197Z"}},"outputs":[],"execution_count":null}]}