{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"},{"sourceId":8210035,"sourceType":"datasetVersion","datasetId":4865295},{"sourceId":207,"sourceType":"modelInstanceVersion","modelInstanceId":146}],"dockerImageVersionId":30498,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport datetime\nimport random\nimport shutil\nimport os, cv2, json, glob\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nimport tensorflow as tf\nimport keras_tuner as kt\nfrom tensorflow import keras\nfrom tensorflow.keras import models, layers\nfrom keras.models import Model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.applications import efficientnet_v2\nfrom keras.optimizers import Adam\nfrom keras_tuner.tuners import Hyperband\nfrom tensorflow.keras.preprocessing.image import img_to_array\nfrom tensorflow.keras.preprocessing import image as kimage","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-23T22:22:18.217954Z","iopub.execute_input":"2024-04-23T22:22:18.218617Z","iopub.status.idle":"2024-04-23T22:22:26.870982Z","shell.execute_reply.started":"2024-04-23T22:22:18.218575Z","shell.execute_reply":"2024-04-23T22:22:26.870115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\nWORK_DIR = '../input/cassava-leaf-disease-classification/'\nOUTPUT_DIR = './'\nif not os.path.exists(OUTPUT_DIR):\n    os.makedirs(OUTPUT_DIR)  \ntarget_folder = \"dataset\"\npath = \"../input/cassava-leaf-disease-classification/train_images\"\nif not os.path.exists(target_folder):\n    os.mkdir(target_folder)\n\nfor label in df['label'].unique():\n    label_folder = os.path.join(target_folder, str(label))\n    if not os.path.exists(label_folder):\n        os.mkdir(label_folder)\n    mask = df['label'] == label\n    rows = df.loc[mask]\n    \n    for _, row in rows.iterrows():\n        image_name = row['image_id']\n        src_path = os.path.join(path, image_name)\n        dst_path = os.path.join(label_folder, image_name)\n        shutil.copy(src_path, dst_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T22:22:26.872910Z","iopub.execute_input":"2024-04-23T22:22:26.873753Z","iopub.status.idle":"2024-04-23T22:25:23.227870Z","shell.execute_reply.started":"2024-04-23T22:22:26.873717Z","shell.execute_reply":"2024-04-23T22:25:23.226526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def img_train(directory, batch_size=32, image_size=(224, 224), validation_split=0.1, seed=123):\n    train_dataset = tf.keras.utils.image_dataset_from_directory(\n        directory,\n        labels=\"inferred\",\n        label_mode=\"categorical\",\n        class_names=None,\n        color_mode=\"rgb\",\n        batch_size=batch_size,  \n        image_size=image_size,  \n        shuffle=True,   \n        seed=seed,\n        validation_split=validation_split,\n        subset=\"training\",\n        interpolation=\"bilinear\",\n        crop_to_aspect_ratio=False\n    )\n    if validation_split != None:\n        valid_dataset = tf.keras.utils.image_dataset_from_directory(\n            directory,\n            labels=\"inferred\",\n            label_mode=\"categorical\",\n            class_names=None,\n            color_mode=\"rgb\",\n            batch_size=batch_size,  \n            image_size=image_size,  \n            shuffle=True,   \n            seed=seed,\n            validation_split=validation_split,\n            subset=\"validation\",\n            interpolation=\"bilinear\",\n            crop_to_aspect_ratio=False\n        )\n        return train_dataset, valid_dataset\n    else:\n        return train_dataset","metadata":{"execution":{"iopub.status.busy":"2024-04-23T22:25:23.229369Z","iopub.execute_input":"2024-04-23T22:25:23.229759Z","iopub.status.idle":"2024-04-23T22:25:23.238460Z","shell.execute_reply.started":"2024-04-23T22:25:23.229727Z","shell.execute_reply":"2024-04-23T22:25:23.237428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = '/kaggle/working/dataset'\ntrain_dataset, test_dataset = img_train(dataset, 16)\nsplit = len(train_dataset)\nsize = int(split * 0.1)\ntrain_dataset = train_dataset.skip(size)\nvalid_dataset = train_dataset.take(size)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T22:25:23.240765Z","iopub.execute_input":"2024-04-23T22:25:23.241090Z","iopub.status.idle":"2024-04-23T22:25:28.161092Z","shell.execute_reply.started":"2024-04-23T22:25:23.241064Z","shell.execute_reply":"2024-04-23T22:25:28.158869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_augmentation():\n    return tf.keras.Sequential([\n        tf.keras.layers.RandomFlip(\"horizontal\", seed=123),\n        tf.keras.layers.RandomTranslation(0.2, 0.2, fill_mode=\"reflect\", seed=123),\n        tf.keras.layers.RandomRotation(0.2, fill_mode=\"reflect\", seed=123),\n        tf.keras.layers.RandomZoom(0.2, seed=123),\n        tf.keras.layers.RandomContrast(0.2, seed=123),\n    ])\ndata_aug = data_augmentation()\naugmented_dataset = train_dataset.repeat(2).map(lambda x, y: (data_aug(x), y))\ntrain_dataset = train_dataset.concatenate(augmented_dataset)\nprint(\"augmentation train dataset size:\", train_dataset.cardinality().numpy())","metadata":{"execution":{"iopub.status.busy":"2024-04-23T22:25:28.162522Z","iopub.execute_input":"2024-04-23T22:25:28.163245Z","iopub.status.idle":"2024-04-23T22:25:28.778955Z","shell.execute_reply.started":"2024-04-23T22:25:28.163209Z","shell.execute_reply":"2024-04-23T22:25:28.777940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SigmoidFocalCrossEntropy(tf.keras.losses.Loss):\n    def __init__(self, alpha=0.25, gamma=2.0, from_logits=False, **kwargs):\n        super().__init__(**kwargs)\n        self.alpha = alpha\n        self.gamma = gamma\n        self.from_logits = from_logits\n\n    def call(self, y_true, y_pred):\n        if self.from_logits:\n            y_pred = tf.sigmoid(y_pred)\n        y_pred = tf.clip_by_value(y_pred, tf.keras.backend.epsilon(), 1 - tf.keras.backend.epsilon())\n        cross_entropy = -y_true * tf.math.log(y_pred) - (1 - y_true) * tf.math.log(1 - y_pred)\n        weight = self.alpha * y_true + (1 - self.alpha) * (1 - y_true)\n        focal_loss = weight * ((1 - y_pred) ** self.gamma) * cross_entropy\n        return tf.reduce_sum(focal_loss, axis=-1)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T22:25:28.780008Z","iopub.execute_input":"2024-04-23T22:25:28.780307Z","iopub.status.idle":"2024-04-23T22:25:28.788692Z","shell.execute_reply.started":"2024-04-23T22:25:28.780281Z","shell.execute_reply":"2024-04-23T22:25:28.787649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(hp):\n    efficientweight = '/kaggle/input/test555/efficientnetv2s.h5'\n    learning_rate = hp.Float('learning_rate', min_value=1e-4, max_value=1e-2, sampling='log')\n    dropout_rate = hp.Float('dropout_rate', min_value=0, max_value=0.5)\n    dense_units = hp.Int('dense_units', min_value=128, max_value=512, step=32)\n    model = tf.keras.applications.efficientnet_v2.EfficientNetV2S(\n        weights=efficientweight,\n        include_top=False,\n        input_shape=(224, 224, 3)\n    )\n    x = tf.keras.layers.GlobalAveragePooling2D()(model.output)\n    x = tf.keras.layers.Dense(dense_units, activation='relu')(x)\n    x = tf.keras.layers.Dropout(dropout_rate)(x)\n    outputs = tf.keras.layers.Dense(5, activation='softmax')(x)\n    model = tf.keras.Model(inputs=model.input, outputs=outputs)\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=learning_rate),\n        loss=SigmoidFocalCrossEntropy(alpha=0.25, gamma=2, from_logits=False),\n        metrics=[tf.keras.metrics.CategoricalAccuracy(name=\"accuracy\")]\n    )\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-04-23T22:25:28.790338Z","iopub.execute_input":"2024-04-23T22:25:28.790734Z","iopub.status.idle":"2024-04-23T22:25:28.800868Z","shell.execute_reply.started":"2024-04-23T22:25:28.790699Z","shell.execute_reply":"2024-04-23T22:25:28.799842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tuner = kt.Hyperband(\n    build_model,\n    objective='val_accuracy',\n    max_epochs=2,  # Decreased from 5\n    hyperband_iterations=1,\n    factor=4,  # Increased from 3\n    directory='my_dir',\n    project_name='keras_tuner_efficientnet'\n)\n\nearly_stop = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss',\n    min_delta=0.01,  # Increased for a less stringent change requirement\n    patience=2,  # Decreased from 3\n    mode='min',\n    verbose=1,\n    restore_best_weights=True\n)\n\ntuner.search(\n    train_dataset, \n    validation_data=valid_dataset, \n    epochs=2,\n    callbacks=[early_stop]\n)\n\nbest_hps = tuner.get_best_hyperparameters()[0]","metadata":{"execution":{"iopub.status.busy":"2024-04-23T22:25:28.801903Z","iopub.execute_input":"2024-04-23T22:25:28.802229Z","iopub.status.idle":"2024-04-23T22:25:29.671777Z","shell.execute_reply.started":"2024-04-23T22:25:28.802204Z","shell.execute_reply":"2024-04-23T22:25:29.669990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Model_Training(tuner, train_dataset, valid_dataset, epochs=10):\n    best_hps = tuner.get_best_hyperparameters()[0]\n    model = tuner.hypermodel.build(best_hps)\n    early_stop = tf.keras.callbacks.EarlyStopping(\n        monitor='val_loss', min_delta=0.001, patience=5, mode='min', verbose=1, restore_best_weights=True\n    )\n    history = model.fit(\n        train_dataset,\n        epochs=epochs,\n        validation_data=valid_dataset,\n        callbacks=[early_stop]\n    )\n    plt.plot(history.history['accuracy'])\n    plt.plot(history.history['val_accuracy'])\n    plt.title('Model accuracy')\n    plt.ylabel('Accuracy')\n    plt.xlabel('Epoch')\n    plt.legend(['Train', 'Test'], loc='upper left')\n    plt.show()\n    plt.clf()\n\n    plt.plot(history.history['loss'])\n    plt.plot(history.history['val_loss'])\n    plt.title('Model loss')\n    plt.ylabel('Loss')\n    plt.xlabel('Epoch')\n    plt.legend(['Train', 'Test'], loc='upper left')\n    plt.show()\n    plt.clf()\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-04-23T22:25:29.672741Z","iopub.status.idle":"2024-04-23T22:25:29.673215Z","shell.execute_reply.started":"2024-04-23T22:25:29.672993Z","shell.execute_reply":"2024-04-23T22:25:29.673025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model_Training(tuner, train_dataset, valid_dataset, epochs=10)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T22:25:29.674890Z","iopub.status.idle":"2024-04-23T22:25:29.675257Z","shell.execute_reply.started":"2024-04-23T22:25:29.675060Z","shell.execute_reply":"2024-04-23T22:25:29.675075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(columns=['image_id','label'])\nfor image_name in os.listdir(WORK_DIR + 'test_images'):\n    image_path = os.path.join(WORK_DIR + 'test_images', image_name)\n    image = tf.keras.preprocessing.image.load_img(image_path)\n    resized_image = image.resize((224, 224))\n    numpied_image = np.expand_dims(resized_image, 0)\n    tensored_image = tf.cast(numpied_image, tf.float32)\n    y_pred = model.predict(tensored_image)\n    y_pred = np.argmax(y_pred, axis=-1)[0]\n    submission.loc[len(submission)] = [image_name, int(y_pred)]\nsubmission.to_csv('submission.csv', index=False)\nfolder = '/kaggle/working/dataset'\ntry:\n    shutil.rmtree(folder)\n    print(\"Folder Deleted\")\nexcept OSError as e:\n    print(f\"Error: {e}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T22:25:29.676681Z","iopub.status.idle":"2024-04-23T22:25:29.677019Z","shell.execute_reply.started":"2024-04-23T22:25:29.676850Z","shell.execute_reply":"2024-04-23T22:25:29.676866Z"},"trusted":true},"execution_count":null,"outputs":[]}]}