{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"},{"sourceId":8200264,"sourceType":"datasetVersion","datasetId":4857783},{"sourceId":207,"sourceType":"modelInstanceVersion","modelInstanceId":146}],"dockerImageVersionId":30498,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport datetime\nimport random\nimport shutil\nimport os, cv2, json\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import models, layers\nfrom keras.models import Model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.applications import efficientnet_v2\nfrom keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import img_to_array\nfrom tensorflow.keras.preprocessing import image as kimage","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-23T03:28:34.607861Z","iopub.execute_input":"2024-04-23T03:28:34.608704Z","iopub.status.idle":"2024-04-23T03:28:34.617818Z","shell.execute_reply.started":"2024-04-23T03:28:34.608660Z","shell.execute_reply":"2024-04-23T03:28:34.616587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\nWORK_DIR = '../input/cassava-leaf-disease-classification/'\nOUTPUT_DIR = './'\nif not os.path.exists(OUTPUT_DIR):\n    os.makedirs(OUTPUT_DIR)  \ntarget_folder = \"dataset\"\npath = \"../input/cassava-leaf-disease-classification/train_images\"\nif not os.path.exists(target_folder):\n    os.mkdir(target_folder)\n\nfor label in df['label'].unique():\n    label_folder = os.path.join(target_folder, str(label))\n    if not os.path.exists(label_folder):\n        os.mkdir(label_folder)\n    mask = df['label'] == label\n    rows = df.loc[mask]\n    \n    for _, row in rows.iterrows():\n        image_name = row['image_id']\n        src_path = os.path.join(path, image_name)\n        dst_path = os.path.join(label_folder, image_name)\n        shutil.copy(src_path, dst_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:28:34.620167Z","iopub.execute_input":"2024-04-23T03:28:34.620490Z","iopub.status.idle":"2024-04-23T03:30:40.847086Z","shell.execute_reply.started":"2024-04-23T03:28:34.620462Z","shell.execute_reply":"2024-04-23T03:30:40.846102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def img_train_dataset(directory, batch_size=32, image_size=(224, 224), validation_split=0.1, seed=123):\n    train_dataset = tf.keras.utils.image_dataset_from_directory(\n        directory,\n        labels=\"inferred\",\n        label_mode=\"categorical\",\n        class_names=None,\n        color_mode=\"rgb\",\n        batch_size=batch_size,  \n        image_size=image_size,  \n        shuffle=True,   \n        seed=seed,\n        validation_split=validation_split,\n        subset=\"training\",\n        interpolation=\"bilinear\",\n        crop_to_aspect_ratio=False\n    )\n    if validation_split != None:\n        valid_dataset = tf.keras.utils.image_dataset_from_directory(\n            directory,\n            labels=\"inferred\",\n            label_mode=\"categorical\",\n            class_names=None,\n            color_mode=\"rgb\",\n            batch_size=batch_size,  \n            image_size=image_size,  \n            shuffle=True,   \n            seed=seed,\n            validation_split=validation_split,\n            subset=\"validation\",\n            interpolation=\"bilinear\",\n            crop_to_aspect_ratio=False\n        )\n        return train_dataset, valid_dataset\n    else:\n        return train_dataset","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:30:40.848702Z","iopub.execute_input":"2024-04-23T03:30:40.849398Z","iopub.status.idle":"2024-04-23T03:30:40.859087Z","shell.execute_reply.started":"2024-04-23T03:30:40.849365Z","shell.execute_reply":"2024-04-23T03:30:40.858245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '/kaggle/working/dataset'\ntrain_dataset, test_dataset = img_train_dataset(path, 16)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:30:40.862243Z","iopub.execute_input":"2024-04-23T03:30:40.862600Z","iopub.status.idle":"2024-04-23T03:30:45.711387Z","shell.execute_reply.started":"2024-04-23T03:30:40.862543Z","shell.execute_reply":"2024-04-23T03:30:45.710381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_samples = len(train_dataset)\nval_size = int(num_samples * 0.1)\ntrain_dataset = train_dataset.skip(val_size)\nvalid_dataset = train_dataset.take(val_size)\nprint(\"Train dataset size:\", train_dataset.cardinality().numpy())\nprint(\"Validation dataset size:\", valid_dataset.cardinality().numpy())","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:30:45.712376Z","iopub.execute_input":"2024-04-23T03:30:45.712628Z","iopub.status.idle":"2024-04-23T03:30:45.727027Z","shell.execute_reply.started":"2024-04-23T03:30:45.712606Z","shell.execute_reply":"2024-04-23T03:30:45.726085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_augmentation():\n    return tf.keras.Sequential([\n        tf.keras.layers.RandomFlip(\"horizontal\", seed=123),\n        tf.keras.layers.RandomTranslation(0.2, 0.2, fill_mode=\"reflect\", seed=123),\n        tf.keras.layers.RandomRotation(0.2, fill_mode=\"reflect\", seed=123),\n        tf.keras.layers.RandomZoom(0.2, seed=123),\n        tf.keras.layers.RandomContrast(0.2, seed=123),\n    ])","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:30:45.728176Z","iopub.execute_input":"2024-04-23T03:30:45.728438Z","iopub.status.idle":"2024-04-23T03:30:45.737892Z","shell.execute_reply.started":"2024-04-23T03:30:45.728417Z","shell.execute_reply":"2024-04-23T03:30:45.737093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_aug = data_augmentation()\naugmented_dataset = train_dataset.repeat(2).map(lambda x, y: (data_aug(x), y))\ntrain_dataset = train_dataset.concatenate(augmented_dataset)\nprint(\"augmentation train dataset size:\", train_dataset.cardinality().numpy())","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:30:45.739036Z","iopub.execute_input":"2024-04-23T03:30:45.739411Z","iopub.status.idle":"2024-04-23T03:30:46.302493Z","shell.execute_reply.started":"2024-04-23T03:30:45.739388Z","shell.execute_reply":"2024-04-23T03:30:46.301543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"efficientweight = '/kaggle/input/test111/efficientnetv2-s_notop.h5'\ndef efficientnet_model(size=(224, 224, 3), classes=10):\n    model = tf.keras.applications.efficientnet_v2.EfficientNetV2S(weights=efficientweight, include_top=False, input_shape=size)\n    x = tf.keras.layers.GlobalAveragePooling2D()(model.output)\n    x = tf.keras.layers.Dense(256, activation='relu')(x)\n    x = tf.keras.layers.Dropout(0.5)(x)\n    output = tf.keras.layers.Dense(classes, activation='softmax')(x)\n    model = tf.keras.models.Model(inputs=model.input, outputs=output)\n    return model\nmodel = efficientnet_model(classes=5)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:30:46.303700Z","iopub.execute_input":"2024-04-23T03:30:46.303973Z","iopub.status.idle":"2024-04-23T03:30:52.090930Z","shell.execute_reply.started":"2024-04-23T03:30:46.303949Z","shell.execute_reply":"2024-04-23T03:30:52.090095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SigmoidFocalCrossEntropy(tf.keras.losses.Loss):\n    def __init__(self, alpha=0.25, gamma=2.0, from_logits=False, **kwargs):\n        super().__init__(**kwargs)\n        self.alpha = alpha\n        self.gamma = gamma\n        self.from_logits = from_logits\n\n    def call(self, y_true, y_pred):\n        if self.from_logits:\n            y_pred = tf.sigmoid(y_pred)\n        \n        y_pred = tf.clip_by_value(y_pred, tf.keras.backend.epsilon(), 1 - tf.keras.backend.epsilon())\n        cross_entropy = -y_true * tf.math.log(y_pred) - (1 - y_true) * tf.math.log(1 - y_pred)\n        weight = self.alpha * y_true + (1 - self.alpha) * (1 - y_true)\n        focal_loss = weight * ((1 - y_pred) ** self.gamma) * cross_entropy\n        return tf.reduce_sum(focal_loss, axis=-1)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:30:52.092112Z","iopub.execute_input":"2024-04-23T03:30:52.092390Z","iopub.status.idle":"2024-04-23T03:30:52.100375Z","shell.execute_reply.started":"2024-04-23T03:30:52.092366Z","shell.execute_reply":"2024-04-23T03:30:52.099423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(model, train_dataset, valid_dataset, epochs=10):\n    try:\n        os.mkdir(save_path)\n    except:\n        pass\n    reduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=1, mode='auto', verbose=0, cooldown=0, min_lr=1e-7)\n    early_stop = tf.keras.callbacks.EarlyStopping(monitor = 'val_loss', min_delta = 0.001, patience = 5, mode = 'min', verbose = 1, restore_best_weights = True)\n    model.compile(\n        optimizer=tf.optimizers.Adam(learning_rate=0.001),\n        loss=SigmoidFocalCrossEntropy(alpha=0.25, gamma=2, from_logits=False),\n        metrics=[tf.keras.metrics.CategoricalAccuracy(name=\"accuracy\"),])\n    \n    history = model.fit(\n            train_dataset,\n            epochs=epochs,\n            validation_data=valid_dataset,\n            callbacks=[reduce_lr,early_stop])\n\n    plt.plot(history.history['accuracy'])\n    plt.plot(history.history['val_accuracy'])\n    plt.title('Model accuracy')\n    plt.ylabel('Accuracy')\n    plt.xlabel('Epoch')\n    plt.legend(['Train', 'Test'], loc='upper left')\n    plt.show()\n    plt.clf()\n    plt.plot(history.history['loss'])\n    plt.plot(history.history['val_loss'])\n    plt.title('Model loss')\n    plt.ylabel('Loss')\n    plt.xlabel('Epoch')\n    plt.legend(['Train', 'Test'], loc='upper left')\n    plt.show()\n    plt.clf()\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:30:52.103707Z","iopub.execute_input":"2024-04-23T03:30:52.104105Z","iopub.status.idle":"2024-04-23T03:30:52.114274Z","shell.execute_reply.started":"2024-04-23T03:30:52.104081Z","shell.execute_reply":"2024-04-23T03:30:52.113351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = train_model(model, train_dataset, valid_dataset, epochs=2)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:30:52.115349Z","iopub.execute_input":"2024-04-23T03:30:52.115609Z","iopub.status.idle":"2024-04-23T05:54:37.639261Z","shell.execute_reply.started":"2024-04-23T03:30:52.115587Z","shell.execute_reply":"2024-04-23T05:54:37.638356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(columns=['image_id','label'])\nfor image_name in os.listdir(WORK_DIR + 'test_images'):\n    image_path = os.path.join(WORK_DIR + 'test_images', image_name)\n    image = tf.keras.preprocessing.image.load_img(image_path)\n    resized_image = image.resize((224, 224))\n    numpied_image = np.expand_dims(resized_image, 0)\n    tensored_image = tf.cast(numpied_image, tf.float32)\n    y_pred = model.predict(tensored_image)\n    y_pred = np.argmax(y_pred, axis=-1)[0]\n    submission.loc[len(submission)] = [image_name, int(y_pred)]\nsubmission.to_csv('submission.csv', index=False)\nfolder = '/kaggle/working/dataset'\ntry:\n    shutil.rmtree(folder)\n    print(\"Folder Deleted\")\nexcept OSError as e:\n    print(f\"Error: {e}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T06:07:14.256834Z","iopub.execute_input":"2024-04-23T06:07:14.257204Z","iopub.status.idle":"2024-04-23T06:07:14.400375Z","shell.execute_reply.started":"2024-04-23T06:07:14.257175Z","shell.execute_reply":"2024-04-23T06:07:14.399419Z"},"trusted":true},"execution_count":null,"outputs":[]}]}