{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"},{"sourceId":161392,"sourceType":"modelInstanceVersion","modelInstanceId":137228,"modelId":159944},{"sourceId":779,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":646,"modelId":55},{"sourceId":170192,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":144802,"modelId":167358},{"sourceId":170201,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":144811,"modelId":167367}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport re\n\nfrom datetime import datetime\n\nimport numpy as np\n\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\n\n\n\nimport tensorflow as tf\n\nimport tensorflow_hub as hub","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-18T03:56:23.465955Z","iopub.execute_input":"2024-11-18T03:56:23.466264Z","iopub.status.idle":"2024-11-18T03:56:38.178773Z","shell.execute_reply.started":"2024-11-18T03:56:23.466230Z","shell.execute_reply":"2024-11-18T03:56:38.178056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nimport matplotlib.pyplot as plt\n\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\n\nfrom tensorflow.keras.applications import EfficientNetB0\n\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D\n\nfrom tensorflow.keras.models import Model\n\nfrom sklearn.preprocessing import LabelEncoder\n\n\n\nlabel_to_disease = pd.read_json('/kaggle/input/cassava-leaf-disease-classification/label_num_to_disease_map.json', typ='series')\n\ntrain_csv = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/train.csv')\n\n\n\ntrain_csv['disease'] = train_csv['label'].map(label_to_disease)\n\ntrain_csv['path'] = '/kaggle/input/cassava-leaf-disease-classification/train_images/' + train_csv['image_id']\n\n\n\ntrain_csv['label_encoded'] = LabelEncoder().fit_transform(train_csv['disease'])\n\n\n\n# Convert 'disease' and 'label' columns to string type\n\ntrain_csv['disease'] = train_csv['disease'].astype(str)\n\ntrain_csv['label'] = train_csv['label'].astype(str)\n\n\n\n# Split the data into train and validation sets with stratified sampling\n\ntrain, valid = train_test_split(train_csv, test_size=0.2, stratify=train_csv['label'])\n\n\n\n\n\n# Data augmentation and preprocessing for training\n\ndatagen_aug = ImageDataGenerator(\n\n    preprocessing_function=preprocess_input,\n\n    rotation_range=45,\n\n    width_shift_range=0.2,\n\n    height_shift_range=0.2,\n\n    shear_range=0.2,\n\n    zoom_range=0.2,\n\n    horizontal_flip=True,\n\n    vertical_flip=True,\n\n    fill_mode='nearest'\n\n)\n\n\n\n# Generator for the training set\n\ntrain_generator = datagen_aug.flow_from_dataframe(\n\n    dataframe=train,\n\n    x_col='path',\n\n    y_col='disease',\n\n    target_size=(224, 224),\n\n    batch_size=32,\n\n    class_mode='categorical',\n\n    shuffle=True\n\n)\n\n\n\n# Generator for the validation set without augmentation\n\ndatagen_valid = ImageDataGenerator(preprocessing_function=preprocess_input)\n\n\n\nvalid_generator = datagen_valid.flow_from_dataframe(\n\n    dataframe=valid,\n\n    x_col='path',\n\n    y_col='disease',\n\n    target_size=(224, 224),\n\n    batch_size=32,\n\n    class_mode='categorical',\n\n    shuffle=False\n\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T03:56:38.180514Z","iopub.execute_input":"2024-11-18T03:56:38.181073Z","iopub.status.idle":"2024-11-18T03:57:19.989151Z","shell.execute_reply.started":"2024-11-18T03:56:38.181036Z","shell.execute_reply":"2024-11-18T03:57:19.988337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_and_preprocess_image(path, label):\n\n    image = tf.io.read_file(path)\n\n    image = tf.image.decode_jpeg(image, channels=3)  # Assuming the images are JPEGs\n\n    image = tf.image.resize(image, [224, 224])  # Resize to the expected input size\n\n    image = image / 255.0  # Normalize to [0, 1]\n\n    return image, label\n\n\n\n# Create datasets\n\ntrain_ds = tf.data.Dataset.from_tensor_slices((train['path'].values, train['label_encoded'].values))\n\nvalid_ds = tf.data.Dataset.from_tensor_slices((valid['path'].values, valid['label_encoded'].values))\n\n\n\n# Map the loading and preprocessing function to the datasets\n\ntrain_ds = train_ds.map(load_and_preprocess_image).batch(32).prefetch(buffer_size=tf.data.experimental.AUTOTUNE)\n\nvalid_ds = valid_ds.map(load_and_preprocess_image).batch(32).prefetch(buffer_size=tf.data.experimental.AUTOTUNE)\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T04:01:21.147422Z","iopub.execute_input":"2024-11-18T04:01:21.148107Z","iopub.status.idle":"2024-11-18T04:01:21.195739Z","shell.execute_reply.started":"2024-11-18T04:01:21.148058Z","shell.execute_reply":"2024-11-18T04:01:21.195010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from tensorflow.keras.models import Model, load_model\n\n# from tensorflow.keras.preprocessing.image import load_img, img_to_array\n\n# from tensorflow.keras.layers import Dense\n\n# from tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, Callback\n\n\n\n# # since cannot load model, redefine and retrain lol \n\n# # Load CropNet feature extractor\n\n# import tensorflow as tf\n\n# import tensorflow_hub as hub\n\n# import numpy as np\n\n\n\n# from tensorflow.keras.applications import DenseNet169\n\n\n\n# class EarlyStoppingCallback(Callback):\n\n#     def on_epoch_end(self, epoch, logs=None):\n\n#         # Check if early stopping has been triggered by checking if patience has run out\n\n#         if self.model.stop_training:\n\n#             print(f\"Early stopping triggered at epoch {epoch + 1}.\")\n\n\n\n# early_stopping = EarlyStopping(\n\n#     monitor='val_loss', \n\n#     patience=3, \n\n#     restore_best_weights=True\n\n# )\n\n\n\n# learning_rate_reduction = tf.keras.callbacks.ReduceLROnPlateau(\n\n#     monitor='val_loss', \n\n#     patience=2, \n\n#     factor=0.5, \n\n#     min_lr=1e-7, \n\n#     verbose=1\n\n# )\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T04:01:21.196850Z","iopub.execute_input":"2024-11-18T04:01:21.197218Z","iopub.status.idle":"2024-11-18T04:01:21.205826Z","shell.execute_reply.started":"2024-11-18T04:01:21.197175Z","shell.execute_reply":"2024-11-18T04:01:21.204949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import tensorflow_hub as hub\n\n# # Re-load the model from TensorFlow Hub\n# model_url = '/kaggle/input/cropnet/tensorflow2/classifier-cassava-disease-v1/1'\n# cropnet_classifier = hub.KerasLayer(model_url)\n# model_cropnet = tf.keras.Sequential([\n#     tf.keras.layers.Input(shape=(224, 224, 3)), \n#     tf.keras.layers.Lambda(lambda x: cropnet_classifier(x))\n# ])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T04:01:21.208244Z","iopub.execute_input":"2024-11-18T04:01:21.209010Z","iopub.status.idle":"2024-11-18T04:01:23.363654Z","shell.execute_reply.started":"2024-11-18T04:01:21.208960Z","shell.execute_reply":"2024-11-18T04:01:23.362879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model_cropnet.compile(\n#     optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n#     loss=tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n#     metrics=['accuracy']\n# )\n\n# history_cropnet = model_cropnet.fit(\n#     train_ds,\n#     validation_data=valid_ds,\n#     epochs=30,\n#     callbacks=[early_stopping, learning_rate_reduction]\n# )\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T04:01:23.364675Z","iopub.execute_input":"2024-11-18T04:01:23.364963Z","iopub.status.idle":"2024-11-18T04:04:18.947694Z","shell.execute_reply.started":"2024-11-18T04:01:23.364933Z","shell.execute_reply":"2024-11-18T04:04:18.946887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # model_cropnet.export('/kaggle/working/cropnet_model_tf')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T04:04:18.948990Z","iopub.execute_input":"2024-11-18T04:04:18.949307Z","iopub.status.idle":"2024-11-18T04:04:22.699976Z","shell.execute_reply.started":"2024-11-18T04:04:18.949274Z","shell.execute_reply":"2024-11-18T04:04:22.699044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !zip -r /kaggle/working/vanilla_cropnet.zip /kaggle/working/cropnet_model_tf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T04:06:50.176382Z","iopub.execute_input":"2024-11-18T04:06:50.177344Z","iopub.status.idle":"2024-11-18T04:06:52.098079Z","shell.execute_reply.started":"2024-11-18T04:06:50.177285Z","shell.execute_reply":"2024-11-18T04:06:52.096934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the TFSMLayer as a layer\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Input, TFSMLayer\nfrom tensorflow.keras.models import Model\nimport numpy as np\nimport pandas as pd\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nimport os\n\n\n#model path\ncropnet_model_path = '/kaggle/input/google_cropnet/tensorflow2/default/1/kaggle/working/cropnet_model_tf'\nlayer = TFSMLayer(cropnet_model_path, call_endpoint='serving_default')\n# Wrap TFSMLayer in a new model for prediction\ninput_layer = Input(shape=(224, 224, 3))\noutput_layer = layer(input_layer)\ncropnet_model = Model(inputs=input_layer, outputs=output_layer)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T04:09:42.428442Z","iopub.execute_input":"2024-11-18T04:09:42.429209Z","iopub.status.idle":"2024-11-18T04:09:44.193668Z","shell.execute_reply.started":"2024-11-18T04:09:42.429167Z","shell.execute_reply":"2024-11-18T04:09:44.192884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\n\n# Set image path and define empty lists to store data\nimage_dir = '/kaggle/input/cassava-leaf-disease-classification/test_images'\npredictions = []\nimage_names = []\n\n# Define image size as per model's expected input\nimg_size = (224, 224)\n\n# Process each image in the folder\nfor filename in os.listdir(image_dir):\n    if filename.endswith(\".jpg\"): \n        # Load and preprocess image\n        img_path = os.path.join(image_dir, filename)\n        img = load_img(img_path, target_size=img_size)\n        img_array = img_to_array(img) / 255.0  # Normalize image to [0, 1] range\n        img_array = np.expand_dims(img_array, axis=0)  # Add batch dimension\n        \n        # Predict using the wrapped model\n        pred = cropnet_model(img_array)\n        \n        # Inspect available keys in the prediction output\n        print(f\"Keys in prediction dictionary for {filename}: {pred.keys()}\")\n        \n        # Access the correct output using the first available key (adjust this if needed)\n        pred_output = list(pred.values())[0]  # Get the first tensor in the dictionary\n        \n        # Print the shape of the output tensor for debugging\n        print(f\"Prediction shape for {filename}: {pred_output.shape}\")\n        \n        # Determine the predicted class based on shape\n        if len(pred_output.shape) == 1:  # 1D output\n            predicted_class = np.argmax(pred_output)\n        else:  # 2D output\n            predicted_class = np.argmax(pred_output, axis=1)[0]\n        \n        predictions.append(predicted_class)\n        image_names.append(filename)\n\n# Create DataFrame for submission\nsubmission_df = pd.DataFrame({\n    'image_id': image_names,\n    'label': predictions\n})\n\n# Save to CSV\nsubmission_df.to_csv('/kaggle/working/submission.csv', index=False)\nprint(\"Submission file created: submission.csv\")\n\n# Display the first few rows to verify\nprint(submission_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T04:10:04.242156Z","iopub.execute_input":"2024-11-18T04:10:04.242550Z","iopub.status.idle":"2024-11-18T04:10:05.504816Z","shell.execute_reply.started":"2024-11-18T04:10:04.242512Z","shell.execute_reply":"2024-11-18T04:10:05.503646Z"}},"outputs":[],"execution_count":null}]}