{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"},{"sourceId":9004644,"sourceType":"datasetVersion","datasetId":5424749}],"dockerImageVersionId":30747,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\n\ndataset_path = '/kaggle/input/cassava-leaf-disease-classification/'\n\nif os.path.exists(dataset_path):\n    print(\"Directory exists. Listing files:\")\n    print(os.listdir(dataset_path))\nelse:\n    print(\"Directory does not exist. Check the dataset path.\")\n\n# Load CSV file\ncsv_path = os.path.join(dataset_path, 'train.csv')\ntest_images_path = os.path.join(dataset_path, 'test_images')\nif os.path.exists(csv_path):\n    df = pd.read_csv(csv_path)\n    print(df.head())\nelse:\n    print(\"train.csv file not found in the specified directory.\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-21T11:23:02.104599Z","iopub.execute_input":"2024-07-21T11:23:02.105178Z","iopub.status.idle":"2024-07-21T11:23:03.162608Z","shell.execute_reply.started":"2024-07-21T11:23:02.105144Z","shell.execute_reply":"2024-07-21T11:23:03.161662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB4\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Dropout, GlobalAveragePooling2D\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\nimport pandas as pd\nimport numpy as np\nimport os\nimport json\n\nfrom tensorflow.keras.applications import EfficientNetB4\nfrom tensorflow.keras.layers import GlobalAveragePooling2D\nfrom tensorflow.keras.models import Model\n\n# Define the local path to the weights file\n\nlocal_weights_path = '/kaggle/input/efficientnet/efficientnetb4_notop.h5'  # Update with your actual path\n\n# Load the model without weights\nbase_model = EfficientNetB4(weights=None, include_top=False, input_shape=(380, 380, 3))\n\n# Load weights from the local file\nbase_model.load_weights(local_weights_path)\n\n# Continue with your model building\nx = base_model.output\nx = GlobalAveragePooling2D()(x)","metadata":{"execution":{"iopub.status.busy":"2024-07-21T17:51:08.153033Z","iopub.execute_input":"2024-07-21T17:51:08.153739Z","iopub.status.idle":"2024-07-21T17:51:11.678131Z","shell.execute_reply.started":"2024-07-21T17:51:08.153709Z","shell.execute_reply":"2024-07-21T17:51:11.677340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(csv_path)\n\njson_path = os.path.join(dataset_path, 'label_num_to_disease_map.json')\nwith open(json_path) as json_file:\n    label_map = json.load(json_file)\n\n# Display the dataset information\nprint(train_df.head())\nprint(label_map)","metadata":{"execution":{"iopub.status.busy":"2024-07-21T11:23:15.258171Z","iopub.execute_input":"2024-07-21T11:23:15.258778Z","iopub.status.idle":"2024-07-21T11:23:15.291194Z","shell.execute_reply.started":"2024-07-21T11:23:15.258744Z","shell.execute_reply":"2024-07-21T11:23:15.290326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest',\n    validation_split=0.2\n)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory='/kaggle/input/cassava-leaf-disease-classification/train_images',\n    x_col='image_id',\n    y_col='label',\n    target_size=(380, 380),  # Larger input size for better feature extraction\n    batch_size=32,\n    class_mode='raw',\n    subset='training'\n)\n\nvalidation_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory='/kaggle/input/cassava-leaf-disease-classification/train_images',\n    x_col='image_id',\n    y_col='label',\n    target_size=(380, 380),\n    batch_size=32,\n    class_mode='raw',\n    subset='validation'\n)","metadata":{"execution":{"iopub.status.busy":"2024-07-21T11:23:15.292300Z","iopub.execute_input":"2024-07-21T11:23:15.292592Z","iopub.status.idle":"2024-07-21T11:24:07.480745Z","shell.execute_reply.started":"2024-07-21T11:23:15.292569Z","shell.execute_reply":"2024-07-21T11:24:07.479754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# base_model = EfficientNetB4(weights='imagenet', include_top=False, input_shape=(380, 380, 3))\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\nx = Dropout(0.5)(x)  # Dropout for regularization\npredictions = Dense(len(label_map), activation='softmax')(x)\n\n# Compile the model\nmodel = Model(inputs=base_model.input, outputs=predictions)","metadata":{"execution":{"iopub.status.busy":"2024-07-21T11:24:07.481931Z","iopub.execute_input":"2024-07-21T11:24:07.482674Z","iopub.status.idle":"2024-07-21T11:24:15.663171Z","shell.execute_reply.started":"2024-07-21T11:24:07.482646Z","shell.execute_reply":"2024-07-21T11:24:15.662413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in base_model.layers:\n    layer.trainable = False\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n# Step 5: Training the Model\n# Using callbacks for better training management\nearly_stopping = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=3, min_lr=1e-6)","metadata":{"execution":{"iopub.status.busy":"2024-07-21T11:24:15.664238Z","iopub.execute_input":"2024-07-21T11:24:15.664509Z","iopub.status.idle":"2024-07-21T11:24:15.690197Z","shell.execute_reply.started":"2024-07-21T11:24:15.664473Z","shell.execute_reply":"2024-07-21T11:24:15.689536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    validation_data=validation_generator,\n    epochs=30,  # Train for 30 epochs initially\n    callbacks=[early_stopping, reduce_lr]\n)","metadata":{"execution":{"iopub.status.busy":"2024-07-21T11:24:15.691160Z","iopub.execute_input":"2024-07-21T11:24:15.691412Z","iopub.status.idle":"2024-07-21T13:04:14.918934Z","shell.execute_reply.started":"2024-07-21T11:24:15.691390Z","shell.execute_reply":"2024-07-21T13:04:14.918068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in base_model.layers[-40:]:  # Unfreeze the last 40 layers\n    layer.trainable = True\n\n# Recompile the model with a lower learning rate for fine-tuning\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=1e-5), loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n# Continue training with fine-tuning\nfine_tune_history = model.fit(\n    train_generator,\n    validation_data=validation_generator,\n    epochs=20,  # Additional 20 epochs for fine-tuning\n    callbacks=[early_stopping, reduce_lr]\n)","metadata":{"execution":{"iopub.status.busy":"2024-07-21T13:04:14.920434Z","iopub.execute_input":"2024-07-21T13:04:14.920754Z","iopub.status.idle":"2024-07-21T17:03:30.073856Z","shell.execute_reply.started":"2024-07-21T13:04:14.920729Z","shell.execute_reply":"2024-07-21T17:03:30.072918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_loss, val_accuracy = model.evaluate(validation_generator)\nprint(f'Validation Accuracy: {val_accuracy * 100:.2f}%')\n\n\ntest_datagen = ImageDataGenerator(rescale=1./255)\ntest_generator = test_datagen.flow_from_directory(\n    test_images_path,\n    target_size=(380, 380),\n    batch_size=1,\n    class_mode=None,\n    shuffle=False\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-07-21T17:14:29.949266Z","iopub.execute_input":"2024-07-21T17:14:29.949733Z","iopub.status.idle":"2024-07-21T17:16:52.180836Z","shell.execute_reply.started":"2024-07-21T17:14:29.949702Z","shell.execute_reply":"2024-07-21T17:16:52.180070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_images_path = os.path.join(dataset_path, 'test_images')\nfrom keras.preprocessing.image import load_img, img_to_array\n\n# Load the image\nimage_path = test_images_path + '/2216849948.jpg'\nimg = load_img(image_path, target_size=(380, 380))\nimg_array = img_to_array(img)\nimg_array = np.expand_dims(img_array, axis=0)  # Create batch dimension\n\n# Rescale the image (if not already done by the ImageDataGenerator)\nimg_array = img_array / 255.0\n\n# Make predictions\npredictions = model.predict(img_array)\npredicted_label = np.argmax(predictions, axis=-1)\n\nprint(f'Predicted Label: {predicted_label[0]}')\n\n# Create a submission DataFrame\nsubmission_df = pd.DataFrame({\n    'image_id': [os.path.basename(image_path)],\n    'label': predicted_label\n})\n\nprint(submission_df.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-07-21T17:26:12.177947Z","iopub.execute_input":"2024-07-21T17:26:12.178295Z","iopub.status.idle":"2024-07-21T17:26:25.847470Z","shell.execute_reply.started":"2024-07-21T17:26:12.178266Z","shell.execute_reply":"2024-07-21T17:26:25.846579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predictions = model.predict(test_generator, steps=len(test_generator), verbose=1)\n# predicted_labels = np.argmax(predictions, axis=-1)\n\n\n# submission_df = pd.DataFrame({\n#     'image_id': [os.path.basename(path) for path in test_generator.filenames],\n#     'label': predicted_labels\n# })\n\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-21T17:27:20.577221Z","iopub.execute_input":"2024-07-21T17:27:20.577609Z","iopub.status.idle":"2024-07-21T17:27:20.585440Z","shell.execute_reply.started":"2024-07-21T17:27:20.577583Z","shell.execute_reply":"2024-07-21T17:27:20.584554Z"},"trusted":true},"execution_count":null,"outputs":[]}]}