{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import cv2\nfrom tqdm import tqdm\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import classification_report, accuracy_score\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.applications import EfficientNetV2B2\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.applications.efficientnet_v2 import preprocess_input\nfrom functools import partial\nfrom sklearn.model_selection import train_test_split\nimport re\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:02:24.135566Z","iopub.execute_input":"2025-04-08T17:02:24.136121Z","iopub.status.idle":"2025-04-08T17:02:24.143325Z","shell.execute_reply.started":"2025-04-08T17:02:24.136072Z","shell.execute_reply":"2025-04-08T17:02:24.141723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Parameters for extracting features\nIMG_RESIZE = (128, 128)  # Resize images for faster processing\nGCS_PATH = '/kaggle/input/cassava-leaf-disease-classification'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:02:24.145080Z","iopub.execute_input":"2025-04-08T17:02:24.145507Z","iopub.status.idle":"2025-04-08T17:02:24.170147Z","shell.execute_reply.started":"2025-04-08T17:02:24.145470Z","shell.execute_reply":"2025-04-08T17:02:24.168885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_tfrecord(example, labeled):\n    tfrecord_format = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"target\": tf.io.FixedLenFeature([], tf.int64)\n    } if labeled else {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"image_name\": tf.io.FixedLenFeature([], tf.string)\n    }\n    example = tf.io.parse_single_example(example, tfrecord_format)\n    image = example['image']\n    if labeled:\n        label = tf.cast(example['target'], tf.int32)\n        return image, label\n    idnum = example['image_name']\n    return image, idnum","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:02:24.172663Z","iopub.execute_input":"2025-04-08T17:02:24.173208Z","iopub.status.idle":"2025-04-08T17:02:24.189271Z","shell.execute_reply.started":"2025-04-08T17:02:24.173156Z","shell.execute_reply":"2025-04-08T17:02:24.188008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reuse decode_image but remove preprocess_input for classic ML\ndef decode_for_sklearn(image_bytes):\n    image = tf.image.decode_jpeg(image_bytes, channels=3)\n    image = tf.image.resize(image, IMG_RESIZE)\n    image = tf.cast(image, tf.uint8)\n    return image.numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:02:24.191350Z","iopub.execute_input":"2025-04-08T17:02:24.191837Z","iopub.status.idle":"2025-04-08T17:02:24.207252Z","shell.execute_reply.started":"2025-04-08T17:02:24.191785Z","shell.execute_reply":"2025-04-08T17:02:24.206082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract color histogram features\ndef extract_hist_features(image_np):\n    hsv = cv2.cvtColor(image_np, cv2.COLOR_RGB2HSV)\n    hist = cv2.calcHist([hsv], [0, 1, 2], None, [4, 4, 4], [0, 180, 0, 256, 0, 256])\n    return cv2.normalize(hist, hist).flatten()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:02:24.208486Z","iopub.execute_input":"2025-04-08T17:02:24.208917Z","iopub.status.idle":"2025-04-08T17:02:24.224232Z","shell.execute_reply.started":"2025-04-08T17:02:24.208862Z","shell.execute_reply":"2025-04-08T17:02:24.222790Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_features_from_tfrecords(tfrecord_files, max_items=None):\n    features, labels = [], []\n    raw_dataset = tf.data.TFRecordDataset(tfrecord_files)\n    parsed_dataset = raw_dataset.map(partial(read_tfrecord, labeled=True))\n\n    for i, (image_bytes, label) in enumerate(tqdm(parsed_dataset, desc=\"Extracting features\")):\n        image_np = decode_for_sklearn(image_bytes.numpy())  # decode and convert to NumPy\n        feat = extract_hist_features(image_np)\n        features.append(feat)\n        labels.append(label.numpy())\n        if max_items and i >= max_items:\n            break\n    return np.array(features), np.array(labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:02:24.225600Z","iopub.execute_input":"2025-04-08T17:02:24.226092Z","iopub.status.idle":"2025-04-08T17:02:24.243559Z","shell.execute_reply.started":"2025-04-08T17:02:24.226038Z","shell.execute_reply":"2025-04-08T17:02:24.242099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAINING_FILENAMES, VALID_FILENAMES = train_test_split(\n    tf.io.gfile.glob(GCS_PATH + '/train_tfrecords/ld_train*.tfrec'),\n    test_size=0.2, random_state=5\n)\n\nTEST_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/test_tfrecords/ld_test*.tfrec')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:02:24.244891Z","iopub.execute_input":"2025-04-08T17:02:24.245391Z","iopub.status.idle":"2025-04-08T17:02:24.277047Z","shell.execute_reply.started":"2025-04-08T17:02:24.245347Z","shell.execute_reply":"2025-04-08T17:02:24.275743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract features from training and validation sets (you can reduce the max_items if needed)\nX_train, y_train = load_features_from_tfrecords(TRAINING_FILENAMES, max_items=3000)\nX_val, y_val = load_features_from_tfrecords(VALID_FILENAMES, max_items=1000)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:02:24.278244Z","iopub.execute_input":"2025-04-08T17:02:24.278582Z","iopub.status.idle":"2025-04-08T17:02:46.125697Z","shell.execute_reply.started":"2025-04-08T17:02:24.278553Z","shell.execute_reply":"2025-04-08T17:02:46.124397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the Decision Tree Classifier\nclf = DecisionTreeClassifier(\n    max_depth=30,\n    criterion='entropy',\n    min_samples_split=5,\n    min_samples_leaf=2,\n    max_features='sqrt',\n    random_state=42\n)\nclf.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:02:46.129070Z","iopub.execute_input":"2025-04-08T17:02:46.129428Z","iopub.status.idle":"2025-04-08T17:02:46.407383Z","shell.execute_reply.started":"2025-04-08T17:02:46.129396Z","shell.execute_reply":"2025-04-08T17:02:46.406245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate\ny_pred = clf.predict(X_val)\nprint(\"Accuracy:\", accuracy_score(y_val, y_pred))\nprint(\"Classification Report:\\n\", classification_report(y_val, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:02:46.408789Z","iopub.execute_input":"2025-04-08T17:02:46.409185Z","iopub.status.idle":"2025-04-08T17:02:46.433630Z","shell.execute_reply.started":"2025-04-08T17:02:46.409149Z","shell.execute_reply":"2025-04-08T17:02:46.432189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Confusion Matrix\ncm = confusion_matrix(y_val, y_pred)\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=range(5), yticklabels=range(5))\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:02:46.434664Z","iopub.execute_input":"2025-04-08T17:02:46.434986Z","iopub.status.idle":"2025-04-08T17:02:46.734206Z","shell.execute_reply.started":"2025-04-08T17:02:46.434943Z","shell.execute_reply":"2025-04-08T17:02:46.733182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_test_features(tfrecord_files, max_items=None):\n    features, image_ids = [], []\n    raw_dataset = tf.data.TFRecordDataset(tfrecord_files)\n    parsed_dataset = raw_dataset.map(partial(read_tfrecord, labeled=False))\n\n    for i, (image_bytes, image_id_bytes) in enumerate(tqdm(parsed_dataset, desc=\"Extracting test features\")):\n        image_np = decode_for_sklearn(image_bytes.numpy())  # decode from bytes to NumPy array\n        feat = extract_hist_features(image_np)\n        features.append(feat)\n        image_id = image_id_bytes.numpy().decode('utf-8')\n        image_ids.append(image_id)\n        if max_items and i >= max_items:\n            break\n    return np.array(features), image_ids","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:02:46.735140Z","iopub.execute_input":"2025-04-08T17:02:46.735452Z","iopub.status.idle":"2025-04-08T17:02:46.741810Z","shell.execute_reply.started":"2025-04-08T17:02:46.735426Z","shell.execute_reply":"2025-04-08T17:02:46.740706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# For Kaggle Submission\n# Load test data\nX_test, test_image_ids = load_test_features(TEST_FILENAMES)\n\n# Predict\ntest_preds = clf.predict(X_test)\n\n# Prepare submission\nsubmission_df = pd.DataFrame({\n    'image_id': test_image_ids,\n    'label': test_preds\n})\n\n# Save to CSV\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:02:46.743121Z","iopub.execute_input":"2025-04-08T17:02:46.743577Z","iopub.status.idle":"2025-04-08T17:02:46.834304Z","shell.execute_reply.started":"2025-04-08T17:02:46.743531Z","shell.execute_reply":"2025-04-08T17:02:46.833052Z"}},"outputs":[],"execution_count":null}]}