{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"isSourceIdPinned":false,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import cv2\nfrom tqdm import tqdm\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import classification_report, accuracy_score\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.applications import EfficientNetV2B2\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.applications.efficientnet_v2 import preprocess_input\nfrom functools import partial\nfrom sklearn.model_selection import train_test_split\nimport re\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:18:56.574632Z","iopub.execute_input":"2025-04-08T17:18:56.575257Z","iopub.status.idle":"2025-04-08T17:18:56.583284Z","shell.execute_reply.started":"2025-04-08T17:18:56.575144Z","shell.execute_reply":"2025-04-08T17:18:56.581880Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Parameters for extracting features\nIMG_RESIZE = (128, 128)  # Resize images for faster processing\nGCS_PATH = '/kaggle/input/cassava-leaf-disease-classification'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:18:56.585085Z","iopub.execute_input":"2025-04-08T17:18:56.585616Z","iopub.status.idle":"2025-04-08T17:18:56.610148Z","shell.execute_reply.started":"2025-04-08T17:18:56.585569Z","shell.execute_reply":"2025-04-08T17:18:56.608692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_tfrecord(example, labeled):\n    tfrecord_format = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"target\": tf.io.FixedLenFeature([], tf.int64)\n    } if labeled else {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"image_name\": tf.io.FixedLenFeature([], tf.string)\n    }\n    example = tf.io.parse_single_example(example, tfrecord_format)\n    image = example['image']\n    if labeled:\n        label = tf.cast(example['target'], tf.int32)\n        return image, label\n    idnum = example['image_name']\n    return image, idnum","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:18:56.612606Z","iopub.execute_input":"2025-04-08T17:18:56.613215Z","iopub.status.idle":"2025-04-08T17:18:56.631757Z","shell.execute_reply.started":"2025-04-08T17:18:56.613137Z","shell.execute_reply":"2025-04-08T17:18:56.630436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reuse decode_image but remove preprocess_input for classic ML\ndef decode_for_sklearn(image_bytes):\n    image = tf.image.decode_jpeg(image_bytes, channels=3)\n    image = tf.image.resize(image, IMG_RESIZE)\n    image = tf.cast(image, tf.uint8)\n    return image.numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:18:56.633526Z","iopub.execute_input":"2025-04-08T17:18:56.633925Z","iopub.status.idle":"2025-04-08T17:18:56.652301Z","shell.execute_reply.started":"2025-04-08T17:18:56.633873Z","shell.execute_reply":"2025-04-08T17:18:56.650930Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract color histogram features\ndef extract_hist_features(image_np):\n    hsv = cv2.cvtColor(image_np, cv2.COLOR_RGB2HSV)\n    hist = cv2.calcHist([hsv], [0, 1, 2], None, [4, 4, 4], [0, 180, 0, 256, 0, 256])\n    return cv2.normalize(hist, hist).flatten()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:18:56.653651Z","iopub.execute_input":"2025-04-08T17:18:56.654255Z","iopub.status.idle":"2025-04-08T17:18:56.671119Z","shell.execute_reply.started":"2025-04-08T17:18:56.654193Z","shell.execute_reply":"2025-04-08T17:18:56.669778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_features_from_tfrecords(tfrecord_files, max_items=None):\n    features, labels = [], []\n    raw_dataset = tf.data.TFRecordDataset(tfrecord_files)\n    parsed_dataset = raw_dataset.map(partial(read_tfrecord, labeled=True))\n\n    for i, (image_bytes, label) in enumerate(tqdm(parsed_dataset, desc=\"Extracting features\")):\n        image_np = decode_for_sklearn(image_bytes.numpy())  # decode and convert to NumPy\n        feat = extract_hist_features(image_np)\n        features.append(feat)\n        labels.append(label.numpy())\n        if max_items and i >= max_items:\n            break\n    return np.array(features), np.array(labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:18:56.672729Z","iopub.execute_input":"2025-04-08T17:18:56.673337Z","iopub.status.idle":"2025-04-08T17:18:56.690573Z","shell.execute_reply.started":"2025-04-08T17:18:56.673277Z","shell.execute_reply":"2025-04-08T17:18:56.689238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAINING_FILENAMES, VALID_FILENAMES = train_test_split(\n    tf.io.gfile.glob(GCS_PATH + '/train_tfrecords/ld_train*.tfrec'),\n    test_size=0.2, random_state=5\n)\n\nTEST_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/test_tfrecords/ld_test*.tfrec')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:18:56.692219Z","iopub.execute_input":"2025-04-08T17:18:56.692598Z","iopub.status.idle":"2025-04-08T17:18:56.742966Z","shell.execute_reply.started":"2025-04-08T17:18:56.692566Z","shell.execute_reply":"2025-04-08T17:18:56.741388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract features from training and validation sets (you can reduce the max_items if needed)\nX_train, y_train = load_features_from_tfrecords(TRAINING_FILENAMES, max_items=3000)\nX_val, y_val = load_features_from_tfrecords(VALID_FILENAMES, max_items=1000)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:18:56.745512Z","iopub.execute_input":"2025-04-08T17:18:56.745906Z","iopub.status.idle":"2025-04-08T17:19:24.961985Z","shell.execute_reply.started":"2025-04-08T17:18:56.745871Z","shell.execute_reply":"2025-04-08T17:19:24.960615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the Decision Tree Classifier\nclf = RandomForestClassifier(\n    n_estimators=100,         # Number of trees in the forest\n    max_depth=30,             # You can adjust this\n    min_samples_split=5,\n    min_samples_leaf=2,\n    max_features='sqrt',\n    random_state=42,\n    n_jobs=-1                 # Use all available CPU cores\n)\nclf.fit(X_train, y_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:19:24.963493Z","iopub.execute_input":"2025-04-08T17:19:24.963863Z","iopub.status.idle":"2025-04-08T17:19:25.897056Z","shell.execute_reply.started":"2025-04-08T17:19:24.963822Z","shell.execute_reply":"2025-04-08T17:19:25.895985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate\ny_pred = clf.predict(X_val)\nprint(\"Accuracy:\", accuracy_score(y_val, y_pred))\nprint(\"Classification Report:\\n\", classification_report(y_val, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:19:25.898323Z","iopub.execute_input":"2025-04-08T17:19:25.898687Z","iopub.status.idle":"2025-04-08T17:19:25.970451Z","shell.execute_reply.started":"2025-04-08T17:19:25.898649Z","shell.execute_reply":"2025-04-08T17:19:25.969483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Confusion Matrix\ncm = confusion_matrix(y_val, y_pred)\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=range(5), yticklabels=range(5))\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:19:25.971360Z","iopub.execute_input":"2025-04-08T17:19:25.971607Z","iopub.status.idle":"2025-04-08T17:19:26.410505Z","shell.execute_reply.started":"2025-04-08T17:19:25.971585Z","shell.execute_reply":"2025-04-08T17:19:26.409236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_test_features(tfrecord_files, max_items=None):\n    features, image_ids = [], []\n    raw_dataset = tf.data.TFRecordDataset(tfrecord_files)\n    parsed_dataset = raw_dataset.map(partial(read_tfrecord, labeled=False))\n\n    for i, (image_bytes, image_id_bytes) in enumerate(tqdm(parsed_dataset, desc=\"Extracting test features\")):\n        image_np = decode_for_sklearn(image_bytes.numpy())  # decode from bytes to NumPy array\n        feat = extract_hist_features(image_np)\n        features.append(feat)\n        image_id = image_id_bytes.numpy().decode('utf-8')\n        image_ids.append(image_id)\n        if max_items and i >= max_items:\n            break\n    return np.array(features), image_ids","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:19:26.411600Z","iopub.execute_input":"2025-04-08T17:19:26.411976Z","iopub.status.idle":"2025-04-08T17:19:26.418495Z","shell.execute_reply.started":"2025-04-08T17:19:26.411946Z","shell.execute_reply":"2025-04-08T17:19:26.417295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# For Kaggle Submission\n# Load test data\nX_test, test_image_ids = load_test_features(TEST_FILENAMES)\n\n# Predict\ntest_preds = clf.predict(X_test)\n\n# Prepare submission\nsubmission_df = pd.DataFrame({\n    'image_id': test_image_ids,\n    'label': test_preds\n})\n\n# Save to CSV\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T17:19:26.419589Z","iopub.execute_input":"2025-04-08T17:19:26.420211Z","iopub.status.idle":"2025-04-08T17:19:26.552869Z","shell.execute_reply.started":"2025-04-08T17:19:26.420134Z","shell.execute_reply":"2025-04-08T17:19:26.551494Z"}},"outputs":[],"execution_count":null}]}