{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":5048,"databundleVersionId":868335,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Full end-to-end script (Kaggle-ready)\n# - 5 preprocessing pipelines\n# - For each pipeline: load 1000 images/class, train/validate split (80/20)\n# - Train Dense NN (BatchNorm + Dropout)\n# - Show validation confusion matrix + training curves\n# - Draw prediction distribution on 3000 random test images + show 20 sample test predictions\n#\n# Paths are set for the Kaggle \"State Farm Distracted Driver Detection\" dataset.\n\nimport os\nimport random\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, accuracy_score\nfrom sklearn.metrics import ConfusionMatrixDisplay\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, callbacks\n\n# ------------------------------\n# Config (Kaggle dataset paths)\n# ------------------------------\nTRAIN_DIR = \"/kaggle/input/state-farm-distracted-driver-detection/imgs/train\"\nTEST_DIR  = \"/kaggle/input/state-farm-distracted-driver-detection/imgs/test\"\n\nIMG_SIZE = 64                # resize: 64x64\nTRAIN_PER_CLASS = 1000       # take up to 1000 images per class from train/\nMAX_TEST_IMAGES = 3000       # number of test images to sample for predictions\nEPOCHS = 25\nBATCH_SIZE = 64\nRANDOM_SEED = 42\n\nnp.random.seed(RANDOM_SEED)\nrandom.seed(RANDOM_SEED)\ntf.random.set_seed(RANDOM_SEED)\n\n# ------------------------------\n# Class names (folder names c0..c9)\n# ------------------------------\n# If your train folders are named differently, update this accordingly.\nclass_names = sorted([d for d in os.listdir(TRAIN_DIR) if os.path.isdir(os.path.join(TRAIN_DIR, d))])\nnum_classes = len(class_names)\nprint(\"Detected classes:\", class_names)\n\n# ------------------------------\n# Preprocessing functions (5 pipelines)\n# ------------------------------\ndef preprocess_standard(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    yuv = cv2.cvtColor(img, cv2.COLOR_RGB2YUV)\n    yuv[:,:,0] = cv2.equalizeHist(yuv[:,:,0])\n    img = cv2.cvtColor(yuv, cv2.COLOR_YUV2RGB)\n    return img.astype(np.float32) / 255.0\n\ndef preprocess_lighting(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    lab = cv2.cvtColor(img, cv2.COLOR_RGB2LAB)\n    l, a, b = cv2.split(lab)\n    clahe = cv2.createCLAHE(clipLimit=3.0, tileGridSize=(8,8))\n    l = clahe.apply(l)\n    lab = cv2.merge((l,a,b))\n    img = cv2.cvtColor(lab, cv2.COLOR_LAB2RGB)\n    gamma = 1.15\n    img = np.power(img/255.0, 1.0/gamma)\n    return np.clip(img, 0.0, 1.0).astype(np.float32)\n\ndef preprocess_noise(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    img = cv2.bilateralFilter(img, d=9, sigmaColor=75, sigmaSpace=75)\n    img = cv2.medianBlur(img, 3)\n    return img.astype(np.float32) / 255.0\n\ndef preprocess_feature(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n    edges = cv2.Canny(gray, 80, 160)\n    edges_rgb = cv2.cvtColor(edges, cv2.COLOR_GRAY2RGB)\n    blended = cv2.addWeighted(img.astype(np.float32), 0.75, edges_rgb.astype(np.float32), 0.25, 0.0)\n    return np.clip(blended / 255.0, 0.0, 1.0).astype(np.float32)\n\ndef preprocess_aug(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    if np.random.rand() > 0.5:\n        img = cv2.flip(img, 1)\n    angle = np.random.uniform(-15, 15)\n    M = cv2.getRotationMatrix2D((IMG_SIZE//2, IMG_SIZE//2), angle, 1.0)\n    img = cv2.warpAffine(img, M, (IMG_SIZE, IMG_SIZE), borderMode=cv2.BORDER_REFLECT)\n    alpha = 1.0 + (np.random.rand() - 0.5) * 0.3\n    beta = int((np.random.rand() - 0.5) * 50)\n    img = cv2.convertScaleAbs(img, alpha=alpha, beta=beta)\n    return img.astype(np.float32) / 255.0\n\nPIPELINES = {\n    \"Standard\": preprocess_standard,\n    \"Lighting\": preprocess_lighting,\n    \"NoiseReduction\": preprocess_noise,\n    \"FeatureEnhancement\": preprocess_feature,\n    \"Augmentation\": preprocess_aug\n}\n\n# ------------------------------\n# Dense model builder (flatten input)\n# ------------------------------\ndef build_dense(input_dim, num_classes, lr=1e-4):\n    model = models.Sequential([\n        layers.Input(shape=(input_dim,)),\n        layers.Dense(1024, activation='relu'),\n        layers.BatchNormalization(),\n        layers.Dropout(0.5),\n        layers.Dense(512, activation='relu'),\n        layers.BatchNormalization(),\n        layers.Dropout(0.5),\n        layers.Dense(num_classes, activation='softmax')\n    ])\n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=lr),\n                  loss='sparse_categorical_crossentropy',\n                  metrics=['accuracy'])\n    return model\n\n# ------------------------------\n# Utilities: load exactly TRAIN_PER_CLASS per class\n# ------------------------------\ndef load_train_sampled(train_dir, preprocess_fn, per_class=TRAIN_PER_CLASS):\n    X, y = [], []\n    classes = sorted([d for d in os.listdir(train_dir) if os.path.isdir(os.path.join(train_dir, d))])\n    for label, cls in enumerate(classes):\n        cls_dir = os.path.join(train_dir, cls)\n        files = sorted(os.listdir(cls_dir))[:per_class]\n        for fname in files:\n            p = os.path.join(cls_dir, fname)\n            img = cv2.imread(p)\n            if img is None:\n                continue\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            X.append(preprocess_fn(img))\n            y.append(label)\n    X = np.array(X, dtype=np.float32)\n    y = np.array(y, dtype=np.int32)\n    return X, y, classes\n\n# ------------------------------\n# Utilities: load N random test images (no labels)\n# ------------------------------\ndef load_test_sample(test_dir, preprocess_fn, max_images=MAX_TEST_IMAGES):\n    all_files = sorted([f for f in os.listdir(test_dir) if os.path.isfile(os.path.join(test_dir, f))])\n    # sample randomly but reproducibly:\n    rng = random.Random(RANDOM_SEED)\n    sample_files = rng.sample(all_files, min(len(all_files), max_images))\n    X_test_raw = []        # raw resized images for display (RGB uint8)\n    X_test_input = []      # preprocessed normalized arrays\n    for fname in tqdm(sample_files, desc=\"Loading test images\"):\n        p = os.path.join(test_dir, fname)\n        img_bgr = cv2.imread(p)\n        if img_bgr is None:\n            continue\n        img_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n        img_resized = cv2.resize(img_rgb, (IMG_SIZE, IMG_SIZE))\n        X_test_raw.append(img_resized)                    # uint8 RGB for display\n        X_test_input.append(preprocess_fn(img_resized))   # normalized float32\n    X_test_raw = np.array(X_test_raw, dtype=np.uint8)\n    X_test_input = np.array(X_test_input, dtype=np.float32)\n    return X_test_raw, X_test_input\n\n# ------------------------------\n# Plot helpers\n# ------------------------------\ndef plot_confusion_matrix_labels(y_true, y_pred, classes, title=\"Confusion Matrix\"):\n    cm = confusion_matrix(y_true, y_pred, labels=np.arange(len(classes)))\n    disp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=classes)\n    plt.figure(figsize=(8,8))\n    disp.plot(cmap=plt.cm.Blues, xticks_rotation=45, ax=plt.gca())\n    plt.title(title)\n    plt.show()\n\ndef plot_training_curves(history, pipeline_name):\n    plt.figure(figsize=(12,4))\n    plt.subplot(1,2,1)\n    plt.plot(history.history.get(\"accuracy\", []), label=\"train_acc\")\n    plt.plot(history.history.get(\"val_accuracy\", []), label=\"val_acc\")\n    plt.xlabel(\"Epoch\"); plt.ylabel(\"Accuracy\"); plt.title(f\"{pipeline_name} Accuracy\"); plt.legend()\n    plt.subplot(1,2,2)\n    plt.plot(history.history.get(\"loss\", []), label=\"train_loss\")\n    plt.plot(history.history.get(\"val_loss\", []), label=\"val_loss\")\n    plt.xlabel(\"Epoch\"); plt.ylabel(\"Loss\"); plt.title(f\"{pipeline_name} Loss\"); plt.legend()\n    plt.show()\n\ndef plot_test_distribution(preds, classes, pipeline_name):\n    plt.figure(figsize=(9,4))\n    sns.countplot(x=preds, order=range(len(classes)))\n    plt.xticks(ticks=range(len(classes)), labels=classes, rotation=45)\n    plt.title(f\"Prediction distribution on test (pipeline={pipeline_name})\")\n    plt.xlabel(\"Predicted class\"); plt.ylabel(\"Count\")\n    plt.show()\n\ndef show_test_samples(X_raw, preds, classes, pipeline_name, n_samples=20):\n    n = min(n_samples, len(X_raw))\n    rng = random.Random(RANDOM_SEED)\n    idxs = rng.sample(range(len(X_raw)), n)\n    cols = 5\n    rows = int(np.ceil(n / cols))\n    plt.figure(figsize=(cols * 3, rows * 3))\n    for i, idx in enumerate(idxs):\n        plt.subplot(rows, cols, i+1)\n        plt.imshow(X_raw[idx])\n        plt.title(f\"P: {classes[preds[idx]]}\")\n        plt.axis(\"off\")\n    plt.suptitle(f\"Sample Test Predictions - {pipeline_name}\", fontsize=16)\n    plt.show()\n\n# ------------------------------\n# Main loop: iterate pipelines\n# ------------------------------\nresults_summary = {}\n\nfor pipeline_name, preprocess_fn in PIPELINES.items():\n    print(\"\\n\" + \"=\"*80)\n    print(f\"PIPELINE: {pipeline_name}\")\n    print(\"=\"*80)\n\n    # 1) Load train (1000 per class) & make train/val split\n    X_all, y_all, classes = load_train_sampled(TRAIN_DIR, preprocess_fn, per_class=TRAIN_PER_CLASS)\n    if X_all.shape[0] == 0:\n        raise RuntimeError(f\"No training images found in {TRAIN_DIR}. Check path.\")\n    print(f\"Loaded train images: {X_all.shape}, labels: {np.unique(y_all).size} classes\")\n\n    X_train, X_val, y_train, y_val = train_test_split(\n        X_all, y_all, test_size=0.2, random_state=RANDOM_SEED, stratify=y_all\n    )\n    print(\"Train shape:\", X_train.shape, \"Val shape:\", X_val.shape)\n\n    # 2) Prepare dense inputs (flatten)\n    X_tr_input = X_train.reshape(X_train.shape[0], -1)\n    X_val_input = X_val.reshape(X_val.shape[0], -1)\n    input_dim = X_tr_input.shape[1]\n    print(\"Dense input dimension:\", input_dim)\n\n    # 3) Build + train model\n    model = build_dense(input_dim=input_dim, num_classes=num_classes, lr=1e-4)\n    model.summary()\n\n    cb_list = [\n        callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=3, verbose=1),\n        callbacks.EarlyStopping(monitor='val_accuracy', patience=5, restore_best_weights=True, verbose=1)\n    ]\n\n    history = model.fit(\n        X_tr_input, y_train,\n        validation_data=(X_val_input, y_val),\n        epochs=EPOCHS,\n        batch_size=BATCH_SIZE,\n        callbacks=cb_list,\n        verbose=1\n    )\n\n    # 4) Validation evaluation: predictions, conf matrix, acc\n    y_val_pred = np.argmax(model.predict(X_val_input, verbose=0), axis=1)\n    val_acc = accuracy_score(y_val, y_val_pred)\n    print(f\"[{pipeline_name}] Validation accuracy = {val_acc:.4f}\")\n\n    plot_confusion_matrix_labels(y_val, y_val_pred, classes, title=f\"Validation Confusion Matrix - {pipeline_name}\")\n    plot_training_curves(history, pipeline_name)\n\n    # 5) Test predictions on sampled 3000 images\n    print(\"Loading & predicting on up to\", MAX_TEST_IMAGES, \"test images (random sample)...\")\n    X_test_raw, X_test_input = load_test_sample(TEST_DIR, preprocess_fn, max_images=MAX_TEST_IMAGES)\n    if X_test_input.shape[0] == 0:\n        print(\"No test images loaded (check TEST_DIR). Skipping test predictions for this pipeline.\")\n        results_summary[pipeline_name] = {\n            \"val_acc\": val_acc,\n            \"num_test\": 0,\n            \"test_preds\": None\n        }\n        continue\n\n    # flatten test input for dense model\n    X_test_flat = X_test_input.reshape(X_test_input.shape[0], -1)\n\n    # Predict in batches\n    preds_prob = model.predict(X_test_flat, batch_size=128, verbose=0)\n    preds = np.argmax(preds_prob, axis=1)\n\n    # 6) Show distribution and sample images\n    plot_test_distribution(preds, classes, pipeline_name)\n    show_test_samples(X_test_raw, preds, classes, pipeline_name, n_samples=20)\n\n    # Save summary\n    results_summary[pipeline_name] = {\n        \"val_acc\": val_acc,\n        \"num_test\": X_test_flat.shape[0],\n        \"test_preds\": preds  # array of predicted class indices for test sample\n    }\n\n# ------------------------------\n# Final summary: validation accuracies bar chart\n# ------------------------------\npipelines_done = list(results_summary.keys())\nval_accs = [results_summary[p][\"val_acc\"] for p in pipelines_done]\n\nplt.figure(figsize=(9,5))\nsns.barplot(x=pipelines_done, y=val_accs, palette=\"mako\")\nplt.ylim(0,1)\nplt.ylabel(\"Validation Accuracy\")\nplt.title(\"Validation Accuracy per Preprocessing Pipeline\")\nplt.xticks(rotation=20)\nplt.show()\n\n# Print numeric summary\nprint(\"\\nNumeric summary:\")\nfor p in pipelines_done:\n    info = results_summary[p]\n    print(f\"{p:20s} | val_acc = {info['val_acc']:.4f} | test_images = {info['num_test']}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T09:58:01.501625Z","iopub.execute_input":"2025-08-21T09:58:01.502133Z","iopub.status.idle":"2025-08-21T10:11:28.642856Z","shell.execute_reply.started":"2025-08-21T09:58:01.502109Z","shell.execute_reply":"2025-08-21T10:11:28.642058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, accuracy_score\nfrom sklearn.metrics import ConfusionMatrixDisplay\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, callbacks\n\n# ------------------------------\n# Config (Kaggle dataset paths)\n# ------------------------------\nTRAIN_DIR = \"/kaggle/input/state-farm-distracted-driver-detection/imgs/train\"\nTEST_DIR  = \"/kaggle/input/state-farm-distracted-driver-detection/imgs/test\"\n\nIMG_SIZE = 64                # resize: 64x64\nTRAIN_PER_CLASS = 1000       # take up to 1000 images per class from train/\nMAX_TEST_IMAGES = 3000       # number of test images to sample for predictions\nEPOCHS = 25\nBATCH_SIZE = 64\nRANDOM_SEED = 42\n\nnp.random.seed(RANDOM_SEED)\nrandom.seed(RANDOM_SEED)\ntf.random.set_seed(RANDOM_SEED)\n\n# ------------------------------\n# Class names (folder names c0..c9)\n# ------------------------------\n# If your train folders are named differently, update this accordingly.\nclass_names = sorted([d for d in os.listdir(TRAIN_DIR) if os.path.isdir(os.path.join(TRAIN_DIR, d))])\nnum_classes = len(class_names)\nprint(\"Detected classes:\", class_names)\n\n# Optional: full readable class names\nclass_full_names = {\n    \"c0\": \"safe driving\",\n    \"c1\": \"texting right\",\n    \"c2\": \"talking on phone right\",\n    \"c3\": \"texting left\",\n    \"c4\": \"talking on phone left\",\n    \"c5\": \"operating radio\",\n    \"c6\": \"drinking\",\n    \"c7\": \"reaching behind\",\n    \"c8\": \"hair & makeup\",\n    \"c9\": \"talking to passenger\"\n}\n\n# ------------------------------\n# Preprocessing functions (5 pipelines)\n# ------------------------------\ndef preprocess_standard(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    yuv = cv2.cvtColor(img, cv2.COLOR_RGB2YUV)\n    yuv[:,:,0] = cv2.equalizeHist(yuv[:,:,0])\n    img = cv2.cvtColor(yuv, cv2.COLOR_YUV2RGB)\n    return img.astype(np.float32) / 255.0\n\ndef preprocess_lighting(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    lab = cv2.cvtColor(img, cv2.COLOR_RGB2LAB)\n    l, a, b = cv2.split(lab)\n    clahe = cv2.createCLAHE(clipLimit=3.0, tileGridSize=(8,8))\n    l = clahe.apply(l)\n    lab = cv2.merge((l,a,b))\n    img = cv2.cvtColor(lab, cv2.COLOR_LAB2RGB)\n    gamma = 1.15\n    img = np.power(img/255.0, 1.0/gamma)\n    return np.clip(img, 0.0, 1.0).astype(np.float32)\n\ndef preprocess_noise(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    img = cv2.bilateralFilter(img, d=9, sigmaColor=75, sigmaSpace=75)\n    img = cv2.medianBlur(img, 3)\n    return img.astype(np.float32) / 255.0\n\ndef preprocess_feature(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n    edges = cv2.Canny(gray, 80, 160)\n    edges_rgb = cv2.cvtColor(edges, cv2.COLOR_GRAY2RGB)\n    blended = cv2.addWeighted(img.astype(np.float32), 0.75, edges_rgb.astype(np.float32), 0.25, 0.0)\n    return np.clip(blended / 255.0, 0.0, 1.0).astype(np.float32)\n\ndef preprocess_aug(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    if np.random.rand() > 0.5:\n        img = cv2.flip(img, 1)\n    angle = np.random.uniform(-15, 15)\n    M = cv2.getRotationMatrix2D((IMG_SIZE//2, IMG_SIZE//2), angle, 1.0)\n    img = cv2.warpAffine(img, M, (IMG_SIZE, IMG_SIZE), borderMode=cv2.BORDER_REFLECT)\n    alpha = 1.0 + (np.random.rand() - 0.5) * 0.3\n    beta = int((np.random.rand() - 0.5) * 50)\n    img = cv2.convertScaleAbs(img, alpha=alpha, beta=beta)\n    return img.astype(np.float32) / 255.0\n\nPIPELINES = {\n    \"Standard\": preprocess_standard,\n    \"Lighting\": preprocess_lighting,\n    \"NoiseReduction\": preprocess_noise,\n    \"FeatureEnhancement\": preprocess_feature,\n    \"Augmentation\": preprocess_aug\n}\n\n# ------------------------------\n# Dense model builder (flatten input)\n# ------------------------------\ndef build_dense(input_dim, num_classes, lr=1e-4):\n    model = models.Sequential([\n        layers.Input(shape=(input_dim,)),\n        layers.Dense(1024, activation='relu'),\n        layers.BatchNormalization(),\n        layers.Dropout(0.5),\n        layers.Dense(512, activation='relu'),\n        layers.BatchNormalization(),\n        layers.Dropout(0.5),\n        layers.Dense(num_classes, activation='softmax')\n    ])\n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=lr),\n                  loss='sparse_categorical_crossentropy',\n                  metrics=['accuracy'])\n    return model\n\n# ------------------------------\n# Utilities: load exactly TRAIN_PER_CLASS per class\n# ------------------------------\ndef load_train_sampled(train_dir, preprocess_fn, per_class=TRAIN_PER_CLASS):\n    X, y = [], []\n    classes = sorted([d for d in os.listdir(train_dir) if os.path.isdir(os.path.join(train_dir, d))])\n    for label, cls in enumerate(classes):\n        cls_dir = os.path.join(train_dir, cls)\n        files = sorted(os.listdir(cls_dir))[:per_class]\n        for fname in files:\n            p = os.path.join(cls_dir, fname)\n            img = cv2.imread(p)\n            if img is None:\n                continue\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            X.append(preprocess_fn(img))\n            y.append(label)\n    X = np.array(X, dtype=np.float32)\n    y = np.array(y, dtype=np.int32)\n    return X, y, classes\n\n# ------------------------------\n# Utilities: load N random test images (no labels)\n# ------------------------------\ndef load_test_sample(test_dir, preprocess_fn, max_images=MAX_TEST_IMAGES):\n    all_files = sorted([f for f in os.listdir(test_dir) if os.path.isfile(os.path.join(test_dir, f))])\n    # sample randomly but reproducibly:\n    rng = random.Random(RANDOM_SEED)\n    sample_files = rng.sample(all_files, min(len(all_files), max_images))\n    X_test_raw = []        # raw resized images for display (RGB uint8)\n    X_test_input = []      # preprocessed normalized arrays\n    for fname in tqdm(sample_files, desc=\"Loading test images\"):\n        p = os.path.join(test_dir, fname)\n        img_bgr = cv2.imread(p)\n        if img_bgr is None:\n            continue\n        img_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n        img_resized = cv2.resize(img_rgb, (IMG_SIZE, IMG_SIZE))\n        X_test_raw.append(img_resized)                    # uint8 RGB for display\n        X_test_input.append(preprocess_fn(img_resized))   # normalized float32\n    X_test_raw = np.array(X_test_raw, dtype=np.uint8)\n    X_test_input = np.array(X_test_input, dtype=np.float32)\n    return X_test_raw, X_test_input\n\n# ------------------------------\n# Plot helpers\n# ------------------------------\ndef plot_confusion_matrix_labels(y_true, y_pred, classes, title=\"Confusion Matrix\"):\n    cm = confusion_matrix(y_true, y_pred, labels=np.arange(len(classes)))\n    disp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=classes)\n    plt.figure(figsize=(8,8))\n    disp.plot(cmap=plt.cm.Blues, xticks_rotation=45, ax=plt.gca())\n    plt.title(title)\n    plt.show()\n\ndef plot_training_curves(history, pipeline_name):\n    plt.figure(figsize=(12,4))\n    plt.subplot(1,2,1)\n    plt.plot(history.history.get(\"accuracy\", []), label=\"train_acc\")\n    plt.plot(history.history.get(\"val_accuracy\", []), label=\"val_acc\")\n    plt.xlabel(\"Epoch\"); plt.ylabel(\"Accuracy\"); plt.title(f\"{pipeline_name} Accuracy\"); plt.legend()\n    plt.subplot(1,2,2)\n    plt.plot(history.history.get(\"loss\", []), label=\"train_loss\")\n    plt.plot(history.history.get(\"val_loss\", []), label=\"val_loss\")\n    plt.xlabel(\"Epoch\"); plt.ylabel(\"Loss\"); plt.title(f\"{pipeline_name} Loss\"); plt.legend()\n    plt.show()\n\ndef plot_test_distribution(preds, classes, pipeline_name):\n    plt.figure(figsize=(9,4))\n    sns.countplot(x=preds, order=range(len(classes)))\n    plt.xticks(ticks=range(len(classes)), labels=classes, rotation=45)\n    plt.title(f\"Prediction distribution on test (pipeline={pipeline_name})\")\n    plt.xlabel(\"Predicted class\"); plt.ylabel(\"Count\")\n    plt.show()\n\ndef show_test_samples(X_raw, preds, classes, pipeline_name, n_samples=20):\n    n = min(n_samples, len(X_raw))\n    rng = random.Random(RANDOM_SEED)\n    idxs = rng.sample(range(len(X_raw)), n)\n    cols = 5\n    rows = int(np.ceil(n / cols))\n    plt.figure(figsize=(cols * 3, rows * 3))\n    for i, idx in enumerate(idxs):\n        plt.subplot(rows, cols, i+1)\n        plt.imshow(X_raw[idx])\n        # ====== تعديل العنوان: يكتب الكلاس + الاسم الكامل ======\n        class_id = preds[idx]\n        class_code = classes[class_id]\n        class_name_full = class_full_names.get(class_code, class_code)\n        plt.title(f\"{class_code} - {class_name_full}\", fontsize=9)\n        plt.axis(\"off\")\n    plt.suptitle(f\"Sample Test Predictions - {pipeline_name}\", fontsize=16)\n    plt.show()\n\n# ------------------------------\n# Main loop: iterate pipelines\n# ------------------------------\nresults_summary = {}\n\nfor pipeline_name, preprocess_fn in PIPELINES.items():\n    print(\"\\n\" + \"=\"*80)\n    print(f\"PIPELINE: {pipeline_name}\")\n    print(\"=\"*80)\n\n    # 1) Load train (1000 per class) & make train/val split\n    X_all, y_all, classes = load_train_sampled(TRAIN_DIR, preprocess_fn, per_class=TRAIN_PER_CLASS)\n    if X_all.shape[0] == 0:\n        raise RuntimeError(f\"No training images found in {TRAIN_DIR}. Check path.\")\n    print(f\"Loaded train images: {X_all.shape}, labels: {np.unique(y_all).size} classes\")\n\n    X_train, X_val, y_train, y_val = train_test_split(\n        X_all, y_all, test_size=0.2, random_state=RANDOM_SEED, stratify=y_all\n    )\n    print(\"Train shape:\", X_train.shape, \"Val shape:\", X_val.shape)\n\n    # 2) Prepare dense inputs (flatten)\n    X_tr_input = X_train.reshape(X_train.shape[0], -1)\n    X_val_input = X_val.reshape(X_val.shape[0], -1)\n    input_dim = X_tr_input.shape[1]\n    print(\"Dense input dimension:\", input_dim)\n\n    # 3) Build + train model\n    model = build_dense(input_dim=input_dim, num_classes=num_classes, lr=1e-4)\n    model.summary()\n\n    cb_list = [\n        callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=3, verbose=1),\n        callbacks.EarlyStopping(monitor='val_accuracy', patience=5, restore_best_weights=True, verbose=1)\n    ]\n\n    history = model.fit(\n        X_tr_input, y_train,\n        validation_data=(X_val_input, y_val),\n        epochs=EPOCHS,\n        batch_size=BATCH_SIZE,\n        callbacks=cb_list,\n        verbose=1\n    )\n\n    # 4) Validation evaluation: predictions, conf matrix, acc\n    y_val_pred = np.argmax(model.predict(X_val_input, verbose=0), axis=1)\n    val_acc = accuracy_score(y_val, y_val_pred)\n    print(f\"[{pipeline_name}] Validation accuracy = {val_acc:.4f}\")\n\n    plot_confusion_matrix_labels(y_val, y_val_pred, classes, title=f\"Validation Confusion Matrix - {pipeline_name}\")\n    plot_training_curves(history, pipeline_name)\n\n    # 5) Test predictions on sampled 3000 images\n    print(\"Loading & predicting on up to\", MAX_TEST_IMAGES, \"test images (random sample)...\")\n    X_test_raw, X_test_input = load_test_sample(TEST_DIR, preprocess_fn, max_images=MAX_TEST_IMAGES)\n    if X_test_input.shape[0] == 0:\n        print(\"No test images loaded (check TEST_DIR). Skipping test predictions for this pipeline.\")\n        results_summary[pipeline_name] = {\n            \"val_acc\": val_acc,\n            \"num_test\": 0,\n            \"test_preds\": None\n        }\n        continue\n\n    # flatten test input for dense model\n    X_test_flat = X_test_input.reshape(X_test_input.shape[0], -1)\n\n    # Predict in batches\n    preds_prob = model.predict(X_test_flat, batch_size=128, verbose=0)\n    preds = np.argmax(preds_prob, axis=1)\n\n    # 6) Show distribution and sample images\n    plot_test_distribution(preds, classes, pipeline_name)\n    show_test_samples(X_test_raw, preds, classes, pipeline_name, n_samples=20)\n\n    # Save summary\n    results_summary[pipeline_name] = {\n        \"val_acc\": val_acc,\n        \"num_test\": X_test_flat.shape[0],\n        \"test_preds\": preds  # array of predicted class indices for test sample\n    }\n\n# ------------------------------\n# Final summary: validation accuracies bar chart\n# ------------------------------\npipelines_done = list(results_summary.keys())\nval_accs = [results_summary[p][\"val_acc\"] for p in pipelines_done]\n\nplt.figure(figsize=(9,5))\nsns.barplot(x=pipelines_done, y=val_accs, palette=\"mako\")\nplt.ylim(0,1)\nplt.ylabel(\"Validation Accuracy\")\nplt.title(\"Validation Accuracy per Preprocessing Pipeline\")\nplt.xticks(rotation=20)\nplt.show()\n\n# Print numeric summary\nprint(\"\\nNumeric summary:\")\nfor p in pipelines_done:\n    info = results_summary[p]\n    print(f\"{p:20s} | val_acc = {info['val_acc']:.4f} | test_images = {info['num_test']}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, accuracy_score\nfrom sklearn.metrics import ConfusionMatrixDisplay\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, callbacks\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.applications.resnet50 import preprocess_input\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout, Input\n\n# ------------------------------\n# Config (Kaggle dataset paths)\n# ------------------------------\nTRAIN_DIR = \"/kaggle/input/state-farm-distracted-driver-detection/imgs/train\"\nTEST_DIR  = \"/kaggle/input/state-farm-distracted-driver-detection/imgs/test\"\n\nIMG_SIZE = 64                # resize: 64x64\nTRAIN_PER_CLASS = 1500       # take up to 1000 images per class from train/\nMAX_TEST_IMAGES = 50000       # number of test images to sample for predictions\nEPOCHS = 30                   # lower for Kaggle runtime\nBATCH_SIZE = 32\nRANDOM_SEED = 42\n\nnp.random.seed(RANDOM_SEED)\nrandom.seed(RANDOM_SEED)\ntf.random.set_seed(RANDOM_SEED)\n\n# ------------------------------\n# Class names (folder names c0..c9)\n# ------------------------------\nclass_names = sorted([d for d in os.listdir(TRAIN_DIR) if os.path.isdir(os.path.join(TRAIN_DIR, d))])\nnum_classes = len(class_names)\nprint(\"Detected classes:\", class_names)\n\n# ------------------------------\n# Preprocessing functions (5 pipelines)\n# ------------------------------\ndef preprocess_standard(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    yuv = cv2.cvtColor(img, cv2.COLOR_RGB2YUV)\n    yuv[:,:,0] = cv2.equalizeHist(yuv[:,:,0])\n    img = cv2.cvtColor(yuv, cv2.COLOR_YUV2RGB)\n    return preprocess_input(img.astype(np.float32))\n\ndef preprocess_lighting(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    lab = cv2.cvtColor(img, cv2.COLOR_RGB2LAB)\n    l, a, b = cv2.split(lab)\n    clahe = cv2.createCLAHE(clipLimit=3.0, tileGridSize=(8,8))\n    l = clahe.apply(l)\n    lab = cv2.merge((l,a,b))\n    img = cv2.cvtColor(lab, cv2.COLOR_LAB2RGB)\n    gamma = 1.15\n    img = np.power(img/255.0, 1.0/gamma) * 255.0\n    return preprocess_input(img.astype(np.float32))\n\ndef preprocess_noise(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    img = cv2.bilateralFilter(img, d=9, sigmaColor=75, sigmaSpace=75)\n    img = cv2.medianBlur(img, 3)\n    return preprocess_input(img.astype(np.float32))\n\ndef preprocess_feature(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n    edges = cv2.Canny(gray, 80, 160)\n    edges_rgb = cv2.cvtColor(edges, cv2.COLOR_GRAY2RGB)\n    blended = cv2.addWeighted(img.astype(np.float32), 0.75, edges_rgb.astype(np.float32), 0.25, 0.0)\n    return preprocess_input(np.clip(blended, 0.0, 255.0).astype(np.float32))\n\ndef preprocess_aug(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    if np.random.rand() > 0.5:\n        img = cv2.flip(img, 1)\n    angle = np.random.uniform(-15, 15)\n    M = cv2.getRotationMatrix2D((IMG_SIZE//2, IMG_SIZE//2), angle, 1.0)\n    img = cv2.warpAffine(img, M, (IMG_SIZE, IMG_SIZE), borderMode=cv2.BORDER_REFLECT)\n    alpha = 1.0 + (np.random.rand() - 0.5) * 0.3\n    beta = int((np.random.rand() - 0.5) * 50)\n    img = cv2.convertScaleAbs(img, alpha=alpha, beta=beta)\n    return preprocess_input(img.astype(np.float32))\n\nPIPELINES = {\n    \"Standard\": preprocess_standard,\n    \"Lighting\": preprocess_lighting,\n    \"NoiseReduction\": preprocess_noise,\n    \"FeatureEnhancement\": preprocess_feature,\n    \"Augmentation\": preprocess_aug\n}\n\n# ------------------------------\n# Transfer Learning: ResNet50\n# ------------------------------\ndef build_resnet50_model(input_shape=(IMG_SIZE, IMG_SIZE, 3), num_classes=num_classes, lr=1e-4):\n    base_model = ResNet50(weights='imagenet', include_top=False, input_shape=input_shape)\n    base_model.trainable = False  # freeze base\n\n    x = base_model.output\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(512, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    predictions = Dense(num_classes, activation='softmax')(x)\n\n    model = Model(inputs=base_model.input, outputs=predictions)\n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=lr),\n                  loss='sparse_categorical_crossentropy',\n                  metrics=['accuracy'])\n    return model\n\n# ------------------------------\n# Utilities: load exactly TRAIN_PER_CLASS per class\n# ------------------------------\ndef load_train_sampled(train_dir, preprocess_fn, per_class=TRAIN_PER_CLASS):\n    X, y = [], []\n    classes = sorted([d for d in os.listdir(train_dir) if os.path.isdir(os.path.join(train_dir, d))])\n    for label, cls in enumerate(classes):\n        cls_dir = os.path.join(train_dir, cls)\n        files = sorted(os.listdir(cls_dir))[:per_class]\n        for fname in files:\n            p = os.path.join(cls_dir, fname)\n            img = cv2.imread(p)\n            if img is None:\n                continue\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            X.append(preprocess_fn(img))\n            y.append(label)\n    return np.array(X, dtype=np.float32), np.array(y, dtype=np.int32), classes\n\n# ------------------------------\n# Utilities: load N random test images (no labels)\n# ------------------------------\ndef load_test_sample(test_dir, preprocess_fn, max_images=MAX_TEST_IMAGES):\n    all_files = sorted([f for f in os.listdir(test_dir) if os.path.isfile(os.path.join(test_dir, f))])\n    rng = random.Random(RANDOM_SEED)\n    sample_files = rng.sample(all_files, min(len(all_files), max_images))\n    X_test_raw, X_test_input = [], []\n    for fname in tqdm(sample_files, desc=\"Loading test images\"):\n        p = os.path.join(test_dir, fname)\n        img_bgr = cv2.imread(p)\n        if img_bgr is None:\n            continue\n        img_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n        img_resized = cv2.resize(img_rgb, (IMG_SIZE, IMG_SIZE))\n        X_test_raw.append(img_resized)\n        X_test_input.append(preprocess_fn(img_resized))\n    return np.array(X_test_raw, dtype=np.uint8), np.array(X_test_input, dtype=np.float32)\n\n# ------------------------------\n# Plot helpers\n# ------------------------------\ndef plot_confusion_matrix_labels(y_true, y_pred, classes, title=\"Confusion Matrix\"):\n    cm = confusion_matrix(y_true, y_pred, labels=np.arange(len(classes)))\n    disp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=classes)\n    plt.figure(figsize=(8,8))\n    disp.plot(cmap=plt.cm.Blues, xticks_rotation=45, ax=plt.gca())\n    plt.title(title)\n    plt.show()\n\ndef plot_training_curves(history, pipeline_name):\n    plt.figure(figsize=(12,4))\n    plt.subplot(1,2,1)\n    plt.plot(history.history.get(\"accuracy\", []), label=\"train_acc\")\n    plt.plot(history.history.get(\"val_accuracy\", []), label=\"val_acc\")\n    plt.xlabel(\"Epoch\"); plt.ylabel(\"Accuracy\"); plt.title(f\"{pipeline_name} Accuracy\"); plt.legend()\n    plt.subplot(1,2,2)\n    plt.plot(history.history.get(\"loss\", []), label=\"train_loss\")\n    plt.plot(history.history.get(\"val_loss\", []), label=\"val_loss\")\n    plt.xlabel(\"Epoch\"); plt.ylabel(\"Loss\"); plt.title(f\"{pipeline_name} Loss\"); plt.legend()\n    plt.show()\n\ndef plot_test_distribution(preds, classes, pipeline_name):\n    plt.figure(figsize=(9,4))\n    sns.countplot(x=preds, order=range(len(classes)))\n    plt.xticks(ticks=range(len(classes)), labels=classes, rotation=45)\n    plt.title(f\"Prediction distribution on test (pipeline={pipeline_name})\")\n    plt.xlabel(\"Predicted class\"); plt.ylabel(\"Count\")\n    plt.show()\n\ndef show_test_samples(X_raw, preds, classes, pipeline_name, n_samples=20):\n    n = min(n_samples, len(X_raw))\n    rng = random.Random(RANDOM_SEED)\n    idxs = rng.sample(range(len(X_raw)), n)\n    cols = 5\n    rows = int(np.ceil(n / cols))\n    plt.figure(figsize=(cols * 3, rows * 3))\n    for i, idx in enumerate(idxs):\n        plt.subplot(rows, cols, i+1)\n        plt.imshow(X_raw[idx])\n        plt.title(f\"{classes[preds[idx]]} ({preds[idx]})\")\n        plt.axis(\"off\")\n    plt.suptitle(f\"Sample Test Predictions - {pipeline_name}\", fontsize=16)\n    plt.show()\n\n# ------------------------------\n# Main loop: iterate pipelines\n# ------------------------------\nresults_summary = {}\n\nfor pipeline_name, preprocess_fn in PIPELINES.items():\n    print(\"\\n\" + \"=\"*80)\n    print(f\"PIPELINE: {pipeline_name}\")\n    print(\"=\"*80)\n\n    X_all, y_all, classes = load_train_sampled(TRAIN_DIR, preprocess_fn, per_class=TRAIN_PER_CLASS)\n    if X_all.shape[0] == 0:\n        raise RuntimeError(f\"No training images found in {TRAIN_DIR}. Check path.\")\n    print(f\"Loaded train images: {X_all.shape}, labels: {np.unique(y_all).size} classes\")\n\n    X_train, X_val, y_train, y_val = train_test_split(\n        X_all, y_all, test_size=0.2, random_state=RANDOM_SEED, stratify=y_all\n    )\n    print(\"Train shape:\", X_train.shape, \"Val shape:\", X_val.shape)\n\n    model = build_resnet50_model(input_shape=(IMG_SIZE, IMG_SIZE, 3), num_classes=num_classes, lr=1e-4)\n    model.summary()\n\n    cb_list = [\n        callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=3, verbose=1),\n        callbacks.EarlyStopping(monitor='val_accuracy', patience=5, restore_best_weights=True, verbose=1)\n    ]\n\n    history = model.fit(\n        X_train, y_train,\n        validation_data=(X_val, y_val),\n        epochs=EPOCHS,\n        batch_size=BATCH_SIZE,\n        callbacks=cb_list,\n        verbose=1\n    )\n\n    y_val_pred = np.argmax(model.predict(X_val, verbose=0), axis=1)\n    val_acc = accuracy_score(y_val, y_val_pred)\n    print(f\"[{pipeline_name}] Validation accuracy = {val_acc:.4f}\")\n\n    plot_confusion_matrix_labels(y_val, y_val_pred, classes, title=f\"Validation Confusion Matrix - {pipeline_name}\")\n    plot_training_curves(history, pipeline_name)\n\n    print(\"Loading & predicting on up to\", MAX_TEST_IMAGES, \"test images (random sample)...\")\n    X_test_raw, X_test_input = load_test_sample(TEST_DIR, preprocess_fn, max_images=MAX_TEST_IMAGES)\n    if X_test_input.shape[0] == 0:\n        print(\"No test images loaded (check TEST_DIR). Skipping test predictions.\")\n        results_summary[pipeline_name] = {\"val_acc\": val_acc, \"num_test\": 0, \"test_preds\": None}\n        continue\n\n    preds_prob = model.predict(X_test_input, batch_size=128, verbose=0)\n    preds = np.argmax(preds_prob, axis=1)\n\n    plot_test_distribution(preds, classes, pipeline_name)\n    show_test_samples(X_test_raw, preds, classes, pipeline_name, n_samples=20)\n\n    results_summary[pipeline_name] = {\"val_acc\": val_acc, \"num_test\": X_test_input.shape[0], \"test_preds\": preds}\n\n# ------------------------------\n# Final summary: validation accuracies bar chart\n# ------------------------------\npipelines_done = list(results_summary.keys())\nval_accs = [results_summary[p][\"val_acc\"] for p in pipelines_done]\n\nplt.figure(figsize=(9,5))\nsns.barplot(x=pipelines_done, y=val_accs, palette=\"mako\")\nplt.ylim(0,1)\nplt.ylabel(\"Validation Accuracy\")\nplt.title(\"Validation Accuracy per Preprocessing Pipeline\")\nplt.xticks(rotation=20)\nplt.show()\n\nprint(\"\\nNumeric summary:\")\nfor p in pipelines_done:\n    info = results_summary[p]\n    print(f\"{p:20s} | val_acc = {info['val_acc']:.4f} | test_images = {info['num_test']}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T16:02:55.976567Z","iopub.execute_input":"2025-08-21T16:02:55.976847Z","iopub.status.idle":"2025-08-21T17:01:08.323460Z","shell.execute_reply.started":"2025-08-21T16:02:55.976825Z","shell.execute_reply":"2025-08-21T17:01:08.322608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==============================\n# Full end-to-end script (Kaggle-ready)\n# State Farm Distracted Driver — 5 Preprocessing Pipelines + ResNet50 (frozen)\n# ==============================\n\nimport os\nimport cv2\nimport math\nimport random\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\n\nfrom sklearn.metrics import confusion_matrix, accuracy_score, ConfusionMatrixDisplay\nfrom tensorflow.keras import callbacks\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.applications.resnet50 import preprocess_input\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\nimport tensorflow as tf\n\n# ------------------------------\n# Config (Kaggle dataset paths)\n# ------------------------------\nTRAIN_DIR = \"/kaggle/input/state-farm-distracted-driver-detection/imgs/train\"\nTEST_DIR  = \"/kaggle/input/state-farm-distracted-driver-detection/imgs/test\"\n\nIMG_SIZE = 224                 # ResNet50 input size\nPER_CLASS_LIMIT = 1500         # 1500 images per class from train/\nSPLIT_RATIOS = (0.70, 0.15, 0.15)  # train/val/test\nMAX_TEST_IMAGES = 10000        # unlabeled test sample size\nEPOCHS = 10                    # زدها لو عندك وقت على كاجل\nBATCH_SIZE = 32\nRANDOM_SEED = 42\nLR = 1e-4\n\nnp.random.seed(RANDOM_SEED)\nrandom.seed(RANDOM_SEED)\ntf.random.set_seed(RANDOM_SEED)\n\n# ------------------------------\n# Classes (folder names c0..c9) + readable names\n# ------------------------------\nclass_names = sorted([d for d in os.listdir(TRAIN_DIR) if os.path.isdir(os.path.join(TRAIN_DIR, d))])\nnum_classes = len(class_names)\nprint(\"Detected classes:\", class_names)\n\nclass_full_names = {\n    \"c0\": \"safe driving\",\n    \"c1\": \"texting right\",\n    \"c2\": \"talking on phone right\",\n    \"c3\": \"texting left\",\n    \"c4\": \"talking on phone left\",\n    \"c5\": \"operating radio\",\n    \"c6\": \"drinking\",\n    \"c7\": \"reaching behind\",\n    \"c8\": \"hair & makeup\",\n    \"c9\": \"talking to passenger\"\n}\n\n# ------------------------------\n# Preprocessing pipelines (return RGB float32 in [0..255]; preprocess_input handles normalization)\n# ------------------------------\ndef preprocess_standard(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    yuv = cv2.cvtColor(img, cv2.COLOR_RGB2YUV)\n    yuv[:, :, 0] = cv2.equalizeHist(yuv[:, :, 0])\n    img = cv2.cvtColor(yuv, cv2.COLOR_YUV2RGB)\n    return img.astype(np.float32)\n\ndef preprocess_lighting(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    lab = cv2.cvtColor(img, cv2.COLOR_RGB2LAB)\n    l, a, b = cv2.split(lab)\n    clahe = cv2.createCLAHE(clipLimit=3.0, tileGridSize=(8,8))\n    l = clahe.apply(l)\n    lab = cv2.merge((l, a, b))\n    img = cv2.cvtColor(lab, cv2.COLOR_LAB2RGB)\n    gamma = 1.15\n    img = np.power(np.clip(img, 0, 255)/255.0, 1.0/gamma) * 255.0\n    return np.clip(img, 0, 255).astype(np.float32)\n\ndef preprocess_noise(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    img = cv2.bilateralFilter(img, d=9, sigmaColor=75, sigmaSpace=75)\n    img = cv2.medianBlur(img, 3)\n    return img.astype(np.float32)\n\ndef preprocess_feature(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n    edges = cv2.Canny(gray, 80, 160)\n    edges_rgb = cv2.cvtColor(edges, cv2.COLOR_GRAY2RGB)\n    blended = cv2.addWeighted(img.astype(np.float32), 0.75, edges_rgb.astype(np.float32), 0.25, 0.0)\n    return np.clip(blended, 0, 255).astype(np.float32)\n\ndef preprocess_aug(img):\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    if np.random.rand() > 0.5:\n        img = cv2.flip(img, 1)\n    angle = np.random.uniform(-15, 15)\n    M = cv2.getRotationMatrix2D((IMG_SIZE//2, IMG_SIZE//2), angle, 1.0)\n    img = cv2.warpAffine(img, M, (IMG_SIZE, IMG_SIZE), borderMode=cv2.BORDER_REFLECT)\n    alpha = 1.0 + (np.random.rand() - 0.5) * 0.3\n    beta = int((np.random.rand() - 0.5) * 50)\n    img = cv2.convertScaleAbs(img, alpha=alpha, beta=beta)\n    return img.astype(np.float32)\n\nPIPELINES = {\n    \"Standard\": preprocess_standard,\n    \"Lighting\": preprocess_lighting,\n    \"NoiseReduction\": preprocess_noise,\n    \"FeatureEnhancement\": preprocess_feature,\n    \"Augmentation\": preprocess_aug\n}\n\n# ------------------------------\n# Build ResNet50 (Frozen) classifier\n# ------------------------------\ndef build_resnet50_model(input_shape=(IMG_SIZE, IMG_SIZE, 3), num_classes=num_classes, lr=LR):\n    base = ResNet50(include_top=False, weights='imagenet', input_shape=input_shape)\n    base.trainable = False\n    x = GlobalAveragePooling2D()(base.output)\n    x = Dense(512, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    out = Dense(num_classes, activation='softmax')(x)\n    model = Model(inputs=base.input, outputs=out)\n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=lr),\n                  loss='sparse_categorical_crossentropy',\n                  metrics=['accuracy'])\n    return model\n\n# ------------------------------\n# Utilities: collect & split paths (per class)\n# ------------------------------\ndef collect_paths_per_class(train_dir, per_class=PER_CLASS_LIMIT, seed=RANDOM_SEED):\n    rng = random.Random(seed)\n    class_to_paths = {}\n    for cls in sorted([d for d in os.listdir(train_dir) if os.path.isdir(os.path.join(train_dir, d))]):\n        cls_dir = os.path.join(train_dir, cls)\n        files = [os.path.join(cls_dir, f) for f in os.listdir(cls_dir) if os.path.isfile(os.path.join(cls_dir, f))]\n        files.sort()\n        if len(files) > per_class:\n            rng.shuffle(files)\n            files = files[:per_class]\n        class_to_paths[cls] = files\n    return class_to_paths\n\ndef split_70_15_15(class_to_paths, ratios=SPLIT_RATIOS, seed=RANDOM_SEED):\n    train_paths, val_paths, test_paths = [], [], []\n    train_labels, val_labels, test_labels = [], [], []\n    rng = random.Random(seed)\n    for label, cls in enumerate(sorted(class_to_paths.keys())):\n        paths = class_to_paths[cls]\n        idxs = list(range(len(paths)))\n        rng.shuffle(idxs)\n        n = len(idxs)\n        n_train = int(round(n * ratios[0]))\n        n_val   = int(round(n * ratios[1]))\n        # ensure total == n\n        n_test  = n - n_train - n_val\n\n        tr = idxs[:n_train]\n        va = idxs[n_train:n_train+n_val]\n        te = idxs[n_train+n_val:]\n\n        for i in tr:\n            train_paths.append(paths[i]); train_labels.append(label)\n        for i in va:\n            val_paths.append(paths[i]);   val_labels.append(label)\n        for i in te:\n            test_paths.append(paths[i]);  test_labels.append(label)\n\n    return (train_paths, np.array(train_labels, dtype=np.int32),\n            val_paths,   np.array(val_labels,   dtype=np.int32),\n            test_paths,  np.array(test_labels,  dtype=np.int32))\n\n# ------------------------------\n# Generator reading from disk + preprocessing + ResNet50 preprocess_input\n# ------------------------------\ndef make_generator(paths, labels, preprocess_fn, batch_size=BATCH_SIZE, shuffle=True):\n    n = len(paths)\n    idxs = np.arange(n)\n    while True:\n        if shuffle:\n            np.random.shuffle(idxs)\n        for start in range(0, n, batch_size):\n            end = min(start + batch_size, n)\n            batch_idx = idxs[start:end]\n            X_batch = np.zeros((len(batch_idx), IMG_SIZE, IMG_SIZE, 3), dtype=np.float32)\n            y_batch = None if labels is None else labels[batch_idx]\n            for j, k in enumerate(batch_idx):\n                p = paths[k]\n                img_bgr = cv2.imread(p)\n                if img_bgr is None:\n                    # fallback empty if read fails\n                    X_batch[j] = np.zeros((IMG_SIZE, IMG_SIZE, 3), dtype=np.float32)\n                    continue\n                img_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n                proc = preprocess_fn(img_rgb)        # float32 RGB [0..255]\n                X_batch[j] = preprocess_input(proc)  # ResNet50 preprocess\n            yield (X_batch, y_batch) if labels is not None else X_batch\n\n# ------------------------------\n# Plot helpers\n# ------------------------------\ndef plot_training_curves(history, pipeline_name):\n    plt.figure(figsize=(12,4))\n    plt.subplot(1,2,1)\n    plt.plot(history.history.get(\"accuracy\", []), label=\"train_acc\")\n    plt.plot(history.history.get(\"val_accuracy\", []), label=\"val_acc\")\n    plt.xlabel(\"Epoch\"); plt.ylabel(\"Accuracy\"); plt.title(f\"{pipeline_name} Accuracy\"); plt.legend()\n    plt.subplot(1,2,2)\n    plt.plot(history.history.get(\"loss\", []), label=\"train_loss\")\n    plt.plot(history.history.get(\"val_loss\", []), label=\"val_loss\")\n    plt.xlabel(\"Epoch\"); plt.ylabel(\"Loss\"); plt.title(f\"{pipeline_name} Loss\"); plt.legend()\n    plt.show()\n\ndef plot_confusion_matrix_labels(y_true, y_pred, classes, title=\"Confusion Matrix\"):\n    cm = confusion_matrix(y_true, y_pred, labels=np.arange(len(classes)))\n    disp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=classes)\n    plt.figure(figsize=(8,8))\n    disp.plot(cmap=plt.cm.Blues, xticks_rotation=45, ax=plt.gca(), colorbar=False)\n    plt.title(title)\n    plt.show()\n\ndef plot_test_distribution(preds, classes, pipeline_name):\n    plt.figure(figsize=(9,4))\n    sns.countplot(x=preds, order=range(len(classes)))\n    plt.xticks(ticks=range(len(classes)), labels=classes, rotation=45)\n    plt.title(f\"Prediction distribution on 10k test (pipeline={pipeline_name})\")\n    plt.xlabel(\"Predicted class\"); plt.ylabel(\"Count\")\n    plt.show()\n\ndef show_test_samples(paths, preds, classes, pipeline_name, n_samples=20):\n    n = min(n_samples, len(paths))\n    rng = random.Random(RANDOM_SEED)\n    idxs = rng.sample(range(len(paths)), n)\n    cols = 5\n    rows = int(np.ceil(n / cols))\n    plt.figure(figsize=(cols * 3, rows * 3))\n    for i, idx in enumerate(idxs):\n        img_bgr = cv2.imread(paths[idx])\n        if img_bgr is None:\n            continue\n        img_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n        img_resized = cv2.resize(img_rgb, (IMG_SIZE, IMG_SIZE))\n        class_id = preds[idx]\n        class_code = classes[class_id]\n        class_name_full = class_full_names.get(class_code, class_code)\n        plt.subplot(rows, cols, i+1)\n        plt.imshow(img_resized)\n        plt.title(f\"{class_code} - {class_name_full}\", fontsize=9)\n        plt.axis(\"off\")\n    plt.suptitle(f\"Sample Test Predictions - {pipeline_name}\", fontsize=16)\n    plt.show()\n\n# ------------------------------\n# Train/Eval per pipeline\n# ------------------------------\nresults_summary = {}\n\nfor pipeline_name, preprocess_fn in PIPELINES.items():\n    print(\"\\n\" + \"=\"*100)\n    print(f\"PIPELINE: {pipeline_name}\")\n    print(\"=\"*100)\n\n    # 1) Collect & split paths 70/15/15\n    class_to_paths = collect_paths_per_class(TRAIN_DIR, per_class=PER_CLASS_LIMIT, seed=RANDOM_SEED)\n    (train_paths, train_labels,\n     val_paths,   val_labels,\n     test_paths,  test_labels) = split_70_15_15(class_to_paths, ratios=SPLIT_RATIOS, seed=RANDOM_SEED)\n\n    print(f\"Train: {len(train_paths)} | Val: {len(val_paths)} | Test: {len(test_paths)}\")\n\n    # 2) Build model\n    model = build_resnet50_model(input_shape=(IMG_SIZE, IMG_SIZE, 3), num_classes=num_classes, lr=LR)\n    model.summary()\n\n    steps_per_epoch = math.ceil(len(train_paths) / BATCH_SIZE)\n    val_steps       = math.ceil(len(val_paths)   / BATCH_SIZE)\n    test_steps      = math.ceil(len(test_paths)  / BATCH_SIZE)\n\n    # 3) Generators\n    train_gen = make_generator(train_paths, train_labels, preprocess_fn, batch_size=BATCH_SIZE, shuffle=True)\n    val_gen   = make_generator(val_paths,   val_labels,   preprocess_fn, batch_size=BATCH_SIZE, shuffle=False)\n    test_gen  = make_generator(test_paths,  test_labels,  preprocess_fn, batch_size=BATCH_SIZE, shuffle=False)\n\n    # 4) Callbacks\n    cb_list = [\n        callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=2, verbose=1),\n        callbacks.EarlyStopping(monitor='val_accuracy', patience=3, restore_best_weights=True, verbose=1)\n    ]\n\n    # 5) Train\n    history = model.fit(\n        train_gen,\n        steps_per_epoch=steps_per_epoch,\n        validation_data=val_gen,\n        validation_steps=val_steps,\n        epochs=EPOCHS,\n        callbacks=cb_list,\n        verbose=1\n    )\n\n    # 6) Validation predictions (Confusion Matrix + Val Accuracy)\n    val_pred_prob = model.predict(val_gen, steps=val_steps, verbose=0)\n    y_val_pred = np.argmax(val_pred_prob, axis=1)\n    # trim to exact length (last partial batch)\n    y_val_pred = y_val_pred[:len(val_paths)]\n    y_val_true = val_labels[:len(val_paths)]\n\n    val_acc = accuracy_score(y_val_true, y_val_pred)\n    print(f\"[{pipeline_name}] Validation accuracy = {val_acc:.4f}\")\n\n    plot_confusion_matrix_labels(y_val_true, y_val_pred, class_names,\n                                 title=f\"Validation Confusion Matrix - {pipeline_name}\")\n    plot_training_curves(history, pipeline_name)\n\n    # 7) Evaluate on internal 15% test (optional metrics)\n    test_pred_prob = model.predict(test_gen, steps=test_steps, verbose=0)\n    y_test_pred = np.argmax(test_pred_prob, axis=1)\n    y_test_pred = y_test_pred[:len(test_paths)]\n    y_test_true = test_labels[:len(test_paths)]\n    test_acc = accuracy_score(y_test_true, y_test_pred)\n    print(f\"[{pipeline_name}] Internal 15% test accuracy = {test_acc:.4f}\")\n\n    # 8) Unlabeled Kaggle test predictions (10,000 random)\n    all_test_files = [f for f in os.listdir(TEST_DIR) if os.path.isfile(os.path.join(TEST_DIR, f))]\n    rng = random.Random(RANDOM_SEED)\n    sample_files = rng.sample(all_test_files, min(len(all_test_files), MAX_TEST_IMAGES))\n    test_unlabeled_paths = [os.path.join(TEST_DIR, f) for f in sample_files]\n\n    # Batch predict\n    preds_all = []\n    batch = []\n    for p in tqdm(test_unlabeled_paths, desc=f\"Predicting 10k test ({pipeline_name})\"):\n        img_bgr = cv2.imread(p)\n        if img_bgr is None:\n            continue\n        img_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n        proc = preprocess_fn(img_rgb)\n        x = preprocess_input(proc)\n        batch.append(x)\n        if len(batch) == 128:\n            batch_np = np.stack(batch, axis=0)\n            prob = model.predict(batch_np, verbose=0)\n            preds_all.extend(np.argmax(prob, axis=1))\n            batch = []\n    if len(batch) > 0:\n        batch_np = np.stack(batch, axis=0)\n        prob = model.predict(batch_np, verbose=0)\n        preds_all.extend(np.argmax(prob, axis=1))\n\n    preds_all = np.array(preds_all, dtype=np.int32)\n    # Trim paths to preds length in case of read failures\n    test_unlabeled_paths = test_unlabeled_paths[:len(preds_all)]\n\n    # 9) Show prediction distribution + 20 samples with class code + full name\n    plot_test_distribution(preds_all, class_names, pipeline_name)\n    show_test_samples(test_unlabeled_paths, preds_all, class_names, pipeline_name, n_samples=20)\n\n    # Save summary\n    results_summary[pipeline_name] = {\n        \"val_acc\": val_acc,\n        \"test_acc_internal\": test_acc,\n        \"num_val\": len(val_paths),\n        \"num_test_internal\": len(test_paths),\n        \"num_test_unlabeled_pred\": len(test_unlabeled_paths)\n    }\n\n# ------------------------------\n# Final summary: compare validation accuracies across 5 pipelines\n# ------------------------------\npipelines_done = list(results_summary.keys())\nval_accs = [results_summary[p][\"val_acc\"] for p in pipelines_done]\n\nplt.figure(figsize=(9,5))\nsns.barplot(x=pipelines_done, y=val_accs)\nplt.ylim(0,1)\nplt.ylabel(\"Validation Accuracy\")\nplt.title(\"Validation Accuracy per Preprocessing Pipeline (ResNet50 Frozen)\")\nplt.xticks(rotation=20)\nplt.show()\n\nprint(\"\\nNumeric summary:\")\nfor p in pipelines_done:\n    info = results_summary[p]\n    print(f\"{p:20s} | val_acc = {info['val_acc']:.4f} | internal_test_acc = {info['test_acc_internal']:.4f} | \"\n          f\"val_n = {info['num_val']:5d} | test_n = {info['num_test_internal']:5d} | unlabeled_test_pred_n = {info['num_test_unlabeled_pred']:5d}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T17:56:42.978798Z","iopub.execute_input":"2025-08-21T17:56:42.979815Z","iopub.status.idle":"2025-08-21T19:41:26.471711Z","shell.execute_reply.started":"2025-08-21T17:56:42.979790Z","shell.execute_reply":"2025-08-21T19:41:26.470884Z"}},"outputs":[],"execution_count":null}]}