{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.18"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceType":"competition","sourceId":132732,"databundleVersionId":16583342}],"dockerImageVersionId":31091,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":414.026879,"end_time":"2026-04-11T02:47:32.68619","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-04-11T02:40:38.659311","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"","metadata":{"papermill":{"duration":0.001809,"end_time":"2026-04-11T02:40:41.622063","exception":false,"start_time":"2026-04-11T02:40:41.620254","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# **Synthetic Image: MobiletNet Classifier w/TPU**\ntrain/valid data not splitted","metadata":{"papermill":{"duration":0.001157,"end_time":"2026-04-11T02:40:41.624671","exception":false,"start_time":"2026-04-11T02:40:41.623514","status":"completed"},"tags":[]}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================================\n# IMAGE CLASSIFIER FOR CSV-BASED LABELS (TPU v5e-8 Compatible)\n# ============================================================================\n\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, optimizers\nfrom tensorflow.keras.applications import MobileNetV2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nfrom pathlib import Path\n\n# TPU Strategy setup\nos.environ['TPU_NAME'] = 'local'\n\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver(tpu='local')\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\n    NUM_REPLICAS = strategy.num_replicas_in_sync\n    print(f\"✅ TPU initialized with {NUM_REPLICAS} replicas\")\nexcept Exception as e:\n    print(f\"TPU error: {e}\")\n    strategy = tf.distribute.get_strategy()\n    NUM_REPLICAS = 1\n    print(\"⚠️  Falling back to CPU/GPU\")\n\nOUTPUT_DIR = '/kaggle/working'  # Adjust as needed","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CSVBasedImageClassifier:\n    def __init__(self, data_dir, csv_path, img_size=(224, 224), max_samples_per_class=None):\n        \"\"\"\n        Args:\n            data_dir: Directory containing all training images\n            csv_path: Path to training.csv with columns (ID, path, y)\n            img_size: Target image size (height, width)\n            max_samples_per_class: Cap samples per class (None = no cap)\n        \"\"\"\n        self.data_dir = data_dir\n        self.csv_path = csv_path\n        self.img_size = img_size\n        self.max_samples_per_class = max_samples_per_class\n        self.class_names = [str(i) for i in range(10)]  # 0-9 digits\n        self.model = None\n        self.base_model = None\n        self.history = None\n        self.class_weights = None\n        \n        # Data containers\n        self.train_paths = None\n        self.val_paths = None\n        self.test_paths = None\n        self.train_labels = None\n        self.val_labels = None\n        self.test_labels = None\n\n    def load_and_split_data(self, test_size=0.15, val_size=0.15, random_state=42):\n        \"\"\"Load image paths and labels from CSV file.\"\"\"\n        \n        # Read CSV file\n        df = pd.read_csv(self.csv_path)\n        print(f\"📄 Loaded CSV: {len(df)} entries\")\n        print(f\"   Columns: {df.columns.tolist()}\")\n        \n        # Verify required columns\n        assert 'path' in df.columns, \"CSV must contain 'path' column\"\n        assert 'y' in df.columns, \"CSV must contain 'y' column\"\n        \n        # Convert y to integer\n        df['y'] = df['y'].astype(int)\n        \n        # Filter valid image files\n        valid_extensions = ('.png', '.jpg', '.jpeg', '.bmp', '.tiff')\n        valid_mask = df['path'].apply(\n            lambda x: str(x).lower().endswith(valid_extensions)\n        )\n        df = df[valid_mask].reset_index(drop=True)\n        print(f\"   Valid images: {len(df)}\")\n        \n        # Apply class cap if specified\n        if self.max_samples_per_class is not None:\n            capped_dfs = []\n            for class_id in range(10):\n                class_df = df[df['y'] == class_id]\n                if len(class_df) > self.max_samples_per_class:\n                    class_df = class_df.sample(\n                        n=self.max_samples_per_class, \n                        random_state=random_state\n                    )\n                capped_dfs.append(class_df)\n            df = pd.concat(capped_dfs, ignore_index=True)\n            print(f\"   After capping: {len(df)} images\")\n        \n        # === FIX: Build full paths correctly ===\n        # Check if paths in CSV already contain the data_dir prefix\n        def build_full_path(path):\n            clean_path = str(path)\n            for prefix in ('Data/Training/', 'Data/'):\n                if clean_path.startswith(prefix):\n                    clean_path = clean_path[len(prefix):]\n                    break\n            return os.path.join(self.data_dir, clean_path)\n\n        \n        all_paths = [build_full_path(p) for p in df['path']]\n        all_labels = df['y'].values\n        \n        # === ADD: Validate paths exist ===\n        missing_count = 0\n        for path in all_paths[:10]:  # Check first few\n            if not os.path.exists(path):\n                missing_count += 1\n                print(f\"   Warning: Missing file example: {path}\")\n        \n        if missing_count > 0:\n            print(f\"   Found missing files. Check your data_dir and CSV paths.\")\n            print(f\"   data_dir = {self.data_dir}\")\n            print(f\"   Sample CSV path: {df['path'].iloc[0]}\")\n            print(f\"   Built full path: {all_paths[0]}\")\n        \n        # Class distribution\n        print(\"\\n📊 Class Distribution:\")\n        for class_id in range(10):\n            count = np.sum(all_labels == class_id)\n            if count > 0:\n                print(f\"  Class {class_id}: {count} images\")\n        \n        if len(all_paths) == 0:\n            raise ValueError(\"No images found — check DATA_DIR and CSV paths.\")\n        \n        # Train / Test split\n        train_val_paths, test_paths, train_val_labels, test_labels = train_test_split(\n            all_paths, all_labels,\n            test_size=test_size,\n            stratify=all_labels,\n            random_state=random_state\n        )\n        \n        # Train / Validation split\n        val_ratio = val_size / (1 - test_size)\n        train_paths, val_paths, train_labels, val_labels = train_test_split(\n            train_val_paths, train_val_labels,\n            test_size=val_ratio,\n            stratify=train_val_labels,\n            random_state=random_state\n        )\n        \n        print(f\"\\n📦 Data Split:\")\n        print(f\"  Train:      {len(train_paths)} images\")\n        print(f\"  Validation: {len(val_paths)} images\")\n        print(f\"  Test:       {len(test_paths)} images\")\n        \n        self.train_paths = train_paths\n        self.val_paths = val_paths\n        self.test_paths = test_paths\n        self.train_labels = train_labels\n        self.val_labels = val_labels\n        self.test_labels = test_labels\n        \n        # Class weights (capped at 5.0)\n        from sklearn.utils.class_weight import compute_class_weight\n        raw_weights = compute_class_weight(\n            'balanced',\n            classes=np.unique(train_labels),\n            y=train_labels\n        )\n        capped_weights = np.minimum(raw_weights, 5.0)\n        self.class_weights = dict(enumerate(capped_weights))\n        print(f\"\\n⚖️  Class Weights: {self.class_weights}\")\n        \n        return train_paths, val_paths, test_paths, train_labels, val_labels, test_labels\n\n    # Rest of your methods remain the same...\n    def _labels_to_sample_weights(self, labels):\n        \"\"\"Convert integer class labels to per-sample weights.\"\"\"\n        return np.array([self.class_weights[l] for l in labels], dtype=np.float32)\n\n\n\n    \n    def create_data_generators(self, batch_size=128):\n        import cv2\n        num_classes = 10\n    \n        def load_images_to_numpy(paths, desc=\"Loading\"):\n            imgs = []\n            for p in paths:\n                img = cv2.imread(p)\n                img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n                img = cv2.resize(img, self.img_size)\n                img = tf.keras.applications.mobilenet_v2.preprocess_input(\n                    img.astype(np.float32)\n                )\n                imgs.append(img)\n            print(f\"  {desc}: {len(imgs)} images loaded\")\n            return np.array(imgs, dtype=np.float32)\n    \n        print(\"📥 Loading images into memory...\")\n        train_imgs = load_images_to_numpy(self.train_paths, \"Train\")\n        val_imgs   = load_images_to_numpy(self.val_paths,   \"Val\")\n        test_imgs  = load_images_to_numpy(self.test_paths,  \"Test\")\n    \n        train_labels_oh = tf.keras.utils.to_categorical(self.train_labels, num_classes)\n        val_labels_oh   = tf.keras.utils.to_categorical(self.val_labels,   num_classes)\n        test_labels_oh  = tf.keras.utils.to_categorical(self.test_labels,  num_classes)\n    \n        train_weights = self._labels_to_sample_weights(self.train_labels)\n        val_weights   = np.ones(len(self.val_labels),   dtype=np.float32)\n        test_weights  = np.ones(len(self.test_labels),  dtype=np.float32)\n    \n        options = tf.data.Options()\n        options.experimental_distribute.auto_shard_policy = (\n            tf.data.experimental.AutoShardPolicy.DATA\n        )\n    \n        def augment(image, label, weight):\n            image = tf.image.random_flip_left_right(image)\n            image = tf.image.random_brightness(image, 0.2)\n            image = tf.image.random_contrast(image, 0.8, 1.2)\n            image = tf.image.random_saturation(image, 0.8, 1.2)\n            return image, label, weight\n    \n        train_ds = (tf.data.Dataset.from_tensor_slices(\n                        (train_imgs, train_labels_oh, train_weights))\n                    .shuffle(2000, reshuffle_each_iteration=True)\n                    .map(augment, num_parallel_calls=tf.data.AUTOTUNE)\n                    .batch(batch_size, drop_remainder=True)\n                    .prefetch(tf.data.AUTOTUNE)\n                    .with_options(options))\n    \n        val_ds = (tf.data.Dataset.from_tensor_slices(\n                      (val_imgs, val_labels_oh, val_weights))\n                  .batch(batch_size, drop_remainder=True)\n                  .prefetch(tf.data.AUTOTUNE)\n                  .with_options(options))\n    \n        test_ds = (tf.data.Dataset.from_tensor_slices(\n                       (test_imgs, test_labels_oh, test_weights))\n                   .batch(batch_size, drop_remainder=False)\n                   .prefetch(tf.data.AUTOTUNE)\n                   .with_options(options))\n\n        train_ds = strategy.experimental_distribute_dataset(train_ds)\n        val_ds   = strategy.experimental_distribute_dataset(val_ds)\n    \n        self.train_generator = train_ds\n        self.val_generator   = val_ds\n        self.test_generator  = test_ds\n    \n        return train_ds, val_ds, test_ds\n        \n\n    # Keep the rest of your methods (build_model, train, predict_test_images, \n    # evaluate_and_report, plot_confusion_matrix, plot_training_history) exactly as they are\n\n    def build_model(self, num_classes=10, dropout_rate=0.3, learning_rate=1e-4):\n        \"\"\"Build MobileNetV2-based classifier inside the TPU/GPU strategy scope.\"\"\"\n        with strategy.scope():\n            self.base_model = MobileNetV2(\n                input_shape=(*self.img_size, 3),\n                include_top=False,\n                weights='imagenet'\n            )\n            self.base_model.trainable = False  # Freeze base initially\n    \n            inputs = keras.Input(shape=(*self.img_size, 3))\n            x = self.base_model(inputs, training=False)\n            x = layers.GlobalAveragePooling2D()(x)\n            x = layers.BatchNormalization()(x)\n            x = layers.Dropout(dropout_rate)(x)\n            x = layers.Dense(256, activation='relu')(x)\n            x = layers.Dropout(dropout_rate)(x)\n            outputs = layers.Dense(num_classes, activation='softmax')(x)\n    \n            self.model = keras.Model(inputs, outputs)\n            self.model.compile(\n                optimizer=optimizers.Adam(learning_rate),\n                loss='categorical_crossentropy',\n                metrics=['accuracy']\n            )\n        print(self.model.summary())\n        return self.model\n\n    \n    def train(self, epochs=10, batch_size=32 * NUM_REPLICAS, learning_rate=1e-4,\n              fine_tune_epochs=100, fine_tune_at=100):\n        \"\"\"\n        Two-phase training:\n          Phase 1 — train head with frozen base.\n          Phase 2 — unfreeze top layers of base for fine-tuning.\n        \"\"\"\n        # Load data if not already done\n        if self.train_paths is None:\n            self.load_and_split_data()\n    \n        train_ds, val_ds, _ = self.create_data_generators(batch_size)\n        self.build_model(learning_rate=learning_rate)\n    \n        callbacks = [\n            keras.callbacks.EarlyStopping(\n                monitor='val_accuracy', patience=100,\n                restore_best_weights=True, verbose=1\n            ),\n            keras.callbacks.ReduceLROnPlateau(\n                monitor='val_loss', factor=0.5,\n                patience=100, min_lr=1e-7, verbose=1\n            ),\n            keras.callbacks.ModelCheckpoint(\n                filepath=os.path.join(OUTPUT_DIR, 'best_model.keras'),\n                monitor='val_accuracy', save_best_only=True, verbose=1\n            )\n        ]\n    \n        print(\"\\n🚀 Phase 1: Training head (base frozen)...\")\n        history1 = self.model.fit(\n            train_ds,\n            validation_data=val_ds,\n            epochs=epochs,\n            callbacks=callbacks,\n        )\n    \n        # --- Phase 2: Fine-tune ---\n        print(f\"\\n🔓 Phase 2: Fine-tuning from layer {fine_tune_at} onward...\")\n        self.base_model.trainable = True\n        for layer in self.base_model.layers[:fine_tune_at]:\n            layer.trainable = False\n    \n        with strategy.scope():\n            self.model.compile(\n                optimizer=optimizers.Adam(learning_rate / 10),\n                loss='categorical_crossentropy',\n                metrics=['accuracy']\n            )\n    \n        history2 = self.model.fit(\n            train_ds,\n            validation_data=val_ds,\n            epochs=fine_tune_epochs,\n            callbacks=callbacks,\n        )\n    \n        # Merge histories\n        self.history = {}\n        for key in history1.history:\n            self.history[key] = history1.history[key] + history2.history.get(key, [])\n    \n        return self.history\n\n    \n    def predict_test_images(self, image_paths, batch_size=64):\n        import cv2\n        imgs = []\n        for p in image_paths:\n            img = cv2.imread(p)\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            img = cv2.resize(img, self.img_size)\n            img = tf.keras.applications.mobilenet_v2.preprocess_input(\n                img.astype(np.float32)\n            )\n            imgs.append(img)\n        imgs = np.array(imgs, dtype=np.float32)\n    \n        ds = (tf.data.Dataset.from_tensor_slices(imgs)\n              .batch(batch_size)\n              .prefetch(tf.data.AUTOTUNE))\n    \n        all_probs = self.model.predict(ds, verbose=1)\n        return np.argmax(all_probs, axis=1).tolist()\n\n\n    def evaluate_and_report(self):\n        if self.model is None:\n            raise RuntimeError(\"Model not trained yet.\")\n        if self.test_paths is None:\n            raise RuntimeError(\"No test split found.\")\n    \n        print(\"\\n📋 Evaluating on test split...\")\n        _, _, test_ds = self.create_data_generators()\n    \n        # Pass the entire test_ds to predict at once, rather than batch by batch\n        all_probs = self.model.predict(test_ds, verbose=1)\n        y_pred = np.argmax(all_probs, axis=1)\n    \n        # Extract true labels from test_ds\n        y_true = []\n        for _, labels_oh, _ in self.test_generator:\n            y_true.extend(np.argmax(labels_oh.numpy(), axis=1))\n        \n        # Align length to account for any dropped remainders\n        y_true = np.array(y_true[:len(y_pred)])  \n    \n        print(\"\\n📊 Classification Report:\")\n        print(classification_report(y_true, y_pred, target_names=self.class_names))\n        self.plot_confusion_matrix(y_true, y_pred)\n    \n    \n    def plot_confusion_matrix(self, y_true, y_pred):\n        \"\"\"Heatmap of the confusion matrix.\"\"\"\n        cm = confusion_matrix(y_true, y_pred)\n        plt.figure(figsize=(10, 8))\n        sns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n                    xticklabels=self.class_names,\n                    yticklabels=self.class_names)\n        plt.title('Confusion Matrix')\n        plt.ylabel('True Label')\n        plt.xlabel('Predicted Label')\n        plt.tight_layout()\n        plt.savefig(os.path.join(OUTPUT_DIR, 'confusion_matrix.png'), dpi=150)\n        plt.show()\n    \n    def plot_training_history(self):\n        \"\"\"Plot accuracy and loss curves across both training phases.\"\"\"\n        if self.history is None:\n            raise RuntimeError(\"No training history. Call train() first.\")\n    \n        fig, axes = plt.subplots(1, 2, figsize=(14, 5))\n    \n        # Accuracy\n        axes[0].plot(self.history['accuracy'],     label='Train Accuracy')\n        axes[0].plot(self.history['val_accuracy'], label='Val Accuracy')\n        axes[0].set_title('Model Accuracy')\n        axes[0].set_xlabel('Epoch')\n        axes[0].set_ylabel('Accuracy')\n        axes[0].legend()\n        axes[0].grid(True)\n    \n        # Loss\n        axes[1].plot(self.history['loss'],     label='Train Loss')\n        axes[1].plot(self.history['val_loss'], label='Val Loss')\n        axes[1].set_title('Model Loss')\n        axes[1].set_xlabel('Epoch')\n        axes[1].set_ylabel('Loss')\n        axes[1].legend()\n        axes[1].grid(True)\n    \n        plt.tight_layout()\n        plt.savefig(os.path.join(OUTPUT_DIR, 'training_history.png'), dpi=150)\n        plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================================\n# USAGE EXAMPLE\n# ============================================================================\n\ndef main():\n    # Configuration\n    DATA_DIR = \"/kaggle/input/competitions/dlmmdd-workshop-synthetic-source-attribution-challenge/Data/Data/Training\"  # Directory with all training images\n    CSV_PATH = \"/kaggle/input/competitions/dlmmdd-workshop-synthetic-source-attribution-challenge/Data/Data/training.csv\"  # CSV with columns: ID, path, y\n    TEST_DIR = \"/kaggle/input/competitions/dlmmdd-workshop-synthetic-source-attribution-challenge/Data/Data/Test\"    # Directory with test images\n    \n    # Initialize classifier\n    classifier = CSVBasedImageClassifier(\n        data_dir=DATA_DIR,\n        csv_path=CSV_PATH,\n        img_size=(224, 224),\n        max_samples_per_class=None  # Optional: cap samples per class\n    )\n    \n    # Train the model\n    classifier.train(epochs=400)\n    \n    # Evaluate on validation/test set\n    classifier.evaluate_and_report()\n    \n    # Plot training history\n    classifier.plot_training_history()\n    \n    # Predict on test images\n    test_image_paths = [os.path.join(TEST_DIR, f) for f in os.listdir(TEST_DIR) if f.endswith(('.png', '.jpg'))]\n    predictions = classifier.predict_test_images(test_image_paths)\n    \n    # Save predictions to CSV\n    results_df = pd.DataFrame({\n        'ID': [Path(p).stem for p in test_image_paths],\n        'TARGET': predictions\n        })\n    results_df.to_csv('submission.csv', index=False)\n\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.024497,"end_time":"2026-04-11T02:47:24.383366","exception":false,"start_time":"2026-04-11T02:47:24.358869","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.02612,"end_time":"2026-04-11T02:47:24.435232","exception":false,"start_time":"2026-04-11T02:47:24.409112","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}