{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":12943083,"sourceType":"datasetVersion","datasetId":8190726},{"sourceId":13061386,"sourceType":"datasetVersion","datasetId":8271157},{"sourceId":13077947,"sourceType":"datasetVersion","datasetId":8282818},{"sourceId":13079217,"sourceType":"datasetVersion","datasetId":8283702},{"sourceId":13080218,"sourceType":"datasetVersion","datasetId":8284422},{"sourceId":13080319,"sourceType":"datasetVersion","datasetId":8284500},{"sourceId":13061722,"sourceType":"datasetVersion","datasetId":8271392}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ====================================================================\n# SCRIPT 2: ADVANCED MASTER SCRIPT FOR MOBILEVIT TRAINING\n# VERSION 6.0: With F1-Score Based Adaptive Augmentation\n# ====================================================================\n# This is the most advanced version of the training script. It uses the\n# performance results (F1 scores) from a previous training run to\n# intelligently apply targeted data augmentation.\n# - LOW F1 classes get AGGRESSIVE augmentation.\n# - HIGH F1 classes get LIGHT augmentation.\n# This focuses the training effort where it is most needed.\n\n# --- Block 2: Imports, Configuration, and Device Setup ---\nimport os\nimport numpy as np\nimport torch\nimport random\nimport pandas as pd\nfrom collections import Counter\nfrom torch.utils.data import Dataset\nfrom transformers import AutoImageProcessor, AutoModelForImageClassification, TrainingArguments, Trainer, EarlyStoppingCallback\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, f1_score\nfrom sklearn.utils import class_weight\nfrom PIL import Image\nfrom tqdm import tqdm\n\nprint(\"\\nStarting the Advanced MobileViT experiment setup...\")\n\n# --- Main Configuration ---\nMODEL_ID = \"apple/mobilevit-small\"\nPROCESSED_DATA_PATH = \"/kaggle/input/birdsong-processed2/02_processed_spectrograms\"\nSAVED_MODELS_PATH = \"/kaggle/working/04_saved_models/\"\nOUTPUT_DIR = \"/kaggle/working/mobilevit_checkpoints/\"\nNUM_EPOCHS = 60\n\n# <<< NEW: Path to your performance report from the previous run >>>\n# This file is used to guide the adaptive augmentation.\nPERFORMANCE_REPORT_PATH = \"/kaggle/input/performance-report/performance_report.md\" \n\n# --- Device Setup ---\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"✅ Using device: {device}\")\n\n# --- Performance Tuning - Enable TF32 for faster matrix math ---\nif torch.cuda.is_available() and torch.cuda.get_device_capability()[0] >= 8:\n    print(\"NVIDIA Ampere GPU detected. Enabling TF32 for performance boost.\")\n    torch.backends.cuda.matmul.allow_tf32 = True\n\n# ====================================================================\n\ndef load_f1_scores(report_path, class_names):\n    \"\"\"\n    Parses the Markdown performance report to extract F1 scores for each class.\n    \"\"\"\n    if not os.path.exists(report_path):\n        print(f\"Warning: Performance report not found at '{report_path}'. Using standard augmentation for all classes.\")\n        return {name: 0.6 for name in class_names} # Default to standard augmentation\n\n    try:\n        with open(report_path, 'r') as f:\n            lines = f.readlines()\n        \n        table_lines = []\n        in_table = False\n        for line in lines:\n            if \"| Class Name\" in line:\n                in_table = True\n                continue\n            if in_table and \"| :\" in line:\n                continue\n            if in_table and line.strip() == \"---\":\n                break\n            if in_table:\n                table_lines.append(line)\n\n        # Use pandas to easily read the markdown table format\n        from io import StringIO\n        table_str = \"\".join(table_lines)\n        df = pd.read_csv(StringIO(table_str), sep='|', skipinitialspace=True)\n        # Clean up dataframe columns and trailing spaces\n        df = df.dropna(axis=1, how='all').iloc[:, 1:-1]\n        df.columns = [col.strip() for col in df.columns]\n        for col in df.columns:\n            if df[col].dtype == 'object':\n                df[col] = df[col].str.strip()\n\n        f1_scores = pd.to_numeric(df['F1-Score'], errors='coerce').fillna(0.5)\n        \n        f1_dict = dict(zip(df['Class Name'], f1_scores))\n        print(\"✅ Successfully loaded F1 scores to guide augmentation.\")\n        return f1_dict\n    except Exception as e:\n        print(f\"Warning: Could not parse performance report. Using standard augmentation. Error: {e}\")\n        return {name: 0.6 for name in class_names}\n\n\n# --- Block 3: Load File Paths, Calculate Weights, and Split ---\nprint(f\"\\nLoading and sampling .npy file paths from local disk: {PROCESSED_DATA_PATH}\")\nall_filepaths = []\nall_labels = []\n\nif os.path.exists(PROCESSED_DATA_PATH):\n    class_names = sorted([d for d in os.listdir(PROCESSED_DATA_PATH) if os.path.isdir(os.path.join(PROCESSED_DATA_PATH, d))])\n    print(f\"Found {len(class_names)} classes.\")\n    class_to_idx = {name: i for i, name in enumerate(class_names)}\n    idx_to_class = {i: name for name, i in class_to_idx.items()}\n\n    # <<< NEW: Load F1 scores based on the class names found >>>\n    f1_scores_by_name = load_f1_scores(PERFORMANCE_REPORT_PATH, class_names)\n    f1_scores_by_idx = {class_to_idx[name]: score for name, score in f1_scores_by_name.items()}\n\n    print(\"\\nLoading all files from each class (no limit)...\")\n    for class_name in class_names:\n        class_path = os.path.join(PROCESSED_DATA_PATH, class_name)\n        label_idx = class_to_idx[class_name]\n        \n        class_npy_files = [f for f in os.listdir(class_path) if f.endswith(\".npy\")]\n        \n        for filename in class_npy_files:\n            all_filepaths.append(os.path.join(class_path, filename))\n            all_labels.append(label_idx)\n\n    print(f\"\\nTotal files found for training/validation: {len(all_filepaths)}\")\n\n    print(\"\\nCalculating class weights...\")\n    class_weights_array = class_weight.compute_class_weight('balanced', classes=np.unique(all_labels), y=np.array(all_labels))\n    class_weights_tensor = torch.tensor(class_weights_array, dtype=torch.float).to(device)\n    print(\"Calculated Class Weights and moved to the active device.\")\n\n    train_paths, val_paths, train_labels, val_labels = train_test_split(\n        all_filepaths, all_labels, test_size=0.2, random_state=42, stratify=all_labels\n    )\n    print(f\"\\nSplit into {len(train_paths)} training and {len(val_paths)} validation samples.\")\n\n    # --- Block 4: Create a Smart PyTorch Dataset with F1-Adaptive Augmentation ---\n    print(\"\\nSetting up the custom PyTorch Dataset with F1-ADAPTIVE data augmentation...\")\n    image_processor = AutoImageProcessor.from_pretrained(MODEL_ID)\n\n    class SpectrogramDataset(Dataset):\n        def __init__(self, filepaths, labels, f1_scores, is_train=False):\n            self.filepaths = filepaths\n            self.labels = labels\n            self.f1_scores = f1_scores\n            self.is_train = is_train\n\n        def __len__(self):\n            return len(self.filepaths)\n        \n        def _apply_spec_augment(self, spectrogram, time_mask_param, freq_mask_param):\n            aug_spec = spectrogram.copy()\n            num_mels, num_timesteps = aug_spec.shape\n            if num_mels > freq_mask_param:\n                freq_mask_start = int(np.random.uniform(0, num_mels - freq_mask_param))\n                aug_spec[freq_mask_start:freq_mask_start + freq_mask_param, :] = 0\n            if num_timesteps > time_mask_param:\n                time_mask_start = int(np.random.uniform(0, num_timesteps - time_mask_param))\n                aug_spec[:, time_mask_start:time_mask_start + time_mask_param] = 0\n            return aug_spec\n\n        def __getitem__(self, idx):\n            file_path = self.filepaths[idx]\n            label = self.labels[idx]\n            \n            try:\n                spectrogram = np.load(file_path)\n\n                # <<< NEW: F1-Score Based Adaptive Augmentation Logic >>>\n                if self.is_train:\n                    f1_score = self.f1_scores.get(label, 0.6) # Default to standard if score not found\n                    if f1_score < 0.5: # LOW F1: Aggressive Augmentation\n                        spectrogram = self._apply_spec_augment(spectrogram, 80, 50)\n                    elif f1_score < 0.8: # MEDIUM F1: Standard Augmentation\n                        spectrogram = self._apply_spec_augment(spectrogram, 40, 30)\n                    else: # HIGH F1: Light Augmentation\n                        spectrogram = self._apply_spec_augment(spectrogram, 20, 15)\n\n                image = np.stack([spectrogram]*3, axis=-1)\n                \n                if (image.max() - image.min()) > 0:\n                    image = (image - image.min()) / (image.max() - image.min()) * 255\n                \n                image = Image.fromarray(image.astype(np.uint8))\n                processed_image = image_processor(images=image, return_tensors=\"pt\")\n                return {\"pixel_values\": processed_image['pixel_values'].squeeze(0), \"labels\": torch.tensor(label, dtype=torch.long)}\n            except Exception as e:\n                tqdm.write(f\"\\nWarning: Error processing {os.path.basename(file_path)}. Skipping. Error: {e}\")\n                return self.__getitem__((idx + 1) % len(self))\n\n    # Pass the f1_scores dictionary to the training dataset\n    train_dataset = SpectrogramDataset(train_paths, train_labels, f1_scores_by_idx, is_train=True)\n    val_dataset = SpectrogramDataset(val_paths, val_labels, f1_scores_by_idx, is_train=False) # No augmentation for validation\n    print(\"✅ Custom datasets created successfully (training set with F1-adaptive augmentation).\")\n\n    # --- Block 5: Model Building and Advanced Training ---\n    print(\"\\nConfiguring and training the MobileViT model...\")\n    label2id = {name: i for i, name in enumerate(class_names)}\n    id2label = {i: name for i, name in enumerate(class_names)}\n    model = AutoModelForImageClassification.from_pretrained(\n        MODEL_ID, num_labels=len(class_names), label2id=label2id, id2label=id2label, ignore_mismatched_sizes=True\n    )\n\n    def compute_metrics(eval_pred):\n        predictions = np.argmax(eval_pred.predictions, axis=1)\n        return {\"accuracy\": accuracy_score(eval_pred.label_ids, predictions), \"f1\": f1_score(eval_pred.label_ids, predictions, average=\"weighted\")}\n\n    class WeightedTrainer(Trainer):\n        def compute_loss(self, model, inputs, return_outputs=False, **kwargs):\n            labels = inputs.get(\"labels\")\n            outputs = model(**inputs)\n            logits = outputs.get(\"logits\")\n            loss_fct = torch.nn.CrossEntropyLoss(weight=class_weights_tensor)\n            loss = loss_fct(logits.view(-1, self.model.config.num_labels), labels.view(-1))\n            return (loss, outputs) if return_outputs else loss\n\n    early_stopping = EarlyStoppingCallback(early_stopping_patience=5)\n\n    training_args = TrainingArguments(\n        output_dir=OUTPUT_DIR,\n        eval_strategy=\"epoch\",\n        save_strategy=\"epoch\",\n        per_device_train_batch_size=16,\n        gradient_accumulation_steps=1,\n        fp16=True,\n        dataloader_num_workers=8,\n        learning_rate=5e-5,\n        lr_scheduler_type='cosine',\n        warmup_ratio=0.1,\n        weight_decay=0.01,\n        num_train_epochs=NUM_EPOCHS,\n        logging_steps=50,\n        load_best_model_at_end=True,\n        metric_for_best_model=\"f1\",\n        report_to=\"tensorboard\",\n    )\n\n    trainer = WeightedTrainer(\n        model=model, args=training_args,\n        train_dataset=train_dataset, eval_dataset=val_dataset,\n        compute_metrics=compute_metrics,\n        callbacks=[early_stopping],\n    )\n\n    trainer.train()\n\n    # --- Block 6: Save the Final Model ---\n    print(\"\\nTraining complete. Saving final model...\")\n    os.makedirs(SAVED_MODELS_PATH, exist_ok=True)\n    final_model_path = os.path.join(SAVED_MODELS_PATH, \"mobilevit_final_model_gpu_weighted\")\n    trainer.save_model(final_model_path)\n    print(f\"✅ MobileViT model successfully saved to: '{final_model_path}'\")\nelse:\n    print(f\"❌ ERROR: Dataset not found at '{PROCESSED_DATA_PATH}'. Please check your path.\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================================\n# SCRIPT 7: CONFUSION MATRIX TABLE GENERATOR\n# VERSION 2.1: With corrected import\n# ====================================================================\n# This script loads a trained MobileViT model and its validation data\n# to generate a detailed confusion matrix in a table format (CSV).\n# This is ideal for analyzing performance on models with many classes.\n\nimport os\nimport numpy as np\nimport torch\nimport pandas as pd\nfrom torch.utils.data import DataLoader, Dataset # <<< THE FIX is here\nfrom transformers import AutoImageProcessor, AutoModelForImageClassification\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix\nfrom PIL import Image\nfrom tqdm import tqdm\n\n# --- Configuration ---\n\n# 1. Path to the folder containing your final, saved model.\nMODEL_PATH = \"/kaggle/input/saved-model2\"\n\n# 2. The original Hugging Face model ID for the image processor.\nORIGINAL_MODEL_ID = \"apple/mobilevit-small\"\n\n# 3. Path to your PROCESSED SPECTROGRAMS.\nPROCESSED_DATA_PATH = \"/kaggle/input/birdsong-processed2/02_processed_spectrograms\"\n\n# 4. The filename for the output CSV table.\nOUTPUT_CSV_NAME = \"confusion_matrix_table.csv\"\n\n# --- Analysis Parameters ---\nINFERENCE_BATCH_SIZE = 32\n\n# ====================================================================\n\ndef get_validation_data(data_path):\n    \"\"\"\n    Loads all file paths and labels and recreates the exact same\n    training/validation split used during training.\n    \"\"\"\n    print(f\"Loading file paths from '{data_path}'...\")\n    all_filepaths, all_labels = [], []\n    \n    if not os.path.isdir(data_path):\n        print(f\"❌ ERROR: Data path not found: {data_path}\")\n        return None, None, None\n\n    class_names = sorted([d for d in os.listdir(data_path) if os.path.isdir(os.path.join(data_path, d))])\n    class_to_idx = {name: i for i, name in enumerate(class_names)}\n\n    for class_name in class_names:\n        class_path = os.path.join(data_path, class_name)\n        label_idx = class_to_idx[class_name]\n        for filename in os.listdir(class_path):\n            if filename.endswith(\".npy\"):\n                all_filepaths.append(os.path.join(class_path, filename))\n                all_labels.append(label_idx)\n\n    _, val_paths, _, val_labels = train_test_split(\n        all_filepaths, all_labels, test_size=0.2, random_state=42, stratify=all_labels\n    )\n    \n    print(f\"Successfully recreated validation split with {len(val_paths)} samples.\")\n    return val_paths, val_labels, class_names\n\nclass InferenceSpectrogramDataset(Dataset):\n    \"\"\"Dataset for loading spectrograms for inference (no augmentation).\"\"\"\n    def __init__(self, filepaths, labels, processor):\n        self.filepaths = filepaths\n        self.labels = labels\n        self.processor = processor\n\n    def __len__(self):\n        return len(self.filepaths)\n\n    def __getitem__(self, idx):\n        file_path = self.filepaths[idx]\n        label = self.labels[idx]\n        try:\n            spectrogram = np.load(file_path)\n            image = np.stack([spectrogram] * 3, axis=-1)\n            if (image.max() - image.min()) > 0:\n                image = (image - image.min()) / (image.max() - image.min()) * 255\n            image = Image.fromarray(image.astype(np.uint8))\n            processed_image = self.processor(images=image, return_tensors=\"pt\")\n            return {\"pixel_values\": processed_image['pixel_values'].squeeze(0), \"labels\": torch.tensor(label, dtype=torch.long)}\n        except Exception:\n            return self.__getitem__((idx + 1) % len(self))\n\ndef generate_predictions(model, dataloader, device):\n    \"\"\"Runs the model on the validation dataloader to get all predictions.\"\"\"\n    model.eval()\n    all_preds, all_true_labels = [], []\n    \n    with torch.no_grad():\n        for batch in tqdm(dataloader, desc=\"Generating Predictions\"):\n            pixel_values = batch['pixel_values'].to(device)\n            labels = batch['labels'].to(device)\n            \n            outputs = model(pixel_values)\n            logits = outputs.logits\n            preds = torch.argmax(logits, dim=-1)\n            \n            all_preds.extend(preds.cpu().numpy())\n            all_true_labels.extend(labels.cpu().numpy())\n            \n    return all_true_labels, all_preds\n\ndef save_confusion_matrix_table(true_labels, preds, class_names, output_filename):\n    \"\"\"\n    Computes the confusion matrix and saves it as a human-readable CSV file.\n    \"\"\"\n    print(\"\\nGenerating confusion matrix table...\")\n    cm = confusion_matrix(true_labels, preds)\n    \n    # Create a Pandas DataFrame from the NumPy matrix\n    # Label the rows and columns with the actual class names\n    cm_df = pd.DataFrame(cm, index=class_names, columns=class_names)\n    \n    # Give a name to the index column for clarity in the CSV\n    cm_df.index.name = 'True Label'\n    \n    try:\n        cm_df.to_csv(output_filename)\n        print(f\"✅ Confusion matrix table saved successfully to '{output_filename}'\")\n        print(\"You can now download this file and open it in a spreadsheet program.\")\n    except Exception as e:\n        print(f\"❌ ERROR: Failed to save CSV file. Reason: {e}\")\n\ndef main():\n    \"\"\"Main execution function.\"\"\"\n    print(\"--- Starting Confusion Matrix Generation (Table Format) ---\")\n    \n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    val_paths, val_labels, class_names = get_validation_data(PROCESSED_DATA_PATH)\n    \n    if val_paths is None:\n        return\n\n    print(f\"\\nLoading trained model from '{MODEL_PATH}'...\")\n    try:\n        processor = AutoImageProcessor.from_pretrained(ORIGINAL_MODEL_ID)\n        model = AutoModelForImageClassification.from_pretrained(MODEL_PATH).to(device)\n        print(\"✅ Model and processor loaded successfully.\")\n    except Exception as e:\n        print(f\"❌ ERROR: Failed to load the model. Please check the path. Reason: {e}\")\n        return\n        \n    val_dataset = InferenceSpectrogramDataset(val_paths, val_labels, processor)\n    val_dataloader = DataLoader(val_dataset, batch_size=INFERENCE_BATCH_SIZE, shuffle=False)\n\n    true_labels, preds = generate_predictions(model, val_dataloader, device)\n    save_confusion_matrix_table(true_labels, preds, class_names, OUTPUT_CSV_NAME)\n    \nif __name__ == '__main__':\n    main()\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================================\n# SCRIPT 9: PREDICTION SCRIPT FOR BIRDSONG CLASSIFICATION\n# ====================================================================\n# This script loads a trained MobileViT model and uses it to predict\n# the species of bird in a single audio file.\n\nimport os\nimport numpy as np\nimport torch\nimport librosa\nfrom transformers import AutoImageProcessor, AutoModelForImageClassification\nfrom PIL import Image\nfrom collections import Counter\n\n# --- Configuration ---\n\n# 1. Path to the folder containing your final, saved model.\nMODEL_PATH = \"/kaggle/input/saved-model2\"\n\n# 2. The original Hugging Face model ID for the image processor.\nORIGINAL_MODEL_ID = \"apple/mobilevit-small\"\n\n# 3. Path to the audio file you want to make a prediction on.\nAUDIO_FILE_TO_PREDICT = \"/kaggle/input/rwbsample3/rwb09.wav\"\n\n# --- Preprocessing Parameters (MUST match your training script) ---\nSAMPLE_RATE = 32000\nCHUNK_DURATION_S = 5\nN_MELS = 224\nN_FFT = 2048\nHOP_LENGTH = 512\n\n# ====================================================================\n\ndef preprocess_audio_for_prediction(file_path):\n    \"\"\"\n    Loads and processes a single audio file into a list of spectrograms,\n    matching the format used for training.\n    \"\"\"\n    try:\n        y, sr = librosa.load(file_path, sr=SAMPLE_RATE)\n        chunk_samples = int(CHUNK_DURATION_S * sr)\n        \n        # Create chunks (padding the last one if necessary)\n        num_chunks = int(np.ceil(len(y) / chunk_samples))\n        chunks = []\n        for i in range(num_chunks):\n            start = i * chunk_samples\n            end = start + chunk_samples\n            chunk = y[start:end]\n            if len(chunk) < chunk_samples:\n                chunk = np.pad(chunk, (0, chunk_samples - len(chunk)), 'constant')\n            chunks.append(chunk)\n\n        # Convert each chunk to a log-Mel spectrogram\n        spectrograms = []\n        for chunk in chunks:\n            mel = librosa.feature.melspectrogram(y=chunk, sr=sr, n_fft=N_FFT, hop_length=HOP_LENGTH, n_mels=N_MELS)\n            log_mel = librosa.power_to_db(mel, ref=np.max)\n            spectrograms.append(log_mel)\n            \n        return spectrograms\n\n    except Exception as e:\n        print(f\"❌ ERROR: Failed to process audio file '{file_path}'. Reason: {e}\")\n        return None\n\ndef predict_bird_in_audio():\n    \"\"\"\n    Main function to load the model, process the audio, and make a prediction.\n    \"\"\"\n    print(\"--- Birdsong Prediction ---\")\n    \n    # --- Step 1: Validate paths ---\n    if not os.path.isdir(MODEL_PATH):\n        print(f\"❌ ERROR: Model folder not found at '{MODEL_PATH}'\")\n        return\n    if not os.path.isfile(AUDIO_FILE_TO_PREDICT):\n        print(f\"❌ ERROR: Audio file not found at '{AUDIO_FILE_TO_PREDICT}'\")\n        return\n\n    # --- Step 2: Load Model and Processor ---\n    print(f\"Loading trained model from '{MODEL_PATH}'...\")\n    try:\n        device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n        processor = AutoImageProcessor.from_pretrained(ORIGINAL_MODEL_ID)\n        model = AutoModelForImageClassification.from_pretrained(MODEL_PATH).to(device)\n        model.eval() # Set the model to evaluation mode\n        print(f\"✅ Model loaded successfully on device: {device}\")\n    except Exception as e:\n        print(f\"❌ ERROR: Failed to load the model. Reason: {e}\")\n        return\n\n    # --- Step 3: Preprocess the Input Audio ---\n    print(f\"\\nProcessing audio file: '{os.path.basename(AUDIO_FILE_TO_PREDICT)}'...\")\n    spectrograms = preprocess_audio_for_prediction(AUDIO_FILE_TO_PREDICT)\n    if spectrograms is None:\n        return\n    print(f\"Audio file was split into {len(spectrograms)} chunk(s) for analysis.\")\n\n    # --- Step 4: Run Inference on Each Chunk ---\n    chunk_predictions = []\n    chunk_confidences = []\n\n    with torch.no_grad():\n        for spec in spectrograms:\n            # Prepare image for the model\n            image = np.stack([spec] * 3, axis=-1)\n            if (image.max() - image.min()) > 0:\n                image = (image - image.min()) / (image.max() - image.min()) * 255\n            image = Image.fromarray(image.astype(np.uint8))\n            \n            inputs = processor(images=image, return_tensors=\"pt\").to(device)\n            \n            # Get model output\n            outputs = model(**inputs)\n            logits = outputs.logits\n            \n            # Get probabilities and top prediction\n            probabilities = torch.nn.functional.softmax(logits, dim=-1)\n            confidence, predicted_class_idx = torch.max(probabilities, dim=-1)\n            \n            chunk_predictions.append(predicted_class_idx.item())\n            chunk_confidences.append(confidence.item())\n\n    # --- Step 5: Aggregate Results and Report Final Prediction ---\n    if not chunk_predictions:\n        print(\"\\nCould not make a prediction. No valid chunks were analyzed.\")\n        return\n\n    # The final prediction is the most common prediction across all chunks (the mode)\n    final_prediction_idx = Counter(chunk_predictions).most_common(1)[0][0]\n    \n    # Get the class name from the model's config\n    final_class_name = model.config.id2label[final_prediction_idx]\n    \n    # Calculate the average confidence for the winning class\n    confidences_for_final_class = [conf for idx, conf in zip(chunk_predictions, chunk_confidences) if idx == final_prediction_idx]\n    avg_confidence = np.mean(confidences_for_final_class)\n\n    print(\"\\n--- Prediction Result ---\")\n    print(f\"Predicted Species: {final_class_name}\")\n    print(f\"Confidence: {avg_confidence:.2%}\")\n    print(\"-------------------------\")\n\nif __name__ == '__main__':\n    predict_bird_in_audio()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================================\n# SCRIPT 11: IN-NOTEBOOK PERFORMANCE REPORT GENERATOR\n# ====================================================================\n# This script loads a trained model, runs it on the validation set,\n# and generates a concise performance report directly in the notebook's\n# output. The report includes overall accuracy, F1-score, and a\n# summary of the most common errors from the confusion matrix.\n\nimport os\nimport numpy as np\nimport torch\nimport pandas as pd\nfrom torch.utils.data import Dataset, DataLoader\nfrom transformers import AutoImageProcessor, AutoModelForImageClassification\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, accuracy_score, f1_score\nfrom PIL import Image\nfrom tqdm import tqdm\n\n# --- Configuration ---\n\n# 1. Path to the folder containing your final, saved model.\nMODEL_PATH = \"/kaggle/input/saved-model2\"\n\n# 2. The original Hugging Face model ID for the image processor.\nORIGINAL_MODEL_ID = \"apple/mobilevit-small\"\n\n# 3. Path to your PROCESSED SPECTROGRAMS.\nPROCESSED_DATA_PATH = \"/kaggle/input/birdsong-processed2/02_processed_spectrograms\"\n\n# --- Analysis Parameters ---\nINFERENCE_BATCH_SIZE = 32\n\n# ====================================================================\n\ndef get_validation_data(data_path):\n    \"\"\"\n    Loads all file paths and labels and recreates the exact same\n    training/validation split used during training.\n    \"\"\"\n    all_filepaths, all_labels = [], []\n    if not os.path.isdir(data_path):\n        print(f\"❌ ERROR: Data path not found: {data_path}\")\n        return None, None, None\n\n    class_names = sorted([d for d in os.listdir(data_path) if os.path.isdir(os.path.join(data_path, d))])\n    class_to_idx = {name: i for i, name in enumerate(class_names)}\n\n    for class_name in class_names:\n        class_path = os.path.join(data_path, class_name)\n        label_idx = class_to_idx[class_name]\n        for filename in os.listdir(class_path):\n            if filename.endswith(\".npy\"):\n                all_filepaths.append(os.path.join(class_path, filename))\n                all_labels.append(label_idx)\n\n    _, val_paths, _, val_labels = train_test_split(\n        all_filepaths, all_labels, test_size=0.2, random_state=42, stratify=all_labels\n    )\n    return val_paths, val_labels, class_names\n\nclass InferenceSpectrogramDataset(Dataset):\n    \"\"\"Dataset for loading spectrograms for inference (no augmentation).\"\"\"\n    def __init__(self, filepaths, labels, processor):\n        self.filepaths = filepaths\n        self.labels = labels\n        self.processor = processor\n\n    def __len__(self):\n        return len(self.filepaths)\n\n    def __getitem__(self, idx):\n        file_path = self.filepaths[idx]\n        label = self.labels[idx]\n        try:\n            spectrogram = np.load(file_path)\n            image = np.stack([spectrogram] * 3, axis=-1)\n            if (image.max() - image.min()) > 0:\n                image = (image - image.min()) / (image.max() - image.min()) * 255\n            image = Image.fromarray(image.astype(np.uint8))\n            processed_image = self.processor(images=image, return_tensors=\"pt\")\n            return {\"pixel_values\": processed_image['pixel_values'].squeeze(0), \"labels\": torch.tensor(label, dtype=torch.long)}\n        except Exception:\n            return self.__getitem__((idx + 1) % len(self))\n\ndef generate_predictions(model, dataloader, device):\n    \"\"\"Runs the model on the validation dataloader to get all predictions.\"\"\"\n    model.eval()\n    all_preds, all_true_labels = [], []\n    \n    with torch.no_grad():\n        for batch in tqdm(dataloader, desc=\"Generating Predictions\"):\n            pixel_values = batch['pixel_values'].to(device)\n            labels = batch['labels'].to(device)\n            \n            outputs = model(pixel_values)\n            logits = outputs.logits\n            preds = torch.argmax(logits, dim=-1)\n            \n            all_preds.extend(preds.cpu().numpy())\n            all_true_labels.extend(labels.cpu().numpy())\n            \n    return all_true_labels, all_preds\n\ndef generate_text_report(true_labels, preds, class_names):\n    \"\"\"\n    Calculates metrics and formats a text-based confusion matrix report.\n    \"\"\"\n    # --- Calculate Overall Metrics ---\n    overall_accuracy = accuracy_score(true_labels, preds)\n    weighted_f1 = f1_score(true_labels, preds, average='weighted')\n\n    # --- Analyze Confusion Matrix ---\n    cm = confusion_matrix(true_labels, preds)\n    cm_df = pd.DataFrame(cm, index=class_names, columns=class_names)\n\n    # Find the top misclassifications\n    misclassifications = []\n    for true_label in cm_df.index:\n        for pred_label in cm_df.columns:\n            if true_label != pred_label:\n                count = cm_df.loc[true_label, pred_label]\n                if count > 0:\n                    misclassifications.append({\n                        \"True Label\": true_label,\n                        \"Predicted As\": pred_label,\n                        \"Times Confused\": count\n                    })\n    \n    misclass_df = pd.DataFrame(misclassifications)\n    top_20_errors = misclass_df.sort_values(by=\"Times Confused\", ascending=False).head(20)\n\n    # --- Print the Formatted Report ---\n    print(\"\\n\\n\" + \"=\"*50)\n    print(\"      MODEL PERFORMANCE & CONFUSION MATRIX REPORT\")\n    print(\"=\"*50)\n    \n    print(\"\\n--- Overall Performance Metrics ---\")\n    print(f\"Overall Accuracy: {overall_accuracy:.2%}\")\n    print(f\"Weighted F1-Score: {weighted_f1:.4f}\")\n    \n    print(\"\\n--- Top 20 Misclassifications ---\")\n    print(\"(True Label -> Predicted As (Count))\")\n    print(\"-------------------------------------\")\n    if top_20_errors.empty:\n        print(\"No misclassifications found!\")\n    else:\n        for _, row in top_20_errors.iterrows():\n            print(f\"- {row['True Label']} -> {row['Predicted As']} ({row['Times Confused']} times)\")\n            \n    print(\"\\n\" + \"=\"*50)\n    print(\"            END OF REPORT\")\n    print(\"=\"*50)\n\ndef main():\n    \"\"\"Main execution function.\"\"\"\n    print(\"--- Starting In-Notebook Performance Report Generation ---\")\n    \n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    val_paths, val_labels, class_names = get_validation_data(PROCESSED_DATA_PATH)\n    \n    if val_paths is None:\n        return\n\n    print(f\"\\nLoading trained model from '{MODEL_PATH}'...\")\n    try:\n        processor = AutoImageProcessor.from_pretrained(ORIGINAL_MODEL_ID)\n        model = AutoModelForImageClassification.from_pretrained(MODEL_PATH).to(device)\n    except Exception as e:\n        print(f\"❌ ERROR: Failed to load the model. Reason: {e}\")\n        return\n        \n    val_dataset = InferenceSpectrogramDataset(val_paths, val_labels, processor)\n    val_dataloader = DataLoader(val_dataset, batch_size=INFERENCE_BATCH_SIZE, shuffle=False)\n\n    true_labels, preds = generate_predictions(model, val_dataloader, device)\n    generate_text_report(true_labels, preds, class_names)\n    \nif __name__ == '__main__':\n    main()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================================\n# SCRIPT 10: AI-POWERED SPECTROGRAM CURATOR (TEACHER-STUDENT)\n# ====================================================================\n# This script implements the \"AI Curation\" (or \"Purist\") strategy from\n# the teacher-student framework. It uses a pre-trained \"teacher\" model\n# to analyze a full spectrogram dataset and create a new, ultra-clean\n# \"gold standard\" dataset.\n#\n# It keeps only the samples that the teacher model classifies CORRECTLY\n# and with HIGH CONFIDENCE.\n\nimport os\nimport shutil\nimport numpy as np\nimport torch\nimport pandas as pd\nfrom torch.utils.data import Dataset, DataLoader\nfrom transformers import AutoImageProcessor, AutoModelForImageClassification\nfrom PIL import Image\nfrom tqdm import tqdm\n\n# --- Configuration ---\n\n# 1. Path to the folder containing your trained \"teacher\" model.\nMODEL_PATH = \"/kaggle/input/saved-model2\"\n\n# 2. The original Hugging Face model ID for the image processor.\nORIGINAL_MODEL_ID = \"apple/mobilevit-small\"\n\n# 3. Path to the full PROCESSED SPECTROGRAMS dataset you want to curate.\nDATA_PATH_TO_CURATE = \"/kaggle/input/birdsong-processed2/02_processed_spectrograms\"\n\n# 4. Path where the NEW, CURATED \"gold standard\" dataset will be created.\nCURATED_DATA_PATH = \"/kaggle/working/03_processed_spectrograms_curated\"\n\n# 5. Path to move the REJECTED spectrograms to for later review.\nREJECTED_DATA_PATH = \"/kaggle/working/03_processed_spectrograms_rejected\"\n\n# 6. The minimum confidence the teacher model must have to keep a sample.\n#    A value of 0.90 means the model must be >90% sure it's correct.\nCONFIDENCE_THRESHOLD = 0.90\n\n# --- Analysis Parameters ---\nINFERENCE_BATCH_SIZE = 32\n\n# ====================================================================\n\nclass CurationDataset(Dataset):\n    \"\"\"A dataset for loading all spectrograms for the curation process.\"\"\"\n    def __init__(self, filepaths, labels, processor):\n        self.filepaths = filepaths\n        self.labels = labels\n        self.processor = processor\n\n    def __len__(self):\n        return len(self.filepaths)\n\n    def __getitem__(self, idx):\n        file_path = self.filepaths[idx]\n        label = self.labels[idx]\n        try:\n            spectrogram = np.load(file_path)\n            image = np.stack([spectrogram] * 3, axis=-1)\n            if (image.max() - image.min()) > 0:\n                image = (image - image.min()) / (image.max() - image.min()) * 255\n            image = Image.fromarray(image.astype(np.uint8))\n            processed_image = self.processor(images=image, return_tensors=\"pt\")\n            return {\n                \"pixel_values\": processed_image['pixel_values'].squeeze(0),\n                \"labels\": torch.tensor(label, dtype=torch.long),\n                \"filepath\": file_path\n            }\n        except Exception:\n            # On error, return the next item to avoid crashing\n            return self.__getitem__((idx + 1) % len(self))\n\ndef curate_with_teacher_model():\n    \"\"\"\n    Analyzes a dataset with a trained model to create a new, curated version.\n    \"\"\"\n    print(\"--- Starting AI-Powered Spectrogram Curation ---\")\n    \n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    print(f\"Using device: {device}\")\n\n    # --- Step 1: Load all file paths and labels ---\n    if not os.path.isdir(DATA_PATH_TO_CURATE):\n        print(f\"❌ ERROR: Data path not found: {DATA_PATH_TO_CURATE}\")\n        return\n\n    all_filepaths, all_labels, class_names = [], [], []\n    class_names = sorted([d for d in os.listdir(DATA_PATH_TO_CURATE) if os.path.isdir(os.path.join(DATA_PATH_TO_CURATE, d))])\n    class_to_idx = {name: i for i, name in enumerate(class_names)}\n    idx_to_class = {i: name for name, i in class_to_idx.items()}\n\n    for class_name in class_names:\n        class_path = os.path.join(DATA_PATH_TO_CURATE, class_name)\n        label_idx = class_to_idx[class_name]\n        for filename in os.listdir(class_path):\n            if filename.endswith(\".npy\"):\n                all_filepaths.append(os.path.join(class_path, filename))\n                all_labels.append(label_idx)\n    \n    print(f\"Found {len(all_filepaths)} total spectrograms to analyze.\")\n\n    # --- Step 2: Load Model and Create Dataloader ---\n    print(f\"\\nLoading teacher model from '{MODEL_PATH}'...\")\n    try:\n        processor = AutoImageProcessor.from_pretrained(ORIGINAL_MODEL_ID)\n        model = AutoModelForImageClassification.from_pretrained(MODEL_PATH).to(device)\n        model.eval()\n    except Exception as e:\n        print(f\"❌ ERROR: Failed to load the model. Reason: {e}\")\n        return\n        \n    curation_dataset = CurationDataset(all_filepaths, all_labels, processor)\n    curation_dataloader = DataLoader(curation_dataset, batch_size=INFERENCE_BATCH_SIZE, shuffle=False)\n\n    # --- Step 3: Generate Predictions and Curate Files ---\n    gold_standard_files = []\n    rejected_files_log = []\n\n    with torch.no_grad():\n        for batch in tqdm(curation_dataloader, desc=\"Curating Dataset\"):\n            pixel_values = batch['pixel_values'].to(device)\n            true_labels = batch['labels']\n            filepaths = batch['filepath']\n            \n            outputs = model(pixel_values)\n            logits = outputs.logits\n            \n            probabilities = torch.nn.functional.softmax(logits, dim=-1)\n            confidences, pred_labels = torch.max(probabilities, dim=-1)\n            \n            for i in range(len(true_labels)):\n                is_correct = pred_labels[i].item() == true_labels[i].item()\n                confidence = confidences[i].item()\n                \n                # The \"Gold Standard\" criteria\n                if is_correct and confidence > CONFIDENCE_THRESHOLD:\n                    gold_standard_files.append(filepaths[i])\n                else:\n                    rejection_reason = \"Low Confidence\" if is_correct else f\"Misclassified as {idx_to_class[pred_labels[i].item()]}\"\n                    rejected_files_log.append({\n                        \"filepath\": filepaths[i],\n                        \"true_label\": idx_to_class[true_labels[i].item()],\n                        \"predicted_label\": idx_to_class[pred_labels[i].item()],\n                        \"confidence\": f\"{confidence:.2%}\",\n                        \"rejection_reason\": rejection_reason\n                    })\n\n    # --- Step 4: Create New Directories and Move Files ---\n    print(f\"\\nCuration complete. Found {len(gold_standard_files)} gold-standard files.\")\n    \n    # Create clean output directories\n    if os.path.exists(CURATED_DATA_PATH): shutil.rmtree(CURATED_DATA_PATH)\n    if os.path.exists(REJECTED_DATA_PATH): shutil.rmtree(REJECTED_DATA_PATH)\n    os.makedirs(CURATED_DATA_PATH)\n    os.makedirs(REJECTED_DATA_PATH)\n\n    print(f\"\\nCopying gold-standard files to '{CURATED_DATA_PATH}'...\")\n    for filepath in tqdm(gold_standard_files, desc=\"Copying good files\"):\n        class_name = os.path.basename(os.path.dirname(filepath))\n        dest_folder = os.path.join(CURATED_DATA_PATH, class_name)\n        os.makedirs(dest_folder, exist_ok=True)\n        shutil.copy(filepath, dest_folder)\n        \n    print(f\"\\nMoving {len(rejected_files_log)} rejected files to '{REJECTED_DATA_PATH}'...\")\n    # NOTE: We move rejected files from their original location.\n    original_rejected_paths = [record['filepath'] for record in rejected_files_log]\n    for filepath in tqdm(original_rejected_paths, desc=\"Moving rejected files\"):\n        class_name = os.path.basename(os.path.dirname(filepath))\n        dest_folder = os.path.join(REJECTED_DATA_PATH, class_name)\n        os.makedirs(dest_folder, exist_ok=True)\n        try:\n            # Important: Use move to take the file from its source location\n            shutil.move(filepath, dest_folder)\n        except FileNotFoundError:\n            # This can happen if a file was already moved due to being in multiple batches (unlikely but possible)\n            tqdm.write(f\"Warning: File {os.path.basename(filepath)} not found at source, may have already been moved.\")\n            continue\n\n\n    # --- Step 5: Save the Rejection Report ---\n    report_df = pd.DataFrame(rejected_files_log)\n    report_filename = \"curation_rejection_report.csv\"\n    report_df.to_csv(report_filename, index=False)\n    \n    print(f\"\\n✅ Curation process finished!\")\n    print(f\"Your new, curated dataset is ready for training at: '{CURATED_DATA_PATH}'\")\n    print(f\"A detailed report of rejected files is saved as: '{report_filename}'\")\n\nif __name__ == '__main__':\n    curate_with_teacher_model()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T08:27:03.135276Z","iopub.execute_input":"2025-09-17T08:27:03.135499Z","iopub.status.idle":"2025-09-17T08:48:12.766316Z","shell.execute_reply.started":"2025-09-17T08:27:03.135479Z","shell.execute_reply":"2025-09-17T08:48:12.765384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    import noisereduce\n    print(\"Libraries seem to be installed.\")\nexcept (NameError, ImportError, RuntimeError):\n    print(\"Installing required libraries: noisereduce, librosa, pandas...\")\n    !pip install noisereduce scipy librosa pandas tqdm -q\n    print(\"\\n✅ Installation complete.\")\n    print(\"‼️ IMPORTANT: Please restart the runtime now before proceeding.\")\n    import sys\n    sys.exit()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================================\n# SCRIPT 12: ENHANCED PREDICTION SCRIPT (TEST-TIME ENHANCEMENT)\n# VERSION 2.0: With Full Audio Analysis\n# ====================================================================\n# This script applies enhancement (noise reduction and normalization)\n# to a new audio file, chunks the entire file, and aggregates\n# predictions from all chunks to produce a final, robust result.\n\n# --- Block 1: Setup and Installation (RUN FIRST, THEN RESTART) ---\n# This block should be in its own cell. After running it, you MUST\n# restart the kernel/runtime before running the main script.\n# ====================================================================\n# MAIN SCRIPT (RUN THIS IN A NEW CELL AFTER RESTARTING)\n# ====================================================================\n\nimport os\nimport numpy as np\nimport torch\nimport librosa\nimport soundfile as sf\nimport noisereduce as nr\nfrom transformers import AutoImageProcessor, AutoModelForImageClassification\nfrom PIL import Image\nfrom collections import Counter\n\n# --- Configuration ---\n\n# 1. Path to the folder containing your final, saved \"student\" model.\nMODEL_PATH = \"/kaggle/input/saved-model2\"\n\n# 2. The original Hugging Face model ID for the image processor.\nORIGINAL_MODEL_ID = \"apple/mobilevit-small\"\n\n# 3. Path to the audio file you want to make a prediction on.\nAUDIO_FILE_TO_PREDICT = \"/kaggle/input/sample/rwb (2).mp3\"\n\n# --- Enhancement & Preprocessing Parameters (MUST match your training setup) ---\nSAMPLE_RATE = 32000\nCHUNK_DURATION_S = 5\nN_MELS = 224\nN_FFT = 2048\nHOP_LENGTH = 512\nNOISE_REDUCTION_STRENGTH = 0.3\n# The target peak volume in decibels for normalization.\nNORMALIZATION_TARGET_DB = -3.0\n\n# ====================================================================\n\ndef enhance_and_chunk_audio(file_path):\n    \"\"\"\n    Loads an audio file, applies noise reduction and normalization,\n    and then splits it into a list of 5-second chunks.\n    \"\"\"\n    try:\n        y, sr = librosa.load(file_path, sr=SAMPLE_RATE)\n\n        # --- Step 1: Enhance Audio ---\n        # Apply noise reduction to the entire signal\n        y_reduced = nr.reduce_noise(y=y, sr=sr, prop_decrease=NOISE_REDUCTION_STRENGTH)\n\n        # Apply peak normalization (a controlled amplitude gain)\n        target_amp = 10**(NORMALIZATION_TARGET_DB / 20)\n        max_amp = np.max(np.abs(y_reduced))\n        if max_amp > 0:\n            y_normalized = y_reduced * (target_amp / max_amp)\n        else:\n            y_normalized = y_reduced # Avoid division by zero on silent clips\n\n        # --- Step 2: Chunk the enhanced audio ---\n        chunk_samples = int(CHUNK_DURATION_S * sr)\n        num_chunks = int(np.ceil(len(y_normalized) / chunk_samples))\n        \n        chunks = []\n        for i in range(num_chunks):\n            start = i * chunk_samples\n            end = start + chunk_samples\n            chunk = y_normalized[start:end]\n            \n            # Pad the last chunk if it's too short\n            if len(chunk) < chunk_samples:\n                chunk = np.pad(chunk, (0, chunk_samples - len(chunk)), 'constant')\n            chunks.append(chunk)\n            \n        return chunks, sr\n\n    except Exception as e:\n        print(f\"❌ ERROR: Failed to process audio file '{file_path}'. Reason: {e}\")\n        return None, None\n\ndef create_spectrogram(audio_chunk, sr):\n    \"\"\"Converts a single audio chunk to a log-Mel spectrogram.\"\"\"\n    mel = librosa.feature.melspectrogram(y=audio_chunk, sr=sr, n_fft=N_FFT, hop_length=HOP_LENGTH, n_mels=N_MELS)\n    log_mel = librosa.power_to_db(mel, ref=np.max)\n    return log_mel\n\ndef predict_bird_in_audio_enhanced():\n    \"\"\"\n    Main function to load model, enhance audio, and make a prediction.\n    \"\"\"\n    print(\"--- Enhanced Birdsong Prediction (Full Analysis) ---\")\n    \n    if not os.path.isdir(MODEL_PATH) or not os.path.isfile(AUDIO_FILE_TO_PREDICT):\n        print(f\"❌ ERROR: Check paths. Model: '{MODEL_PATH}', Audio: '{AUDIO_FILE_TO_PREDICT}'\")\n        return\n\n    # --- Step 1: Load Model and Processor ---\n    print(f\"Loading trained model from '{MODEL_PATH}'...\")\n    try:\n        device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n        processor = AutoImageProcessor.from_pretrained(ORIGINAL_MODEL_ID)\n        model = AutoModelForImageClassification.from_pretrained(MODEL_PATH).to(device)\n        model.eval()\n        print(f\"✅ Model loaded successfully on device: {device}\")\n    except Exception as e:\n        print(f\"❌ ERROR: Failed to load the model. Reason: {e}\")\n        return\n\n    # --- Step 2: Enhance Audio and Chunk it ---\n    print(f\"\\nProcessing and enhancing audio file: '{os.path.basename(AUDIO_FILE_TO_PREDICT)}'...\")\n    audio_chunks, sr = enhance_and_chunk_audio(AUDIO_FILE_TO_PREDICT)\n    if audio_chunks is None:\n        return\n    print(f\"✅ Audio enhanced and split into {len(audio_chunks)} chunk(s).\")\n\n    # --- Step 3: Create Spectrograms and Run Inference on all chunks ---\n    chunk_predictions = []\n    chunk_confidences = []\n\n    with torch.no_grad():\n        for chunk in audio_chunks:\n            spectrogram = create_spectrogram(chunk, sr)\n            \n            image = np.stack([spectrogram] * 3, axis=-1)\n            if (image.max() - image.min()) > 0:\n                image = (image - image.min()) / (image.max() - image.min()) * 255\n            image = Image.fromarray(image.astype(np.uint8))\n            \n            inputs = processor(images=image, return_tensors=\"pt\").to(device)\n            outputs = model(**inputs)\n            logits = outputs.logits\n            \n            probabilities = torch.nn.functional.softmax(logits, dim=-1)\n            confidence, predicted_class_idx = torch.max(probabilities, dim=-1)\n            \n            chunk_predictions.append(predicted_class_idx.item())\n            chunk_confidences.append(confidence.item())\n\n    # --- Step 4: Aggregate Results and Report Final Prediction ---\n    if not chunk_predictions:\n        print(\"\\nCould not make a prediction. No valid chunks were analyzed.\")\n        return\n\n    final_prediction_idx = Counter(chunk_predictions).most_common(1)[0][0]\n    final_class_name = model.config.id2label[final_prediction_idx]\n    \n    confidences_for_final_class = [conf for idx, conf in zip(chunk_predictions, chunk_confidences) if idx == final_prediction_idx]\n    avg_confidence = np.mean(confidences_for_final_class)\n\n    print(\"\\n--- Prediction Result ---\")\n    print(f\"Predicted Species: {final_class_name}\")\n    print(f\"Confidence: {avg_confidence:.2%}\")\n    print(\"-------------------------\")\n\nif __name__ == '__main__':\n    predict_bird_in_audio_enhanced()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T08:24:19.805797Z","iopub.execute_input":"2025-09-17T08:24:19.806115Z","iopub.status.idle":"2025-09-17T08:24:21.088889Z","shell.execute_reply.started":"2025-09-17T08:24:19.806090Z","shell.execute_reply":"2025-09-17T08:24:21.088232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nfrom IPython.display import FileLink\n\n# --- 1. SETUP: Create a dummy directory and files to zip (for demonstration) ---\n# In your real code, you would skip this part and just use your existing directory.\n\n# Name of the directory you want to zip\ndirectory_to_zip = 'my_submission_files'\nfull_directory_path = f'/kaggle/working/03_processed_spectrograms_curated'\n\n# Create the directory if it doesn't exist\nos.makedirs(full_directory_path, exist_ok=True)\n\nprint(f\"Created directory: '{full_directory_path}'\")\n\n# Create some dummy files inside the directory\nwith open(os.path.join(full_directory_path, 'submission.csv'), 'w') as f:\n    f.write('id,label\\n')\n    f.write('1,0.5\\n')\n    f.write('2,0.9\\n')\n\nwith open(os.path.join(full_directory_path, 'log.txt'), 'w') as f:\n    f.write('Training completed.\\n')\n    f.write('Final accuracy: 95%\\n')\n\nprint(\"Created dummy files for demonstration.\")\nprint(\"-\" * 30)\n\n\n# --- 2. CORE LOGIC: Zip the directory ---\n\n# The name of the output zip file (without the .zip extension)\n# It will be saved in /kaggle/working/\noutput_filename = 'final_submission_archive'\n\n# The directory to be archived.\n# shutil.make_archive will zip the contents of this directory.\n# `root_dir` is the directory to zip.\n# `base_dir` is what's inside the root_dir to start from (here, we zip everything)\nshutil.make_archive(\n    base_name=f'/kaggle/working/{output_filename}', # Path and name of the output zip file (without .zip)\n    format='zip',                                 # The archive format\n    root_dir=full_directory_path                  # The directory to be zipped\n)\n\nprint(f\"Successfully created zip archive: /kaggle/working/{output_filename}.zip\")\nprint(\"-\" * 30)\n\n\n# --- 3. GENERATE DOWNLOAD LINK ---\n\n# Create a link to the zip file you just created\nzip_file_path = f'/kaggle/working/{output_filename}.zip'\n\nprint(\"Generating download link...\")\ndisplay(FileLink(zip_file_path))","metadata":{"trusted":true,"execution":{"execution_failed":"2025-09-17T09:29:58.212Z"}},"outputs":[],"execution_count":null}]}