{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"isSourceIdPinned":false,"sourceType":"competition"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Importing Modules","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport librosa\nimport librosa.display\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, MultiLabelBinarizer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.multiclass import OneVsRestClassifier\nimport sklearn # Import sklearn itself to get its version\n\n# Print versions for reproducibility and debugging\nprint(f\"numpy version: {np.__version__}\")\nprint(f\"pandas version: {pd.__version__}\")\nprint(f\"librosa version: {librosa.__version__}\")\nprint(f\"scikit-learn version: {sklearn.__version__}\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-22T03:29:41.621082Z","iopub.execute_input":"2025-05-22T03:29:41.621905Z","iopub.status.idle":"2025-05-22T03:29:41.627483Z","shell.execute_reply.started":"2025-05-22T03:29:41.621882Z","shell.execute_reply":"2025-05-22T03:29:41.626772Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Defining base paths for input data","metadata":{}},{"cell_type":"code","source":"# IMPORTANT: Adjust BASE_INPUT_PATH if your data is not in the same directory as your notebook.\n# On Kaggle, it's typically '../input/birdclef-2025/'. Locally, it might be './data/' or similar.\nBASE_INPUT_PATH = '/kaggle/input/birdclef-2025/'  # Updated for Kaggle environment\n\n# Construct full paths to specific data directories\nTRAIN_AUDIO_PATH = os.path.join(BASE_INPUT_PATH, 'train_audio')\nTEST_SOUNDSCAPES_PATH = os.path.join(BASE_INPUT_PATH, 'test_soundscapes')\nTRAIN_SOUNDSCAPES_PATH = os.path.join(BASE_INPUT_PATH, 'train_soundscapes') # Unlabeled data\n\n# Define constants for audio processing and feature extraction\nSR = 32000 \nDURATION = 5  \nN_MFCC = 20   \n\n# Print the defined paths and constants for verification\nprint(\"File paths and constants defined:\")\nprint(f\"  Base input path: {BASE_INPUT_PATH}\")\nprint(f\"  Train audio path: {TRAIN_AUDIO_PATH}\")\nprint(f\"  Test soundscapes path: {TEST_SOUNDSCAPES_PATH}\")\nprint(f\"  Train soundscapes path: {TRAIN_SOUNDSCAPES_PATH}\")\nprint(f\"Audio Processing Parameters:\")\nprint(f\"  Sample Rate (SR): {SR} Hz\")\nprint(f\"  Snippet Duration (DURATION): {DURATION} seconds\")\nprint(f\"  Number of MFCCs (N_MFCC): {N_MFCC}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T03:29:44.344912Z","iopub.execute_input":"2025-05-22T03:29:44.345581Z","iopub.status.idle":"2025-05-22T03:29:44.351119Z","shell.execute_reply.started":"2025-05-22T03:29:44.345559Z","shell.execute_reply":"2025-05-22T03:29:44.350369Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Loading the essential metadata files using pandas","metadata":{}},{"cell_type":"code","source":"# These files provide information about the training audio, species, and submission format.\n\ntrain_metadata_df = pd.read_csv(os.path.join(BASE_INPUT_PATH, 'train.csv'))\ntaxonomy_df = pd.read_csv(os.path.join(BASE_INPUT_PATH, 'taxonomy.csv'))\nsample_submission_df = pd.read_csv(os.path.join(BASE_INPUT_PATH, 'sample_submission.csv'))\nprint(\"Metadata loaded successfully from specified paths.\")\n\n\n# Display the first few rows of each DataFrame to verify successful loading and inspect data structure.\nprint(\"\\n--- train_metadata_df (first 5 rows) ---\")\nprint(train_metadata_df.head())\nprint(f\"\\nShape of train_metadata_df: {train_metadata_df.shape}\")\n\nprint(\"\\n--- taxonomy_df (first 5 rows) ---\")\nprint(taxonomy_df.head())\nprint(f\"\\nShape of taxonomy_df: {taxonomy_df.shape}\")\n\nprint(\"\\n--- sample_submission_df (first 5 rows) ---\")\nprint(sample_submission_df.head())\nprint(f\"\\nShape of sample_submission_df: {sample_submission_df.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T03:29:48.412815Z","iopub.execute_input":"2025-05-22T03:29:48.413058Z","iopub.status.idle":"2025-05-22T03:29:48.564141Z","shell.execute_reply.started":"2025-05-22T03:29:48.413043Z","shell.execute_reply":"2025-05-22T03:29:48.563543Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Identifying all unique species that the model needs to predict","metadata":{}},{"cell_type":"code","source":"# This list will define the target columns for our multi-label classification.\n\ninitial_species_candidates = []\n\n# Strategy: Prioritize getting the species list from sample_submission.csv columns.\nif 'row_id' in sample_submission_df.columns:\n    # Exclude 'row_id' as it's not a species column\n    initial_species_candidates = [col for col in sample_submission_df.columns if col != 'row_id']\n    print(\"Strategy: Species list successfully derived from sample_submission.csv columns.\")\nelif 'species_code' in taxonomy_df.columns:\n    # Fallback 1: If sample_submission didn't provide species columns, use taxonomy.csv\n    initial_species_candidates = taxonomy_df['species_code'].unique().tolist()\n    print(\"Strategy: Species list derived from taxonomy.csv 'species_code' column (fallback).\")\nelif 'primary_label' in train_metadata_df.columns:\n    # Fallback 2: If neither of the above worked, use primary_label from train.csv\n    initial_species_candidates = train_metadata_df['primary_label'].unique().tolist()\n    print(\"Strategy: Species list derived from train.csv 'primary_label' column (secondary fallback).\")\nelse:\n    # Ultimate fallback: If no species columns found, initialize as empty.\n    # This will trigger the hardcoded dummy species later if still empty.\n    initial_species_candidates = []\n    print(\"Warning: No clear source for species list found in standard dataframes.\")\n\n\n# Clean and finalize ALL_SPECIES list:\n\nALL_SPECIES = sorted(list(set(str(s).strip() for s in initial_species_candidates if pd.notna(s) and str(s).strip())))\n\n# Critical check: If after all attempts, the species list is still empty,\n\nif not ALL_SPECIES:\n    ALL_SPECIES = ['dummy_species_1', 'dummy_species_2', 'dummy_species_3']\n    print(\"CRITICAL WARNING: ALL_SPECIES list is empty after all attempts. Using hardcoded dummy species.\")\n    print(\"Please verify your data files and their contents.\")\n\n# Create a set version of ALL_SPECIES for efficient membership checking later (e.g., when parsing secondary labels).\nALL_SPECIES_SET = set(ALL_SPECIES)\n\n# Print summary information about the identified species list\nprint(f\"\\n--- Species List Summary ---\")\nprint(f\"Total unique species identified for prediction: {len(ALL_SPECIES)}\")\nprint(f\"First 10 species in the list: {ALL_SPECIES[:10]}\")\nprint(f\"Last 10 species in the list: {ALL_SPECIES[-10:]}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T03:29:53.241307Z","iopub.execute_input":"2025-05-22T03:29:53.242080Z","iopub.status.idle":"2025-05-22T03:29:53.249152Z","shell.execute_reply.started":"2025-05-22T03:29:53.242059Z","shell.execute_reply":"2025-05-22T03:29:53.248312Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Feature Extraction Function","metadata":{}},{"cell_type":"code","source":"def extract_features(file_path, sr=SR, duration=DURATION, n_mfcc=N_MFCC):\n\n    try:\n        # Load the audio file. 'sr=sr' resamples the audio to our target sample rate.\n\n        y, current_sr = librosa.load(file_path, sr=sr, mono=True)\n\n        # Ensure the audio snippet has a fixed length.\n\n        target_len_samples = duration * sr\n        if len(y) > target_len_samples:\n            # If audio is longer than target duration, truncate it.\n            y = y[:target_len_samples]\n        else:\n            # If audio is shorter, pad with zeros to reach the target duration.\n\n            y = np.pad(y, (0, target_len_samples - len(y)), 'constant')\n\n\n        mfccs = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=n_mfcc)\n\n\n        return np.mean(mfccs.T, axis=0)\n\n    except Exception as e:\n        # Handle any errors during audio processing (e.g., corrupted file, invalid format).\n        # Print an error message and return an array of zeros to maintain consistent feature dimensions.\n        print(f\"Error processing {file_path}: {e}\")\n        return np.zeros(n_mfcc)\n\nprint(\"Feature extraction function 'extract_features' defined.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T03:29:58.377158Z","iopub.execute_input":"2025-05-22T03:29:58.377814Z","iopub.status.idle":"2025-05-22T03:29:58.383511Z","shell.execute_reply.started":"2025-05-22T03:29:58.377793Z","shell.execute_reply":"2025-05-22T03:29:58.382645Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Preparing Data for Models","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.model_selection import train_test_split\n\nprint(\"Preparing data for model training...\")\n\nfeatures_list = [] # To store the extracted features for each audio file\nlabels_list = []   # To store the corresponding labels (species codes) for each audio file\n\n# Helper function to safely parse and clean secondary_labels\n# This handles the string representation of lists and filters for valid species\ndef parse_and_clean_labels(label_str_list_repr, all_species_set):\n    \"\"\"\n    Parses a string representation of a list of labels and cleans them.\n\n    Args:\n        label_str_list_repr (str): A string that might represent a Python list (e.g., \"['speciesA', 'speciesB']\").\n        all_species_set (set): A set of all known valid species codes for filtering.\n\n    Returns:\n        list: A list of cleaned and validated species codes. Returns an empty list if parsing fails\n              or no valid labels are found.\n    \"\"\"\n    if isinstance(label_str_list_repr, str) and label_str_list_repr.startswith('[') and label_str_list_repr.endswith(']'):\n        try:\n            # Safely evaluate the string as a Python literal (list).\n            # Using ast.literal_eval is safer than eval() for untrusted input,\n            # but for competition data, eval() is often used for convenience.\n            # Here, we'll use eval() as it's common in Kaggle notebooks for this specific format.\n            parsed_list = eval(label_str_list_repr)\n            # Ensure parsed_list is actually a list before iterating\n            if not isinstance(parsed_list, list):\n                return []\n            # Clean each label and keep only those present in our ALL_SPECIES_SET\n            return [str(s).strip() for s in parsed_list if str(s).strip() and str(s).strip() in all_species_set]\n        except Exception as e:\n            # print(f\"Warning: Could not parse secondary_labels '{label_str_list_repr}': {e}\")\n            return []\n    return [] # Return empty list if input is not a string list representation\n\n# --- Data Processing Loop ---\n# IMPORTANT CHANGE: Now processing the FULL dataset for better performance.\n# This will take significantly longer than processing a small subset.\nMAX_FILES_TO_PROCESS = len(train_metadata_df) # Process all available training files\n\nprocessed_count = 0\n# Iterate through the entire training metadata DataFrame\nfor index, row in train_metadata_df.iterrows(): # Removed .head(MAX_FILES_TO_PROCESS)\n    file_name = row['filename']\n    primary_label = str(row['primary_label']).strip() # Ensure primary label is clean string\n\n    current_file_labels = set() # Use a set to collect unique labels for the current file\n\n    # Add primary label if it's valid and in our ALL_SPECIES_SET\n    if primary_label and primary_label in ALL_SPECIES_SET:\n        current_file_labels.add(primary_label)\n\n    # Process secondary labels\n    secondary_labels_str = row.get('secondary_labels', '[]') # Get secondary labels, default to '[]' if missing\n    parsed_secondary_labels = parse_and_clean_labels(secondary_labels_str, ALL_SPECIES_SET)\n    for sec_label in parsed_secondary_labels:\n        current_file_labels.add(sec_label) # Add cleaned and validated secondary labels\n\n    # Construct the full path to the audio file\n    full_audio_path = os.path.join(TRAIN_AUDIO_PATH, file_name)\n\n    # Extract features for the current audio file\n    if os.path.exists(full_audio_path):\n        current_features = extract_features(full_audio_path)\n    else:\n        # If audio file not found, print a warning and return a zero feature vector.\n        # This is important for robustness, especially with large datasets.\n        print(f\"Warning: Audio file not found at {full_audio_path}. Using zero features.\")\n        current_features = np.zeros(N_MFCC) # Use N_MFCC as the feature dimension\n\n    # Only add features and labels if there's at least one valid label.\n    # This prevents training on samples with no relevant species information.\n    if current_file_labels:\n        features_list.append(current_features)\n        labels_list.append(list(current_file_labels)) # Convert set to list for MultiLabelBinarizer\n    else:\n        print(f\"Skipping {file_name}: No valid primary or secondary labels found.\")\n\n    processed_count += 1\n    if processed_count % 1000 == 0: # Print progress every 1000 files now\n        print(f\"Processed {processed_count}/{MAX_FILES_TO_PROCESS} files...\")\n\nprint(f\"\\nFinished processing {processed_count} files for feature extraction.\")\n\n# Convert lists of features and labels into NumPy arrays\nX = np.array(features_list)\n\n# Initialize MultiLabelBinarizer with our comprehensive list of ALL_SPECIES\nmlb = MultiLabelBinarizer(classes=ALL_SPECIES)\n# Transform the list of lists of labels into a binary matrix (one-hot encoded for multi-label)\ny_transformed = mlb.fit_transform(labels_list)\n\n# Scale the features using StandardScaler\n# This is important for many ML algorithms (e.g., Logistic Regression, SVM, Neural Networks)\n# to ensure features with larger values don't dominate the learning process.\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\n# Split the data into training and validation sets\n# This allows us to evaluate model performance on unseen data during development.\n# test_size=0.2 means 20% of the data will be used for validation.\n# random_state for reproducibility of the split.\nif X_scaled.shape[0] > 1: # Ensure there's enough data to split\n    X_train, X_val, y_train, y_val = train_test_split(X_scaled, y_transformed, test_size=0.2, random_state=42)\n    print(\"\\nData split into training and validation sets.\")\nelse:\n    # If not enough data for splitting (e.g., only 1 file processed),\n    # use the entire dataset as training and create dummy validation sets.\n    X_train, y_train = X_scaled, y_transformed\n    X_val, y_val = X_scaled.copy(), y_transformed.copy() # Use copies to avoid modifying original\n    print(\"Warning: Insufficient data for train-validation split. Using all data for training and dummy validation.\")\n\n\n# Print the shapes of the resulting datasets to verify their dimensions\nprint(f\"\\nShape of X (all features): {X.shape}\")\nprint(f\"Shape of y (all transformed labels): {y_transformed.shape}\")\nprint(f\"Shape of X_scaled (all scaled features): {X_scaled.shape}\")\nprint(f\"Shape of X_train: {X_train.shape}, Shape of y_train: {y_train.shape}\")\nprint(f\"Shape of X_val: {X_val.shape}, Shape of y_val: {y_val.shape}\")\nprint(f\"Number of features (MFCCs): {X_train.shape[1]}\")\nprint(f\"Number of output classes (species): {y_train.shape[1]}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T03:30:04.204080Z","iopub.execute_input":"2025-05-22T03:30:04.204383Z","iopub.status.idle":"2025-05-22T03:51:37.674412Z","shell.execute_reply.started":"2025-05-22T03:30:04.204359Z","shell.execute_reply":"2025-05-22T03:51:37.673276Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7. Model Training and Hyperparameter Comparison (Setup and Summary","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score # make_scorer is no longer needed as RandomizedSearchCV is removed\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.svm import LinearSVC\n\n# RandomizedSearchCV, uniform, randint are no longer needed as manual comparison is used\nfrom sklearn.model_selection import train_test_split # Still needed for X_val, y_val split in Cell 6\n\nimport warnings\nimport sqlite3\n\n# ONLY Suppress the OperationalError related to readonly database and history writing.\nwarnings.filterwarnings('ignore', category=UserWarning, message='.*attempt to write a readonly database.*')\nwarnings.filterwarnings('ignore', category=UserWarning, message='History will not be written to the database.')\nwarnings.filterwarnings('ignore', category=UserWarning, module='IPython.core.history')\nwarnings.filterwarnings('ignore', category=UserWarning, module='sqlite3')\n\n\nprint(\"Starting model training and hyperparameter comparison setup...\")\n\n# --- Helper function for safe AUC calculation ---\ndef calculate_safe_auc(y_true, y_pred_proba, class_names):\n    \"\"\"\n    Calculates the macro-averaged ROC AUC score, safely handling classes\n    that do not have both positive and negative samples in y_true.\n    \"\"\"\n    class_auc_scores = []\n    for i in range(y_true.shape[1]):\n        if len(np.unique(y_true[:, i])) > 1:\n            try:\n                class_auc = roc_auc_score(y_true[:, i], y_pred_proba[:, i])\n                class_auc_scores.append(class_auc)\n            except ValueError as ve:\n                pass # Skip this class for AUC calculation\n\n    if not class_auc_scores:\n        print(\"  No classes were eligible for ROC AUC calculation. Returning 0.0.\")\n        return 0.0\n    return np.mean(class_auc_scores)\n\n# Dictionaries to store trained models and their scores\nmodels = {}\nvalidation_auc_scores = {}\ntraining_auc_scores = {}\nbest_hyperparameters = {} # To store the best params found for each model type\n\nprint(\"\\n--- Model Training and Evaluation Summary (Best Hyperparameters) ---\")\nprint(\"---------------------------------------------\")\nprint(f\"{'Model':<25} | {'Training AUC':<15} | {'Validation AUC':<15} | {'Best Params':<50}\")\nprint(\"---------------------------------------------\")\nfor model_name in models.keys():\n    train_auc = training_auc_scores.get(model_name, 'N/A')\n    val_auc = validation_auc_scores.get(model_name, 'N/A')\n    \n    # Retrieve the best params from the best_hyperparameters dictionary\n    params_to_display = str(best_hyperparameters.get(model_name, 'N/A'))\n\n    train_auc_str = f\"{train_auc:.4f}\" if isinstance(train_auc, float) else str(train_auc)\n    val_auc_str = f\"{val_auc:.4f}\" if isinstance(val_auc, float) else str(val_auc)\n\n    print(f\"{model_name:<25} | {train_auc_str:<15} | {val_auc_str:<15} | {params_to_display:<50}\")\nprint(\"---------------------------------------------\")\n\n\n# Find the best overall model based on validation AUC\nif validation_auc_scores:\n    best_overall_model_name = max(validation_auc_scores, key=validation_auc_scores.get)\n    print(f\"\\nBest overall performing model on validation set: {best_overall_model_name} (AUC: {validation_auc_scores[best_overall_model_name]:.4f})\")\nelse:\n    print(\"\\nNo models were trained or evaluated successfully.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T04:04:24.443687Z","iopub.execute_input":"2025-05-22T04:04:24.444649Z","iopub.status.idle":"2025-05-22T04:04:24.456570Z","shell.execute_reply.started":"2025-05-22T04:04:24.444630Z","shell.execute_reply":"2025-05-22T04:04:24.455830Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7.1. Logistic Regression Training and Comparison","metadata":{}},{"cell_type":"code","source":"import warnings\n\n# Suppress ALL warnings in this specific cell, with more targeted filters for common sklearn warnings\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn.multiclass')\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn.model_selection._validation')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Only one class present in y_true')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Label not .* is present in all training examples')\n\nprint(\"\\n--- Tuning and Training Logistic Regression (OneVsRest) ---\")\n\n# Define three different sets of hyperparameters to compare manually\nlog_reg_param_sets = [\n    # Set 1: Good all-rounder, balanced for imbalanced classes\n    {'solver': 'liblinear', 'C': 1.0, 'class_weight': 'balanced', 'max_iter': 2000},\n    # Set 2: Slightly less regularization, different solver, no class weight\n    {'solver': 'saga', 'C': 0.7, 'class_weight': None, 'max_iter': 2000},\n    # Set 3: More regularization, different solver, balanced class weight\n    {'solver': 'lbfgs', 'C': 0.5, 'class_weight': 'balanced', 'max_iter': 2000},\n]\n\nbest_log_reg_val_auc = -1.0\nbest_log_reg_model = None\nbest_log_reg_params = None\n\n# List to store results for plotting\nlog_reg_results_for_plot = []\n\nif X_train.shape[0] > 0 and y_train.shape[0] > 0:\n    for i, params in enumerate(log_reg_param_sets):\n        # Removed detailed print for each set as requested\n        # print(f\"\\n  Trying Logistic Regression with params (Set {i+1}): {params}\")\n        \n        # Create the OneVsRestClassifier with the current LogisticRegression parameters\n        log_reg_clf = OneVsRestClassifier(LogisticRegression(random_state=42, **params))\n        \n        # Fit the model on the full training data\n        log_reg_clf.fit(X_train, y_train)\n\n        # Calculate Training AUC\n        y_pred_proba_train = log_reg_clf.predict_proba(X_train)\n        auc_train = calculate_safe_auc(y_train, y_pred_proba_train, ALL_SPECIES)\n        # Removed print(f\"    Training AUC: {auc_train:.4f}\")\n\n        # Calculate Validation AUC\n        auc_val = 0.0 # Default if no validation data\n        if X_val.shape[0] > 0:\n            y_pred_proba_val = log_reg_clf.predict_proba(X_val)\n            auc_val = calculate_safe_auc(y_val, y_pred_proba_val, ALL_SPECIES)\n            # Removed print(f\"    Validation AUC: {auc_val:.4f}\")\n\n            # Check if this model is the best so far based on Validation AUC\n            if auc_val > best_log_reg_val_auc:\n                best_log_reg_val_auc = auc_val\n                best_log_reg_model = log_reg_clf\n                best_log_reg_params = params\n        else:\n            # If no validation data, use training AUC as fallback for best model selection\n            if auc_train > best_log_reg_val_auc:\n                best_log_reg_val_auc = auc_train\n                best_log_reg_model = log_reg_clf\n                best_log_reg_params = params\n        \n        # Store results for plotting\n        log_reg_results_for_plot.append({\n            'Set': f'Set {i+1}',\n            'Params': params,\n            'Training AUC': auc_train,\n            'Validation AUC': auc_val\n        })\n\nelse:\n    print(\"  Skipping Logistic Regression: Insufficient training data.\")\n\n# After trying all parameter sets, store the best model found\nif best_log_reg_model:\n    models['LogisticRegression'] = best_log_reg_model\n    validation_auc_scores['LogisticRegression'] = best_log_reg_val_auc\n    training_auc_scores['LogisticRegression'] = auc_train # Store the training AUC of the best model\n    best_hyperparameters['LogisticRegression'] = best_log_reg_params\n    print(f\"\\n  Best Logistic Regression setup chosen: {best_log_reg_params}\") # Only print best setup\n\n    # --- Plotting Hyperparameter Performance ---\n    fig, ax = plt.subplots(figsize=(10, 6))\n    \n    set_labels = [res['Set'] for res in log_reg_results_for_plot]\n    train_aucs = [res['Training AUC'] for res in log_reg_results_for_plot]\n    val_aucs = [res['Validation AUC'] for res in log_reg_results_for_plot]\n\n    x = np.arange(len(set_labels)) # the label locations\n    width = 0.35 # the width of the bars\n\n    rects1 = ax.bar(x - width/2, train_aucs, width, label='Training AUC', color='skyblue')\n    rects2 = ax.bar(x + width/2, val_aucs, width, label='Validation AUC', color='lightcoral')\n\n    # Add some text for labels, title and custom x-axis tick labels, etc.\n    ax.set_ylabel('AUC Score')\n    ax.set_title('Logistic Regression Hyperparameter Performance Comparison')\n    ax.set_xticks(x)\n    ax.set_xticklabels(set_labels)\n    ax.legend()\n    ax.set_ylim(0, 1) # AUC scores are between 0 and 1\n\n    # Add exact values on top of bars\n    def autolabel(rects):\n        for rect in rects:\n            height = rect.get_height()\n            ax.annotate(f'{height:.3f}',\n                        xy=(rect.get_x() + rect.get_width() / 2, height),\n                        xytext=(0, 3),  # 3 points vertical offset\n                        textcoords=\"offset points\",\n                        ha='center', va='bottom')\n\n    autolabel(rects1)\n    autolabel(rects2)\n\n    fig.tight_layout()\n    plt.show()\n\nelse:\n    print(\"\\n  No Logistic Regression model trained due to insufficient data.\")\n\nprint(\"\\nFinished Logistic Regression training and comparison.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T04:04:28.861598Z","iopub.execute_input":"2025-05-22T04:04:28.862438Z","iopub.status.idle":"2025-05-22T04:12:47.727300Z","shell.execute_reply.started":"2025-05-22T04:04:28.862416Z","shell.execute_reply":"2025-05-22T04:12:47.726399Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7.2. K-Nearest Neighbors Training and Comparison","metadata":{}},{"cell_type":"code","source":"import warnings\n\n# Suppress ALL warnings in this specific cell, with more targeted filters for common sklearn warnings\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn.multiclass')\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn.model_selection._validation')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Only one class present in y_true')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Label not .* is present in all training examples')\n\nprint(\"\\n--- Tuning and Training K-Nearest Neighbors (OneVsRest) ---\")\n\n# Define three different sets of hyperparameters to compare manually\nknn_param_sets = [\n    # Set 1: Default-ish, common choice\n    {'n_neighbors': 5, 'weights': 'uniform'},\n    # Set 2: Slightly more neighbors, distance weighting\n    {'n_neighbors': 7, 'weights': 'distance'},\n    # Set 3: Fewer neighbors, uniform weighting\n    {'n_neighbors': 3, 'weights': 'uniform'},\n]\n\nbest_knn_val_auc = -1.0\nbest_knn_model = None\nbest_knn_params = None\n\nknn_results_for_plot = []\n\nif X_train.shape[0] > 0 and y_train.shape[0] > 0:\n    for i, params in enumerate(knn_param_sets):\n        # Removed detailed print for each set\n        \n        # Create the OneVsRestClassifier with the current KNeighborsClassifier parameters\n        knn_clf = OneVsRestClassifier(KNeighborsClassifier(**params))\n        \n        # Fit the model on the full training data\n        knn_clf.fit(X_train, y_train)\n\n        # Calculate Training AUC\n        y_pred_proba_train = knn_clf.predict_proba(X_train)\n        auc_train = calculate_safe_auc(y_train, y_pred_proba_train, ALL_SPECIES)\n        \n        # Calculate Validation AUC\n        auc_val = 0.0\n        if X_val.shape[0] > 0:\n            y_pred_proba_val = knn_clf.predict_proba(X_val)\n            auc_val = calculate_safe_auc(y_val, y_pred_proba_val, ALL_SPECIES)\n\n            # Check if this model is the best so far based on Validation AUC\n            if auc_val > best_knn_val_auc:\n                best_knn_val_auc = auc_val\n                best_knn_model = knn_clf\n                best_knn_params = params\n        else:\n            if auc_train > best_knn_val_auc:\n                best_knn_val_auc = auc_train\n                best_knn_model = knn_clf\n                best_knn_params = params\n                \n        knn_results_for_plot.append({\n            'Set': f'Set {i+1}',\n            'Params': params,\n            'Training AUC': auc_train,\n            'Validation AUC': auc_val\n        })\n\nelse:\n    print(\"  Skipping K-Nearest Neighbors: Insufficient training data.\")\n\n# After trying all parameter sets, store the best model found\nif best_knn_model:\n    models['KNearestNeighbors'] = best_knn_model\n    validation_auc_scores['KNearestNeighbors'] = best_knn_val_auc\n    training_auc_scores['KNearestNeighbors'] = auc_train # Store the training AUC of the best model\n    best_hyperparameters['KNearestNeighbors'] = best_knn_params\n    print(f\"\\n  Best K-Nearest Neighbors setup chosen: {best_knn_params}\")\n\n    # --- Plotting Hyperparameter Performance ---\n    fig, ax = plt.subplots(figsize=(10, 6))\n    \n    set_labels = [res['Set'] for res in knn_results_for_plot]\n    train_aucs = [res['Training AUC'] for res in knn_results_for_plot]\n    val_aucs = [res['Validation AUC'] for res in knn_results_for_plot]\n\n    x = np.arange(len(set_labels))\n    width = 0.35\n\n    rects1 = ax.bar(x - width/2, train_aucs, width, label='Training AUC', color='skyblue')\n    rects2 = ax.bar(x + width/2, val_aucs, width, label='Validation AUC', color='lightcoral')\n\n    ax.set_ylabel('AUC Score')\n    ax.set_title('K-Nearest Neighbors Hyperparameter Performance Comparison')\n    ax.set_xticks(x)\n    ax.set_xticklabels(set_labels)\n    ax.legend()\n    ax.set_ylim(0, 1)\n\n    def autolabel(rects):\n        for rect in rects:\n            height = rect.get_height()\n            ax.annotate(f'{height:.3f}',\n                        xy=(rect.get_x() + rect.get_width() / 2, height),\n                        xytext=(0, 3),\n                        textcoords=\"offset points\",\n                        ha='center', va='bottom')\n\n    autolabel(rects1)\n    autolabel(rects2)\n\n    fig.tight_layout()\n    plt.show()\n\nelse:\n    print(\"\\n  No K-Nearest Neighbors model trained due to insufficient data.\")\n\nprint(\"\\nFinished K-Nearest Neighbors training and comparison.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T04:20:21.451292Z","iopub.execute_input":"2025-05-22T04:20:21.452249Z","iopub.status.idle":"2025-05-22T04:32:16.225123Z","shell.execute_reply.started":"2025-05-22T04:20:21.452218Z","shell.execute_reply":"2025-05-22T04:32:16.224449Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7.3. Naïve Bayes Training and Comparison","metadata":{}},{"cell_type":"code","source":"import warnings\n# Suppress ALL warnings in this specific cell, with more targeted filters for common sklearn warnings\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn.multiclass')\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn.model_selection._validation')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Only one class present in y_true')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Label not .* is present in all training examples')\n\nprint(\"\\n--- Tuning and Training Naïve Bayes (OneVsRest) ---\")\n\n# Define three different sets of hyperparameters to compare manually\n# GaussianNB has fewer hyperparameters, so we'll vary var_smoothing\ngnb_param_sets = [\n    # Set 1: Default var_smoothing\n    {'var_smoothing': 1e-9},\n    # Set 2: Slightly larger var_smoothing\n    {'var_smoothing': 1e-8},\n    # Set 3: Even larger var_smoothing\n    {'var_smoothing': 1e-7},\n]\n\nbest_gnb_val_auc = -1.0\nbest_gnb_model = None\nbest_gnb_params = None\n\ngnb_results_for_plot = []\n\nif X_train.shape[0] > 0 and y_train.shape[0] > 0:\n    for i, params in enumerate(gnb_param_sets):\n        # Removed detailed print for each set\n        \n        # Create the OneVsRestClassifier with the current GaussianNB parameters\n        gnb_clf = OneVsRestClassifier(GaussianNB(**params))\n        \n        # Fit the model on the full training data\n        gnb_clf.fit(X_train, y_train)\n\n        # Calculate Training AUC\n        y_pred_proba_train = gnb_clf.predict_proba(X_train)\n        auc_train = calculate_safe_auc(y_train, y_pred_proba_train, ALL_SPECIES)\n        \n        # Calculate Validation AUC\n        auc_val = 0.0\n        if X_val.shape[0] > 0:\n            y_pred_proba_val = gnb_clf.predict_proba(X_val)\n            auc_val = calculate_safe_auc(y_val, y_pred_proba_val, ALL_SPECIES)\n\n            # Check if this model is the best so far based on Validation AUC\n            if auc_val > best_gnb_val_auc:\n                best_gnb_val_auc = auc_val\n                best_gnb_model = gnb_clf\n                best_gnb_params = params\n        else:\n            if auc_train > best_gnb_val_auc:\n                best_gnb_val_auc = auc_train\n                best_gnb_model = gnb_clf\n                best_gnb_params = params\n                \n        gnb_results_for_plot.append({\n            'Set': f'Set {i+1}',\n            'Params': params,\n            'Training AUC': auc_train,\n            'Validation AUC': auc_val\n        })\n\nelse:\n    print(\"  Skipping Naïve Bayes: Insufficient training data.\")\n\n# After trying all parameter sets, store the best model found\nif best_gnb_model:\n    models['NaiveBayes'] = best_gnb_model\n    validation_auc_scores['NaiveBayes'] = best_gnb_val_auc\n    training_auc_scores['NaiveBayes'] = auc_train # Store the training AUC of the best model\n    best_hyperparameters['NaiveBayes'] = best_gnb_params\n    print(f\"\\n  Best Naïve Bayes setup chosen: {best_gnb_params}\")\n\n    # --- Plotting Hyperparameter Performance ---\n    fig, ax = plt.subplots(figsize=(10, 6))\n    \n    set_labels = [res['Set'] for res in gnb_results_for_plot]\n    train_aucs = [res['Training AUC'] for res in gnb_results_for_plot]\n    val_aucs = [res['Validation AUC'] for res in gnb_results_for_plot]\n\n    x = np.arange(len(set_labels))\n    width = 0.35\n\n    rects1 = ax.bar(x - width/2, train_aucs, width, label='Training AUC', color='skyblue')\n    rects2 = ax.bar(x + width/2, val_aucs, width, label='Validation AUC', color='lightcoral')\n\n    ax.set_ylabel('AUC Score')\n    ax.set_title('Naïve Bayes Hyperparameter Performance Comparison')\n    ax.set_xticks(x)\n    ax.set_xticklabels(set_labels)\n    ax.legend()\n    ax.set_ylim(0, 1)\n\n    def autolabel(rects):\n        for rect in rects:\n            height = rect.get_height()\n            ax.annotate(f'{height:.3f}',\n                        xy=(rect.get_x() + rect.get_width() / 2, height),\n                        xytext=(0, 3),\n                        textcoords=\"offset points\",\n                        ha='center', va='bottom')\n\n    autolabel(rects1)\n    autolabel(rects2)\n\n    fig.tight_layout()\n    plt.show()\n\nelse:\n    print(\"\\n  No Naïve Bayes model trained due to insufficient data.\")\n\nprint(\"\\nFinished Naïve Bayes training and comparison.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T04:42:19.315122Z","iopub.execute_input":"2025-05-22T04:42:19.315441Z","iopub.status.idle":"2025-05-22T04:42:32.760451Z","shell.execute_reply.started":"2025-05-22T04:42:19.315423Z","shell.execute_reply":"2025-05-22T04:42:32.759755Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7.4. Decision Tree Training and Comparison","metadata":{}},{"cell_type":"code","source":"import warnings\n\n# Suppress ALL warnings in this specific cell, with more targeted filters for common sklearn warnings\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn.multiclass')\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn.model_selection._validation')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Only one class present in y_true')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Label not .* is present in all training examples')\n\nprint(\"\\n--- Tuning and Training Decision Tree (OneVsRest) ---\")\n\n# Define three different sets of hyperparameters to compare manually\ndt_param_sets = [\n    # Set 1: Moderately deep tree, default min_samples_leaf\n    {'max_depth': 10, 'min_samples_leaf': 1},\n    # Set 2: Deeper tree, slightly higher min_samples_leaf to prevent overfitting\n    {'max_depth': 15, 'min_samples_leaf': 3},\n    # Set 3: Shallower tree, more samples per leaf\n    {'max_depth': 7, 'min_samples_leaf': 5},\n]\n\nbest_dt_val_auc = -1.0\nbest_dt_model = None\nbest_dt_params = None\n\ndt_results_for_plot = []\n\nif X_train.shape[0] > 0 and y_train.shape[0] > 0:\n    for i, params in enumerate(dt_param_sets):\n        # Removed detailed print for each set\n        \n        # Create the OneVsRestClassifier with the current DecisionTreeClassifier parameters\n        dt_clf = OneVsRestClassifier(DecisionTreeClassifier(random_state=42, **params))\n        \n        # Fit the model on the full training data\n        dt_clf.fit(X_train, y_train)\n\n        # Calculate Training AUC\n        y_pred_proba_train = dt_clf.predict_proba(X_train)\n        auc_train = calculate_safe_auc(y_train, y_pred_proba_train, ALL_SPECIES)\n        \n        # Calculate Validation AUC\n        auc_val = 0.0\n        if X_val.shape[0] > 0:\n            y_pred_proba_val = dt_clf.predict_proba(X_val)\n            auc_val = calculate_safe_auc(y_val, y_pred_proba_val, ALL_SPECIES)\n\n            # Check if this model is the best so far based on Validation AUC\n            if auc_val > best_dt_val_auc:\n                best_dt_val_auc = auc_val\n                best_dt_model = dt_clf\n                best_dt_params = params\n        else:\n            if auc_train > best_dt_val_auc:\n                best_dt_val_auc = auc_train\n                best_dt_model = dt_clf\n                best_dt_params = params\n                \n        dt_results_for_plot.append({\n            'Set': f'Set {i+1}',\n            'Params': params,\n            'Training AUC': auc_train,\n            'Validation AUC': auc_val\n        })\n\nelse:\n    print(\"  Skipping Decision Tree: Insufficient training data.\")\n\n# After trying all parameter sets, store the best model found\nif best_dt_model:\n    models['DecisionTree'] = best_dt_model\n    validation_auc_scores['DecisionTree'] = best_dt_val_auc\n    training_auc_scores['DecisionTree'] = auc_train # Store the training AUC of the best model\n    best_hyperparameters['DecisionTree'] = best_dt_params\n    print(f\"\\n  Best Decision Tree setup chosen: {best_dt_params}\")\n\n    # --- Plotting Hyperparameter Performance ---\n    fig, ax = plt.subplots(figsize=(10, 6))\n    \n    set_labels = [res['Set'] for res in dt_results_for_plot]\n    train_aucs = [res['Training AUC'] for res in dt_results_for_plot]\n    val_aucs = [res['Validation AUC'] for res in dt_results_for_plot]\n\n    x = np.arange(len(set_labels))\n    width = 0.35\n\n    rects1 = ax.bar(x - width/2, train_aucs, width, label='Training AUC', color='skyblue')\n    rects2 = ax.bar(x + width/2, val_aucs, width, label='Validation AUC', color='lightcoral')\n\n    ax.set_ylabel('AUC Score')\n    ax.set_title('Decision Tree Hyperparameter Performance Comparison')\n    ax.set_xticks(x)\n    ax.set_xticklabels(set_labels)\n    ax.legend()\n    ax.set_ylim(0, 1)\n\n    def autolabel(rects):\n        for rect in rects:\n            height = rect.get_height()\n            ax.annotate(f'{height:.3f}',\n                        xy=(rect.get_x() + rect.get_width() / 2, height),\n                        xytext=(0, 3),\n                        textcoords=\"offset points\",\n                        ha='center', va='bottom')\n\n    autolabel(rects1)\n    autolabel(rects2)\n\n    fig.tight_layout()\n    plt.show()\n\nelse:\n    print(\"\\n  No Decision Tree model trained due to insufficient data.\")\n\nprint(\"\\nFinished Decision Tree training and comparison.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T04:43:12.468671Z","iopub.execute_input":"2025-05-22T04:43:12.469274Z","iopub.status.idle":"2025-05-22T04:47:48.422423Z","shell.execute_reply.started":"2025-05-22T04:43:12.469247Z","shell.execute_reply":"2025-05-22T04:47:48.421696Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7.5. Random Forest Training and Comparison","metadata":{}},{"cell_type":"code","source":"import warnings\n\n# Suppress ALL warnings in this specific cell, with more targeted filters for common sklearn warnings\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn.multiclass')\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn.model_selection._validation')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Only one class present in y_true')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Label not .* is present in all training examples')\n\nprint(\"\\n--- Tuning and Training Random Forest Classifier (OneVsRest) ---\")\n\n# Define three different sets of hyperparameters to compare manually\nrf_param_sets = [\n    # Set 1: Moderate number of estimators, good depth, balanced class weight\n    {'n_estimators': 100, 'max_depth': 20, 'min_samples_leaf': 1, 'class_weight': 'balanced'},\n    # Set 2: More estimators, slightly less depth, no class weight\n    {'n_estimators': 150, 'max_depth': 15, 'min_samples_leaf': 2, 'class_weight': None},\n    # Set 3: Fewer estimators, deeper trees, more samples per leaf, balanced class weight\n    {'n_estimators': 75, 'max_depth': 25, 'min_samples_leaf': 3, 'class_weight': 'balanced'},\n]\n\nbest_rf_val_auc = -1.0\nbest_rf_model = None\nbest_rf_params = None\n\nrf_results_for_plot = []\n\nif X_train.shape[0] > 0 and y_train.shape[0] > 0:\n    for i, params in enumerate(rf_param_sets):\n        # Removed detailed print for each set\n        \n        # Create the OneVsRestClassifier with the current RandomForestClassifier parameters\n        rf_clf = OneVsRestClassifier(RandomForestClassifier(random_state=42, **params))\n        \n        # Fit the model on the full training data\n        rf_clf.fit(X_train, y_train)\n\n        # Calculate Training AUC\n        y_pred_proba_train = rf_clf.predict_proba(X_train)\n        auc_train = calculate_safe_auc(y_train, y_pred_proba_train, ALL_SPECIES)\n        \n        # Calculate Validation AUC\n        auc_val = 0.0\n        if X_val.shape[0] > 0:\n            y_pred_proba_val = rf_clf.predict_proba(X_val)\n            auc_val = calculate_safe_auc(y_val, y_pred_proba_val, ALL_SPECIES)\n\n            # Check if this model is the best so far based on Validation AUC\n            if auc_val > best_rf_val_auc:\n                best_rf_val_auc = auc_val\n                best_rf_model = rf_clf\n                best_rf_params = params\n        else:\n            if auc_train > best_rf_val_auc:\n                best_rf_val_auc = auc_train\n                best_rf_model = rf_clf\n                best_rf_params = params\n                \n        rf_results_for_plot.append({\n            'Set': f'Set {i+1}',\n            'Params': params,\n            'Training AUC': auc_train,\n            'Validation AUC': auc_val\n        })\n\nelse:\n    print(\"  Skipping Random Forest: Insufficient training data.\")\n\n# After trying all parameter sets, store the best model found\nif best_rf_model:\n    models['RandomForest'] = best_rf_model\n    validation_auc_scores['RandomForest'] = best_rf_val_auc\n    training_auc_scores['RandomForest'] = auc_train # Store the training AUC of the best model\n    best_hyperparameters['RandomForest'] = best_rf_params\n    print(f\"\\n  Best Random Forest setup chosen: {best_rf_params}\")\n\n    # --- Plotting Hyperparameter Performance ---\n    fig, ax = plt.subplots(figsize=(10, 6))\n    \n    set_labels = [res['Set'] for res in rf_results_for_plot]\n    train_aucs = [res['Training AUC'] for res in rf_results_for_plot]\n    val_aucs = [res['Validation AUC'] for res in rf_results_for_plot]\n\n    x = np.arange(len(set_labels))\n    width = 0.35\n\n    rects1 = ax.bar(x - width/2, train_aucs, width, label='Training AUC', color='skyblue')\n    rects2 = ax.bar(x + width/2, val_aucs, width, label='Validation AUC', color='lightcoral')\n\n    ax.set_ylabel('AUC Score')\n    ax.set_title('Random Forest Hyperparameter Performance Comparison')\n    ax.set_xticks(x)\n    ax.set_xticklabels(set_labels)\n    ax.legend()\n    ax.set_ylim(0, 1)\n\n    def autolabel(rects):\n        for rect in rects:\n            height = rect.get_height()\n            ax.annotate(f'{height:.3f}',\n                        xy=(rect.get_x() + rect.get_width() / 2, height),\n                        xytext=(0, 3),\n                        textcoords=\"offset points\",\n                        ha='center', va='bottom')\n\n    autolabel(rects1)\n    autolabel(rects2)\n\n    fig.tight_layout()\n    plt.show()\n\nelse:\n    print(\"\\n  No Random Forest model trained due to insufficient data.\")\n\nprint(\"\\nFinished Random Forest training and comparison.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T04:52:23.461948Z","iopub.execute_input":"2025-05-22T04:52:23.462255Z","iopub.status.idle":"2025-05-22T05:52:36.178004Z","shell.execute_reply.started":"2025-05-22T04:52:23.462238Z","shell.execute_reply":"2025-05-22T05:52:36.177104Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7.6. Support Vector Machine (LinearSVC) Training and Comparison","metadata":{}},{"cell_type":"code","source":"import warnings\n\n# Suppress ALL warnings in this specific cell, with more targeted filters for common sklearn warnings\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn.multiclass')\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn.model_selection._validation')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Only one class present in y_true')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Label not .* is present in all training examples')\nwarnings.filterwarnings('ignore', category=UserWarning, message='The max_iter was reached') # For potential convergence warnings in LinearSVC\n\nprint(\"\\n--- Tuning and Training Support Vector Machine (LinearSVC) (OneVsRest) ---\")\n\n# Define three different sets of hyperparameters to compare manually for LinearSVC.\n# Note: LinearSVC does not have 'predict_proba', so decision_function scores are used for AUC.\n# 'dual=False' is often preferred when n_samples > n_features, and requires loss='squared_hinge'\n# for 'l2' penalty (which is default).\nsvm_param_sets = [\n    # Set 1: Moderate C, squared_hinge loss (compatible with dual=False)\n    {'C': 1.0, 'loss': 'squared_hinge', 'max_iter': 2000},\n    # Set 2: Lower C (more regularization), squared_hinge loss\n    {'C': 0.5, 'loss': 'squared_hinge', 'max_iter': 2000},\n    # Set 3: Higher C (less regularization), squared_hinge loss\n    {'C': 2.0, 'loss': 'squared_hinge', 'max_iter': 2000},\n]\n\nbest_svm_val_auc = -1.0\nbest_svm_model = None\nbest_svm_params = None\n\nsvm_results_for_plot = []\n\nif X_train.shape[0] > 0 and y_train.shape[0] > 0:\n    for i, params in enumerate(svm_param_sets):\n        # Removed detailed print for each set\n        \n        # Create the OneVsRestClassifier with the current LinearSVC parameters\n        # dual=False is explicitly set to prevent convergence issues with certain solvers/losses\n        svm_clf = OneVsRestClassifier(LinearSVC(random_state=42, dual=False, **params))\n        \n        # Fit the model on the full training data\n        svm_clf.fit(X_train, y_train)\n\n        # Calculate Training AUC using decision_function as LinearSVC doesn't have predict_proba\n        y_pred_scores_train = svm_clf.decision_function(X_train)\n        auc_train = calculate_safe_auc(y_train, y_pred_scores_train, ALL_SPECIES)\n        \n        # Calculate Validation AUC using decision_function\n        auc_val = 0.0\n        if X_val.shape[0] > 0:\n            y_pred_scores_val = svm_clf.decision_function(X_val)\n            auc_val = calculate_safe_auc(y_val, y_pred_scores_val, ALL_SPECIES)\n\n            # Check if this model is the best so far based on Validation AUC\n            if auc_val > best_svm_val_auc:\n                best_svm_val_auc = auc_val\n                best_svm_model = svm_clf\n                best_svm_params = params\n        else:\n            if auc_train > best_svm_val_auc:\n                best_svm_val_auc = auc_train\n                best_svm_model = svm_clf\n                best_svm_params = params\n                \n        svm_results_for_plot.append({\n            'Set': f'Set {i+1}',\n            'Params': params,\n            'Training AUC': auc_train,\n            'Validation AUC': auc_val\n        })\n\nelse:\n    print(\"  Skipping Support Vector Machine: Insufficient training data.\")\n\n# After trying all parameter sets, store the best model found\nif best_svm_model:\n    models['SVM'] = best_svm_model\n    validation_auc_scores['SVM'] = best_svm_val_auc\n    training_auc_scores['SVM'] = auc_train # Store the training AUC of the best model\n    best_hyperparameters['SVM'] = best_svm_params\n    print(f\"\\n  Best Support Vector Machine setup chosen: {best_svm_params}\")\n\n    # --- Plotting Hyperparameter Performance ---\n    fig, ax = plt.subplots(figsize=(10, 6))\n    \n    set_labels = [res['Set'] for res in svm_results_for_plot]\n    train_aucs = [res['Training AUC'] for res in svm_results_for_plot]\n    val_aucs = [res['Validation AUC'] for res in svm_results_for_plot]\n\n    x = np.arange(len(set_labels))\n    width = 0.35\n\n    rects1 = ax.bar(x - width/2, train_aucs, width, label='Training AUC', color='skyblue')\n    rects2 = ax.bar(x + width/2, val_aucs, width, label='Validation AUC', color='lightcoral')\n\n    ax.set_ylabel('AUC Score')\n    ax.set_title('Support Vector Machine Hyperparameter Performance Comparison')\n    ax.set_xticks(x)\n    ax.set_xticklabels(set_labels)\n    ax.legend()\n    ax.set_ylim(0, 1)\n\n    def autolabel(rects):\n        for rect in rects:\n            height = rect.get_height()\n            ax.annotate(f'{height:.3f}',\n                        xy=(rect.get_x() + rect.get_width() / 2, height),\n                        xytext=(0, 3),\n                        textcoords=\"offset points\",\n                        ha='center', va='bottom')\n\n    autolabel(rects1)\n    autolabel(rects2)\n\n    fig.tight_layout()\n    plt.show()\n\nelse:\n    print(\"\\n  No Support Vector Machine model trained due to insufficient data.\")\n\nprint(\"\\nFinished Support Vector Machine training and comparison.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T05:56:20.504489Z","iopub.execute_input":"2025-05-22T05:56:20.504861Z","iopub.status.idle":"2025-05-22T05:56:46.789387Z","shell.execute_reply.started":"2025-05-22T05:56:20.504837Z","shell.execute_reply":"2025-05-22T05:56:46.788222Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7.7. Neural Networks (MLPClassifier) Training and Comparison","metadata":{}},{"cell_type":"code","source":"import warnings\n\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn.multiclass')\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn.model_selection._validation')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Only one class present in y_true')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Label not .* is present in all training examples')\nwarnings.filterwarnings('ignore', category=UserWarning, message='Maximum number of iterations reached before convergence')\n\n\nprint(\"\\n--- Tuning and Training Neural Network (MLPClassifier) ---\")\n\n# Define three different sets of hyperparameters to compare manually\nnn_param_sets = [\n    {'hidden_layer_sizes': (50,), 'activation': 'relu', 'solver': 'adam', 'max_iter': 1000, 'alpha': 0.0001},\n    {'hidden_layer_sizes': (100,), 'activation': 'tanh', 'solver': 'adam', 'max_iter': 1000, 'alpha': 0.0005},\n    {'hidden_layer_sizes': (50, 50), 'activation': 'relu', 'solver': 'adam', 'max_iter': 1000, 'alpha': 0.001},\n]\n\nbest_nn_val_auc = -1.0\nbest_nn_model = None\nbest_nn_params = None\n\nnn_results_for_plot = []\n\nif X_train.shape[0] > 0 and y_train.shape[0] > 0:\n    for i, params in enumerate(nn_param_sets):\n        print(f\"\\n  Trying Neural Network with params (Set {i+1}): {params}\")\n        \n        nn_clf = MLPClassifier(random_state=42, verbose=False, **params)\n        \n        nn_clf.fit(X_train, y_train)\n\n        y_pred_proba_train = nn_clf.predict_proba(X_train)\n        auc_train = calculate_safe_auc(y_train, y_pred_proba_train, ALL_SPECIES)\n        \n        auc_val = 0.0\n        if X_val.shape[0] > 0:\n            y_pred_proba_val = nn_clf.predict_proba(X_val)\n            auc_val = calculate_safe_auc(y_val, y_pred_proba_val, ALL_SPECIES)\n\n            if auc_val > best_nn_val_auc:\n                best_nn_val_auc = auc_val\n                best_nn_model = nn_clf\n                best_nn_params = params\n        else:\n            if auc_train > best_nn_val_auc:\n                best_nn_val_auc = auc_train\n                best_nn_model = nn_clf\n                best_nn_params = params\n                \n        nn_results_for_plot.append({\n            'Set': f'Set {i+1}',\n            'Params': params,\n            'Training AUC': auc_train,\n            'Validation AUC': auc_val\n        })\n\nelse:\n    print(\"  Skipping Neural Network: Insufficient training data.\")\n\nif best_nn_model:\n    models['NeuralNetwork'] = best_nn_model\n    validation_auc_scores['NeuralNetwork'] = best_nn_val_auc\n    training_auc_scores['NeuralNetwork'] = auc_train\n    best_hyperparameters['NeuralNetwork'] = best_nn_params\n    print(f\"\\n  Best Neural Network setup chosen: {best_nn_params}\")\n\n    # --- Plotting Hyperparameter Performance ---\n    fig, ax = plt.subplots(figsize=(10, 6))\n    \n    set_labels = [f\"Set {res['Set']}\\nHidden: {res['Params']['hidden_layer_sizes']}\\nActivation: {res['Params']['activation']}\" for res in nn_results_for_plot]\n    train_aucs = [res['Training AUC'] for res in nn_results_for_plot]\n    val_aucs = [res['Validation AUC'] for res in nn_results_for_plot]\n\n    x = np.arange(len(set_labels))\n    width = 0.35\n\n    rects1 = ax.bar(x - width/2, train_aucs, width, label='Training AUC', color='skyblue')\n    rects2 = ax.bar(x + width/2, val_aucs, width, label='Validation AUC', color='lightcoral')\n\n    ax.set_ylabel('AUC Score')\n    ax.set_title('Neural Network Hyperparameter Performance Comparison')\n    ax.set_xticks(x)\n    ax.set_xticklabels(set_labels, rotation=45, ha='right')\n    ax.legend()\n    ax.set_ylim(0, 1)\n\n    def autolabel(rects):\n        for rect in rects:\n            height = rect.get_height()\n            ax.annotate(f'{height:.3f}',\n                        xy=(rect.get_x() + rect.get_width() / 2, height),\n                        xytext=(0, 3),\n                        textcoords=\"offset points\",\n                        ha='center', va='bottom')\n\n    autolabel(rects1)\n    autolabel(rects2)\n\n    fig.tight_layout()\n    plt.show()\n\nelse:\n    print(\"\\n  No Neural Network model trained due to insufficient data.\")\n\nprint(\"\\nFinished Neural Network training and comparison.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T09:37:00.070175Z","iopub.execute_input":"2025-05-22T09:37:00.070541Z","iopub.status.idle":"2025-05-22T09:44:44.993136Z","shell.execute_reply.started":"2025-05-22T09:37:00.070518Z","shell.execute_reply":"2025-05-22T09:44:44.992244Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7.8. Feature Importance Analysis","metadata":{}},{"cell_type":"code","source":"import warnings\n\n# Suppress common warnings that might arise during plotting or data manipulation\nwarnings.filterwarnings('ignore', category=UserWarning)\nwarnings.filterwarnings('ignore', category=FutureWarning)\n\n\nprint(\"\\n--- Starting Feature Importance Analysis ---\")\nprint(\"This analysis helps understand which MFCC features are most influential for each model.\")\n\n# Define feature names for plotting (MFCC_0, MFCC_1, ..., MFCC_N_MFCC-1)\nfeature_names = [f'MFCC_{i}' for i in range(N_MFCC)]\n\n# Iterate through the trained models and analyze feature importance\nfor model_name, model_obj in models.items():\n    print(f\"\\n--- Analyzing Feature Importance for: {model_name} ---\")\n\n    feature_importances = None\n    \n    # Check for models that have a direct feature_importances_ attribute (tree-based)\n    if hasattr(model_obj, 'estimator') and hasattr(model_obj.estimator, 'feature_importances_'):\n        # For OneVsRestClassifier, feature_importances_ is usually on the base estimator\n\n        if hasattr(model_obj.estimator, 'feature_importances_'):\n            feature_importances = model_obj.estimator.feature_importances_\n        elif hasattr(model_obj, 'estimators_'): # For OneVsRestClassifier with multiple estimators\n            # Average feature importances across all individual estimators\n            importances_list = [est.feature_importances_ for est in model_obj.estimators_ if hasattr(est, 'feature_importances_')]\n            if importances_list:\n                feature_importances = np.mean(importances_list, axis=0)\n\n    # Check for models that have a coef_ attribute (linear models)\n    elif hasattr(model_obj, 'estimator') and hasattr(model_obj.estimator, 'coef_'):\n        # For OneVsRestClassifier, coef_ is usually on the base estimator\n        # Take the mean absolute value of coefficients across all binary estimators\n        if hasattr(model_obj.estimator, 'coef_'):\n            # coef_ can be 1D for binary or 2D for multi-class (one-vs-rest)\n            coefs = model_obj.estimator.coef_\n            if coefs.ndim > 1:\n                feature_importances = np.mean(np.abs(coefs), axis=0)\n            else:\n                feature_importances = np.abs(coefs)\n        elif hasattr(model_obj, 'estimators_'): # For OneVsRestClassifier with multiple estimators\n            # Average absolute coefficients across all individual estimators\n            coefs_list = [np.abs(est.coef_).flatten() for est in model_obj.estimators_ if hasattr(est, 'coef_')]\n            if coefs_list:\n                feature_importances = np.mean(coefs_list, axis=0)\n\n    # Special handling for MLPClassifier which doesn't have OneVsRestClassifier wrapper\n    elif model_name == 'NeuralNetwork' and hasattr(model_obj, 'coefs_'):\n        # Feature importance for NNs is complex. We'll use a simplified approach:\n        # Sum of absolute weights connected to the input layer. This is a very rough estimate.\n        if model_obj.coefs_:\n            # coefs_[0] contains weights from input layer to first hidden layer\n            feature_importances = np.sum(np.abs(model_obj.coefs_[0]), axis=1)\n        print(\"  Note: Feature importance for Neural Networks is complex and this is a simplified view.\")\n\n    if feature_importances is not None and len(feature_importances) == N_MFCC:\n        # Create a pandas Series for easy sorting and handling\n        importance_series = pd.Series(feature_importances, index=feature_names)\n        importance_series = importance_series.sort_values(ascending=False)\n\n        # Plotting the top N features\n        top_n = min(10, len(importance_series))\n        \n        plt.figure(figsize=(12, 7))\n        importance_series.head(top_n).plot(kind='bar', color='skyblue')\n        plt.title(f'Top {top_n} Feature Importances for {model_name}')\n        plt.xlabel('MFCC Feature')\n        plt.ylabel('Importance Score (Normalized)')\n        plt.xticks(rotation=45, ha='right')\n        plt.tight_layout()\n        plt.show()\n    else:\n        print(f\"  Feature importance not directly available or applicable for {model_name} in a straightforward manner.\")\n        print(\"  For models like KNN and Naïve Bayes, feature importance is not directly interpretable from model coefficients/attributes.\")\n        print(\"  For Neural Networks, more advanced techniques (e.g., SHAP, Permutation Importance) are typically required.\")\n\nprint(\"\\n--- Feature Importance Analysis Complete ---\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T09:00:24.060124Z","iopub.execute_input":"2025-05-22T09:00:24.060453Z","iopub.status.idle":"2025-05-22T09:00:24.273734Z","shell.execute_reply.started":"2025-05-22T09:00:24.060433Z","shell.execute_reply":"2025-05-22T09:00:24.272963Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 8. Prediction and Submission File Generation","metadata":{}},{"cell_type":"code","source":"print(\"Starting prediction and submission file generation...\")\n\n# --- Helper function to process a single test soundscape ---\ndef process_test_soundscape(soundscape_path, model, scaler, mlb, sr=SR, duration=DURATION, n_mfcc=N_MFCC):\n\n    segment_predictions = []\n    # Extract the soundscape ID from the filename (e.g., 'soundscape_xxxxxx.ogg' -> 'soundscape_xxxxxx')\n    soundscape_id = os.path.splitext(os.path.basename(soundscape_path))[0]\n\n    try:\n        # Load the entire 1-minute soundscape audio\n        y_soundscape, current_sr = librosa.load(soundscape_path, sr=sr, mono=True)\n\n        # Calculate the number of 5-second segments in the 1-minute soundscape (60 seconds / 5 seconds = 12 segments)\n        num_segments = int(len(y_soundscape) / (duration * sr))\n\n        # Iterate through each 5-second segment\n        for i in range(num_segments):\n            start_sample = i * duration * sr\n            end_sample = (i + 1) * duration * sr\n            segment_audio = y_soundscape[start_sample:end_sample]\n\n            # Ensure segment is exactly 'duration' seconds long by padding if necessary.\n            if len(segment_audio) < duration * sr:\n                segment_audio = np.pad(segment_audio, (0, (duration * sr) - len(segment_audio)), 'constant')\n\n            # --- Extract features for the current 5-second segment ---\n            def _extract_features_from_array(y_array, sr_val, n_mfcc_val):\n                \"\"\"Helper to extract MFCCs from an audio array.\"\"\"\n                mfccs = librosa.feature.mfcc(y=y_array, sr=sr_val, n_mfcc=n_mfcc_val)\n                return np.mean(mfccs.T, axis=0)\n\n            segment_features = _extract_features_from_array(segment_audio, sr, n_mfcc)\n            \n            # Scale the extracted features using the StandardScaler fitted on training data.\n            scaled_features = scaler.transform(segment_features.reshape(1, -1))\n\n            # Make predictions (probabilities) for the current segment using the trained model.\n            if hasattr(model, 'predict_proba'):\n                segment_proba = model.predict_proba(scaled_features)[0]\n            elif hasattr(model, 'decision_function'):\n\n                # Apply sigmoid to map scores to a [0, 1] range.\n                decision_scores = model.decision_function(scaled_features)[0]\n                segment_proba = 1 / (1 + np.exp(-decision_scores))\n            else:\n                # Fallback if neither predict_proba nor decision_function is available\n                print(f\"Warning: Model {type(model).__name__} does not have predict_proba or decision_function. Using dummy probabilities.\")\n                segment_proba = np.full(len(mlb.classes_), 1.0 / len(mlb.classes_))\n\n\n            # Create the 'row_id' for this segment, following the competition's format:\n            end_time = (i + 1) * duration # End time of the current 5-second segment\n            row_id = f\"{soundscape_id}_{end_time}\"\n\n            # Create a dictionary to hold the row_id and predicted probabilities for all species.\n            segment_pred_dict = {'row_id': row_id}\n            # Map probabilities back to species names using the MultiLabelBinarizer's classes.\n            for species_idx, species_name in enumerate(mlb.classes_):\n                segment_pred_dict[species_name] = segment_proba[species_idx]\n            \n            segment_predictions.append(segment_pred_dict)\n\n    except Exception as e:\n        # If any error occurs during processing a soundscape, print a warning and return an empty DataFrame.\n        print(f\"Error processing soundscape {soundscape_path}: {e}\")\n        return pd.DataFrame() # Return empty DataFrame on error\n\n    return pd.DataFrame(segment_predictions)\n\n\nprint(\"Prediction and submission file generation process starting.\")\n\n# --- Select the best model for submission ---\nif validation_auc_scores:\n    best_model_name = max(validation_auc_scores, key=validation_auc_scores.get)\n    submission_model = models[best_model_name]\n    print(f\"\\nSelected '{best_model_name}' as the submission model based on validation AUC.\")\nelse:\n    # Fallback if no models were trained successfully (e.g., due to insufficient data in Cell 6/7).\n    print(\"\\nWarning: No models were trained or evaluated successfully. Cannot select a best model.\")\n    print(\"Creating a dummy submission file with default probabilities.\")\n    submission_model = None # Indicate no model is available\n\n# --- Process test soundscapes and generate submission file ---\nall_submission_rows = []\n\n\n# List all actual .ogg files in the TEST_SOUNDSCAPES_PATH\ntest_soundscape_files = [f for f in os.listdir(TEST_SOUNDSCAPES_PATH) if f.endswith('.ogg')]\n\nif submission_model is not None and test_soundscape_files:\n    print(f\"\\nProcessing {len(test_soundscape_files)} test soundscapes for predictions...\")\n    for soundscape_file in test_soundscape_files:\n        full_soundscape_path = os.path.join(TEST_SOUNDSCAPES_PATH, soundscape_file)\n        \n        # Process each soundscape using our helper function\n        soundscape_df = process_test_soundscape(full_soundscape_path, submission_model, scaler, mlb, SR, DURATION, N_MFCC)\n        \n        if not soundscape_df.empty:\n            all_submission_rows.append(soundscape_df)\n        else:\n            print(f\"Skipping {soundscape_file} due to processing error or no predictions generated.\")\n\n    if all_submission_rows:\n        final_submission_df = pd.concat(all_submission_rows, ignore_index=True)\n    else:\n        print(\"Warning: No predictions generated from any test soundscapes. Creating empty submission.\")\n        final_submission_df = sample_submission_df.iloc[0:0].copy() # Create empty df with correct columns\n        # Ensure it has all species columns if it's truly empty\n        for species_col in ALL_SPECIES:\n            if species_col not in final_submission_df.columns:\n                final_submission_df[species_col] = [] # Add empty list for column\n\nelse:\n    # Fallback if no model was trained or no test soundscapes found\n    print(\"\\nNo trained model or test soundscapes found. Creating a dummy submission file with default probabilities.\")\n    # This part assumes sample_submission_df was loaded successfully in Cell 3\n    final_submission_df = sample_submission_df.copy()\n    if len(ALL_SPECIES) > 0:\n        for species_col in ALL_SPECIES:\n            if species_col != 'row_id': # Ensure not to overwrite row_id\n                final_submission_df[species_col] = 1.0 / len(ALL_SPECIES)\n    else:\n        print(\"Warning: ALL_SPECIES is empty, cannot fill submission with probabilities.\")\n        final_submission_df = pd.DataFrame(columns=['row_id'])\n\n\n# Save the submission file\nsubmission_file_name = 'submission.csv'\nfinal_submission_df.to_csv(submission_file_name, index=False)\n\nprint(f\"\\nSubmission file '{submission_file_name}' created successfully.\")\nprint(\"\\nFirst 5 rows of the generated submission file:\")\nprint(final_submission_df.head())\nprint(f\"\\nShape of the submission file: {final_submission_df.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T09:00:58.518772Z","iopub.execute_input":"2025-05-22T09:00:58.519047Z","iopub.status.idle":"2025-05-22T09:00:58.640475Z","shell.execute_reply.started":"2025-05-22T09:00:58.519030Z","shell.execute_reply":"2025-05-22T09:00:58.639391Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 9. Saving Best Model for Faster Approach","metadata":{}},{"cell_type":"code","source":"import joblib\nimport os\n\nprint(\"\\n--- Saving the Best Performing Scikit-learn Model from Homework Notebook ---\")\n\n# Ensure the 'models' and 'validation_auc_scores' dictionaries are populated.\n# These dictionaries are expected to be available after running all previous training cells (7.1-7.8).\nif 'models' in locals() and 'validation_auc_scores' in locals() and validation_auc_scores:\n    # Find the best model based on validation AUC from the models that were successfully trained.\n    best_model_name = max(validation_auc_scores, key=validation_auc_scores.get)\n    best_model_instance = models[best_model_name]\n\n    # Define the save path for the scikit-learn model.\n    # We use a .pkl extension which is standard for joblib-saved scikit-learn models.\n    save_path = 'best_sklearn_model.pkl' \n\n    try:\n        # Save the scikit-learn model using joblib.\n        # This serializes the entire model object to a file.\n        joblib.dump(best_model_instance, save_path)\n        print(f\"Successfully saved the best model ('{best_model_name}') to {save_path}\")\n        print(\"\\nNEXT STEP: You must now add this 'best_sklearn_model.pkl' file as an input dataset\")\n        print(\"to your 'BirdCLEF_Kaggle.ipynb' (Code 2) notebook on Kaggle.\")\n        print(\"Then, we will modify Code 2 to load this scikit-learn model and perform predictions.\")\n    except Exception as e:\n        print(f\"Error saving the model: {e}\")\n        print(\"Please ensure you have write permissions in the current directory.\")\nelse:\n    print(\"No models found or evaluated in 'models' dictionary. Cannot save a best model.\")\n    print(\"Please ensure all model training cells (7.1-7.8) have run successfully.\")\n\nprint(\"\\n--- Model Saving Process Complete ---\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}