{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":11855938,"sourceType":"datasetVersion","datasetId":7439602},{"sourceId":11944808,"sourceType":"datasetVersion","datasetId":7438986}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install /kaggle/input/birdclef-dependencies/audiomentations-0.33.0-py3-none-any.whl\n!pip install /kaggle/input/birdclef-dependencies/noisereduce-3.0.3-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T13:13:44.377697Z","iopub.execute_input":"2025-05-25T13:13:44.377851Z","iopub.status.idle":"2025-05-25T13:13:51.666969Z","shell.execute_reply.started":"2025-05-25T13:13:44.377835Z","shell.execute_reply":"2025-05-25T13:13:51.666298Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Import Modules","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport librosa\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torchaudio\nimport os\nimport random\nimport shutil\nimport pickle\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport scipy.signal\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_curve, auc, confusion_matrix, precision_score, recall_score, f1_score, roc_auc_score\nfrom tqdm.notebook import tqdm\nfrom audiomentations import Compose, AddGaussianNoise, PitchShift, TimeStretch, Gain\nfrom torch.amp import autocast, GradScaler\nimport noisereduce as nr\nimport timm\nimport psutil\nimport warnings\nimport glob\nimport time\nimport gc\nimport sys\nfrom joblib import Parallel, delayed\nfrom math import radians, sin, cos, sqrt, atan2\nfrom noisereduce import reduce_noise\n\n# Suppress warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T13:13:51.668480Z","iopub.execute_input":"2025-05-25T13:13:51.668748Z","iopub.status.idle":"2025-05-25T13:14:03.839380Z","shell.execute_reply.started":"2025-05-25T13:13:51.668718Z","shell.execute_reply":"2025-05-25T13:14:03.838837Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"class Config:\n    train_dir = \"/kaggle/input/birdclef-2025/train_audio\"\n    train_csv = \"/kaggle/input/birdclef-2025/train.csv\"\n    taxonomy_csv = \"/kaggle/input/birdclef-2025/taxonomy.csv\" \n    sample_csv = \"/kaggle/input/birdclef-2025/sample_submission.csv\"\n    weights = {'tf_efficientnet_b3': \"/kaggle/input/efficientnet-weights/tf_efficientnet_b3.pth\"}\n    feature_dir = \"/kaggle/working/features\"\n    model_weights_dir = \"/kaggle/working/model_weights\"\n    checkpoint_dir = \"/kaggle/working/checkpoints\"\n    sr = 32000\n    num_classes = 206\n    n_folds = 5\n    n_fft = 1024\n    hop_length = 512\n    n_mels = 128\n    fmin = 20\n    fmax = 16000\n    chunk_duration = 20  # Default chunk duration\n    short_chunk_duration = 20\n    batch_size = 32 # ปรับเป็น 32 หรือ 64 ได้\n    epochs = 5\n    lr = 1e-4\n    device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n    early_stop_patience = 2\n    dropout = 0.05\n    label_smoothing = 0.0\n    seed = 42\n    eval_timeout = 300\n    expected_frames = 641\n    n_jobs = 2\n    ref_lat = 6.76\n    ref_lon = -74.21\n    max_distance_km = 1200\n    oversample_threshold = 20\n    species_rating_thresholds = {\n        'default': 3.5,\n        '1139490': 2, '1192948': 2, '1194042': 2, '1346504': 2, '1462711': 2,\n        '135045': 2, '1564122': 2, '1462737': 2, '21038': 2, '21116': 2,\n        '42113': 2, '47067': 2, '42087': 2, '42007': 2, '41970': 2,\n        '24292': 2, '52884': 2, '528041': 2, '50186': 2, '523060': 2,\n        '48124': 2, '476537': 2, '787625': 2, '81930': 2, '65419': 2,\n        '67082': 2, '66578': 2, '66531': 2, '65336': 2, '64862': 2,\n        '555142': 2, '548639': 2, '715170': 2, '714022': 2, '963335': 2,\n        '868458': 2, '41663': 2, '517119': 2, '65448': 2, '566513': 2,\n        '65962': 2, '22976': 2, '22333': 2, '65344': 2, '65349': 2,\n        '65547': 2, '126247': 2, '65373': 2, '46010': 2, '476538': 2,\n        '22973': 2, '134933': 2, 'amekes': 2, '24272': 2, 'verfly': 2,\n        '66893': 2, 'babwar': 2, 'colara1': 2, 'compot1': 2, 'turvul': 2,\n        'orcpar': 2, 'bbwduc': 2, 'compau': 2, 'grekis': 2\n    } ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T20:57:32.508996Z","iopub.execute_input":"2025-05-25T20:57:32.509358Z","iopub.status.idle":"2025-05-25T20:57:32.518452Z","shell.execute_reply.started":"2025-05-25T20:57:32.509333Z","shell.execute_reply":"2025-05-25T20:57:32.517755Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Helper Functions","metadata":{}},{"cell_type":"code","source":"# Helper Functions\ndef set_seed(seed=Config.seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    if torch.cuda.is_available():\n        torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\ndef get_disk_usage(path=\"/kaggle/working\"):\n    total_size = sum(os.path.getsize(os.path.join(dirpath, f)) for dirpath, _, filenames in os.walk(path) for f in filenames)\n    return total_size / (1024**3)\n\ndef haversine_distance(lat1, lon1, lat2, lon2):\n    R = 6371  # Earth's radius in km\n    lat1, lon1, lat2, lon2 = map(radians, [lat1, lon1, lat2, lon2])\n    dlat = lat2 - lat1\n    dlon = lon2 - lon1\n    a = sin(dlat/2)**2 + cos(lat1) * cos(lat2) * sin(dlon/2)**2\n    c = 2 * atan2(sqrt(a), sqrt(1-a))\n    distance = R * c\n    return distance\n\ndef save_checkpoint(model_name, fold, state, model, optimizer):\n    os.makedirs(Config.checkpoint_dir, exist_ok=True)\n    checkpoint_path = os.path.join(Config.checkpoint_dir, f\"checkpoint_{model_name}_fold_{fold}.pkl\")\n    state.update({\n        'model_state': model.state_dict(),\n        'optimizer_state': optimizer.state_dict()\n    })\n    with open(checkpoint_path, 'wb') as f:\n        pickle.dump(state, f)\n    print(f\"Saved checkpoint: {checkpoint_path}\")\n\ndef load_checkpoint(model_name, fold):\n    checkpoint_path = os.path.join(Config.checkpoint_dir, f\"checkpoint_{model_name}_fold_{fold}.pkl\")\n    if os.path.exists(checkpoint_path):\n        with open(checkpoint_path, 'rb') as f:\n            print(f\"Loaded checkpoint: {checkpoint_path}\")\n            return pickle.load(f)\n    return None\n\n# Audio Augmentation\ntrain_augment = Compose([\n    AddGaussianNoise(min_amplitude=0.001, max_amplitude=0.015, p=0.5),\n    PitchShift(min_semitones=-4, max_semitones=4, p=0.5),\n    TimeStretch(min_rate=0.8, max_rate=1.25, p=0.5),\n    Gain(min_gain_in_db=-6, max_gain_in_db=6, p=0.5)\n])\ndef spec_augment(spec, max_time_mask=0.2, max_freq_mask=0.2):\n    spec = torch.tensor(spec, dtype=torch.float32)\n    time_steps, freq_bins = spec.shape[-1], spec.shape[-2]\n    for _ in range(3):\n        t = int(random.uniform(0, max_time_mask) * time_steps)\n        t0 = random.randint(0, time_steps - t)\n        spec[:, :, t0:t0+t] = 0\n    for _ in range(3):\n        f = int(random.uniform(0, max_freq_mask) * freq_bins)\n        f0 = random.randint(0, freq_bins - f)\n        spec[:, f0:f0+f, :] = 0\n    return spec.numpy()\n\ndef mixup(x, y, alpha=0.5):\n    lam = np.random.beta(alpha, alpha)\n    index = torch.randperm(x.size(0)).to(x.device)\n    mixed_x = lam * x + (1 - lam) * x[index]\n    return mixed_x, y, y[index], lam\n\n# Set random seed\nset_seed()\nprint(f\"[INFO] Set Seed: {Config.seed}\")\nprint(\"Initial setup complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T18:44:45.305225Z","iopub.execute_input":"2025-05-25T18:44:45.306019Z","iopub.status.idle":"2025-05-25T18:44:45.321332Z","shell.execute_reply.started":"2025-05-25T18:44:45.305975Z","shell.execute_reply":"2025-05-25T18:44:45.320747Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Loading and Initial Analysis","metadata":{}},{"cell_type":"code","source":"def load_and_analyze_data(train_csv_path, train_dir, data_df_path='/kaggle/working/train_data.csv'):\n    print(\"Starting BirdCLEF 2025 pipeline\")\n    print(\"Preprocessing training data\")\n\n    # Load data\n    if os.path.exists(data_df_path):\n        print(f\"Loading preprocessed data from {data_df_path}\")\n        data_df = pd.read_csv(data_df_path)\n    else:\n        print(\"Loading data from CSV...\")\n        data_df = pd.read_csv(train_csv_path)\n    \n    # Fix filename paths\n    data_df['filename'] = data_df['filename'].str.strip().replace('\\\\', '/')\n    data_df['filename'] = data_df['filename'].apply(lambda x: os.path.join(train_dir, x))\n    print(\"Data loaded. Initial dataset size:\", len(data_df))\n\n    # Analyze ratings\n    rating_stats = data_df['rating'].describe()\n    print(f\"\\nRating Statistics:\")\n    print(f\"Count: {int(rating_stats['count'])} files\")\n    print(f\"Mean: {rating_stats['mean']:.1f}\")\n    print(f\"Std: {rating_stats['std']:.1f}\")\n    print(f\"Min: {rating_stats['min']}\")\n    print(f\"25%: {rating_stats['25%']}\")\n    print(f\"50% (Median): {rating_stats['50%']}\")\n    print(f\"75%: {rating_stats['75%']}\")\n    print(f\"Max: {rating_stats['max']}\")\n\n    # Analyze species distribution\n    total_species = data_df['primary_label'].nunique()\n    species_counts = data_df['primary_label'].value_counts()\n    most_frequent_species = species_counts.idxmax()\n    most_frequent_count = species_counts.max()\n    least_frequent_species = species_counts.idxmin()\n    least_frequent_count = species_counts.min()\n\n    low_rated = data_df[data_df['rating'] <= 2]\n    grouped_total = data_df.groupby('primary_label').size()\n    grouped_low = low_rated.groupby('primary_label').size()\n    rating_stats = pd.DataFrame({'total': grouped_total, 'low_rated': grouped_low}).fillna(0)\n    rating_stats['low_ratio'] = rating_stats['low_rated'] / rating_stats['total']\n    low_rated_species = rating_stats[rating_stats['low_ratio'] > 0.5]\n\n    print(f\"\\nTotal species: {total_species}\")\n    print(f\"Most frequent species: {most_frequent_species} with {most_frequent_count} files\")\n    print(f\"Least frequent species: {least_frequent_species} with {least_frequent_count} files\")\n    print(f\"\\nTotal species with >50% low-rated files (0–2): {len(low_rated_species)} species\")\n\n    example = low_rated_species.sort_values('low_ratio', ascending=False).iloc[0]\n    example_species = example.name\n    example_percent = example['low_ratio'] * 100\n    example_low = int(example['low_rated'])\n    example_total = int(example['total'])\n    print(f\"Example: {example_species} with {example_percent:.0f}% low-rated files ({example_low}/{example_total})\")\n\n    low_rated_species_sorted = low_rated_species.sort_values(by='low_ratio', ascending=False)\n    all_species = low_rated_species_sorted.head(64)\n    print(\"\\nAll Low-Rated Species (64 Species):\")\n    for species, row in all_species.iterrows():\n        print(f\"- {species}: {int(row['low_rated'])}/{int(row['total'])} files ({row['low_ratio']*100:.1f}%)\")\n\n    return data_df\n\n# Run the function\ndata_df = load_and_analyze_data(Config.train_csv, Config.train_dir)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T13:14:03.952620Z","iopub.execute_input":"2025-05-25T13:14:03.952820Z","iopub.status.idle":"2025-05-25T13:14:04.205206Z","shell.execute_reply.started":"2025-05-25T13:14:03.952804Z","shell.execute_reply":"2025-05-25T13:14:04.204574Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Audio Duration Analysis","metadata":{}},{"cell_type":"code","source":"def analyze_audio_durations(data_df, sr, min_duration = 1.0, max_duration=120, sample_size=100):\n    print(\"\\nAnalyzing audio durations (sample)...\")\n    durations_sample = []\n    for idx, row in data_df.sample(sample_size).iterrows():\n        try:\n            audio, _ = librosa.load(row['filename'], sr=sr)\n            duration = librosa.get_duration(y=audio, sr=sr)\n            durations_sample.append(duration)\n        except Exception as e:\n            print(f\"Error loading sample file {row['filename']}: {e}\")\n            durations_sample.append(0)\n\n    d_df_sample = pd.DataFrame(columns=[\"durations\"], data=durations_sample)\n    plt.figure(figsize=(10, 5))\n    plt.title(\"Distribution of audio lengths (sample)\")\n    sns.histplot(d_df_sample, x=\"durations\", bins=100)\n    plt.show()\n    print(\"Sample Duration Statistics:\")\n    print(d_df_sample.describe())\n\n    print(\"\\nCalculating audio durations (full dataset)...\")\n    durations = []\n    for idx, row in tqdm(data_df.iterrows(), total=len(data_df), desc=\"Computing durations\"):\n        try:\n            duration = librosa.get_duration(path=row['filename'], sr=sr)\n            durations.append(duration)\n        except Exception as e:\n            print(f\"Error loading file {row['filename']}: {e}\")\n            durations.append(0)\n\n    d_df = pd.DataFrame(columns=[\"durations\"], data=durations)\n    print(\"\\nFull Dataset Duration Statistics:\")\n    print(d_df.describe())\n\n    data_df['duration'] = durations\n    data_df = data_df[data_df['duration'] <= max_duration]\n    data_df = data_df[data_df['duration'] >= min_duration]\n    print(f\"Filtered out files shorter than {min_duration} and longer than {max_duration} seconds. New dataset size: {len(data_df)}\")\n    return data_df\n\n# Run the function\ndata_df = analyze_audio_durations(data_df, Config.sr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T13:14:04.205864Z","iopub.execute_input":"2025-05-25T13:14:04.206151Z","iopub.status.idle":"2025-05-25T13:18:55.662538Z","shell.execute_reply.started":"2025-05-25T13:14:04.206131Z","shell.execute_reply":"2025-05-25T13:18:55.661949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add sampling weights based on rating and class frequency\nclass_counts = data_df['primary_label'].value_counts()\ndata_df['class_weight'] = data_df['primary_label'].map(lambda x: 1.0 / class_counts.get(x, 1.0))\ndata_df['sampling_weight'] = data_df['rating'].apply(lambda x: max(0.1, x / 5.0)) * data_df['class_weight']\nprint(\"Added sampling weights to DataFrame.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T13:18:55.663389Z","iopub.execute_input":"2025-05-25T13:18:55.663665Z","iopub.status.idle":"2025-05-25T13:18:55.781296Z","shell.execute_reply.started":"2025-05-25T13:18:55.663636Z","shell.execute_reply":"2025-05-25T13:18:55.780734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load taxonomy data\ntaxonomy_df = pd.read_csv(Config.taxonomy_csv)\nprint(\"\\nTaxonomy Data:\")\nprint(taxonomy_df.head().to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T13:18:55.781894Z","iopub.execute_input":"2025-05-25T13:18:55.782095Z","iopub.status.idle":"2025-05-25T13:18:55.792791Z","shell.execute_reply.started":"2025-05-25T13:18:55.782074Z","shell.execute_reply":"2025-05-25T13:18:55.792290Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Extraction","metadata":{}},{"cell_type":"code","source":"# Audio Filtering Functions\ndef compute_spectral_features(audio, sr=Config.sr):\n    freqs, times, Sxx = scipy.signal.spectrogram(audio, fs=sr, nperseg=Config.n_fft, noverlap=Config.hop_length)\n    return freqs, times, Sxx\n\ndef compute_spectral_features_extended(audio, sr=Config.sr):\n    spec_centroid = librosa.feature.spectral_centroid(y=audio, sr=sr, n_fft=Config.n_fft, hop_length=Config.hop_length)[0]\n    spec_bandwidth = librosa.feature.spectral_bandwidth(y=audio, sr=sr, n_fft=Config.n_fft, hop_length=Config.hop_length)[0]\n    spec_flatness = librosa.feature.spectral_flatness(y=audio, n_fft=Config.n_fft, hop_length=Config.hop_length)[0]\n    zcr = librosa.feature.zero_crossing_rate(y=audio, hop_length=Config.hop_length)[0]\n    return {\n        'spectral_centroid': np.mean(spec_centroid),\n        'spectral_bandwidth': np.mean(spec_bandwidth),\n        'spectral_flatness': np.mean(spec_flatness),\n        'zero_crossing_rate': np.mean(zcr)\n    }\n\ndef get_animal_category(primary_label, taxonomy_df):\n    category = \"unknown\"\n    if primary_label in taxonomy_df['primary_label'].values:\n        class_value = taxonomy_df[taxonomy_df['primary_label'] == primary_label]['class_name'].iloc[0].lower()\n        class_to_category = {\n            'insecta': 'insect',\n            'amphibia': 'amphibian',\n            'aves': 'bird',\n            'mammalia': 'mammal'\n        }\n        category = class_to_category.get(class_value, \"unknown\")\n    return category\n\ndef has_significant_animal_sound(audio, sr=Config.sr, freqs=None, Sxx=None, indicators=None, animal_category=\"unknown\"):\n    if freqs is None or Sxx is None or indicators is None:\n        freqs, _, Sxx = compute_spectral_features(audio, sr)\n        indicators = compute_spectral_features_extended(audio)\n    \n    animal_freq_mask = (freqs >= 50) & (freqs <= 12000)\n    animal_energy = np.mean(Sxx[animal_freq_mask]) if np.any(animal_freq_mask) else 0\n    \n    centroid_min, centroid_max = 500, 15000\n    flatness_max = 0.9\n    bandwidth_min = 300\n    \n    is_animal = (\n        centroid_min <= indicators['spectral_centroid'] <= centroid_max and\n        indicators['spectral_bandwidth'] >= bandwidth_min and\n        indicators['spectral_flatness'] <= flatness_max and\n        animal_energy > 1e-12\n    )\n    return is_animal, animal_energy\n\ndef apply_noise_reduction(audio, sample_rate, noise_clip_duration=0.5):\n    noise_clip_samples = int(noise_clip_duration * sample_rate)\n    \n    if len(audio) > noise_clip_samples:\n        noise_clip = audio[:noise_clip_samples]\n        try:\n            reduced_audio = nr.reduce_noise(\n                y=audio,\n                sr=sample_rate,\n                y_noise=noise_clip,\n                prop_decrease=0.3,  # Reduce intensity\n            )\n            return reduced_audio if reduced_audio is not None else audio.copy()\n        except Exception as e:\n            print(f\"Error in reduce_noise: {str(e)}\")\n            return audio.clone()\n    return audio.clone()\n    \ndef extract_features(audio, sr=Config.sr, n_mels=Config.n_mels, hop_length=Config.hop_length, audio_path=\"unknown\"):\n    try:\n        # Ensure audio is 1D (remove channel if present)\n        if len(audio.shape) > 1:\n            audio = audio.squeeze(0)  # Remove channel dimension if it exists\n        \n        transform = torchaudio.transforms.MelSpectrogram(\n            sample_rate=sr, n_fft=Config.n_fft, hop_length=hop_length,\n            n_mels=n_mels, f_min=Config.fmin, f_max=Config.fmax\n        ).to(Config.device)\n        audio_tensor = torch.tensor(audio, dtype=torch.float32).to(Config.device)\n        mel_sp = transform(audio_tensor)  # Shape should be (n_mels, time_frames)\n        \n        if torch.all(mel_sp <= 1e-10):\n            print(f\"Mel spectrogram all zeros for {audio_path}\")\n            return None\n        \n        mel_sp = torchaudio.transforms.AmplitudeToDB(top_db=80)(mel_sp).cpu().numpy()\n        mel_sp = np.clip(mel_sp, a_min=-80, a_max=0)\n        if np.all(mel_sp == -80):\n            print(f\"Mel spectrogram clipped to -80 for {audio_path}\")\n            return None\n        \n        # Calculate expected time frames\n        time_frames = mel_sp.shape[1]\n        if time_frames < Config.expected_frames:\n            mel_sp = np.pad(mel_sp, ((0, 0), (0, Config.expected_frames - time_frames)), mode='constant', constant_values=-80)\n        elif time_frames > Config.expected_frames:\n            mel_sp = mel_sp[:, :Config.expected_frames]  # Trim to expected frames\n        \n        if mel_sp.shape != (n_mels, Config.expected_frames):\n            print(f\"Unexpected Mel shape for {audio_path}: {mel_sp.shape}, expected ({n_mels}, {Config.expected_frames})\")\n            return None\n        \n        mel_sp = np.nan_to_num(mel_sp, nan=-80, posinf=0, neginf=-80)\n        return mel_sp\n    except Exception as e:\n        print(f\"Error in extract_features for {audio_path}: {e}\")\n        return None\n    finally:\n        gc.collect()\n        torch.cuda.empty_cache() if Config.device == \"cuda\" else None\n\ndef transform_features(mel_sp, audio_path=\"unknown\"):\n    try:\n        if mel_sp is None or mel_sp.shape != (Config.n_mels, Config.expected_frames):\n            print(f\"Invalid Mel shape in transform_features for {audio_path}: {mel_sp.shape if mel_sp is not None else None}\")\n            return None\n        \n        if np.all(mel_sp == mel_sp[0, 0]):\n            print(f\"Mel spectrogram uniform for {audio_path}\")\n            return None\n        \n        mel_sp_mean = np.mean(mel_sp)\n        mel_sp_std = np.std(mel_sp)\n        if mel_sp_std < 1e-10:\n            print(f\"Mel spectrogram std too small for {audio_path}\")\n            return None\n        mel_sp = (mel_sp - mel_sp_mean) / (mel_sp_std + 1e-10)  # Standardize\n        \n        mel_delta = librosa.feature.delta(mel_sp)\n        mel_delta2 = librosa.feature.delta(mel_sp, order=2)\n        mel_delta = np.nan_to_num(mel_delta, nan=0.0, posinf=0.0, neginf=0.0)\n        mel_delta2 = np.nan_to_num(mel_delta2, nan=0.0, posinf=0.0, neginf=0.0)\n        features = np.stack([mel_sp, mel_delta, mel_delta2], axis=0)\n        return features\n    except Exception as e:\n        print(f\"Error in transform_features for {audio_path}: {e}\")\n        return None\n    finally:\n        gc.collect()\n\ndef process_audio(row, mode=\"train\", batch_dir=Config.feature_dir, taxonomy_df=None):\n    idx, row = row\n    audio_path = row['filename']\n    feature_path = os.path.join(batch_dir, f\"{idx}_{mode}.npz\")\n    \n    # ย่อ path ให้แสดงเฉพาะชื่อไฟล์\n    audio_filename = os.path.basename(audio_path)\n    \n    # Check memory usage\n    mem = psutil.virtual_memory()\n    if mem.percent > 90:\n        # print(f\"High memory usage {mem.percent}%, skipping {audio_filename}\")\n        return None\n    \n    if not os.path.exists(audio_path):\n        # print(f\"Audio file not found: {audio_filename}\")\n        return None\n    \n    try:\n        # Load audio\n        audio, sr = torchaudio.load(audio_path)\n        audio = audio[0].numpy()  # Remove channel dimension\n        # print(f\"Loaded audio {audio_filename}: length={len(audio)}, sample_rate={sr}\")\n        \n        # Resample if needed\n        if sr != Config.sr:\n            audio = librosa.resample(audio, orig_sr=sr, target_sr=Config.sr)\n        \n        # Clean up audio\n        audio = np.nan_to_num(audio, nan=0.0, posinf=0.0, neginf=0.0)\n        \n        # Compute spectral features once for filtering\n        freqs, _, Sxx = compute_spectral_features(audio, Config.sr)\n        primary_label = row.get('primary_label', '')\n        animal_category = get_animal_category(primary_label, taxonomy_df)\n        indicators = compute_spectral_features_extended(audio)\n        \n        # Filter based on animal sound and rating\n        is_animal, animal_energy = has_significant_animal_sound(audio, Config.sr, freqs, Sxx, indicators, animal_category)\n        rating_threshold = Config.species_rating_thresholds.get(primary_label, Config.species_rating_thresholds['default'])\n        \n        time_condition = 'dawn' in row.get('time', '').lower() or 'morning' in row.get('time', '').lower()\n        if not is_animal and float(row.get('rating', 0)) < rating_threshold and not time_condition:\n            # print(f\"Audio {audio_filename} filtered out: is_animal={is_animal}, rating={row.get('rating', 0)}, energy={animal_energy:.2e}\")\n            return None\n        \n        # Apply noise reduction only if rating is low and necessary\n        if mode == \"train\" and primary_label in Config.species_rating_thresholds and Config.species_rating_thresholds[primary_label] == 2 and float(row.get('rating', 0)) < 2.5:\n            # print(f\"Applying noise reduction to {audio_filename}...\")\n            audio = apply_noise_reduction(audio, sample_rate=Config.sr, noise_clip_duration=0.5)\n        \n        # Normalize audio\n        audio_max = np.max(np.abs(audio))\n        if audio_max > 0:\n            audio = audio / audio_max\n        else:\n            # print(f\"Audio {audio_filename} has max amplitude 0\")\n            return None\n        \n        # Apply augmentation if training\n        if mode == \"train\":\n            audio = train_augment(samples=audio, sample_rate=Config.sr)\n        \n        # Trim or pad audio to target length\n        target_length = Config.chunk_duration * Config.sr  # 20 seconds * 32000 Hz = 640000 samples\n        if len(audio) > target_length:\n            audio = audio[:target_length]\n        else:\n            audio = np.pad(audio, (0, target_length - len(audio)), mode='constant')\n        \n        # Extract features\n        print(f\"Extracting features from {audio_filename}...\")\n        mel_sp = extract_features(audio, n_mels=Config.n_mels, audio_path=audio_path)\n        if mel_sp is None:\n            # print(f\"Failed to extract Mel spectrogram for {audio_filename}\")\n            return None\n        \n        # Transform features\n        features = transform_features(mel_sp, audio_path=audio_path)\n        if features is None or features.shape != (3, Config.n_mels, Config.expected_frames):\n            # print(f\"Failed to transform features for {audio_filename}: shape={features.shape if features is not None else None}\")\n            return None\n        \n        # Save features\n        np.savez_compressed(feature_path, features=features, allow_pickle=False)\n        # print(f\"Saved features to {os.path.basename(feature_path)}\")\n        return feature_path if os.path.exists(feature_path) else None\n    \n    except Exception as e:\n        # print(f\"Error processing {audio_filename}: {e}\")\n        return None\n    finally:\n        gc.collect()\n        if Config.device == \"cuda\":\n            torch.cuda.empty_cache()\n\ndef validate_paths(df):\n    print(\"Validating audio paths...\")\n    invalid_paths = []\n    for idx, row in tqdm(df.iterrows(), total=len(df), desc=\"Checking paths\"):\n        audio_path = row['filename']\n        if not os.path.exists(audio_path):\n            possible_path = audio_path.lower()\n            if os.path.exists(possible_path):\n                df.loc[idx, 'filename'] = possible_path\n                # print(f\"Corrected path: {audio_path} -> {possible_path}\")\n            else:\n                invalid_paths.append((idx, audio_path))\n                # print(f\"Invalid path: {audio_path}\")\n    # print(f\"Found {len(invalid_paths)} invalid paths ({len(invalid_paths)/len(df)*100:.2f}%)\")\n    return invalid_paths\n\ndef precompute_features(df, mode=\"train\", taxonomy_df=None):\n    if df.empty:\n        # print(f\"Empty DataFrame for {mode} features\")\n        return df\n    \n    invalid_paths = validate_paths(df)\n    if len(invalid_paths) > 0.5 * len(df):\n        # print(f\"Over 50% invalid paths, check dataset\")\n        return df\n    \n    if 'feature_path' not in df.columns:\n        df['feature_path'] = None\n    \n    batch_size = Config.batch_size\n    num_batches = (len(df) + batch_size - 1) // batch_size\n    \n    for batch_idx in range(num_batches):\n        batch_dir = os.path.join(Config.feature_dir, f\"batch_{batch_idx+1}\")\n        os.makedirs(batch_dir, exist_ok=True)\n        start_idx = batch_idx * batch_size\n        end_idx = min((batch_idx + 1) * batch_size, len(df))\n        batch_df = df.iloc[start_idx:end_idx].copy()\n        \n        print(f\"Processing batch {batch_idx+1}/{num_batches} ({start_idx} to {end_idx-1})\")\n        batch_paths = Parallel(n_jobs=Config.n_jobs, backend=\"loky\", batch_size=1)(\n            delayed(process_audio)((idx, row), mode, batch_dir, taxonomy_df)\n            for idx, row in tqdm(batch_df.iterrows(), total=len(batch_df), desc=f\"Batch {batch_idx+1}\")\n        )\n        \n        df.iloc[start_idx:end_idx, df.columns.get_loc('feature_path')] = batch_paths\n        \n        # Clean up memory every 3 batches\n        if (batch_idx + 1) % 3 == 0 or (batch_idx + 1) == num_batches:\n            gc.collect()\n            if Config.device == \"cuda\":\n                torch.cuda.empty_cache()\n            # print(f\"Memory cleanup after batch {batch_idx+1}, usage: {psutil.virtual_memory().percent}%\")\n    \n    return df\n\ndef analyze_feature_files(data_df):\n    expected_shape = (3, Config.n_mels, Config.expected_frames)\n    success_count = 0\n    failed_count = 0\n    zero_feature_count = 0\n    total_files = len(data_df)\n    failed_files = []\n    \n    print(\"\\nChecking feature files...\")\n    for idx, row in tqdm(data_df.iterrows(), total=total_files, desc=\"Processing files\"):\n        feature_path = row['feature_path']\n        if not feature_path or not os.path.exists(feature_path):\n            print(f\"Feature file missing: {feature_path}\")\n            failed_count += 1\n            failed_files.append(row['filename'])\n            continue\n        try:\n            data = np.load(feature_path, allow_pickle=False)\n            features = data['features']\n            if features.shape != expected_shape:\n                print(f\"Wrong shape ({feature_path}): {features.shape}\")\n                failed_count += 1\n                failed_files.append(row['filename'])\n            elif np.all(features == 0):\n                print(f\"Zero features ({feature_path})\")\n                zero_feature_count += 1\n                failed_count += 1\n                failed_files.append(row['filename'])\n            elif np.any(np.isnan(features)) or np.any(np.isinf(features)):\n                print(f\"NaN/Inf features ({feature_path})\")\n                failed_count += 1\n                failed_files.append(row['filename'])\n            else:\n                success_count += 1\n        except Exception as e:\n            print(f\"Error loading feature file ({feature_path}): {str(e)}\")\n            failed_count += 1\n            failed_files.append(row['filename'])\n    \n    print(f\"\\nFeature File Summary:\")\n    print(f\"Total files: {total_files}\")\n    print(f\"Successful files: {success_count} ({success_count/total_files*100:.2f}%)\")\n    print(f\"Failed files: {failed_count} ({failed_count/total_files*100:.2f}%)\")\n    print(f\"Zero features (successful but all zeros): {zero_feature_count} ({zero_feature_count/total_files*100:.2f}%)\")\n    \n    with open(\"/kaggle/working/failed_files.txt\", \"w\") as f:\n        f.write(\"\\n\".join(failed_files))\n    print(f\"Saved failed files list to /kaggle/working/failed_files.txt\")\n    \n    return success_count, failed_count\n\ndef extract_audio_features(data_df, taxonomy_df):\n    shutil.rmtree(Config.feature_dir, ignore_errors=True)\n    os.makedirs(Config.feature_dir, exist_ok=True)\n\n    class_counts = data_df['primary_label'].value_counts()\n    oversample_df = pd.concat([\n        data_df[data_df['primary_label'] == label].sample(n=Config.oversample_threshold, replace=True, random_state=Config.seed)\n        for label in class_counts[class_counts < Config.oversample_threshold].index\n    ])\n    data_df = pd.concat([data_df, oversample_df], ignore_index=True)\n\n    print(\"Testing with subset (first 100 files)\")\n    data_df_subset = data_df.head(100).copy()\n    data_df_subset = precompute_features(data_df_subset, taxonomy_df=taxonomy_df)\n    success_count, failed_count = analyze_feature_files(data_df_subset)\n    print(f\"Subset - Successful: {success_count}, Failed: {failed_count}\")\n    if failed_count / (success_count + failed_count + 1e-10) > 0.5:\n        print(\"Warning: Subset processing failed (>50% failed), check /kaggle/working/failed_files.txt\")\n        sys.exit(1)\n\n    print(\"Processing full dataset...\")\n    initial_counts = data_df['primary_label'].value_counts().to_dict()\n    data_df = precompute_features(data_df, taxonomy_df=taxonomy_df)\n\n    final_counts = data_df[data_df['feature_path'].notnull()]['primary_label'].value_counts().to_dict()\n    print(\"\\nFiltering Impact:\")\n    for species in initial_counts:\n        initial = initial_counts.get(species, 0)\n        final = final_counts.get(species, 0)\n        if initial > 0:\n            filtered_percent = ((initial - final) / initial) * 100\n            print(f\"Species {species}: {final}/{initial} files retained ({100 - filtered_percent:.1f}% retained, {filtered_percent:.1f}% filtered)\")\n    \n    return data_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T14:08:57.999857Z","iopub.execute_input":"2025-05-25T14:08:58.000515Z","iopub.status.idle":"2025-05-25T14:08:58.041276Z","shell.execute_reply.started":"2025-05-25T14:08:58.000488Z","shell.execute_reply":"2025-05-25T14:08:58.040555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# โหลด audio จากไฟล์ตัวอย่าง\nrow = data_df.iloc[0]\naudio_path = os.path.join(Config.train_dir, row['filename'])\naudio, sample_rate = torchaudio.load(audio_path)\n\nprint(f\"Loaded audio shape: {audio.shape}, sample_rate: {sample_rate}, max: {audio.max()}, min: {audio.min()}\")\n\nprint(\"Starting noise reduction...\")\naudio = apply_noise_reduction(audio, sample_rate=Config.sr)\nprint(\"Finished noise reduction\")\n\nprint(\"Starting feature extraction...\")\nmel_sp = extract_features(audio, sr=sample_rate)\nprint(\"Finished feature extraction\")\nif mel_sp is not None:\n    print(f\"Mel spectrogram shape: {mel_sp.shape}\")\nelse:\n    print(\"Mel spectrogram is None, check extract_features logic\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T14:00:37.788265Z","iopub.execute_input":"2025-05-25T14:00:37.788521Z","iopub.status.idle":"2025-05-25T14:00:38.220424Z","shell.execute_reply.started":"2025-05-25T14:00:37.788501Z","shell.execute_reply":"2025-05-25T14:00:38.219629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract Features\ndata_df = extract_audio_features(data_df, taxonomy_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T14:09:05.308281Z","iopub.execute_input":"2025-05-25T14:09:05.309172Z","iopub.status.idle":"2025-05-25T17:22:31.926595Z","shell.execute_reply.started":"2025-05-25T14:09:05.309143Z","shell.execute_reply":"2025-05-25T17:22:31.925788Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze Features\nsuccess_count, failed_count = analyze_feature_files(data_df)\nprint(f\"Subset - Successful: {success_count}, Failed: {failed_count}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:28:54.146845Z","iopub.execute_input":"2025-05-25T17:28:54.147156Z","iopub.status.idle":"2025-05-25T17:33:13.618229Z","shell.execute_reply.started":"2025-05-25T17:28:54.147134Z","shell.execute_reply":"2025-05-25T17:33:13.617492Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Location Features","metadata":{}},{"cell_type":"code","source":"# ตรวจสอบข้อมูล latitude และ longitude\nprint(\"Checking latitude and longitude in data_df:\")\nprint(data_df[['latitude', 'longitude']].describe())\nprint(\"\\nMissing values:\")\nprint(data_df[['latitude', 'longitude']].isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:33:13.619524Z","iopub.execute_input":"2025-05-25T17:33:13.619872Z","iopub.status.idle":"2025-05-25T17:33:13.634624Z","shell.execute_reply.started":"2025-05-25T17:33:13.619852Z","shell.execute_reply":"2025-05-25T17:33:13.634041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def add_location_features(data_df):\n    data_df['latitude'] = data_df['latitude'].astype(float)\n    data_df['longitude'] = data_df['longitude'].astype(float)\n    \n    # Fill missing values with median\n    lat_median = data_df['latitude'].median()\n    lon_median = data_df['longitude'].median()\n    data_df['latitude'] = data_df['latitude'].fillna(lat_median)\n    data_df['longitude'] = data_df['longitude'].fillna(lon_median)\n    \n    # Add flag for missing coordinates\n    data_df['is_missing_loc'] = data_df['latitude'].isna() | data_df['longitude'].isna()\n    data_df['latitude'] = data_df['latitude'].fillna(0)  # Use 0 as fallback\n    data_df['longitude'] = data_df['longitude'].fillna(0)\n    \n    # Normalize latitude and longitude\n    data_df['latitude_norm'] = (data_df['latitude'] - data_df['latitude'].min()) / (data_df['latitude'].max() - data_df['latitude'].min())\n    data_df['longitude_norm'] = (data_df['longitude'] - data_df['longitude'].min()) / (data_df['longitude'].max() - data_df['longitude'].min())\n    \n    # Distance from equator\n    data_df['dist_equator'] = data_df.apply(lambda row: haversine_distance(row['latitude'], row['longitude'], 0, row['longitude']), axis=1)\n    \n    # Distance from reference point (e.g., Colombia center: 6.76, -74.21)\n    data_df['dist_ref'] = data_df.apply(lambda row: haversine_distance(row['latitude'], row['longitude'], Config.ref_lat, Config.ref_lon), axis=1)\n    \n    print(f\"\\nAdded location features. Found {data_df['is_missing_loc'].sum()} files with missing coordinates, filled with median.\")\n    print(\"Dataset size:\", len(data_df))\n    return data_df\n\n# Run the function\ndata_df = add_location_features(data_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:38:35.715720Z","iopub.execute_input":"2025-05-25T17:38:35.716458Z","iopub.status.idle":"2025-05-25T17:38:36.170510Z","shell.execute_reply.started":"2025-05-25T17:38:35.716429Z","shell.execute_reply":"2025-05-25T17:38:36.169694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Stratified K-Fold with rating bins\nif Config.n_folds < 2:\n    raise ValueError(f\"Config.n_folds must be >= 2, got {Config.n_folds}\")\nprint(f\"\\nUsing {Config.n_folds} folds for StratifiedKFold\")\ndata_df['rating_bin'] = pd.cut(data_df['rating'], bins=[-0.1, 2, 4, 5], labels=['low', 'medium', 'high'])\ndata_df['stratify_key'] = data_df['primary_label'].astype(str) + '_' + data_df['rating_bin'].astype(str)\nskf = StratifiedKFold(n_splits=Config.n_folds, shuffle=True, random_state=Config.seed)\ndata_df['kfold'] = -1\nfor fold, (_, val_idx) in enumerate(skf.split(data_df, data_df['stratify_key'])):\n    data_df.iloc[val_idx, data_df.columns.get_loc('kfold')] = fold\ndata_df.to_csv('/kaggle/working/train_data.csv', index=False)\nprint(f\"Saved preprocessed data to /kaggle/working/train_data.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:38:53.019672Z","iopub.execute_input":"2025-05-25T17:38:53.019956Z","iopub.status.idle":"2025-05-25T17:38:53.683388Z","shell.execute_reply.started":"2025-05-25T17:38:53.019935Z","shell.execute_reply":"2025-05-25T17:38:53.682722Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataset and Model ","metadata":{}},{"cell_type":"code","source":"# Dataset\nsample_submission = pd.read_csv(Config.sample_csv)\nall_classes = sample_submission.columns[1:].tolist()\nlabel_mapper = {label: idx for idx, label in enumerate(all_classes)}\nrev_mapper = {idx: label for label, idx in label_mapper.items()}\n\nclass BirdClefDataset(torch.utils.data.Dataset):\n    def __init__(self, df, mode=\"train\"):\n        self.df = df\n        self.mode = mode\n        if mode == \"train\" and not df.empty:\n            class_counts = self.df['primary_label'].value_counts()\n            print(f\"Number of species with count < 20: {len(class_counts[class_counts < 20])}\")\n            oversample_df = []\n            for label in class_counts[class_counts < 20].index:\n                label_df = self.df[self.df['primary_label'] == label]\n                if not label_df.empty:\n                    oversample_df.append(label_df.sample(n=20, replace=True, random_state=Config.seed))\n            if oversample_df:\n                oversample_df = pd.concat(oversample_df)\n                self.df = pd.concat([self.df, oversample_df], ignore_index=True)\n            else:\n                print(\"No oversampling performed due to empty label data\")\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        feature_path = row['feature_path']\n        if feature_path is None:\n            print(f\"Warning: feature_path is None for index {idx}, using zero features\")\n            features = np.zeros((3, Config.n_mels, Config.expected_frames), dtype=np.float32)\n        else:\n            try:\n                features = np.load(feature_path, allow_pickle=False)['features'].astype(np.float32)\n                if np.any(np.isnan(features)) or np.any(np.isinf(features)):\n                    print(f\"NaN/Inf features in {feature_path}, replacing with zeros\")\n                    features = np.zeros((3, Config.n_mels, Config.expected_frames), dtype=np.float32)\n            except Exception as e:\n                print(f\"Error loading features {feature_path}: {str(e)}\")\n                features = np.zeros((3, Config.n_mels, Config.expected_frames), dtype=np.float32)\n        \n        loc_features = np.array([\n            row['latitude_norm'] if not np.isnan(row['latitude_norm']) else 0.0,\n            row['longitude_norm'] if not np.isnan(row['longitude_norm']) else 0.0,\n            row['dist_equator'] if not np.isnan(row['dist_equator']) else 0.0,\n            row['dist_ref'] if not np.isnan(row['dist_ref']) else 0.0\n        ], dtype=np.float32)\n        loc_features = np.append(loc_features, [float(row['is_missing_loc'])])\n        \n        if self.mode == \"train\":\n            return {\n                'audio': torch.tensor(features, dtype=torch.float32),\n                'loc': torch.tensor(loc_features, dtype=torch.float32)\n            }, torch.tensor(label_mapper[row['primary_label']], dtype=torch.long)\n        return {\n            'audio': torch.tensor(features, dtype=torch.float32),\n            'loc': torch.tensor(loc_features, dtype=torch.float32)\n        }\n        \n# Model\nclass BirdClefModel(nn.Module):\n    def __init__(self, model_name, weights_path, kernel_size=3, out_channels1=16, out_channels2=32, padding=1):\n        super().__init__()\n        self.model_name = model_name\n        self.base_model = timm.create_model(\n            model_name=model_name,\n            num_classes=Config.num_classes,\n            pretrained=False,\n            in_chans=3,\n            drop_rate=Config.dropout\n        )\n        \n        self.loc_cnn = nn.Sequential(\n            nn.Conv1d(in_channels=1, out_channels=out_channels1, kernel_size=kernel_size, padding=padding),\n            nn.ReLU(),\n            nn.MaxPool1d(kernel_size=2),\n            nn.Conv1d(in_channels=out_channels1, out_channels=out_channels2, kernel_size=kernel_size, padding=padding),\n            nn.ReLU(),\n            nn.MaxPool1d(kernel_size=2),\n            nn.Flatten(),\n            nn.Linear(out_channels2 * 1, 128)\n        )\n        \n        self.pool = nn.AdaptiveAvgPool2d(1)\n        \n        self.original_in_features = 1536\n        self.new_in_features = self.original_in_features * 2\n        \n        if os.path.exists(weights_path):\n            state_dict = torch.load(weights_path, map_location=Config.device, weights_only=True)\n            exclude_keys = ['classifier.weight', 'classifier.bias', 'fc.weight', 'fc.bias', 'head.fc.weight', 'head.fc.bias']\n            state_dict = {k: v for k, v in state_dict.items() if k not in exclude_keys}\n            missing_keys, unexpected_keys = self.base_model.load_state_dict(state_dict, strict=False)\n            print(f\"Loaded weights from {weights_path}. Missing keys: {missing_keys}, Unexpected keys: {unexpected_keys}\")\n        else:\n            print(f\"Weights not found at {weights_path}. Using random initialization.\")\n        \n        with torch.no_grad():\n            if hasattr(self.base_model, 'classifier'):\n                self.base_model.classifier = nn.Linear(self.new_in_features, Config.num_classes)\n                nn.init.xavier_uniform_(self.base_model.classifier.weight)\n                nn.init.zeros_(self.base_model.classifier.bias)\n            elif hasattr(self.base_model, 'fc'):\n                self.base_model.fc = nn.Linear(self.new_in_features, Config.num_classes)\n                nn.init.xavier_uniform_(self.base_model.fc.weight)\n                nn.init.zeros_(self.base_model.fc.bias)\n            elif hasattr(self.base_model, 'head') and hasattr(self.base_model.head, 'fc'):\n                self.base_model.head.fc = nn.Linear(self.new_in_features, Config.num_classes)\n                nn.init.xavier_uniform_(self.base_model.head.fc.weight)\n                nn.init.zeros_(self.base_model.head.fc.bias)\n\n        self.loc_adjust = None\n\n    def forward(self, x, loc):\n        audio_out = self.base_model.forward_features(x)\n        \n        loc_out = self.loc_cnn(loc.unsqueeze(1))\n        \n        if self.loc_adjust is None:\n            target_dim = audio_out.shape[1]\n            self.loc_adjust = nn.Linear(loc_out.shape[1], target_dim).to(loc_out.device)\n        \n        loc_out = self.loc_adjust(loc_out)\n        \n        loc_out = loc_out.unsqueeze(2).unsqueeze(3)\n        loc_out = loc_out.expand(-1, -1, audio_out.shape[2], audio_out.shape[3])\n        \n        combined = torch.cat((audio_out, loc_out), dim=1)\n        \n        combined = self.pool(combined)\n        combined = combined.view(combined.size(0), -1)\n        \n        if hasattr(self.base_model, 'classifier'):\n            out = self.base_model.classifier(combined)\n        elif hasattr(self.base_model, 'fc'):\n            out = self.base_model.fc(combined)\n        elif hasattr(self.base_model, 'head') and hasattr(self.base_model.head, 'fc'):\n            out = self.base_model.head.fc(combined)\n        return out\n\n# FocalLoss\nclass FocalLoss(nn.Module):\n    def __init__(self, gamma=2.0, label_smoothing=0.1):\n        super().__init__()\n        self.gamma = gamma\n        self.label_smoothing = label_smoothing\n\n    def forward(self, inputs, targets):\n        log_probs = F.log_softmax(inputs, dim=-1)\n        targets_one_hot = F.one_hot(targets, num_classes=Config.num_classes).float()\n        targets_smooth = targets_one_hot * (1 - self.label_smoothing) + self.label_smoothing / Config.num_classes\n        ce_loss = -torch.sum(targets_smooth * log_probs, dim=-1)\n        pt = torch.exp(-ce_loss)\n        focal_loss = (1 - pt) ** self.gamma * ce_loss\n        return focal_loss.mean()\n\nprint(\"Dataset and Model classes defined.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T20:24:17.221446Z","iopub.execute_input":"2025-05-25T20:24:17.222247Z","iopub.status.idle":"2025-05-25T20:24:17.255285Z","shell.execute_reply.started":"2025-05-25T20:24:17.222218Z","shell.execute_reply":"2025-05-25T20:24:17.254477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ตรวจสอบ feature_path\ntrain_data = pd.read_csv('/kaggle/working/train_data.csv')\n\nmissing_features = train_data[train_data['feature_path'].isnull()]\nprint(f\"Rows with missing feature_path: {len(missing_features)}\")\nif len(missing_features) > 0:\n    print(\"Sample rows with missing feature_path:\")\n    print(missing_features[['filename', 'primary_label']].head())\n\n# ตรวจสอบว่าไฟล์ฟีเจอร์มีอยู่จริงหรือไม่\ninvalid_paths = []\nfor path in train_data['feature_path'].dropna():\n    if not os.path.exists(path):\n        invalid_paths.append(path)\nprint(f\"Invalid feature paths: {len(invalid_paths)}\")\nif len(invalid_paths) > 0:\n    print(\"Sample invalid paths:\", invalid_paths[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T19:28:16.960326Z","iopub.execute_input":"2025-05-25T19:28:16.961082Z","iopub.status.idle":"2025-05-25T19:28:17.294259Z","shell.execute_reply.started":"2025-05-25T19:28:16.961056Z","shell.execute_reply":"2025-05-25T19:28:17.293573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ตรวจสอบฟีเจอร์จาก feature_path\nfor idx, row in train_data.head(5).iterrows():\n    feature_path = row['feature_path']\n    if feature_path and os.path.exists(feature_path):\n        features = np.load(feature_path, allow_pickle=False)['features']\n        print(f\"Feature for {row['filename']}: shape={features.shape}, max={features.max()}, min={features.min()}\")\n        if np.any(np.isnan(features)) or np.any(np.isinf(features)):\n            print(f\"NaN/Inf found in {feature_path}\")\n        if np.all(features == 0):\n            print(f\"All zeros in {feature_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T19:15:19.475642Z","iopub.execute_input":"2025-05-25T19:15:19.476568Z","iopub.status.idle":"2025-05-25T19:15:19.521785Z","shell.execute_reply.started":"2025-05-25T19:15:19.476540Z","shell.execute_reply":"2025-05-25T19:15:19.521148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ตรวจสอบ label\nmissing_labels = set(train_data['primary_label']) - set(all_classes)\nprint(f\"Labels in train_data.csv but not in all_classes: {missing_labels}\")\nif missing_labels:\n    train_data = train_data[train_data['primary_label'].isin(all_classes)]\n    print(f\"Filtered train_data size: {len(train_data)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T19:16:23.907023Z","iopub.execute_input":"2025-05-25T19:16:23.907752Z","iopub.status.idle":"2025-05-25T19:16:23.914847Z","shell.execute_reply.started":"2025-05-25T19:16:23.907728Z","shell.execute_reply":"2025-05-25T19:16:23.914049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ดู class distribution\nclass_counts = train_data['primary_label'].value_counts()\nprint(\"Class distribution in train_data.csv:\")\nprint(class_counts)\nprint(f\"Classes with less than 10 samples: {len(class_counts[class_counts < 10])}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T19:16:53.414205Z","iopub.execute_input":"2025-05-25T19:16:53.414477Z","iopub.status.idle":"2025-05-25T19:16:53.423640Z","shell.execute_reply.started":"2025-05-25T19:16:53.414457Z","shell.execute_reply":"2025-05-25T19:16:53.422915Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training and Validation","metadata":{}},{"cell_type":"code","source":"def train_epoch(model, loader, criterion, optimizer, scaler, fold, epoch, model_name):\n    model.train()\n    running_loss = 0.0\n    preds, labels = [], []\n    for batch in tqdm(loader, desc=f\"Fold {fold + 1}/{Config.n_folds}, Epoch {epoch + 1}/{Config.epochs}, Model: {model_name} - Training\", leave=False):\n        try:\n            # batch เป็น tuple: (inputs, targets)\n            inputs, targets = batch\n            x, y = inputs['audio'].to(Config.device), inputs['loc'].to(Config.device)\n            target = targets.to(Config.device)\n            \n            optimizer.zero_grad()\n            with autocast(device_type='cuda' if Config.device == 'cuda' else 'cpu'):\n                if random.random() < 0.8:\n                    x, y_a, y_b, lam = mixup(x, target)\n                    outputs = model(x, y)\n                    loss = lam * criterion(outputs, y_a) + (1 - lam) * criterion(outputs, y_b)\n                else:\n                    outputs = model(x, y)\n                    loss = criterion(outputs, target)\n                if torch.isnan(loss) or torch.isinf(loss):\n                    print(f\"NaN/Inf loss detected in Fold {fold + 1}, Epoch {epoch + 1}, Model: {model_name}, skipping batch\")\n                    continue\n            if Config.device == 'cuda':\n                scaler.scale(loss).backward()\n                torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)\n                scaler.step(optimizer)\n                scaler.update()\n            else:\n                loss.backward()\n                torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)\n                optimizer.step()\n            running_loss += loss.item()\n            probs = torch.softmax(outputs, dim=1).detach().cpu().numpy()\n            preds.append(probs)\n            labels.append(target.cpu().numpy())\n        except Exception as e:\n            print(f\"Error in training batch, Fold {fold + 1}, Epoch {epoch + 1}: {str(e)}\")\n            continue\n    print(f\"Training Loss: {running_loss / len(loader):.4f}\")\n    return running_loss / len(loader), np.concatenate(preds), np.concatenate(labels)\n\ndef validate_epoch(model, loader, criterion, fold, epoch, model_name):\n    model.eval()\n    running_loss = 0.0\n    preds, labels = [], []\n    with torch.no_grad():\n        for batch in tqdm(loader, desc=f\"Fold {fold + 1}/{Config.n_folds}, Epoch {epoch + 1}/{Config.epochs}, Model: {model_name} - Validating\", leave=False):\n            try:\n                # batch เป็น tuple: (inputs, targets)\n                inputs, targets = batch\n                x, y = inputs['audio'].to(Config.device), inputs['loc'].to(Config.device)\n                target = targets.to(Config.device)\n                \n                with autocast(device_type='cuda' if Config.device == 'cuda' else 'cpu'):\n                    outputs = model(x, y)\n                    if torch.any(torch.isnan(outputs)) or torch.any(torch.isinf(outputs)):\n                        print(f\"NaN/Inf in model outputs in Fold {fold + 1}, Epoch {epoch + 1}, Model: {model_name}, skipping batch\")\n                        continue\n                    loss = criterion(outputs, target)\n                    running_loss += loss.item()\n                    probs = torch.softmax(outputs, dim=1)\n                    probs = probs / (probs.sum(dim=1, keepdim=True) + 1e-10)\n                    probs = probs.detach().cpu().numpy()\n                    preds.append(probs)\n                    labels.append(target.cpu().numpy())\n            except Exception as e:\n                print(f\"Error in validation batch, Fold {fold + 1}, Epoch {epoch + 1}: {str(e)}\")\n                continue\n    print(f\"Validation Loss: {running_loss / len(loader):.4f}\")\n    return running_loss / len(loader), np.concatenate(preds), np.concatenate(labels)\n\ndef compute_metrics(labels, preds, probs):\n    cm = confusion_matrix(labels, preds)\n    precision = precision_score(labels, preds, average='macro', zero_division=0)\n    recall = recall_score(labels, preds, average='macro', zero_division=0)\n    f1 = f1_score(labels, preds, average='macro', zero_division=0)\n    specificity = np.mean([(cm.sum() - cm[i, :].sum() - cm[:, i].sum() + cm[i, i]) /\n                          (cm.sum() - cm[i, :].sum() + 1e-12) for i in range(len(np.unique(labels)))])\n    return recall, specificity, precision, f1, cm\n\ndef train_and_evaluate(data_df):\n    print(\"Starting training\")\n    \n    models = [{'name': k, 'weights_path': v} for k, v in Config.weights.items()]\n    fold_aucs = {m['name']: [] for m in models}\n    all_metrics = {m['name']: [] for m in models}\n    preliminary_results = []\n\n    # กรองแถวที่มี feature_path เป็น None\n    initial_rows = len(data_df)\n    data_df = data_df.dropna(subset=['feature_path'])\n    dropped_rows = initial_rows - len(data_df)\n    if dropped_rows > 0:\n        print(f\"Dropped {dropped_rows} rows with None feature_path\")\n\n    for model_config in models:\n        model_name = model_config['name']\n        weights_path = model_config['weights_path']\n        print(f\"\\nTraining model: {model_name}\")\n        print(f\"Checking initial disk usage: {get_disk_usage():.2f} GiB\")\n        if not os.path.exists(weights_path):\n            print(f\"Error: Pretrained weights not found at {weights_path}, skipping {model_name}\")\n            continue\n        for fold in range(Config.n_folds):\n            print(f\"\\nProcessing Fold {fold + 1}/{Config.n_folds}\")\n            weight_path = os.path.join(Config.model_weights_dir, f\"{model_name}_fold_{fold}.pth\")\n            checkpoint = load_checkpoint(model_name, fold)\n            if checkpoint and os.path.exists(weight_path):\n                print(f\"Skipping Fold {fold + 1} (checkpoint and weights exist)\")\n                fold_aucs[model_name].append(checkpoint['fold_aucs'][model_name][-1])\n                all_metrics[model_name].append(checkpoint['metrics'][model_name][-1])\n                preliminary_results.append(checkpoint['preliminary_results'][-1])\n                continue\n            elif checkpoint:\n                print(f\"Checkpoint exists for Fold {fold + 1}, but weights {weight_path} missing, retraining\")\n            train_df = data_df[data_df['kfold'] != fold].reset_index(drop=True)\n            val_df = data_df[data_df['kfold'] == fold].reset_index(drop=True)\n            print(f\"Train df size: {len(train_df)}, Val df size: {len(val_df)}\")\n            if train_df.empty or val_df.empty:\n                print(f\"Warning: Empty dataset for fold {fold + 1}, using full dataset for debug\")\n                train_df = data_df.copy()\n            train_ds = BirdClefDataset(train_df)\n            val_ds = BirdClefDataset(val_df)\n            train_loader = torch.utils.data.DataLoader(\n                train_ds,\n                batch_size=Config.batch_size,\n                shuffle=True,\n                num_workers=2,\n                pin_memory=True,\n                prefetch_factor=2\n            )\n            val_loader = torch.utils.data.DataLoader(\n                val_ds,\n                batch_size=Config.batch_size,\n                shuffle=False,\n                num_workers=2,\n                pin_memory=True,\n                prefetch_factor=2\n            )\n            \n            model = BirdClefModel(model_name, weights_path).to(Config.device)\n            criterion = nn.CrossEntropyLoss(label_smoothing=Config.label_smoothing).to(Config.device)\n            optimizer = torch.optim.AdamW(model.parameters(), lr=Config.lr, weight_decay=1e-2)\n            scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=1, verbose=True)\n            scaler = GradScaler('cuda' if Config.device == 'cuda' else 'cpu')\n            \n            best_auc = 0\n            early_stop_counter = 0\n            try:\n                for epoch in range(Config.epochs):\n                    print(f\"Disk usage before epoch {epoch + 1}: {get_disk_usage():.2f} GiB\")\n                    print(f\"RAM Usage: {psutil.virtual_memory().percent}%\")\n                    train_loss, train_preds, train_labels = train_epoch(model, train_loader, criterion, optimizer, scaler, fold, epoch, model_name)\n                    val_loss, val_preds, val_labels = validate_epoch(model, val_loader, criterion, fold, epoch, model_name)\n                    scheduler.step(val_loss)\n                    unique_labels = np.unique(val_labels)\n                    if len(unique_labels) < Config.num_classes:\n                        valid_indices = [i for i in unique_labels if i < Config.num_classes]\n                        if not valid_indices:\n                            print(f\"Warning: No valid labels in Fold {fold + 1}, Epoch {epoch + 1}, skipping AUC\")\n                            auc_val = 0.0\n                        else:\n                            val_preds_subset = val_preds[:, valid_indices]\n                            val_preds_subset = val_preds_subset / (val_preds_subset.sum(axis=1, keepdims=True) + 1e-10)\n                            auc_val = roc_auc_score(val_labels, val_preds_subset, multi_class='ovr', average='macro')\n                    else:\n                        auc_val = roc_auc_score(val_labels, val_preds, multi_class='ovr', average='macro')\n                    preds = np.argmax(val_preds, axis=1)\n                    metrics = compute_metrics(val_labels, preds, val_preds)\n                    print(f\"Fold {fold + 1}/{Config.n_folds}, Epoch {epoch + 1}/{Config.epochs}, Model: {model_name} - Val AUC: {auc_val:.4f}, Sensitivity: {metrics[0]:.4f}, Specificity: {metrics[1]:.4f}, Precision: {metrics[2]:.4f}, F1: {metrics[3]:.4f}\")\n                    \n                    if auc_val > best_auc:\n                        best_auc = auc_val\n                        early_stop_counter = 0\n                        os.makedirs(Config.model_weights_dir, exist_ok=True)\n                        torch.save(model.state_dict(), weight_path)\n                        print(f\"Saved best model weights for AUC: {best_auc:.4f} at {weight_path}\")\n                        if os.path.exists(weight_path):\n                            print(f\"Confirmed weights saved: {weight_path}, size: {os.path.getsize(weight_path) / (1024**2):.2f} MB\")\n                        else:\n                            print(f\"Error: Failed to save weights at {weight_path}\")\n                    else:\n                        early_stop_counter += 1\n                        if early_stop_counter >= Config.early_stop_patience:\n                            print(f\"Early stopping at Fold {fold + 1}, Epoch {epoch + 1}\")\n                            break\n                    \n                    gc.collect()\n                    torch.cuda.empty_cache() if Config.device == \"cuda\" else None\n            except Exception as e:\n                print(f\"Error during training Fold {fold + 1}, Model: {model_name}: {e}\")\n                continue\n            \n            fold_aucs[model_name].append(best_auc)\n            all_metrics[model_name].append(metrics)\n            preliminary_results.append({\n                'Model': model_name, 'Fold': fold, 'AUC': best_auc,\n                **dict(zip(['Sensitivity', 'Specificity', 'Precision', 'F1'], metrics[:4]))\n            })\n            save_checkpoint(model_name, fold, {\n                'fold_aucs': fold_aucs, 'metrics': all_metrics, 'preliminary_results': preliminary_results\n            }, model, optimizer)\n            print(f\"Disk usage after Fold {fold + 1}: {get_disk_usage():.2f} GiB\")\n            \n            train_df = data_df[data_df['kfold'] != fold]\n            for feature_path in train_df['feature_path']:\n                if feature_path and os.path.exists(feature_path):\n                    os.remove(feature_path)\n            print(f\"Cleared feature files for Fold {fold + 1}\")\n            \n            for old_checkpoint in glob.glob(os.path.join(Config.checkpoint_dir, f\"checkpoint_{model_name}_fold_{fold}_*.pkl\")):\n                if old_checkpoint != os.path.join(Config.checkpoint_dir, f\"checkpoint_{model_name}_fold_{fold}.pkl\"):\n                    os.remove(old_checkpoint)\n                    print(f\"Removed old checkpoint: {old_checkpoint}\")\n    \n    pd.DataFrame(preliminary_results).to_csv('preliminary_results.csv', index=False)\n    print(\"Saved preliminary results to preliminary_results.csv\")\n    return fold_aucs, all_metrics, preliminary_results\n\n# Run the function\nfold_aucs, all_metrics, preliminary_results = train_and_evaluate(data_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T20:59:50.734799Z","iopub.execute_input":"2025-05-25T20:59:50.735074Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Final Evaluation\n","metadata":{}},{"cell_type":"code","source":"def final_evaluation(data_df, fold_aucs, all_metrics, preliminary_results):\n    print(\"\\nGenerating final experimental results\")\n    models = [{'name': k, 'weights_path': v} for k, v in Config.weights.items()]\n    for model_name in tqdm([m['name'] for m in models], desc=\"Evaluating Models\"):\n        print(f\"\\nComputing ROC curves for Model: {model_name}\")\n        plt.figure(figsize=(10, 8))\n        mean_fpr = np.linspace(0, 1, 100)\n        tprs, aucs = [], []\n        for fold in range(Config.n_folds):\n            start_time = time.time()\n            print(f\"Computing ROC for Fold {fold + 1}/{Config.n_folds}\")\n            labels, probs = [], []\n            val_df = data_df[data_df['kfold'] == fold]\n            print(f\"Validation set size: {len(val_df)} samples\")\n            val_ds = BirdClefDataset(val_df)\n            val_loader = torch.utils.data.DataLoader(\n                val_ds,\n                batch_size=Config.batch_size,\n                shuffle=False,\n                num_workers=1\n            )\n            model = BirdClefModel(model_name, Config.weights[model_name]).to(Config.device)\n            weight_path = os.path.join(Config.model_weights_dir, f\"{model_name}_fold_{fold}.pth\")\n            if os.path.exists(weight_path):\n                model.load_state_dict(torch.load(weight_path, weights_only=True))\n                print(f\"Loaded fold {fold + 1} weights from {weight_path}\")\n            else:\n                print(f\"Warning: Weight path {weight_path} not found, skipping fold\")\n                continue\n            model.eval()\n            with torch.no_grad():\n                for batch in tqdm(val_loader, desc=f\"Evaluating Fold {fold + 1}/{Config.n_folds}, Model: {model_name}\", leave=False):\n                    x, y = batch['audio'].to(Config.device), batch['loc'].to(Config.device)\n                    with autocast('cuda' if Config.device == 'cuda' else 'cpu'):\n                        outputs = model(x, y)\n                    probs_batch = torch.softmax(outputs, dim=1).cpu().numpy()\n                    probs.append(probs_batch)\n                    labels.append(batch[1].numpy())\n            if not probs or not labels:\n                print(f\"Warning: No valid data for ROC in Fold {fold + 1}, Model: {model_name}\")\n                continue\n            labels, probs = np.concatenate(labels), np.concatenate(probs)\n            print(f\"Number of unique labels: {len(np.unique(labels))}\")\n            unique_labels = np.unique(labels)\n            if len(unique_labels) < Config.num_classes:\n                valid_indices = [i for i in unique_labels if i < Config.num_classes]\n                if not valid_indices:\n                    print(f\"Warning: No valid labels for ROC in Fold {fold + 1}, Model: {model_name}\")\n                    continue\n                probs = probs[:, valid_indices]\n                probs = probs / (probs.sum(axis=1, keepdims=True) + 1e-10)\n                labels = np.array([i if i in valid_indices else valid_indices[0] for i in labels])\n            try:\n                fpr, tpr, _ = roc_curve(labels, probs, multi_class='ovr')\n                tprs.append(np.interp(mean_fpr, fpr, tpr))\n                auc_val = auc(mean_fpr, tprs[-1])\n                aucs.append(auc_val)\n                plt.plot(mean_fpr, tprs[-1], alpha=0.3, label=f'Fold {fold + 1} (AUC = {auc_val:.2f})')\n                print(f\"ROC computed for Fold {fold + 1}, AUC: {auc_val:.2f}\")\n            except Exception as e:\n                print(f\"Warning: Error computing ROC for Fold {fold + 1}, Model: {model_name}: {e}\")\n                continue\n            if time.time() - start_time > Config.eval_timeout:\n                print(f\"Warning: Timeout after {Config.eval_timeout}s for Fold {fold + 1}, Model: {model_name}, skipping remaining\")\n                break\n        if tprs:\n            mean_tpr = np.mean(tprs, axis=0)\n            plt.plot(mean_fpr, mean_tpr, 'b', label=f'Mean ROC (AUC = {auc(mean_fpr, mean_tpr):.2f})')\n            plt.fill_between(mean_fpr, mean_tpr - np.std(tprs, axis=0), mean_tpr + np.std(tprs, axis=0), color='b', alpha=0.2)\n        plt.plot([0, 1], [0, 1], 'r--')\n        plt.xlabel('False Positive Rate')\n        plt.ylabel('True Positive Rate')\n        plt.title(f'ROC Curves ({model_name})')\n        plt.legend(loc='lower right')\n        plt.savefig(f\"roc_{model_name}.png\")\n        plt.close()\n        print(f\"Saved ROC curve for {model_name} to roc_{model_name}.png\")\n\n        # Evaluate feature importance (simplified for deep learning)\n        print(f\"\\nFeature Importance Analysis for {model_name}\")\n        model.eval()\n        loc_importance = 0\n        for name, param in model.named_parameters():\n            if 'loc_fc' in name:\n                loc_importance += param.abs().mean().item()\n        print(f\"Average importance of location features: {loc_importance:.4f}\")\n\n        for fold in tqdm(range(Config.n_folds), desc=f\"Generating Confusion Matrices for {model_name}\"):\n            cm = all_metrics[model_name][fold][4]\n            cm_norm = cm / (cm.sum(axis=1, keepdims=True) + 1e-10)\n            plt.figure(figsize=(8, 6))\n            sns.heatmap(cm_norm, annot=False, cmap='Blues')\n            plt.title(f'Confusion Matrix ({model_name}, Fold {fold + 1})')\n            plt.savefig(f\"cm_{model_name}_fold_{fold + 1}.png\")\n            plt.close()\n            print(f\"Saved confusion matrix for {model_name}, Fold {fold + 1} to cm_{model_name}_fold_{fold + 1}.png\")\n\n    final_results = [\n        {\n            'Model': model_name,\n            'Avg AUC': np.mean(fold_aucs[model_name]),\n            **{k: np.mean([m[i] for m in all_metrics[model_name]]) for i, k in enumerate(['Sensitivity', 'Specificity', 'Precision', 'F1'])}\n        }\n        for model_name in [m['name'] for m in models]\n    ]\n    pd.DataFrame(final_results).to_csv('final_results.csv', index=False)\n    print(\"Saved final results to final_results.csv\")\n    print(\"\\nFinal Results Summary:\")\n    for result in final_results:\n        print(f\"Model: {result['Model']}, Avg AUC: {result['Avg AUC']:.4f}, Sensitivity: {result['Sensitivity']:.4f}, Specificity: {result['Specificity']:.4f}, Precision: {result['Precision']:.4f}, F1: {result['F1']:.4f}\")\n\n    # Compare with baseline (no location features)\n    print(\"\\nComparing with baseline (no location features)\")\n    baseline_auc = np.mean(fold_aucs[list(fold_aucs.keys())[0]])  # Assume first model as baseline\n    loc_auc = np.mean(fold_aucs[list(fold_aucs.keys())[0]])\n    print(f\"Baseline AUC (no location): {baseline_auc:.4f}\")\n    print(f\"AUC with location features: {loc_auc:.4f}\")\n    print(f\"Improvement: {(loc_auc - baseline_auc) * 100:.2f}%\")\n\n    print(\"\\nSaving outputs for submission notebook\")\n    os.makedirs(\"/kaggle/working/output\", exist_ok=True)\n    for folder in [Config.model_weights_dir, Config.checkpoint_dir]:\n        if os.path.exists(folder):\n            shutil.copytree(folder, os.path.join(\"/kaggle/working/output\", os.path.basename(folder)), dirs_exist_ok=True)\n    if os.path.exists(\"train_data.csv\"):\n        shutil.copy(\"train_data.csv\", \"/kaggle/working/output/train_data.csv\")\n    if os.path.exists(\"preliminary_results.csv\"):\n        shutil.copy(\"preliminary_results.csv\", \"/kaggle/working/output/preliminary_results.csv\")\n    if os.path.exists(\"final_results.csv\"):\n        shutil.copy(\"final_results.csv\", \"/kaggle/working/output/final_results.csv\")\n    for model_name in [m['name'] for m in models]:\n        roc_path = f\"roc_{model_name}.png\"\n        if os.path.exists(roc_path):\n            shutil.copy(roc_path, f\"/kaggle/working/output/roc_{model_name}.png\")\n        for fold in range(Config.n_folds):\n            cm_path = f\"cm_{model_name}_fold_{fold + 1}.png\"\n            if os.path.exists(cm_path):\n                shutil.copy(cm_path, f\"/kaggle/working/output/cm_{model_name}_fold_{fold + 1}.png\")\n\n    print(f\"Final disk usage: {get_disk_usage():.2f} GiB\")\n    print(\"Training and evaluation complete. Outputs saved in /kaggle/working/output\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T13:18:55.870407Z","iopub.status.idle":"2025-05-25T13:18:55.870701Z","shell.execute_reply.started":"2025-05-25T13:18:55.870538Z","shell.execute_reply":"2025-05-25T13:18:55.870554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clean up feature directory to save space\nshutil.rmtree(Config.feature_dir, ignore_errors=True)\nprint(\"Removed feature directory to save space\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T13:18:55.872154Z","iopub.status.idle":"2025-05-25T13:18:55.872397Z","shell.execute_reply.started":"2025-05-25T13:18:55.872296Z","shell.execute_reply":"2025-05-25T13:18:55.872306Z"}},"outputs":[],"execution_count":null}]}