{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv(\"/kaggle/input/birdclef-2025/train.csv\")\nfreq_table = df[\"primary_label\"].value_counts()\nprint(freq_table.head(20)) \n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-28T02:32:07.581414Z","iopub.execute_input":"2025-04-28T02:32:07.581728Z","iopub.status.idle":"2025-04-28T02:32:07.708129Z","shell.execute_reply.started":"2025-04-28T02:32:07.581707Z","shell.execute_reply":"2025-04-28T02:32:07.707254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top10 = ['grekis', 'compau', 'trokin', 'roahaw', 'banana', 'whtdov', 'socfly1', 'yeofly1', 'bobfly1', 'wbwwre1']\n\n# Positive = grekis\npos_df = df[df['primary_label'] == 'grekis'].copy()\npos_df['label'] = 1\n\n# Negative = rest of top 10 (excluding grekis)\nneg_df = df[df['primary_label'].isin(top10) & (df['primary_label'] != 'grekis')].copy()\nneg_df['label'] = 0\n\nfrom sklearn.utils import resample\n\n# Downsample negatives to match positives\nneg_df_balanced = resample(neg_df, \n                           replace=False, \n                           n_samples=len(pos_df), \n                           random_state=42)\n\ncombined_df = pd.concat([pos_df, neg_df_balanced]).sample(frac=1, random_state=42)\n\n# Combine\ncombined_df = pd.concat([pos_df, neg_df], ignore_index=True).sample(frac=1, random_state=42)\n\n# Preview\ncombined_df[['primary_label', 'filename', 'label']].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T02:32:07.709597Z","iopub.execute_input":"2025-04-28T02:32:07.709881Z","iopub.status.idle":"2025-04-28T02:32:07.744104Z","shell.execute_reply.started":"2025-04-28T02:32:07.709851Z","shell.execute_reply":"2025-04-28T02:32:07.743244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"neg_df = df[df['primary_label'].isin(top10) & (df['primary_label'] != 'grekis')].copy()\nneg_df['label'] = 0\n\nfrom sklearn.utils import resample\n\nneg_df_balanced = resample(neg_df, \n                           replace=False, \n                           n_samples=len(pos_df), \n                           random_state=42)\n\ncombined_df = pd.concat([pos_df, neg_df_balanced]).sample(frac=1, random_state=42)\ncombined_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T02:32:07.744797Z","iopub.execute_input":"2025-04-28T02:32:07.745103Z","iopub.status.idle":"2025-04-28T02:32:07.776409Z","shell.execute_reply.started":"2025-04-28T02:32:07.745084Z","shell.execute_reply":"2025-04-28T02:32:07.775537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport librosa\n\ndef extract_binary_features(filepath, sr=32000, n_fft=1024, hop_length=512, percentile=85):\n    \"\"\"\n    Converts an audio file into a 3200-dim binary frequency-bin vector using adaptive thresholding.\n    \n    Args:\n        filepath (str): Full path to the audio file\n        sr (int): Sampling rate\n        n_fft (int): FFT window size\n        hop_length (int): Number of samples between frames\n        percentile (float): Percentile threshold for activity in a frequency bin (0–100)\n\n    Returns:\n        np.ndarray: Binary vector of shape (3200,)\n    \"\"\"\n    # Load audio\n    y, sr_actual = librosa.load(filepath, sr=sr)\n    \n    # Compute STFT (complex)\n    S = librosa.stft(y, n_fft=n_fft, hop_length=hop_length)\n    \n    # Power and convert to decibels\n    S_power = np.abs(S) ** 2\n    S_db = librosa.power_to_db(S_power, ref=np.max)\n\n    # Frequency values\n    freqs = librosa.fft_frequencies(sr=sr, n_fft=n_fft)\n\n    # Frequency binning: 0 to 16000 Hz, in 5 Hz intervals → 3200 bins\n    bin_edges = np.linspace(0, 16000, 3201)\n    binary_vector = np.zeros(3200, dtype=int)\n\n    # Compute threshold based on global energy distribution\n    energy_per_freq = S_db.max(axis=1)  # Max across time for each frequency\n    adaptive_threshold = np.percentile(energy_per_freq, percentile)\n\n    # Assign 1s where max energy in bin > threshold\n    for i in range(3200):\n        f_start = bin_edges[i]\n        f_end = bin_edges[i+1]\n        bin_mask = (freqs >= f_start) & (freqs < f_end)\n        \n        if np.any(bin_mask):\n            max_energy = np.max(energy_per_freq[bin_mask])\n            if max_energy > adaptive_threshold:\n                binary_vector[i] = 1\n\n    return binary_vector","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2025-04-28T02:32:07.777383Z","iopub.execute_input":"2025-04-28T02:32:07.777607Z","iopub.status.idle":"2025-04-28T02:32:07.785927Z","shell.execute_reply.started":"2025-04-28T02:32:07.777590Z","shell.execute_reply":"2025-04-28T02:32:07.784894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = \"/kaggle/input/birdclef-2025/train_audio/\" + combined_df.iloc[0][\"filename\"]\nvec = extract_binary_features(path)\nprint(vec.shape) \nprint(vec.sum()) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T02:32:07.788037Z","iopub.execute_input":"2025-04-28T02:32:07.788344Z","iopub.status.idle":"2025-04-28T02:32:07.880540Z","shell.execute_reply.started":"2025-04-28T02:32:07.788322Z","shell.execute_reply":"2025-04-28T02:32:07.879592Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Updated Binary Vector by Chunking the Audio files and Aggregating them using Median","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport librosa\nimport soundfile as sf\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import f1_score\nfrom xgboost import XGBClassifier\nfrom tqdm import tqdm\n\n# Settings\nthresholds_to_try = np.linspace(70, 90, 10)  # 10 thresholds\nn_splits = 5\nAUDIO_BASE = \"/kaggle/input/birdclef-2025/train_audio/\"\nN_FILES = 100  \nresults = {}\n\ndef extract_binary_features(path, percentile=85):\n    y, sr = librosa.load(path, sr=32000)\n    S = librosa.stft(y, n_fft=1024, hop_length=512)\n    S_power = np.abs(S) ** 2\n    S_db = librosa.power_to_db(S_power, ref=np.max)\n    freqs = librosa.fft_frequencies(sr=sr, n_fft=1024)\n    bin_edges = np.linspace(0, 16000, 3201)\n    binary_vector = np.zeros(3200, dtype=int)\n    energy_per_freq = S_db.max(axis=1)\n    adaptive_threshold = np.percentile(energy_per_freq, percentile)\n\n    for j in range(3200):\n        f_start = bin_edges[j]\n        f_end = bin_edges[j+1]\n        bin_mask = (freqs >= f_start) & (freqs < f_end)\n        if np.any(bin_mask):\n            max_energy = np.max(energy_per_freq[bin_mask])\n            if max_energy > adaptive_threshold:\n                binary_vector[j] = 1\n\n    return binary_vector\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T02:38:24.841419Z","iopub.execute_input":"2025-04-28T02:38:24.841745Z","iopub.status.idle":"2025-04-28T02:38:25.111556Z","shell.execute_reply.started":"2025-04-28T02:38:24.841720Z","shell.execute_reply":"2025-04-28T02:38:25.110657Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Implemented Threshold Tuning for Optimised Threshold","metadata":{}},{"cell_type":"code","source":"# Main Loop\nfor thresh in thresholds_to_try:\n    print(f\"\\n🔵 Threshold = {thresh:.1f}%\")\n\n    X = []\n    y = []\n\n    for i in tqdm(range(N_FILES), desc=f\"Extracting with {thresh:.1f}\"):\n        path = AUDIO_BASE + combined_df.iloc[i][\"filename\"]\n\n        try:\n            vec = extract_binary_features(path, percentile=thresh)\n            X.append(vec)\n            y.append(combined_df.iloc[i]['label'])\n        except Exception as e:\n            print(f\"❌ Failed for file {i}: {e}\")\n\n    X = np.stack(X)\n    y = np.array(y)\n\n    # 5-Fold Cross-Validation\n    f1_scores = []\n\n    skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n\n    for train_idx, val_idx in skf.split(X, y):\n        X_train, X_val = X[train_idx], X[val_idx]\n        y_train, y_val = y[train_idx], y[val_idx]\n\n        model = XGBClassifier(\n            n_estimators=500,\n            scale_pos_weight=(len(y_train) - sum(y_train)) / sum(y_train),\n            use_label_encoder=False,\n            eval_metric='logloss',\n            random_state=42\n        )\n\n        model.fit(X_train, y_train)\n        y_pred = model.predict(X_val)\n        f1 = f1_score(y_val, y_pred)\n        f1_scores.append(f1)\n\n    avg_f1 = np.mean(f1_scores)\n    print(f\"✅ Average F1 @ {thresh:.1f}% = {avg_f1:.4f}\")\n\n    results[thresh] = avg_f1\n\n# Summary\nprint(\"\\n📊 Final Results:\")\nfor k, v in results.items():\n    print(f\"Threshold {k:.1f}% -> Avg F1: {v:.4f}\")\n\n# Plot\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(8,5))\nplt.plot(list(results.keys()), list(results.values()), marker='o')\nplt.xlabel('Percentile Threshold')\nplt.ylabel('Average F1-Score (5-Fold CV)')\nplt.title('Threshold vs F1-Score')\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T02:38:27.454979Z","iopub.execute_input":"2025-04-28T02:38:27.455295Z","iopub.status.idle":"2025-04-28T02:40:40.460046Z","shell.execute_reply.started":"2025-04-28T02:38:27.455273Z","shell.execute_reply":"2025-04-28T02:40:40.459101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = \"/kaggle/input/birdclef-2025/train_audio/\" + combined_df.iloc[0][\"filename\"]\nvec = extract_binary_features(path, percentile=76.7)\nprint(vec.shape) \nprint(vec.sum()) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T02:40:44.796097Z","iopub.execute_input":"2025-04-28T02:40:44.796418Z","iopub.status.idle":"2025-04-28T02:40:44.875162Z","shell.execute_reply.started":"2025-04-28T02:40:44.796396Z","shell.execute_reply":"2025-04-28T02:40:44.874245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\nplt.figure(figsize=(6,4))\nsns.countplot(x=y)\nplt.title('Positive vs Negative Samples (After Processing)')\nplt.xlabel('Class (0=Negative, 1=Positive)')\nplt.ylabel('Count')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T02:51:17.492806Z","iopub.execute_input":"2025-04-28T02:51:17.493190Z","iopub.status.idle":"2025-04-28T02:51:18.231538Z","shell.execute_reply.started":"2025-04-28T02:51:17.493166Z","shell.execute_reply":"2025-04-28T02:51:18.230575Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Implementing Optimised Threshold for Training Data","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport soundfile as sf\nimport librosa\nfrom tqdm import tqdm\n\n# Storage\nX = []\ny = []\n\n# Full base path to audio files\nAUDIO_BASE = \"/kaggle/input/birdclef-2025/train_audio/\"\n\nfor _, row in tqdm(combined_df.iterrows(), total=len(combined_df)):\n    full_path = os.path.join(AUDIO_BASE, row['filename'])\n\n    try:\n        # Load audio\n        y_raw, sr = sf.read(full_path)\n        samples_per_chunk = sr * 5  # 5 seconds\n        num_chunks = len(y_raw) // samples_per_chunk\n\n        if num_chunks == 0:\n            continue  # skip if too short\n\n        # Extract binary vectors per chunk\n        chunk_features = []\n\n        for i in range(num_chunks):\n            start = i * samples_per_chunk\n            end = start + samples_per_chunk\n            chunk = y_raw[start:end]\n\n            # Extract features for chunk\n            S = librosa.stft(chunk, n_fft=1024, hop_length=512)\n            S_power = np.abs(S) ** 2\n            S_db = librosa.power_to_db(S_power, ref=np.max)\n            freqs = librosa.fft_frequencies(sr=sr, n_fft=1024)\n            bin_edges = np.linspace(0, 16000, 3201)\n            binary_vector = np.zeros(3200, dtype=int)\n            energy_per_freq = S_db.max(axis=1)\n            adaptive_threshold = np.percentile(energy_per_freq, 85)\n\n            for j in range(3200):\n                f_start = bin_edges[j]\n                f_end = bin_edges[j+1]\n                bin_mask = (freqs >= f_start) & (freqs < f_end)\n                if np.any(bin_mask):\n                    max_energy = np.max(energy_per_freq[bin_mask])\n                    if max_energy > adaptive_threshold:\n                        binary_vector[j] = 1\n\n            chunk_features.append(binary_vector)\n\n        # Now aggregate across all chunks\n        chunk_features = np.array(chunk_features)\n\n        # median aggregation across chunks\n        final_vector = np.median(chunk_features, axis=0)\n\n        # binarize again (keep 0 or 1 only)\n        final_vector = (final_vector > 0.5).astype(int)\n\n        # Save final vector\n        X.append(final_vector)\n        y.append(row['label'])\n\n    except Exception as e:\n        print(f\"❌ Failed for {row['filename']}: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T02:40:47.073305Z","iopub.execute_input":"2025-04-28T02:40:47.074028Z","iopub.status.idle":"2025-04-28T02:51:14.092066Z","shell.execute_reply.started":"2025-04-28T02:40:47.074002Z","shell.execute_reply":"2025-04-28T02:51:14.091108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = np.stack(X)\ny = np.array(y)\n\nprint(\"✅ Final Feature Matrix shape:\", X.shape)\nprint(\"✅ Final Labels shape:\", y.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T02:51:24.239597Z","iopub.execute_input":"2025-04-28T02:51:24.240571Z","iopub.status.idle":"2025-04-28T02:51:24.260495Z","shell.execute_reply.started":"2025-04-28T02:51:24.240543Z","shell.execute_reply":"2025-04-28T02:51:24.259652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import f1_score, classification_report, confusion_matrix\nimport numpy as np\n\nn_splits = 5\nskf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n\nf1_scores_rf = []\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(X, y)):\n    print(f\"\\n🔵 Fold {fold+1}/{n_splits}\")\n\n    X_train, X_val = X[train_idx], X[val_idx]\n    y_train, y_val = y[train_idx], y[val_idx]\n\n    clf_rf = RandomForestClassifier(\n        n_estimators=500,\n        class_weight='balanced',\n        random_state=42,\n        n_jobs=-1  # use all cores\n    )\n\n    clf_rf.fit(X_train, y_train)\n    y_pred_rf = clf_rf.predict(X_val)\n\n    f1 = f1_score(y_val, y_pred_rf)\n    print(f\"✅ Fold {fold+1} F1-Score: {f1:.4f}\")\n\n    f1_scores_rf.append(f1)\n\n    # (Optional) Print confusion matrix\n    # print(confusion_matrix(y_val, y_pred_rf))\n\n# Final average\navg_f1_rf = np.mean(f1_scores_rf)\nprint(f\"\\n📊 Average F1-Score across {n_splits} folds: {avg_f1_rf:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T02:51:26.340668Z","iopub.execute_input":"2025-04-28T02:51:26.341028Z","iopub.status.idle":"2025-04-28T02:51:40.061374Z","shell.execute_reply.started":"2025-04-28T02:51:26.341002Z","shell.execute_reply":"2025-04-28T02:51:40.060381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(8,5))\nplt.bar(range(1, len(f1_scores_rf)+1), f1_scores_rf, color='skyblue')\nplt.ylim(0.7,0.8)\nplt.xlabel('Fold')\nplt.ylabel('F1 Score')\nplt.title('Fold-wise F1 Scores (Random Forest)')\nplt.grid(axis='y')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T02:52:33.256114Z","iopub.execute_input":"2025-04-28T02:52:33.256416Z","iopub.status.idle":"2025-04-28T02:52:33.427788Z","shell.execute_reply.started":"2025-04-28T02:52:33.256394Z","shell.execute_reply":"2025-04-28T02:52:33.426895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\ncm = confusion_matrix(y_val, y_pred_rf)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm)\ndisp.plot()\nplt.title(\"Confusion Matrix on Validation Fold\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T02:52:36.213629Z","iopub.execute_input":"2025-04-28T02:52:36.213969Z","iopub.status.idle":"2025-04-28T02:52:36.404588Z","shell.execute_reply.started":"2025-04-28T02:52:36.213945Z","shell.execute_reply":"2025-04-28T02:52:36.403615Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission Format","metadata":{}},{"cell_type":"code","source":"import os\nimport librosa\nimport numpy as np\nimport pandas as pd\n\nAUDIO_BASE_TEST = '/kaggle/input/birdclef-2025/test_soundscapes/'\n\n# Class labels\nclass_labels = sorted(os.listdir('/kaggle/input/birdclef-2025/train_audio/'))\n\n# Your known target bird\ntarget_bird = 'grekis'  # model trained for grekis detection\n\n# Submission dataframe\nsubmission_rows = []\n\n# For each soundscape\ntest_soundscapes = [os.path.join(AUDIO_BASE_TEST, f) for f in sorted(os.listdir(AUDIO_BASE_TEST)) if f.endswith('.ogg')]\n\nfor soundscape_path in test_soundscapes:\n    sig, sr = librosa.load(soundscape_path, sr=32000)\n\n    samples_per_chunk = sr * 5\n\n    for i in range(0, len(sig), samples_per_chunk):\n        chunk = sig[i:i+samples_per_chunk]\n        if len(chunk) < samples_per_chunk:\n            continue  # skip short last chunk\n\n        # extract binary features\n        try:\n            S = librosa.stft(chunk, n_fft=1024, hop_length=512)\n            S_power = np.abs(S)**2\n            S_db = librosa.power_to_db(S_power, ref=np.max)\n            freqs = librosa.fft_frequencies(sr=sr, n_fft=1024)\n            bin_edges = np.linspace(0, 16000, 3201)\n            binary_vector = np.zeros(3200, dtype=int)\n            energy_per_freq = S_db.max(axis=1)\n            adaptive_threshold = np.percentile(energy_per_freq, 76.7)\n\n            for j in range(3200):\n                f_start = bin_edges[j]\n                f_end = bin_edges[j+1]\n                bin_mask = (freqs >= f_start) & (freqs < f_end)\n                if np.any(bin_mask):\n                    max_energy = np.max(energy_per_freq[bin_mask])\n                    if max_energy > adaptive_threshold:\n                        binary_vector[j] = 1\n\n            # Predict\n            pred_prob = clf_rf.predict_proba(binary_vector.reshape(1, -1))[:,1][0]\n\n            # Build prediction row\n            row_id = os.path.basename(soundscape_path).split('.')[0] + f'_{(i//samples_per_chunk+1)*5}'\n            row = [row_id]\n\n            for bird in class_labels:\n                if bird == target_bird:\n                    row.append(pred_prob)\n                else:\n                    row.append(0.001)  # small probability for all other birds\n\n            submission_rows.append(row)\n\n        except Exception as e:\n            print(f\"❌ Error in {soundscape_path}: {e}\")\n\n# Create dataframe\nsubmission_df = pd.DataFrame(submission_rows, columns=['row_id'] + class_labels)\n\n# Save\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T02:52:40.158000Z","iopub.execute_input":"2025-04-28T02:52:40.158704Z","iopub.status.idle":"2025-04-28T02:52:40.194649Z","shell.execute_reply.started":"2025-04-28T02:52:40.158671Z","shell.execute_reply":"2025-04-28T02:52:40.193385Z"}},"outputs":[],"execution_count":null}]}