{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":91844,"databundleVersionId":11361821}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cài thêm nếu chưa có\n!pip install librosa scikit-learn pandas numpy soundfile --quiet\n\nimport os\nimport glob\nimport numpy as np\nimport pandas as pd\nimport librosa\n\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.metrics import roc_auc_score, accuracy_score\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T12:15:21.882733Z","iopub.execute_input":"2025-04-21T12:15:21.883139Z","iopub.status.idle":"2025-04-21T12:15:29.150819Z","shell.execute_reply.started":"2025-04-21T12:15:21.883106Z","shell.execute_reply":"2025-04-21T12:15:29.149476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE = '/kaggle/input/birdclef-2025/'\nmeta = pd.read_csv(os.path.join(BASE, 'train.csv'))\nprint(meta.head())\n# Ví dụ thống kê số lượng samples mỗi loài\nprint(meta.primary_label.value_counts().head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T12:17:57.00857Z","iopub.execute_input":"2025-04-21T12:17:57.00942Z","iopub.status.idle":"2025-04-21T12:17:57.231591Z","shell.execute_reply.started":"2025-04-21T12:17:57.009388Z","shell.execute_reply":"2025-04-21T12:17:57.230524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_features(path, n_mels=64, sr=32000, duration=5):\n    \"\"\"\n    - Load audio, cắt hoặc dãn về đúng duration (giây)\n    - Tính mel-spectrogram và trả về vector trung bình theo thời gian\n    \"\"\"\n    y, _ = librosa.load(path, sr=sr, mono=True, duration=duration)\n    # Nếu file quá ngắn, pad thêm 0\n    if len(y) < sr * duration:\n        y = np.pad(y, (0, sr*duration - len(y)))\n    melspec = librosa.feature.melspectrogram(y, sr=sr, n_mels=n_mels)\n    log_melspec = librosa.power_to_db(melspec)\n    # Trả về vector feature shape=(n_mels,)\n    return log_melspec.mean(axis=1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T12:23:17.779746Z","iopub.execute_input":"2025-04-21T12:23:17.780123Z","iopub.status.idle":"2025-04-21T12:23:17.78719Z","shell.execute_reply.started":"2025-04-21T12:23:17.780097Z","shell.execute_reply":"2025-04-21T12:23:17.786033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lấy sample ngẫu nhiên để demo (thay bằng meta.sample nếu dataset nhỏ)\nSAMPLE_SIZE = 5000\ndf = meta.sample(SAMPLE_SIZE, random_state=42).reset_index(drop=True)\n\n# Trích feature và label\nX, y = [], []\nfor i, row in df.iterrows():\n    audio_path = os.path.join(BASE, 'train_audio', row.filename)\n    feat = extract_features(audio_path)\n    X.append(feat)\n    y.append(row.primary_label)\n\nX = np.vstack(X)\n# Chuyển label sang số\nlabels, y_enc = np.unique(y, return_inverse=True)\n\n# Chia train/val\nX_train, X_val, y_train, y_val = train_test_split(\n    X, y_enc, test_size=0.2, random_state=42\n)\nprint(\"Train:\", X_train.shape, \"Val:\", X_val.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T12:35:24.524078Z","iopub.execute_input":"2025-04-21T12:35:24.525344Z","iopub.status.idle":"2025-04-21T12:37:08.719739Z","shell.execute_reply.started":"2025-04-21T12:35:24.525305Z","shell.execute_reply":"2025-04-21T12:37:08.718686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelBinarizer\nfrom sklearn.metrics import roc_auc_score\nimport numpy as np\n\n# 1. Lấy ra các lớp mà RF đã học\nclasses = rf.classes_\nprint(f\"RF learned {len(classes)} classes\")\n\n# 2. Khởi tạo và fit LabelBinarizer với đúng các lớp đó\nlb = LabelBinarizer()\nlb.fit(classes)\n\n# 3. Chuyển y_val thành one-hot theo classes\ny_true_bin = lb.transform(y_val)       # shape (n_samples, n_classes)\n\n# 4. Lấy xác suất dự đoán\ny_pred_prob = rf.predict_proba(X_val)  # shape (n_samples, n_classes)\n\n# 5. Tính AUC cho mỗi class, bỏ qua các class không có cả positive & negative\naucs = []\nskipped = []\nfor i, cls in enumerate(classes):\n    yt = y_true_bin[:, i]\n    ys = y_pred_prob[:, i]\n    if len(np.unique(yt)) == 2:\n        aucs.append(roc_auc_score(yt, ys))\n    else:\n        skipped.append(cls)\n\nprint(f\"Skipped {len(skipped)} classes (only one label present): {skipped}\")\n# 6. Tính macro‑average trên các class còn lại\nmacro_auc = np.mean(aucs)\nprint(f\"Validation ROC‑AUC (macro over {len(aucs)}/{len(classes)} classes): {macro_auc:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T12:43:46.145018Z","iopub.execute_input":"2025-04-21T12:43:46.145432Z","iopub.status.idle":"2025-04-21T12:43:46.512781Z","shell.execute_reply.started":"2025-04-21T12:43:46.145404Z","shell.execute_reply":"2025-04-21T12:43:46.511814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Duyệt tất cả file soundscape (*.ogg)\ntest_paths = sorted(glob.glob(os.path.join(BASE, 'train_soundscapes', '*.ogg')))\n\nrows = []\nfor path in test_paths:\n    feat = extract_features(path)\n    probs = rf.predict_proba(feat.reshape(1, -1))[0]\n\n    fname = os.path.basename(path)\n    # với mỗi class, tạo 1 dòng row_id và target\n    for idx, p in enumerate(probs):\n        rows.append({\n            'row_id': f\"{fname}_{labels[idx]}\",\n            'target': p\n        })\n\nsub_df = pd.DataFrame(rows)\nsub_df.to_csv('submission.csv', index=False)\nprint(\"Đã tạo submission.csv với\", len(sub_df), \"dòng\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T12:44:06.423431Z","iopub.execute_input":"2025-04-21T12:44:06.423858Z","iopub.status.idle":"2025-04-21T12:57:12.674106Z","shell.execute_reply.started":"2025-04-21T12:44:06.423831Z","shell.execute_reply":"2025-04-21T12:57:12.673145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# 1) Đọc file submission\nsub = pd.read_csv('submission.csv')\n\n# 2) Xem 10 dòng đầu\nprint(sub.head(10))\n\n# 3) Và 10 dòng cuối\nprint(sub.tail(10))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T13:06:15.085402Z","iopub.execute_input":"2025-04-21T13:06:15.08573Z","iopub.status.idle":"2025-04-21T13:06:17.057081Z","shell.execute_reply.started":"2025-04-21T13:06:15.085709Z","shell.execute_reply":"2025-04-21T13:06:17.056156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tách phần file name từ row_id\nsub['file'] = sub['row_id'].apply(lambda x: x.split('_')[0])\n\n# Chọn một file bất kỳ, ví dụ file đầu tiên\nfile0 = sub['file'].unique()[0]\nprint(\"File mẫu:\", file0)\n\n# Lấy các dự đoán cho file đó, sắp xếp theo target giảm dần\npreds0 = sub[sub['file'] == file0].sort_values('target', ascending=False)\nprint(preds0.head(5))   # 5 loài đầu với xác suất cao nhất\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T13:06:54.814001Z","iopub.execute_input":"2025-04-21T13:06:54.815144Z","iopub.status.idle":"2025-04-21T13:06:55.794768Z","shell.execute_reply.started":"2025-04-21T13:06:54.815106Z","shell.execute_reply":"2025-04-21T13:06:55.793545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# 1. Dự đoán nhãn và xác suất\ny_pred      = rf.predict(X_val)            # array shape (n_samples,)\ny_pred_prob = rf.predict_proba(X_val)      # array shape (n_samples, n_classes_trained)\n\n# 2. Chuẩn bị mapping: cột j của y_pred_prob tương ứng với lớp rf.classes_[j]\nclass_order = rf.classes_                  # array các integer labels RF học được\nclass_labels = labels[class_order]         # array tên loài tương ứng\n\n# 3. In 10 mẫu đầu\nfor i in range(10):\n    true_lbl = labels[y_val[i]]            # nhãn thật (tên loài)\n    pred_idx = y_pred[i]                   # integer label do RF trả về\n    \n    # Tìm vị trí cột trong class_order\n    col = np.where(class_order == pred_idx)[0][0]\n    \n    pred_lbl = class_labels[col]           # tên loài dự đoán\n    prob     = y_pred_prob[i, col]         # xác suất cột đó\n    \n    print(f\"Mẫu {i}: True = {true_lbl}, Pred = {pred_lbl} (p={prob:.3f})\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T13:08:03.295561Z","iopub.execute_input":"2025-04-21T13:08:03.295893Z","iopub.status.idle":"2025-04-21T13:08:03.626919Z","shell.execute_reply.started":"2025-04-21T13:08:03.295869Z","shell.execute_reply":"2025-04-21T13:08:03.625919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}