{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":21669,"databundleVersionId":1692278}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport cv2\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom sklearn.model_selection import StratifiedGroupKFold\nimport tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB1\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\nimport shutil\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-19T19:20:44.782788Z","iopub.execute_input":"2026-03-19T19:20:44.783102Z","iopub.status.idle":"2026-03-19T19:21:08.897903Z","shell.execute_reply.started":"2026-03-19T19:20:44.783077Z","shell.execute_reply":"2026-03-19T19:21:08.897149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"config = {\n    'SR': 48000,\n    'DURATION': 10,\n    'N_MELS': 224,\n    'TIME_STEPS': 400,\n    'BATCH_SIZE': 32,\n    'EPOCHS': 20,\n    'SEED': 42,\n    'NUM_CLASSES': 24,\n    'N_FOLDS': 5\n}\n\nDIR_TRAIN = '/kaggle/input/competitions/rfcx-species-audio-detection/train'\nDIR_TEST = '/kaggle/input/competitions/rfcx-species-audio-detection/test'\nCSV_TRAIN = '/kaggle/input/competitions/rfcx-species-audio-detection/train_tp.csv'\n\nOUTPUT_DIR = Path('/kaggle/working/train_images/')\n\nnp.random.seed(config['SEED'])\ntf.random.set_seed(config['SEED'])\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-19T19:21:08.899327Z","iopub.execute_input":"2026-03-19T19:21:08.899943Z","iopub.status.idle":"2026-03-19T19:21:08.904827Z","shell.execute_reply.started":"2026-03-19T19:21:08.899917Z","shell.execute_reply":"2026-03-19T19:21:08.904177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv(CSV_TRAIN)\ndf_train['image_name'] = df_train.apply(lambda row: f\"{row['recording_id']}_{row.name}.png\", axis=1)\n\ndef create_spectrograms(df, output_dir, config):\n    if output_dir.exists():\n        print(\"Images already exist. Skipping...\")\n        return\n        \n    output_dir.mkdir(parents=True, exist_ok=True)\n    print(\"Creating spectrograms...\")\n    \n    for idx, row in tqdm(df.iterrows(), total=len(df)):\n        file_path = os.path.join(DIR_TRAIN, f\"{row['recording_id']}.flac\")\n        try:\n            audio, sr = librosa.load(file_path, sr=config['SR'])\n            \n            t_min, t_max = row['t_min'], row['t_max']\n            center_sec = (t_min + t_max) / 2\n            half_window = (config['DURATION'] * config['SR']) // 2\n            center_idx = int(center_sec * config['SR'])\n            \n            start_idx = max(0, center_idx - half_window)\n            end_idx = start_idx + (config['DURATION'] * config['SR'])\n            \n            if end_idx > len(audio):\n                end_idx = len(audio)\n                start_idx = max(0, end_idx - (config['DURATION'] * config['SR']))\n                \n            chunk = audio[start_idx:end_idx]\n            \n            melspec = librosa.feature.melspectrogram(\n                y=chunk, sr=config['SR'], n_mels=config['N_MELS'], fmin=50, fmax=14000\n            )\n            melspec = librosa.power_to_db(melspec, top_db=80)\n            \n            eps = 1e-6\n            mu = melspec.mean()\n            sigma = melspec.std()\n            z = (melspec - mu) / (sigma + eps)\n            \n            lo, hi = z.min(), z.max()\n            img = 255.0 * (z - lo) / (hi - lo + eps)\n            \n            img_uint8 = img.astype(np.uint8)\n            img_uint8 = cv2.resize(img_uint8, (config['TIME_STEPS'], config['N_MELS']))\n            \n            cv2.imwrite(str(output_dir / row['image_name']), img_uint8)\n        except Exception as e:\n            print(f\"Error: {e}\")\n\ncreate_spectrograms(df_train, OUTPUT_DIR, config)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-19T19:21:08.905778Z","iopub.execute_input":"2026-03-19T19:21:08.906360Z","iopub.status.idle":"2026-03-19T19:21:08.927909Z","shell.execute_reply.started":"2026-03-19T19:21:08.906328Z","shell.execute_reply":"2026-03-19T19:21:08.927258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RFCXDataGen(tf.keras.utils.Sequence):\n    def __init__(self, df, img_dir, batch_size, is_train=True):\n        self.df = df.reset_index(drop=True)\n        self.img_dir = img_dir\n        self.batch_size = batch_size\n        self.is_train = is_train\n        self.indices = np.arange(len(self.df))\n        if self.is_train:\n            np.random.shuffle(self.indices)\n            \n    def __len__(self):\n        return int(np.ceil(len(self.df) / self.batch_size))\n        \n    def __getitem__(self, index):\n        batch_idx = self.indices[index*self.batch_size : (index+1)*self.batch_size]\n        batch_df = self.df.iloc[batch_idx]\n        \n        X, y = [],[]\n        for _, row in batch_df.iterrows():\n            img_path = self.img_dir / row['image_name']\n            img = cv2.imread(str(img_path), cv2.IMREAD_GRAYSCALE)\n            if img is None:\n                img = np.zeros((config['N_MELS'], config['TIME_STEPS']), dtype=np.uint8)\n                \n            img = np.stack((img, img, img), axis=-1)\n            \n            if self.is_train and np.random.rand() > 0.5:\n                img = img[:, ::-1, :]\n                \n            X.append(img)\n            y.append(int(row['species_id']))\n            \n        X = np.array(X, dtype=np.float32)\n        y = tf.keras.utils.to_categorical(y, num_classes=config['NUM_CLASSES'])\n        return X, y\n        \n    def on_epoch_end(self):\n        if self.is_train:\n            np.random.shuffle(self.indices)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-19T19:21:08.929702Z","iopub.execute_input":"2026-03-19T19:21:08.929991Z","iopub.status.idle":"2026-03-19T19:24:40.351257Z","shell.execute_reply.started":"2026-03-19T19:21:08.929950Z","shell.execute_reply":"2026-03-19T19:24:40.350487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_model():\n    base_model = EfficientNetB1(\n        include_top=False, \n        weights='imagenet',\n        input_shape=(config['N_MELS'], config['TIME_STEPS'], 3) \n    )\n    base_model.trainable = True\n\n    model = Sequential([\n        base_model,\n        GlobalAveragePooling2D(),\n        BatchNormalization(),\n        Dropout(0.3),\n        Dense(config['NUM_CLASSES'], activation='softmax') \n    ])\n    \n    model.compile(\n        optimizer=Adam(learning_rate=1e-3),\n        loss='categorical_crossentropy',\n        metrics=['accuracy']\n    )\n    return model\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-19T19:24:42.791312Z","iopub.execute_input":"2026-03-19T19:24:42.791568Z","iopub.status.idle":"2026-03-19T19:24:42.796517Z","shell.execute_reply.started":"2026-03-19T19:24:42.791539Z","shell.execute_reply":"2026-03-19T19:24:42.795929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sgkf = StratifiedGroupKFold(n_splits=config['N_FOLDS'])\n\ntrained_models_paths =[]\n\nfor fold, (train_idx, val_idx) in enumerate(sgkf.split(df_train, df_train['species_id'], groups=df_train['recording_id'])):\n    train_df = df_train.iloc[train_idx]\n    val_df = df_train.iloc[val_idx]\n    \n    train_gen = RFCXDataGen(train_df, OUTPUT_DIR, config['BATCH_SIZE'], is_train=True)\n    val_gen = RFCXDataGen(val_df, OUTPUT_DIR, config['BATCH_SIZE'], is_train=False)\n    \n    model = get_model()\n    model_path = f'rfcx_model_fold_{fold}.keras'\n    \n    callbacks =[\n        ModelCheckpoint(model_path, monitor='val_loss', save_best_only=True),\n        ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=2),\n        EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\n    ]\n    \n    model.fit(train_gen, validation_data=val_gen, epochs=config['EPOCHS'], callbacks=callbacks, verbose=1)\n    trained_models_paths.append(model_path)\n    \n    del model\n    tf.keras.backend.clear_session()\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-19T19:24:42.797399Z","iopub.execute_input":"2026-03-19T19:24:42.797601Z","iopub.status.idle":"2026-03-19T19:27:36.136204Z","shell.execute_reply.started":"2026-03-19T19:24:42.797584Z","shell.execute_reply":"2026-03-19T19:27:36.134489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if OUTPUT_DIR.exists():\n    shutil.rmtree(OUTPUT_DIR)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-19T19:27:36.137292Z","iopub.execute_input":"2026-03-19T19:27:36.137588Z","iopub.status.idle":"2026-03-19T19:27:36.172538Z","shell.execute_reply.started":"2026-03-19T19:27:36.137560Z","shell.execute_reply":"2026-03-19T19:27:36.171995Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_submission_ensemble(model_paths):\n    files =[f for f in os.listdir(DIR_TEST) if f.endswith('.flac')]\n    models =[tf.keras.models.load_model(path) for path in model_paths]\n    \n    submission_data =[]\n    print(\"Predicting test data...\")\n    \n    for filename in tqdm(files):\n        path = os.path.join(DIR_TEST, filename)\n        recording_id = Path(filename).stem\n        \n        y_full, sr = librosa.load(path, sr=config['SR'])\n        step = config['SR'] * config['DURATION']\n        segments = int(np.ceil(len(y_full) / step))\n        \n        batch_imgs =[]\n        for i in range(segments):\n            if (i + 1) * step > len(y_full):\n                chunk = y_full[len(y_full) - step : len(y_full)]\n            else:\n                chunk = y_full[i * step : (i + 1) * step]\n                \n            melspec = librosa.feature.melspectrogram(\n                y=chunk, sr=config['SR'], n_mels=config['N_MELS'], fmin=50, fmax=14000\n            )\n            melspec = librosa.power_to_db(melspec, top_db=80)\n            \n            eps = 1e-6\n            mu = melspec.mean()\n            sigma = melspec.std()\n            z = (melspec - mu) / (sigma + eps)\n            lo, hi = z.min(), z.max()\n            img = 255.0 * (z - lo) / (hi - lo + eps)\n            \n            img = cv2.resize(img.astype(np.uint8), (config['TIME_STEPS'], config['N_MELS']))\n            img = np.stack((img, img, img), axis=-1).astype(np.float32)\n            batch_imgs.append(img)\n            \n        batch_imgs = np.array(batch_imgs)\n        \n        ensemble_preds =[]\n        for model in models:\n            preds = model.predict(batch_imgs, verbose=0)\n            max_preds = np.max(preds, axis=0)\n            ensemble_preds.append(max_preds)\n            \n        final_preds = np.mean(ensemble_preds, axis=0)\n        submission_data.append([recording_id] + final_preds.tolist())\n            \n    cols =['recording_id'] + [f's{i}' for i in range(24)]\n    df_sub = pd.DataFrame(submission_data, columns=cols)\n    df_sub = df_sub.sort_values('recording_id') \n    df_sub.to_csv('submission.csv', index=False)\n    print(\"Saved submission.csv\")\n\npredict_submission_ensemble(trained_models_paths)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-19T19:27:36.173297Z","iopub.execute_input":"2026-03-19T19:27:36.173559Z","iopub.status.idle":"2026-03-19T19:39:59.579538Z","shell.execute_reply.started":"2026-03-19T19:27:36.173527Z","shell.execute_reply":"2026-03-19T19:39:59.578788Z"}},"outputs":[],"execution_count":null}]}