{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom tqdm import tqdm\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport librosa.display\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import LabelEncoder\nimport json\n\nfrom pathlib import Path\n\nimport gc\n\nfrom sklearn.model_selection import train_test_split\nfrom keras.utils import to_categorical\nfrom keras.layers import Input, Convolution2D, MaxPooling2D, GlobalAveragePooling2D, Dense, Dropout, Flatten, BatchNormalization, Reshape, GRU\nfrom keras.models import Model, load_model, Sequential\nfrom keras.optimizers import Adam\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\ndataset_path = Path(\"../input/birdclef-2021/\")\n\nMAX_LENGTH = 40\n\ntrain_metadata_file = dataset_path / \"train_metadata.csv\"\ntrain_dir_short = dataset_path / \"train_short_audio\"\ntrain_data_df = pd.read_csv(train_metadata_file)\ntrain_data_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T07:37:56.981582Z","iopub.execute_input":"2025-11-16T07:37:56.981843Z","iopub.status.idle":"2025-11-16T07:38:12.066222Z","shell.execute_reply.started":"2025-11-16T07:37:56.981822Z","shell.execute_reply":"2025-11-16T07:38:12.065405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TEST_AUDIO_PATH = os.path.join(dataset_path, \"test_soundscapes\")\ntest_meta = pd.read_csv(os.path.join(dataset_path, \"test.csv\"))\nuse_train_as_test = len(test_meta) < 10\naudio_source = os.path.join(dataset_path, \"train_soundscapes\") if use_train_as_test else TEST_AUDIO_PATH\nif use_train_as_test:\n    print(\"Case when train is the test\")\n    test_meta = pd.read_csv(os.path.join(dataset_path, \"train_soundscape_labels.csv\"))\n\nprint(f\"Total segments to process: {len(test_meta)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T07:38:46.673000Z","iopub.execute_input":"2025-11-16T07:38:46.673294Z","iopub.status.idle":"2025-11-16T07:38:46.684253Z","shell.execute_reply.started":"2025-11-16T07:38:46.673272Z","shell.execute_reply":"2025-11-16T07:38:46.683559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data_df = (train_data_df[[\"primary_label\"]]\n                 .assign(path = train_data_df.apply(\n                    lambda row: Path(dataset_path) / row[\"primary_label\"] / row[\"filename\"],\n                    axis=1\n                )))\ntrain_data_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:14:22.398093Z","iopub.execute_input":"2025-11-14T12:14:22.398332Z","iopub.status.idle":"2025-11-14T12:14:23.508445Z","shell.execute_reply.started":"2025-11-14T12:14:22.398314Z","shell.execute_reply":"2025-11-14T12:14:23.507695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LABEL_TO_NUM = {l: n for n, l in enumerate(train_data_df.primary_label.unique())}\nNUM_TO_LABEL = {n: l for l, n in LABEL_TO_NUM.items()}\nNUM_CLASSES = len(LABEL_TO_NUM)\nNUM_CLASSES","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:14:23.509319Z","iopub.execute_input":"2025-11-14T12:14:23.509611Z","iopub.status.idle":"2025-11-14T12:14:23.521479Z","shell.execute_reply.started":"2025-11-14T12:14:23.509590Z","shell.execute_reply":"2025-11-14T12:14:23.520833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"encoder = LabelEncoder() \n\nlabels = encoder.fit_transform(train_data_df[\"primary_label\"].unique())\nindexes = encoder.inverse_transform(labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:14:23.522349Z","iopub.execute_input":"2025-11-14T12:14:23.522657Z","iopub.status.idle":"2025-11-14T12:14:23.539622Z","shell.execute_reply.started":"2025-11-14T12:14:23.522617Z","shell.execute_reply":"2025-11-14T12:14:23.538845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"value_counts = train_data_df[\"primary_label\"].value_counts()\n\ntruncate_counts = value_counts.map(lambda x: x if x <= MAX_LENGTH else MAX_LENGTH)\n\nvalue_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:14:23.540400Z","iopub.execute_input":"2025-11-14T12:14:23.540715Z","iopub.status.idle":"2025-11-14T12:14:23.563273Z","shell.execute_reply.started":"2025-11-14T12:14:23.540690Z","shell.execute_reply":"2025-11-14T12:14:23.562695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"truncate_counts.sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:14:23.564020Z","iopub.execute_input":"2025-11-14T12:14:23.564328Z","iopub.status.idle":"2025-11-14T12:14:23.581406Z","shell.execute_reply.started":"2025-11-14T12:14:23.564308Z","shell.execute_reply":"2025-11-14T12:14:23.580782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\ndirs = os.listdir(train_dir_short)\n\nfiles_by_label = {}\n\n\nfor label in tqdm(dirs):\n    folder = train_dir_short / label\n    sorted_files = sorted(folder.iterdir(), key=lambda f: f.stat().st_size)\n    files_by_label[label] = [str(f) for f in sorted_files[:MAX_LENGTH]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:14:23.584801Z","iopub.execute_input":"2025-11-14T12:14:23.585016Z","iopub.status.idle":"2025-11-14T12:17:38.560934Z","shell.execute_reply.started":"2025-11-14T12:14:23.584999Z","shell.execute_reply":"2025-11-14T12:17:38.560248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SAMPLE_RATE = 32000\nN_MFCC = 40\nHOP_LENGTH = 512 \n\n\ndef load_and_convert_to_mfcc(file_path):\n    audio, _ = librosa.load(file_path, sr=SAMPLE_RATE)\n    mfcc = librosa.feature.mfcc(y=audio, sr=SAMPLE_RATE, n_mfcc=N_MFCC, hop_length=HOP_LENGTH).astype(np.float16)\n    return mfcc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:17:38.561757Z","iopub.execute_input":"2025-11-14T12:17:38.562025Z","iopub.status.idle":"2025-11-14T12:17:38.566744Z","shell.execute_reply.started":"2025-11-14T12:17:38.562005Z","shell.execute_reply":"2025-11-14T12:17:38.566095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from joblib import Parallel, delayed\nimport os\n\ntrain_metadata = pd.read_csv(dataset_path / \"train_metadata.csv\")\n\ntrain_data = []\nnot_in = []\n\ndef process_file(file_path, label):\n    if os.path.exists(file_path):\n        mfcc = load_and_convert_to_mfcc(file_path)\n        return {'mfcc': mfcc, 'label': label}\n    else:\n        print(f\"Файл не найден: {file_path}\")\n        return None\n\nresults = Parallel(n_jobs=-1)(\n    delayed(process_file)(file_path, label)\n    for label, files in tqdm(files_by_label.items())\n    for file_path in files\n)\n\n# Supprimer les None (fichiers non trouvés)\ntrain_data = [r for r in results if r is not None]\nprint(f\"len train data: {len(train_data)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:17:38.567673Z","iopub.execute_input":"2025-11-14T12:17:38.567922Z","iopub.status.idle":"2025-11-14T12:25:06.034380Z","shell.execute_reply.started":"2025-11-14T12:17:38.567906Z","shell.execute_reply":"2025-11-14T12:25:06.033691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:25:06.035199Z","iopub.execute_input":"2025-11-14T12:25:06.035437Z","iopub.status.idle":"2025-11-14T12:25:06.046799Z","shell.execute_reply.started":"2025-11-14T12:25:06.035419Z","shell.execute_reply":"2025-11-14T12:25:06.045919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_lens = sorted([i['mfcc'].shape[1] for i in train_data])\n\nplt.plot(data_lens)\nplt.show()\n\nstats = pd.Series(data_lens)\n\nprint(stats.describe())\n\n\nMAX_LEN = 2000\ngreater_then_max_len = [i for i in data_lens if i > MAX_LEN]\namount_greater_max = len(greater_then_max_len)\n\nprint(amount_greater_max)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:25:06.083080Z","iopub.execute_input":"2025-11-14T12:25:06.083306Z","iopub.status.idle":"2025-11-14T12:25:06.394239Z","shell.execute_reply.started":"2025-11-14T12:25:06.083289Z","shell.execute_reply":"2025-11-14T12:25:06.393587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_padding_mfcc(d):\n    mfcc = librosa.util.fix_length(d['mfcc'], size=MAX_LEN, axis=1)\n    mfcc = mfcc[..., np.newaxis]\n    return mfcc, d['label']\n\nresults = Parallel(n_jobs=-1)(\n    delayed(process_padding_mfcc)(d) for d in tqdm(train_data, desc=\"Processing MFCCs\")\n)\n\n# Séparer MFCCs et labels\nmfccs_padded, labels = zip(*results)\n\n\nfor i, d in enumerate(train_data):\n    d['mfcc'] = mfccs_padded[i]  # remplace MFCC par le pad/crop + channel","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:25:06.394972Z","iopub.execute_input":"2025-11-14T12:25:06.395190Z","iopub.status.idle":"2025-11-14T12:25:16.138125Z","shell.execute_reply.started":"2025-11-14T12:25:06.395173Z","shell.execute_reply":"2025-11-14T12:25:16.137229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(mfccs_padded[0][0]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:25:16.139107Z","iopub.execute_input":"2025-11-14T12:25:16.139346Z","iopub.status.idle":"2025-11-14T12:25:16.143881Z","shell.execute_reply.started":"2025-11-14T12:25:16.139328Z","shell.execute_reply":"2025-11-14T12:25:16.143025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data[0]['mfcc'].shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:25:16.144726Z","iopub.execute_input":"2025-11-14T12:25:16.145676Z","iopub.status.idle":"2025-11-14T12:25:16.164723Z","shell.execute_reply.started":"2025-11-14T12:25:16.145632Z","shell.execute_reply":"2025-11-14T12:25:16.163874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = [d['label'] for d in train_data]\nlabels_categorical = pd.get_dummies(labels).values\n\nprint(labels_categorical.shape)\nprint(labels_categorical)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:25:16.165519Z","iopub.execute_input":"2025-11-14T12:25:16.165758Z","iopub.status.idle":"2025-11-14T12:25:16.189123Z","shell.execute_reply.started":"2025-11-14T12:25:16.165735Z","shell.execute_reply":"2025-11-14T12:25:16.188375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels_as_nums = [LABEL_TO_NUM[i['label']] for i in train_data]\nlabels_categorical = to_categorical(labels_as_nums, num_classes=NUM_CLASSES)\nprint(f\"labels_categorical shape = {labels_categorical.shape}\")\nprint(f\"labels_categorical shape = {labels_categorical}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:25:16.190038Z","iopub.execute_input":"2025-11-14T12:25:16.190356Z","iopub.status.idle":"2025-11-14T12:25:16.246805Z","shell.execute_reply.started":"2025-11-14T12:25:16.190330Z","shell.execute_reply":"2025-11-14T12:25:16.245986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = np.stack([i['mfcc'] for i in train_data], axis=0)\nX_train, X_val, y_train, y_val = train_test_split(\n    X, labels_categorical, test_size=0.3, random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:25:16.247761Z","iopub.execute_input":"2025-11-14T12:25:16.248070Z","iopub.status.idle":"2025-11-14T12:25:17.751255Z","shell.execute_reply.started":"2025-11-14T12:25:16.248043Z","shell.execute_reply":"2025-11-14T12:25:17.750692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import tensorflow_addons as tfa\n\n# f1_score = tfa.metrics.F1Score(num_classes=NUM_CLASSES, average='macro', threshold=0.5)\n\n\nkernel_size = 3\npool_size = 2\nconv_depth_1 = 32\nconv_depth_2 = 64  \nconv_depth_3 = 128\ndrop_prob_1 = 0.25 \ndrop_prob_2 = 0.5 \nhidden_size = 512\n\n\nbatch_size = 32\nnum_epochs = 3\n\n\ninp = Input(shape=(X_train.shape[1:]))\n\n# CNN feature extractor\nx = Convolution2D(conv_depth_1, (kernel_size + 2, kernel_size + 2), activation='relu', padding='same')(inp)\nx = BatchNormalization()(x)\nx = Convolution2D(conv_depth_1, (kernel_size + 2, kernel_size + 2), activation='relu', padding='same')(x)\nx = BatchNormalization()(x)\nx = MaxPooling2D((pool_size,pool_size))(x)\nx = Dropout(0.25)(x)    \n    \n\nx = Convolution2D(conv_depth_1, (kernel_size + 1, kernel_size + 1), activation='relu', padding='same')(x)\nx = BatchNormalization()(x)\nx = Convolution2D(conv_depth_1, (kernel_size + 1, kernel_size + 1), activation='relu', padding='same')(x)\nx = BatchNormalization()(x)\nx = MaxPooling2D((pool_size,pool_size))(x)\nx = Dropout(0.25)(x)\n\nx = Convolution2D(conv_depth_2, (kernel_size, kernel_size), activation='relu', padding='same')(x)\nx = BatchNormalization()(x)\nx = Convolution2D(conv_depth_2, (kernel_size, kernel_size), activation='relu', padding='same')(x)\nx = BatchNormalization()(x)\nx = MaxPooling2D((pool_size,pool_size))(x)\nx = Dropout(0.3)(x)\n\n\n# Reshape correctement avec int()\nx = GlobalAveragePooling2D()(x)\n\n# Recurrent part\nx = Dense(256, activation='relu')(x)\nx = Dropout(0.5)(x)\n# Dense output\noutput = Dense(NUM_CLASSES, activation='softmax')(x)\n\nmodel = Model(inputs=inp, outputs=output)\n\nmodel.summary()\n\nearly_stop = EarlyStopping(\nmonitor='val_loss',\npatience=5,\nverbose=1)\n\nmodel.compile(loss='categorical_crossentropy', \n                optimizer='adam', \n                metrics=['accuracy']) \nhistory = model.fit(\nX_train, y_train,\nbatch_size=batch_size,\nepochs=num_epochs,\nverbose=1,\nvalidation_data=[X_val, y_val],\ncallbacks=[early_stop])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:25:23.304899Z","iopub.execute_input":"2025-11-14T12:25:23.305167Z","iopub.status.idle":"2025-11-14T12:32:05.866935Z","shell.execute_reply.started":"2025-11-14T12:25:23.305140Z","shell.execute_reply":"2025-11-14T12:32:05.866008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(X_val, y_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T12:32:05.868761Z","iopub.execute_input":"2025-11-14T12:32:05.869008Z","iopub.status.idle":"2025-11-14T12:32:13.994003Z","shell.execute_reply.started":"2025-11-14T12:32:05.868990Z","shell.execute_reply":"2025-11-14T12:32:13.993229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEGMENT_SECONDS = 5\nSEGMENT_SAMPLES = SEGMENT_SECONDS * SAMPLE_RATE\n\nX_test_segments = []\nrow_ids = []\n\ntest_dir = audio_source\n\n# Lister uniquement les fichiers .ogg\nogg_files = [f for f in test_dir.iterdir() if f.suffix.lower() == \".ogg\"]\n\nprint(f\"Nombre de fichiers .ogg à traiter : {len(ogg_files)}\")\n\nfor f in tqdm(ogg_files, desc=\"Traitement segments\"):\n    audio, _ = librosa.load(f, sr=SAMPLE_RATE)\n    num_segments = max(1, len(audio) // SEGMENT_SAMPLES)\n\n    for i in range(num_segments):\n        start = i * SEGMENT_SAMPLES\n        end = start + SEGMENT_SAMPLES\n        segment_audio = audio[start:end]\n\n        # MFCC\n        mfcc = librosa.feature.mfcc(\n            y=segment_audio, sr=SAMPLE_RATE, n_mfcc=N_MFCC, hop_length=HOP_LENGTH\n        )\n        mfcc = librosa.util.fix_length(mfcc, size=MAX_LEN, axis=1)\n        mfcc = mfcc[..., np.newaxis]  # ajouter dimension channel\n\n        X_test_segments.append(mfcc)\n        row_ids.append(f\"{'_'.join(f.stem.split('_')[0:2])}_{i*SEGMENT_SECONDS + 5}\")\n\n# Convertir en array numpy\nif len(X_test_segments) > 0:\n    X_test_segments = np.stack(X_test_segments, axis=0)\nelse:\n    raise ValueError(\"⚠️ Aucun segment à traiter !\")\n\n# -------------------------------\n# Prédictions avec seuil\n# -------------------------------\nlower_limit = 0.25  # seuil pour sélectionner plusieurs oiseaux\n\nres_birds = []\n\npreds_segments = model.predict(X_test_segments, batch_size=32, verbose=1)\n\nfor pred in preds_segments:\n    selected_indices = np.where(pred > lower_limit)[0]\n    predicted_labels = [NUM_TO_LABEL[i] for i in selected_indices]\n    birds = ' '.join(predicted_labels) if predicted_labels else 'nocall'\n    res_birds.append(birds)\n\n# -------------------------------\n# Créer le CSV de submission\n# -------------------------------\nsubmission_df = pd.DataFrame({\n    \"row_id\": row_ids,\n    \"birds\": res_birds\n})\n\nsubmission_df.to_csv(\"submission.csv\", index=False)\nprint(\"Fichier submission.csv généré avec seuil et row_id correct !\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T13:06:47.953463Z","iopub.execute_input":"2025-11-14T13:06:47.954054Z","iopub.status.idle":"2025-11-14T13:07:38.428440Z","shell.execute_reply.started":"2025-11-14T13:06:47.954017Z","shell.execute_reply":"2025-11-14T13:07:38.427803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}