{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","ExecuteTime":{"end_time":"2023-11-19T11:47:47.349503700Z","start_time":"2023-11-19T11:47:47.267985700Z"},"execution":{"iopub.status.busy":"2023-11-20T07:05:02.854915Z","iopub.execute_input":"2023-11-20T07:05:02.856215Z","iopub.status.idle":"2023-11-20T07:05:03.331702Z","shell.execute_reply.started":"2023-11-20T07:05:02.856166Z","shell.execute_reply":"2023-11-20T07:05:03.330220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Загрузка датасета","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/birdclef-2021/train_metadata.csv')\ntrain = train.query('rating>=4')\n\ntrain_path = '../input/birdclef-2021/train_short_audio'\n\ntest_path = '../input/birdclef-2021/test_soundscapes'\ntest_pd_path = '../input/birdclef-2021/sample_submission.csv'\n\nother_train_path = '../input/birdclef-2021/train_soundscapes'\nother_train_pd_path = '../input/birdclef-2021/train_soundscape_labels.csv'","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-11-19T11:47:47.694663600Z","start_time":"2023-11-19T11:47:47.284980Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-20T07:08:19.759614Z","iopub.execute_input":"2023-11-20T07:08:19.760100Z","iopub.status.idle":"2023-11-20T07:08:20.157363Z","shell.execute_reply.started":"2023-11-20T07:08:19.760061Z","shell.execute_reply":"2023-11-20T07:08:20.156033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Извлечение метаданных","metadata":{}},{"cell_type":"code","source":"# Second, assume that birds with the most training samples are also the most common\n# A species needs at least 200 recordings with a rating above 4 to be considered common\nbirds_count = {}\nfor bird_species, count in zip(train.primary_label.unique(),\n                               train.groupby('primary_label')['primary_label'].count().values):\n    birds_count[bird_species] = count\nmost_represented_birds = [key for key,value in birds_count.items() if value >= 200]\n\ntrain = train.query('primary_label in @most_represented_birds')","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-11-19T11:47:47.743105200Z","start_time":"2023-11-19T11:47:47.698666100Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-20T07:08:23.639424Z","iopub.execute_input":"2023-11-20T07:08:23.639845Z","iopub.status.idle":"2023-11-20T07:08:23.696103Z","shell.execute_reply.started":"2023-11-20T07:08:23.639812Z","shell.execute_reply":"2023-11-20T07:08:23.694819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Конфигурация","metadata":{}},{"cell_type":"code","source":"class Config:\n    def __init__(self):\n        self.sample_rate = 32000\n        self.n_fft = 1024\n        self.hop_length = 512\n        self.n_mels = 128\n        self.fmin = 0\n        self.fmax = self.sample_rate // 2\n                \n        self.audio_duration = 5\n        self.audio_length = self.sample_rate * self.audio_duration\n        \n        self.n_classes = None\n    \n    def set_n_classes(self, n: int):\n        self.n_classes = n\n\n\nconfig = Config()\n        ","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-11-19T11:47:47.757291600Z","start_time":"2023-11-19T11:47:47.746292700Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-20T07:08:35.732283Z","iopub.execute_input":"2023-11-20T07:08:35.732719Z","iopub.status.idle":"2023-11-20T07:08:35.741202Z","shell.execute_reply.started":"2023-11-20T07:08:35.732686Z","shell.execute_reply":"2023-11-20T07:08:35.739758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Извлечение спектрограмм для файла\n\nУ коллег описаны [разные способы получения информации](https://www.kaggle.com/code/halilbrahimhatun/birdclef-audio-feature-extraction-with-librosa) из аудио-файла. Среди прочих наиболее информативным показалось оконное преобразование Фурье.","metadata":{}},{"cell_type":"code","source":"import librosa\nfrom PIL import Image\n\ndef process_audio(full_audio: np.ndarray, sr: float) -> np.ndarray:\n    # audio_duration = librosa.get_duration(y=full_audio, sr=config.sample_rate)\n    # tempo, beats = librosa.beat.beat_track(y=full_audio, sr=sr)\n    # beat_times = librosa.frames_to_time(beats, sr=sr)\n    audio_len = len(full_audio)\n    db_chunks = []\n    for begin_index in range(0, audio_len, config.audio_length):\n        end_index = begin_index + config.audio_length\n        if end_index > audio_len:\n            break\n        beat_audio = full_audio[begin_index:end_index]\n#         d_beat = np.abs(librosa.stft(beat_audio, n_fft=config.n_fft, hop_length=config.hop_length))\n#         # Convert an amplitude spectrogram to Decibels-scaled spectrogram.\n#         db_beat = librosa.amplitude_to_db(d_beat, ref=np.max)\n#         # reshape to smaller size (from 513x313 to 200x120):\n#         image = Image.fromarray(db_beat * 255)\n#         image = image.resize((200, 120), Image.Resampling.BICUBIC)\n#         # convert back to array:\n#         db_beat = np.array(image, dtype='float32')\n        \n        # melspec:\n        d_beat = librosa.feature.melspectrogram(y=beat_audio, sr=sr, n_mels=config.n_mels, fmin=config.fmin, fmax=config.fmax)\n        db_beat = librosa.power_to_db(d_beat, ref=np.max)\n        # Normalize\n        db_beat -= db_beat.min()\n        db_beat /= db_beat.max()\n\n        if not db_beat.max() == 1.0 or not db_beat.min() == 0.0:\n            continue\n\n        db_beat = np.expand_dims(db_beat, -1)\n\n        # Add new dimension for batch size\n        db_beat = np.expand_dims(db_beat, 0)\n        \n        if len(db_chunks) == 0:\n            db_chunks = db_beat\n        else:\n            db_chunks = np.vstack((db_chunks, db_beat))\n    return db_chunks\n\n\ndef process_file(file_path: str, duration: int = None) -> tuple[np.ndarray, float]:\n    if duration:\n        full_audio, sr = librosa.load(file_path, sr=config.sample_rate, duration=duration)\n    else:\n        full_audio, sr = librosa.load(file_path, sr=config.sample_rate)\n    return full_audio, sr","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-11-19T14:35:52.565527300Z","start_time":"2023-11-19T14:35:52.471231100Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-20T07:08:35.743406Z","iopub.execute_input":"2023-11-20T07:08:35.743909Z","iopub.status.idle":"2023-11-20T07:08:35.756731Z","shell.execute_reply.started":"2023-11-20T07:08:35.743863Z","shell.execute_reply":"2023-11-20T07:08:35.755492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n\ndef load_train() -> tuple[np.ndarray, np.ndarray]:\n    samples = []\n    labels = []\n    with tqdm(total=len(train)) as pbar:\n        for idx, row in train.iterrows():\n            pbar.update(1)\n            file_path = f'{train_path}/{row.primary_label}/{row.filename}'\n            full_audio, sr = process_file(file_path, 15)\n            chunks = process_audio(full_audio, sr)\n            chunks_len = len(chunks)\n            try:\n                if len(samples) == 0:\n                    samples = chunks\n                else:\n                    samples = np.vstack((samples, chunks))\n                for _ in range(chunks_len):\n                    if len(labels) == 0:\n                        labels = row.primary_label\n                    else:\n                        labels = np.vstack((labels, row.primary_label))\n            except Exception:\n                print(chunks)\n                continue\n            \n                \n                \n    return samples, labels\n\n\ndef list_files(path):\n    return [os.path.join(path, f) for f in os.listdir(path) if f.rsplit('.', 1)[-1] in ['ogg']]\n\ndef load_test() -> np.ndarray:\n    samples = np.array([])\n    files_list = list_files(test_path)\n    if len(files_list) == 0:\n        files_list = list_files(other_train_path)\n    \n    with tqdm(total=len(files_list)) as pbar:\n        for file_path in files_list:\n            pbar.update(1)\n            full_audio, sr = process_file(file_path)\n            samples = np.append(samples, process_audio(full_audio, sr))\n    return samples","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-11-19T14:35:52.585661600Z","start_time":"2023-11-19T14:35:52.498734800Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-20T07:08:35.758124Z","iopub.execute_input":"2023-11-20T07:08:35.758711Z","iopub.status.idle":"2023-11-20T07:08:35.775263Z","shell.execute_reply.started":"2023-11-20T07:08:35.758679Z","shell.execute_reply":"2023-11-20T07:08:35.773913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, Y_train_labels = load_train()\n# X_test = load_test()","metadata":{"collapsed":false,"is_executing":true,"ExecuteTime":{"start_time":"2023-11-19T14:35:52.508838400Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-20T07:16:15.766086Z","iopub.execute_input":"2023-11-20T07:16:15.766827Z","iopub.status.idle":"2023-11-20T07:22:45.412639Z","shell.execute_reply.started":"2023-11-20T07:16:15.766787Z","shell.execute_reply":"2023-11-20T07:22:45.411401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LABELS = np.unique(Y_train_labels)\n\nfrom sklearn.preprocessing import LabelEncoder\n\nencoder = LabelEncoder()\nY_train_encoded = encoder.fit_transform(Y_train_labels)\n# Y_test = encoder.transform(test['label'].to_numpy())\n\nconfig.set_n_classes(encoder.classes_)\n\n# Make data categorical\nfrom keras.utils import to_categorical\n\n# Y_train_length = Y_train_encoded.shape[0]\n# Y = np.concatenate(\n#     (np.array(Y_train_encoded), \n#     np.array(Y_test))\n# )\nY_train = to_categorical(Y_train_encoded, num_classes=len(config.n_classes))","metadata":{"collapsed":false,"is_executing":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-20T07:27:57.793323Z","iopub.execute_input":"2023-11-20T07:27:57.793719Z","iopub.status.idle":"2023-11-20T07:27:57.803097Z","shell.execute_reply.started":"2023-11-20T07:27:57.793690Z","shell.execute_reply":"2023-11-20T07:27:57.801839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras import models, optimizers, losses\nfrom keras.layers import (Convolution2D, BatchNormalization, Flatten, MaxPool2D, Activation, Input, Dense)\nfrom keras.activations import softmax\nfrom keras.metrics import F1Score\n\ndef get_model():\n    inp = Input(shape=(X_train.shape[1], X_train.shape[2],1))\n    x = Convolution2D(32, (3, 3), padding=\"same\")(inp)\n    x = BatchNormalization()(x)\n    x = Activation(\"relu\")(x)\n    x = MaxPool2D()(x)\n\n    x = Convolution2D(64, (3,3), padding=\"same\")(x)\n    x = BatchNormalization()(x)\n    x = Activation(\"relu\")(x)\n    x = MaxPool2D()(x)\n\n    x = Convolution2D(128, (3,3), padding=\"same\")(x)\n    x = BatchNormalization()(x)\n    x = Activation(\"relu\")(x)\n    x = MaxPool2D()(x)\n\n    x = Convolution2D(256, (3,3), padding=\"same\")(x)\n    x = BatchNormalization()(x)\n    x = Activation(\"relu\")(x)\n    x = MaxPool2D()(x)\n\n    x = Flatten()(x)\n    x = Dense(64)(x)\n    x = BatchNormalization()(x)\n    x = Activation(\"relu\")(x)\n\n    classes_num = len(encoder.classes_)\n    # x = Dense(60, activation='relu')(x)\n    out = Dense(classes_num, activation=softmax)(x)\n\n    model = models.Model(inputs=inp, outputs=out)\n    opt = optimizers.Adam()\n    model.compile(optimizer=opt, loss=losses.categorical_crossentropy, metrics=['acc', F1Score(average='macro')])\n    return model","metadata":{"collapsed":false,"is_executing":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-20T07:27:57.805468Z","iopub.execute_input":"2023-11-20T07:27:57.805981Z","iopub.status.idle":"2023-11-20T07:27:57.821091Z","shell.execute_reply.started":"2023-11-20T07:27:57.805933Z","shell.execute_reply":"2023-11-20T07:27:57.819991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\nimport os\nfrom sklearn.utils import shuffle\n\nX_train, Y_train = shuffle(X_train, Y_train, random_state=1337)\n\nPREDICTION_FOLDER = \"predictions\"\nif not os.path.exists(PREDICTION_FOLDER):\n    os.mkdir(PREDICTION_FOLDER)\n\nreduce = ReduceLROnPlateau(monitor='val_loss', patience=2, verbose=1, factor=0.5)\ncheckpoint = ModelCheckpoint('best_model.h5', monitor='val_loss', verbose=1, save_best_only=True)\nearly = EarlyStopping(monitor=\"val_loss\", mode=\"min\", patience=5)\n\nmodel = get_model()\nhistory = model.fit(X_train, Y_train,\n                    validation_split=0.2,\n                    batch_size=64,\n                    epochs=50,\n                    callbacks=[early, checkpoint]\n                    )","metadata":{"collapsed":false,"is_executing":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-11-20T07:27:57.823329Z","iopub.execute_input":"2023-11-20T07:27:57.823807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_audio(file_path: str, model: models.Model, data: dict):\n    full_audio, sr = process_file(file_path)\n\n    audio_chunks = process_audio(full_audio, sr) \n    seconds = 0\n    for audio_chunk in audio_chunks:\n        seconds += 5\n        if len(audio_chunk.shape) == 3:\n            audio_chunk = np.expand_dims(audio_chunk, 0)\n        chunk_prediction = model.predict(audio_chunk)[0]\n        idx = chunk_prediction.argmax()\n        species = LABELS[idx]\n        score = chunk_prediction[idx]\n        if score >= 0.25:\n            data['birds'].append(species)\n        else:\n            data['birds'].append('nocall')\n\n        data['row_id'].append(file_path.split(os.sep)[-1].rsplit('_', 1)[0] + '_' + str(seconds))","metadata":{"collapsed":false,"is_executing":true,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_test_soundscapes(model: models.Model):\n    IS_TEST = True\n    test_audio_list = list_files(test_path)\n    if len(test_audio_list) == 0:\n        test_audio_list = list_files(other_train_path)\n        IS_TEST = False\n    data = {'row_id': [], 'birds': []}\n    for file_path in test_audio_list:\n        predict_audio(file_path, model, data)\n    results = pd.DataFrame(data, columns = ['row_id', 'birds'])\n    results.to_csv(\"submission.csv\", index=False)","metadata":{"collapsed":false,"is_executing":true,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_test_soundscapes(model)","metadata":{},"execution_count":null,"outputs":[]}]}