{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30733,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n#import numpy as np # linear algebra\n#import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-05T15:54:51.732876Z","iopub.execute_input":"2024-06-05T15:54:51.733316Z","iopub.status.idle":"2024-06-05T15:54:51.739146Z","shell.execute_reply.started":"2024-06-05T15:54:51.733285Z","shell.execute_reply":"2024-06-05T15:54:51.737865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_path = '/kaggle/input/birdclef-2024/train_audio'","metadata":{"execution":{"iopub.status.busy":"2024-06-07T13:37:04.687744Z","iopub.execute_input":"2024-06-07T13:37:04.688520Z","iopub.status.idle":"2024-06-07T13:37:04.693177Z","shell.execute_reply.started":"2024-06-07T13:37:04.688480Z","shell.execute_reply":"2024-06-07T13:37:04.691998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport librosa\nimport numpy as np\nfrom joblib import Parallel, delayed\nfrom sklearn.preprocessing import LabelEncoder\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\n\n# only use 15 second of each audio\nMAX_AUDIO_LENGTH = 15\n\ndef add_noise(audio, noise_factor=0.005):\n    noise = np.random.randn(len(audio))\n    augmented_audio = audio + noise_factor * noise\n    return augmented_audio\n\ndef shift_audio(audio, shift_max=0.2, shift_direction='both'):\n    shift = np.random.randint(len(audio) * shift_max)\n    if shift_direction == 'right':\n        shift = -shift\n    elif shift_direction == 'both':\n        direction = np.random.randint(0, 2)\n        if direction == 1:\n            shift = -shift\n    augmented_audio = np.roll(audio, shift)\n    return augmented_audio\n\ndef change_volume(audio, volume_change=0.1):\n    change = np.random.uniform(low=1.0-volume_change, high=1.0+volume_change)\n    return audio * change\n\ndef process_file_with_augmentation(file_path, label, augment=False):\n    audio, sample_rate = librosa.load(file_path, sr=16000, duration=MAX_AUDIO_LENGTH)\n    if augment:\n        audio = add_noise(audio)\n        audio = shift_audio(audio)\n        audio = change_volume(audio)\n    mfccs = librosa.feature.mfcc(y=audio, sr=sample_rate, n_mfcc=40)\n    mfccs_scaled = np.mean(mfccs.T, axis=0)\n    return mfccs_scaled, label\n\ndef load_audio_data_with_augmentation(data_dir, augment=False):\n    labels = []\n    features = []\n    for label in os.listdir(data_dir):\n        folder_path = os.path.join(data_dir, label)\n        if os.path.isdir(folder_path):\n            results = Parallel(n_jobs=-1)(delayed(process_file_with_augmentation)(os.path.join(folder_path, file), label, augment)\n                                         for file in os.listdir(folder_path) if file.endswith('.ogg'))\n            for feature, lbl in results:\n                features.append(feature)\n                labels.append(lbl)\n    return np.array(features), np.array(labels)\n\ndata_dir = full_path\nfeatures, labels = load_audio_data_with_augmentation(data_dir, augment=True)\n\n\nle = LabelEncoder()\nlabels_encoded = le.fit_transform(labels)\nlabels_categorical = to_categorical(labels_encoded)\n\n\nX_train, X_test, y_train, y_test = train_test_split(features, labels_categorical, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-07T13:37:07.210373Z","iopub.execute_input":"2024-06-07T13:37:07.211099Z","iopub.status.idle":"2024-06-07T13:45:41.102283Z","shell.execute_reply.started":"2024-06-07T13:37:07.211062Z","shell.execute_reply":"2024-06-07T13:45:41.100973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\n\nmodel = Sequential()\nmodel.add(Dense(512, input_shape=(40,), activation='relu'))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(256, activation='relu'))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(len(le.classes_), activation='softmax'))\n\nmodel.compile(optimizer=Adam(), loss='categorical_crossentropy', metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2024-06-07T13:45:41.105029Z","iopub.execute_input":"2024-06-07T13:45:41.105479Z","iopub.status.idle":"2024-06-07T13:45:41.211037Z","shell.execute_reply.started":"2024-06-07T13:45:41.105438Z","shell.execute_reply":"2024-06-07T13:45:41.209888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"early_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n\nhistory = model.fit(X_train, y_train, epochs=400, batch_size=64, validation_data=(X_test, y_test), callbacks=[early_stopping])\n","metadata":{"execution":{"iopub.status.busy":"2024-06-07T13:45:41.212513Z","iopub.execute_input":"2024-06-07T13:45:41.212843Z","iopub.status.idle":"2024-06-07T13:47:41.713019Z","shell.execute_reply.started":"2024-06-07T13:45:41.212811Z","shell.execute_reply":"2024-06-07T13:47:41.711933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = model.evaluate(X_test, y_test, verbose=0)\nprint(f'Test loss: {score[0]}')\nprint(f'Test accuracy: {score[1]}')","metadata":{"execution":{"iopub.status.busy":"2024-06-07T13:48:09.683037Z","iopub.execute_input":"2024-06-07T13:48:09.684014Z","iopub.status.idle":"2024-06-07T13:48:09.990425Z","shell.execute_reply.started":"2024-06-07T13:48:09.683976Z","shell.execute_reply":"2024-06-07T13:48:09.989391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndef predict_unlabeled_audio(model, le, unlabeled_dir, output_file):\n    predictions = []\n    filenames = os.listdir(unlabeled_dir)\n    for filename in filenames:\n        file_path = os.path.join(unlabeled_dir, filename)\n        audio, sample_rate = librosa.load(file_path, sr=16000, duration=MAX_AUDIO_LENGTH)\n        mfccs = librosa.feature.mfcc(y=audio, sr=sample_rate, n_mfcc=40)\n        mfccs_scaled = np.mean(mfccs.T, axis=0)\n        mfccs_scaled = np.expand_dims(mfccs_scaled, axis=0)\n        prediction = model.predict(mfccs_scaled)[0]\n        predictions.append([f\"soundscape_{filename}_15\"] + prediction.tolist())\n    \n    column_names = ['row_id'] + le.classes_.tolist()\n    df = pd.DataFrame(predictions, columns=column_names)\n    df.to_csv(output_file, index=False)\n\nunlabeled_dir = '/kaggle/input/birdclef-2024/unlabeled_soundscapes'\noutput_file = 'submission.csv'\npredict_unlabeled_audio(model, le, unlabeled_dir, output_file)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T13:51:52.031490Z","iopub.execute_input":"2024-06-07T13:51:52.032129Z","iopub.status.idle":"2024-06-07T13:52:24.726581Z","shell.execute_reply.started":"2024-06-07T13:51:52.032086Z","shell.execute_reply":"2024-06-07T13:52:24.725510Z"},"trusted":true},"execution_count":null,"outputs":[]}]}