{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30733,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport librosa\nimport numpy as np\nfrom joblib import Parallel, delayed\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.utils import to_categorical\n\n\nMAX_AUDIO_LENGTH = 15\n\nMAX_MFCC_LENGTH = 640  \n\ndef process_file(file_path, label):\n    audio, sample_rate = librosa.load(file_path, sr=32000, duration=MAX_AUDIO_LENGTH)\n    mfccs = librosa.feature.mfcc(y=audio, sr=sample_rate, n_mfcc=40)\n    if mfccs.shape[1] < MAX_MFCC_LENGTH:\n        mfccs = np.pad(mfccs, ((0, 0), (0, MAX_MFCC_LENGTH - mfccs.shape[1])), mode='constant')\n    else:\n        mfccs = mfccs[:, :MAX_MFCC_LENGTH]\n    return mfccs.T, label\n\ndef load_audio_data(data_dir):\n    features = []\n    labels = []\n    for label in os.listdir(data_dir):\n        folder_path = os.path.join(data_dir, label)\n        if os.path.isdir(folder_path):\n            results = Parallel(n_jobs=-1)(delayed(process_file)(os.path.join(folder_path, file), label)\n                                         for file in os.listdir(folder_path) if file.endswith('.ogg'))\n            for feature, lbl in results:\n                features.append(feature)\n                labels.append(lbl)\n    return features, labels\n\ndata_dir = '/kaggle/input/birdclef-2024/train_audio'\nfeatures, labels = load_audio_data(data_dir)\n\n\nle = LabelEncoder()\nlabels_encoded = le.fit_transform(labels)\nlabels_categorical = to_categorical(labels_encoded)\n\n\nX_train, X_test, y_train, y_test = train_test_split(features, labels_categorical, test_size=0.2, random_state=42)\n\n\nX_train = np.array(X_train)\nX_test = np.array(X_test)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:18:23.247054Z","iopub.execute_input":"2024-06-09T08:18:23.247312Z","iopub.status.idle":"2024-06-09T08:26:49.102484Z","shell.execute_reply.started":"2024-06-09T08:18:23.247288Z","shell.execute_reply":"2024-06-09T08:26:49.101392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout, Flatten\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n\nimport tensorflow as tf\nprint(\"Num GPUs Available: \", len(tf.config.experimental.list_physical_devices('GPU')))\n\nmodel = Sequential()\nmodel.add(LSTM(128, input_shape=(X_train.shape[1], X_train.shape[2]), return_sequences=True))\nmodel.add(Dropout(0.3))\nmodel.add(LSTM(128, return_sequences=True))\nmodel.add(Dropout(0.3))\nmodel.add(Flatten())\nmodel.add(Dense(256, activation='relu'))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(len(le.classes_), activation='softmax'))\n\n\nmodel.compile(optimizer=Adam(learning_rate=0.0001), loss='categorical_crossentropy', metrics=['accuracy'])\n\n\nearly_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n\n\nhistory = model.fit(X_train, y_train, epochs=200, batch_size=64, validation_data=(X_test, y_test), callbacks=[early_stopping])\n","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:26:49.104758Z","iopub.execute_input":"2024-06-09T08:26:49.105779Z","iopub.status.idle":"2024-06-09T08:36:52.489699Z","shell.execute_reply.started":"2024-06-09T08:26:49.105743Z","shell.execute_reply":"2024-06-09T08:36:52.488877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:37:30.552112Z","iopub.execute_input":"2024-06-09T08:37:30.552511Z","iopub.status.idle":"2024-06-09T08:37:30.584024Z","shell.execute_reply.started":"2024-06-09T08:37:30.552460Z","shell.execute_reply":"2024-06-09T08:37:30.583171Z"},"trusted":true},"execution_count":null,"outputs":[]}]}