{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"sourceType":"competition"},{"sourceId":33246,"databundleVersionId":3221581,"sourceType":"competition"},{"sourceId":44224,"databundleVersionId":5188730,"sourceType":"competition"},{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":61125034,"sourceType":"kernelVersion"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# any pip install needed\n!pip install -Uq fastkaggle","metadata":{"_uuid":"91aeb1ad-2614-4775-8057-820dad0fc4da","_cell_guid":"00396690-0f28-4ab3-9d14-d2c2d7800649","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-25T23:52:13.169539Z","iopub.execute_input":"2024-04-25T23:52:13.169911Z","iopub.status.idle":"2024-04-25T23:52:21.906781Z","shell.execute_reply.started":"2024-04-25T23:52:13.169886Z","shell.execute_reply":"2024-04-25T23:52:21.905547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Input packages needed to process data.\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom fastkaggle import *\n# for audio\nimport librosa\nfrom scipy import signal\nfrom IPython import display\n\n# ml, train test split\nimport keras\nfrom sklearn.model_selection import train_test_split\n\n#plotting\nimport gc\nimport matplotlib.pyplot as plt\n\n# weight and biases\nimport wandb\n# This is code from somebody else if it works perfect if not I die\n# Try to get the API key from Kaggle secrets\ntry:\n    from kaggle_secrets import UserSecretsClient\n    user_secrets = UserSecretsClient()\n    api_key = user_secrets.get_secret(\"WANDB\")\n    # Login to wandb with the API key\n    wandb.login(key=api_key)\n    # Set anonymous mode to None\n    anonymous = None\nexcept:\n    # If Kaggle secrets are not available, set anonymous mode to 'must'\n    anonymous = 'must'\n    # Login to wandb anonymously and relogin if needed\n    wandb.login(anonymous=anonymous, relogin=True)","metadata":{"_uuid":"bab719ff-a238-4b1f-b119-c88b5f03ca9c","_cell_guid":"ffc43e9f-012e-4bff-82c1-34c293530ee1","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-26T01:21:24.255566Z","iopub.execute_input":"2024-04-26T01:21:24.255987Z","iopub.status.idle":"2024-04-26T01:21:25.720909Z","shell.execute_reply.started":"2024-04-26T01:21:24.255955Z","shell.execute_reply":"2024-04-26T01:21:25.719960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Start with loading data**\n- Import data from the 2024 birdclef data cite. \n- Filter audio?\n- convert audio to spectogram\n- train set, test set.","metadata":{"_uuid":"ccfb6e59-6411-4754-9ac9-7c68352e99a7","_cell_guid":"a36357a2-344a-4b35-80b6-b644565c14cb","trusted":true}},{"cell_type":"code","source":"start_dir = '/kaggle/input/birdclef-2024'","metadata":{"_uuid":"c5fd0540-ffba-466f-96af-b4e9ebfbfba9","_cell_guid":"fbc62437-b9f8-499e-a36a-1823d9a7ce81","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-25T23:52:22.791902Z","iopub.execute_input":"2024-04-25T23:52:22.792309Z","iopub.status.idle":"2024-04-25T23:52:22.797366Z","shell.execute_reply.started":"2024-04-25T23:52:22.792276Z","shell.execute_reply":"2024-04-25T23:52:22.796126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Band pass filtering**\n\nBird's sound (either call or songs) should fall into a rather specific range. Filter the raw audio file by a certain frequency band seem to be a good place to start clean/preprocess data.","metadata":{}},{"cell_type":"code","source":"def filter_audio(data, fs, low = 1000, high = 8000, order=5):\n    \"\"\"Band pass filter of auido data, deflaut pass band is 1kHz to 8kHz\"\"\"\n    # Passband decided base on a quick google search of birds calls frequency \n    # range.\n    nyq = 0.5 * fs\n    lc_norm = low/ nyq\n    hc_norm = high/ nyq\n    sos = signal.butter(order, [lc_norm, hc_norm], btype='band', output='sos')\n    y = signal.sosfiltfilt(sos, data)\n    return y","metadata":{"execution":{"iopub.status.busy":"2024-04-25T23:52:22.800145Z","iopub.execute_input":"2024-04-25T23:52:22.800475Z","iopub.status.idle":"2024-04-25T23:52:22.808830Z","shell.execute_reply.started":"2024-04-25T23:52:22.800450Z","shell.execute_reply":"2024-04-25T23:52:22.807976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Choose a random file and test filtering.\naudio, sr = librosa.load('/kaggle/input/birdclef-2024/train_audio/asbfly/XC175797.ogg')\nflt_aud = filter_audio(audio, sr)\nlibrosa.display.waveshow(audio,sr = sr)\nlibrosa.display.waveshow(flt_aud,sr = sr)\ndisplay.display(display.Audio(flt_aud, rate = sr))\ndisplay.display(display.Audio(audio, rate = sr))","metadata":{"execution":{"iopub.status.busy":"2024-04-25T23:53:04.984198Z","iopub.execute_input":"2024-04-25T23:53:04.985075Z","iopub.status.idle":"2024-04-25T23:53:06.123294Z","shell.execute_reply.started":"2024-04-25T23:53:04.985044Z","shell.execute_reply":"2024-04-25T23:53:06.122182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing:\n1. Load meta data file, remove duplicate files from data list.","metadata":{}},{"cell_type":"code","source":"root = '/kaggle/input/birdclef-2024'\ntrain_meta = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\nfiltered = train_meta.drop_duplicates({'primary_label','latitude','longitude','type','author'})\nprint(len(train_meta),len(filtered))","metadata":{"execution":{"iopub.status.busy":"2024-04-26T00:08:55.179278Z","iopub.execute_input":"2024-04-26T00:08:55.179649Z","iopub.status.idle":"2024-04-26T00:08:55.198536Z","shell.execute_reply.started":"2024-04-26T00:08:55.179621Z","shell.execute_reply":"2024-04-26T00:08:55.197511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2. Apply scikit learn train test split function, split filtered dataframe into training set and testing set.","metadata":{}},{"cell_type":"code","source":"train_pd, test_pd = train_test_split(filtered)\nprint(len(train_pd), len(test_pd))","metadata":{"execution":{"iopub.status.busy":"2024-04-26T00:18:08.292078Z","iopub.execute_input":"2024-04-26T00:18:08.293019Z","iopub.status.idle":"2024-04-26T00:18:08.305949Z","shell.execute_reply.started":"2024-04-26T00:18:08.292981Z","shell.execute_reply":"2024-04-26T00:18:08.304879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"3. To run the data through a classifier we should first convert them to what is called a Mel-frequency cepstrum(MFCC).\n- Seems like audio file of varying length can be converted into a image-like plot, but the longer the audio is to create the MFCC, the more danger of over fitting. ","metadata":{}},{"cell_type":"code","source":"filelist[:10]","metadata":{"execution":{"iopub.status.busy":"2024-04-26T01:09:02.639169Z","iopub.execute_input":"2024-04-26T01:09:02.640097Z","iopub.status.idle":"2024-04-26T01:09:02.647782Z","shell.execute_reply.started":"2024-04-26T01:09:02.640062Z","shell.execute_reply":"2024-04-26T01:09:02.646767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_data(root_dir,filelist):\n    # load audio data from file given the root dir an a filelist of names.\n    audio_list = [[] for x in range(len(filelist))]\n    sr = [0] * len(filelist)\n    for i,file in enumerate(filelist):\n        fullpath = os.path.join(root_dir,'train_audio', file)\n        print(fullpath,i)\n        print(len(audio_list))\n        audio_list[i], sr[i] = librosa.load(fullpath)\n    return audio_list, sr\nfilelist = train_pd['filename']\naud_list, sr_list = load_data(root, filelist[:5])","metadata":{"execution":{"iopub.status.busy":"2024-04-26T01:14:43.938366Z","iopub.execute_input":"2024-04-26T01:14:43.938815Z","iopub.status.idle":"2024-04-26T01:14:44.247944Z","shell.execute_reply.started":"2024-04-26T01:14:43.938787Z","shell.execute_reply":"2024-04-26T01:14:44.246761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plotMel(signal):\n    gc.enable()\n    # MK_spectrogram modified\n    N_FFT = 1024         # \n    HOP_SIZE = 1024      #  \n    N_MELS = 128          # Higher   \n    WIN_SIZE = 1024      # \n    WINDOW_TYPE = 'hann' # \n    FEATURE = 'mel'      # \n    FMIN = 1400\n\n    fig = plt.figure(1,frameon=False)\n    fig.set_size_inches(6,6)\n\n    ax = plt.Axes(fig, [0., 0., 1., 1.])\n    ax.set_axis_off()\n    fig.add_axes(ax)\n    \n    S = librosa.feature.melspectrogram(y=signal, sr=sr,\n                                        n_fft=N_FFT,\n                                        hop_length=HOP_SIZE, \n                                        n_mels=N_MELS, \n                                        htk=True, \n                                        fmin=FMIN, # higher limit ##high-pass filter freq.\n                                        fmax=sr/2) # AMPLITUDE\n    librosa.display.specshow(librosa.power_to_db(S**2,ref=np.max), fmin=FMIN) #power = S**2\n\nplotMel(aud_list[0])\n#     fig.savefig(directory)\n#     plt.ioff()\n    #plt.show(block=False)\n#     fig.clf()\n#     ax.cla()\n#     plt.clf()\n#     plt.close('all')","metadata":{"execution":{"iopub.status.busy":"2024-04-26T01:21:28.655429Z","iopub.execute_input":"2024-04-26T01:21:28.655866Z","iopub.status.idle":"2024-04-26T01:21:28.897744Z","shell.execute_reply.started":"2024-04-26T01:21:28.655835Z","shell.execute_reply":"2024-04-26T01:21:28.896469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# walk through data dir and organize data for training. \n# Need to find some library to setup data set and stuff...\n# I can get a quick filelist first\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2024-04-25T23:52:33.791454Z","iopub.status.idle":"2024-04-25T23:52:33.791959Z","shell.execute_reply.started":"2024-04-25T23:52:33.791695Z","shell.execute_reply":"2024-04-25T23:52:33.791716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## put the cross validate code here incase I loss it.\nfrom sklearn.model_selection import KFold, StratifiedKFold, train_test_split\n# functionalize kfold cross validation for our purpose:\ndef kfold_experiment(all_paths, all_labels, num_folds = 5, shuffle = True): #figure out what proper input is as we go\n    # Define the K-fold Cross Validator\n    kfold = StratifiedKFold(n_splits=num_folds, shuffle= shuffle)\n    auc_per_fold = []\n    loss_per_fold = []\n\n\n    fold_no = 1\n    for train, test in kfold.split(all_paths, all_labels):\n        # build dataset first.\n        train_ds = build_dataset(all_paths[train], all_labels[train], \n                                 batch_size = CFG.batch_size, shuffle = True, \n                                 augment= CFG.augment, augment_aud = CFG.augment_aud)\n        # Setup the model\n        inp = keras.layers.Input(shape=(None, None, 3))\n        backbone = keras_cv.models.EfficientNetV2Backbone.from_preset(CFG.preset)\n        out = keras_cv.models.ImageClassifier(\n        backbone= backbone,\n        num_classes = CFG.num_classes,\n        name = 'classifier')(inp)\n        model = keras.models.Model(inputs = inp, outputs = out)\n        # compile model\n        model.compile(optimizer = \"adam\",\n                     loss = keras.losses.CategoricalCrossentropy(label_smoothing=0.02),\n                     metrics = [keras.metrics.AUC(name='auc')])\n        # Generate a print\n        print('------------------------------------------------------------------------')\n        print(f'Training for fold {fold_no} ...')\n        # Fit data to model\n        callbacks = [\n                    lr_cb, \n                    ckpt_cb,\n                    ]\n        with tf.device(\"/GPU:0\"):\n            history = model.fit(train_ds,\n                               validation_data = valid_ds,\n                               epochs = 5,\n                               callbacks = callbacks, # issue with callbacks, haven't figured out yet./\n                               verbose = 1\n                               )\n        # model evaluation\n        scores = model.evaluate(valid_ds, verbose=1)\n        print(f'Score for fold {fold_no}: {model.metrics_names[0]} of {scores[0]}; {model.metrics_names[1]} of {scores[1]*100}%')\n        auc_per_fold.append(scores[1])\n        loss_per_fold.append(scores[0])\n        #update counter\n        fold_no +=1\n        return auc_per_fold, loss_per_fold\n\n# test functionalize k-fold k-fold cross-validation for model selection\n# for all data do this\nall_paths = np.concatenate((train_paths, valid_paths), axis=0)\nall_labels = np.concatenate((train_labels, valid_labels), axis=0)\n\n# test code with a small subset\nall_paths, _, all_labels,_ = train_test_split(all_paths, all_labels, train_size=0.05, stratify=all_labels)\nauc, loss = kfold_experiment(all_paths, all_labels,num_folds = 2)","metadata":{},"execution_count":null,"outputs":[]}]}