{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Motivation \n- The orginal goal to write this notebook and  using onset_detection \n  to detect the most intersting 5s in the audio file is to extract features \n  from embeddings vectors of the bird-vocalization-classifier.\n\n- To do this extraction and infer the pre-trained model using all the dataset will take so much time for me with the resources I have. \n- so, I need to select the most intersting 5s in audio files.\n- I planified to share a notebook after doing the infering, but after the a disccusion with SPS444, here : https://www.kaggle.com/competitions/birdclef-2023/discussion/397213\n  I take a decesion to write and share thid notebook.\n","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as hub\nimport tensorflow_io as tfio\nimport soundfile as sf\n\nimport pandas as pd\nimport numpy as np\nimport librosa\nimport glob\nfrom scipy import signal\nimport random\n\nimport csv\nimport io\nimport os\n#import noisereduce as nr\n\nfrom IPython.display import Audio\n#\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Dense\n#\nimport shutil","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-30T15:41:49.856091Z","iopub.execute_input":"2023-03-30T15:41:49.856615Z","iopub.status.idle":"2023-03-30T15:42:00.702904Z","shell.execute_reply.started":"2023-03-30T15:41:49.856568Z","shell.execute_reply":"2023-03-30T15:42:00.701112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = hub.load('https://kaggle.com/models/google/bird-vocalization-classifier/frameworks/tensorFlow2/variations/bird-vocalization-classifier/versions/1')","metadata":{"execution":{"iopub.status.busy":"2023-03-27T13:38:20.288696Z","iopub.execute_input":"2023-03-27T13:38:20.289486Z","iopub.status.idle":"2023-03-27T13:38:27.298653Z","shell.execute_reply.started":"2023-03-27T13:38:20.289446Z","shell.execute_reply":"2023-03-27T13:38:27.297567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pandas as pd\ndf= pd.read_csv(\"/kaggle/input/onset-events-birdclef2023/frames-features-birdCLEF2023 (2).csv\")\n\n#label_data = df[df['pri_label'] == label]\nbins = np.arange(101)\nplt.hist(df['onset_events'], bins=bins)\nplt.title(f\"Onset event distribution for all\")\nplt.xlabel(\"Onset events\")\nplt.ylabel(\"Count\")\n#plt.xlim(0,5)\n#plt.ylim(0,50)\nplt.show()\n\n# Group the dataframe by pri_label and count the number of occurrences of each label\ncount_df = df.groupby('pri_label').size().reset_index(name='count')\nplt.hist(count_df['count'], bins=500)\nplt.title(f\"Onset event distribution for count\")\nplt.xlabel(\"Onset events\")\nplt.ylabel(\"Count\")\nplt.xlim(0,2000)\nplt.show()\n# Print the new dataframe\n\npd.set_option('display.max_rows', 500)\nprint(count_df.sort_values(by=[\"count\"]))","metadata":{"execution":{"iopub.status.busy":"2023-03-30T15:58:10.408534Z","iopub.execute_input":"2023-03-30T15:58:10.409422Z","iopub.status.idle":"2023-03-30T15:58:12.464364Z","shell.execute_reply.started":"2023-03-30T15:58:10.409360Z","shell.execute_reply":"2023-03-30T15:58:12.462911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pandas as pd\ndf= pd.read_csv(\"/kaggle/input/onset-events-birdclef2023/frames-features-birdCLEF2023 (2).csv\")\nlabels = df['pri_label'].unique()\nfor label in labels:\n    label_data = df[df['pri_label'] == label]\n    print(len(label_data))\n    plt.hist(label_data['onset_events'], bins=100)\n    plt.title(f\"Onset event distribution for {label}\")\n    plt.xlabel(\"Onset events\")\n    plt.ylabel(\"Count\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-28T12:21:59.822034Z","iopub.execute_input":"2023-03-28T12:21:59.822805Z","iopub.status.idle":"2023-03-28T12:23:38.480801Z","shell.execute_reply.started":"2023-03-28T12:21:59.822757Z","shell.execute_reply":"2023-03-28T12:23:38.478809Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## function realated to audio file \n\ndef frame_audio(\n      audio_array: np.ndarray,\n      window_size_s: float = 5.0,\n      hop_size_s: float = 5.0,\n      sample_rate = 32000,\n      ) -> np.ndarray:\n    \n    \"\"\"Helper function for framing audio for inference.\"\"\"\n    \"\"\" using tf.signal \"\"\"\n    if window_size_s is None or window_size_s < 0:\n        return audio_array[np.newaxis, :]\n    frame_length = int(window_size_s * sample_rate)\n    hop_length = int(hop_size_s * sample_rate)\n    framed_audio = tf.signal.frame(audio_array, frame_length, hop_length, pad_end=True)\n    return framed_audio\n\ndef ensure_sample_rate(waveform, original_sample_rate,\n                       desired_sample_rate=32000):\n    \"\"\"Resample waveform if required.\"\"\"\n    if original_sample_rate != desired_sample_rate:\n        waveform = tfio.audio.resample(waveform, original_sample_rate, desired_sample_rate)\n    return desired_sample_rate, waveform\ndef f_high(y,sr):\n    b,a = signal.butter(10, 2000/(sr/2), btype='highpass')\n    yf = signal.lfilter(b,a,y)\n    return yf","metadata":{"execution":{"iopub.status.busy":"2023-03-27T13:59:32.606496Z","iopub.execute_input":"2023-03-27T13:59:32.606877Z","iopub.status.idle":"2023-03-27T13:59:32.615459Z","shell.execute_reply.started":"2023-03-27T13:59:32.606843Z","shell.execute_reply":"2023-03-27T13:59:32.614444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## load file, ensure sample rate, and frame audio\ndef frame_file(file_name)-> np.ndarray:\n    audio, sample_rate = librosa.load(file_name,sr=32000)\n    audio2 = f_high(audio,32000)\n    #print(sample_rate)\n    framed_audio =  frame_audio(audio2)\n    return framed_audio\nframe = frame_file(\"/kaggle/input/birdclef-2023/train_audio/afghor1/XC156639.ogg\")\nlen(frame[:,0])\nframe[0,:].shape\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-27T14:02:58.475358Z","iopub.execute_input":"2023-03-27T14:02:58.475726Z","iopub.status.idle":"2023-03-27T14:02:58.546678Z","shell.execute_reply.started":"2023-03-27T14:02:58.475689Z","shell.execute_reply":"2023-03-27T14:02:58.545715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\n\ndef list_files(path):\n    # Create an empty list to store file paths and parent directory names\n    file_data = []\n\n    # Loop through all files in the directory path\n    for root, dirs, files in os.walk(path):\n        for file in files:\n            # Check if the file has a .ogg extension\n            if file.endswith('.ogg'):\n                # Get the full path of the file\n                full_path = os.path.join(root, file)\n                # Get the name of the parent directory of the file\n                parent_dir = os.path.basename(root)\n                # Append the file path and parent directory name to the list\n                #file_data.append((full_path, parent_dir))\n                file_data.append((full_path, parent_dir))\n\n    # Create a pandas DataFrame from the list of file paths and parent directory names\n    df = pd.DataFrame(file_data, columns=['full_path', 'parent_dir'])\n    return df\n\n# Call the list_files function with the directory path as an argument\ndf = list_files('/kaggle/input/birdclef-2023/train_audio/')\n## print unique value\nuv = df['parent_dir'].nunique()\nuv_count = df['parent_dir'].count()\nuv_value_counts = df['parent_dir'].value_counts()\n\n\npd.set_option('display.max_rows', 500)\nprint(uv)\nprint(uv_count)\n#print(uv_value_counts)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-27T13:38:27.370789Z","iopub.execute_input":"2023-03-27T13:38:27.371402Z","iopub.status.idle":"2023-03-27T13:38:27.674351Z","shell.execute_reply.started":"2023-03-27T13:38:27.371365Z","shell.execute_reply":"2023-03-27T13:38:27.673229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#new_row = pd.DataFrame(columns=[f\"col_{i}\" for i in range(1280)], index=[0])\n#new_row.iloc[0,:] = np.zeros(1280)\n#new_row2 = pd.DataFrame(columns=[\"pri_label\",\"file_id\",\"frame_count\",\"dB\"], index=[0])\n#embed_vec = pd.concat([new_row, new_row2], axis=1)\n\ndf_frames = pd.DataFrame(columns=[\"pri_label\",\"file_id\",\"frame_count\",\"dB\",\"onset_events\"], index=[0])\nnew_row2 =  pd.DataFrame(columns=[\"pri_label\",\"file_id\",\"frame_count\",\"dB\",\"onset_events\"], index=[0])\n\n\n\n\nrow_counter = 0;\n#genrate 5s files \nfor index, row in  df.iterrows():\n    #print('\\r{}'.format(index), end='')\n       \n    # extract file id \n    file_id = row['full_path'].split(\".ogg\")[0].split(\"/\")[-1]\n    # create ouput bird's label folder like the tree in  train_audio\n   # if not(os.path.isdir('/kaggle/working/5s_train_audio/'+row['parent_dir']+'/')):\n    #    os.mkdir('/kaggle/working/5s_train_audio/'+row['parent_dir']+'/')\n    # load the file from input folder\n    #audio, sample_rate = librosa.load(row['full_path'])\n    # frame the audio \n    frames = frame_file(row['full_path'])\n    #audio, sample_rate = librosa.load(row['full_path'],sr=32000)\n    #save every frame to an audio file to be used later to genrate embedded vectors\n    print(index,len(frames[:,0]))\n    for frame_count in range(len(frames[:,0])):\n        \n    #for frame_count in range(1):\n        #sf.write('/kaggle/working/5s_train_audio/'+row['parent_dir']+'/'+file_id+'_'+str(frame_count)+'.ogg', frames[frame_count,:], 32000, format='ogg')\n        #logits, embeddings = model.infer_tf(frames[frame_count,tf.newaxis])\n        rms = librosa.feature.rms(y=frames[frame_count,tf.newaxis])\n        dB = librosa.core.amplitude_to_db(rms, ref=1.0)\n        onset_events = len(librosa.onset.onset_detect(y=frames[frame_count,:].numpy(),sr=32000))\n        # Print the sound level in decibels\n        #print(rms)\n        #print(\"Sound level (dB): \", dB[0][0].mean())\n        \n        #numpy_array = np.array(embeddings)\n        #df_temp1 = pd.DataFrame(numpy_array)\n\n        #print(embeddings.shape)\n        #print(numpy_array.shape)\n        #print(embed_vec.shape)\n        \n        new_row2.loc[0][\"pri_label\"] = row['parent_dir']\n        new_row2.loc[0][\"file_id\"] = file_id\n        new_row2.loc[0][\"frame_count\"] = frame_count\n        new_row2.loc[0][\"dB\"] = dB[0][0].mean()/100\n        new_row2.loc[0][\"onset_events\"] = onset_events\n        if (row_counter == 0):\n            df_frames.loc[0] = new_row2.loc[0]\n            row_counter = 1\n        else:\n            df_frames = df_frames.append( new_row2)\n        #row_embed_vect = row_embed_vect+1\n        ","metadata":{"execution":{"iopub.status.busy":"2023-03-27T14:03:07.161821Z","iopub.execute_input":"2023-03-27T14:03:07.162483Z","iopub.status.idle":"2023-03-27T14:03:11.387971Z","shell.execute_reply.started":"2023-03-27T14:03:07.162444Z","shell.execute_reply":"2023-03-27T14:03:11.386527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_frames.to_csv(\"frames-features-birdCLEF2023.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-03-27T13:44:29.494154Z","iopub.execute_input":"2023-03-27T13:44:29.494754Z","iopub.status.idle":"2023-03-27T13:44:29.503176Z","shell.execute_reply.started":"2023-03-27T13:44:29.494707Z","shell.execute_reply":"2023-03-27T13:44:29.502248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_frames\n","metadata":{"execution":{"iopub.status.busy":"2023-03-27T14:03:14.251319Z","iopub.execute_input":"2023-03-27T14:03:14.251710Z","iopub.status.idle":"2023-03-27T14:03:14.269552Z","shell.execute_reply.started":"2023-03-27T14:03:14.251653Z","shell.execute_reply":"2023-03-27T14:03:14.268509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio, sample_rate = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/yetgre1/XC247367.ogg\",sr=32000)\n\n#Audio(data=audio, rate=32000)\ndisplay(Audio(data=audio, rate=32000))\nlibrosa.display.waveshow(audio, sr=32000)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-03-27T14:12:28.768442Z","iopub.execute_input":"2023-03-27T14:12:28.769193Z","iopub.status.idle":"2023-03-27T14:12:29.254987Z","shell.execute_reply.started":"2023-03-27T14:12:28.769153Z","shell.execute_reply":"2023-03-27T14:12:29.253937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}