{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Import and Install Dependencies","metadata":{}},{"cell_type":"markdown","source":"## 1.2 Load Dependencies","metadata":{}},{"cell_type":"code","source":"\nimport os\nfrom matplotlib import pyplot as plt\nimport tensorflow as tf \nimport pandas as pd\nimport librosa\nimport numpy as np\nimport math\nfrom tqdm import tqdm\nfrom joblib import Parallel, delayed  \nimport multiprocessing\n\n","metadata":{"tags":[],"jupyter":{"source_hidden":true}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parameters\n\nclip_duration_sec = 5\nframe_length=1024\nframe_step=128\n\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Build Data Loading Function","metadata":{}},{"cell_type":"markdown","source":"## 2.1 Define Paths to Files","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/mnt/disks/birdclef-2022/all_data_df.csv\")\n# Shuffle the dataframe\ndf = df.sample(frac=1).reset_index(drop=True)\ndf","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Proc\ndifferent_birds = list(df[\"files\"])\ndf[\"bird_type\"] = df.apply(lambda x: x[\"files\"].split(\"/\")[-2], axis=1)\ndf","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"files\"][0]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"files\"] = df.apply(lambda x: x[\"files\"].replace(\n    \"/Users/viktorcikojevic/coding/identify_bird_calls/birdclef-2022/train_audio\", \n    \"/mnt/disks/birdclef-2022/birdclef-2022-data/train_audio\"\n), axis=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example\ndf_rinduc = df[df[\"bird_type\"] == \"rinduc\"]\ndf_rinduc = df_rinduc.reset_index(drop=True)\ndf_rinduc","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfile_name = df_rinduc[\"files\"][4]\nprint(file_name)\nwav, sample_rate = librosa.load(file_name, duration=5, sr=16000 )\nplt.plot(wav)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_spectrogram(wav, clip_duration_sec=5):\n    # 16000 data = 1 seconds\n    num_data = int(16000 * clip_duration_sec)\n    wav = wav[:num_data]\n    zero_padding = tf.zeros([num_data] - tf.shape(wav), dtype=tf.float32)\n    wav = tf.concat([zero_padding, wav],0)\n    spectrogram = tf.signal.stft(wav, frame_length=frame_length, frame_step=frame_step)\n    spectrogram = tf.abs(spectrogram)\n    spectrogram = spectrogram / tf.reduce_max(spectrogram)\n    spectrogram = tf.expand_dims(spectrogram, axis=2)\n    return spectrogram.numpy()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_grid = 10\n\ndef moving_average(x, w):\n    arr =  np.convolve(x, np.ones(w), 'valid') / w\n    return arr / np.max(arr)\n\ndf_rinduc = df[df[\"bird_type\"] == \"sora\"]\ndf_rinduc = df_rinduc.reset_index(drop=True)\n\nfig, axs = plt.subplots(n_grid, 3, figsize=(15,30))\nfor i in range(n_grid):\n    indx = i + 10\n    file_name = df_rinduc[\"files\"][indx]\n    print(file_name)\n    wav, sample_rate = librosa.load(file_name, duration=5, sr=16000, offset=0 )\n    wav = wav - np.average(wav)\n    \n    axs[i,0].plot(wav)\n    spectrogram = np.abs(np.fft.fftn(wav))\n    spectrogram *= 1. / np.max(spectrogram)\n    spectrogram = spectrogram[0:int(len(spectrogram)/2)]\n    \n    axs[i, 1].plot(spectrogram)\n    \n    \n    spectrogram = moving_average(spectrogram, 400)\n    axs[i, 2].plot(spectrogram[::100])\n    x_shape = np.shape(spectrogram)[0]\n    x = np.arange(1, x_shape+1)\n    spectrogram = spectrogram * np.exp(-(2000 / x)**5)\n    spectrogram = spectrogram[::100]\n    axs[i, 2].plot(spectrogram)\n    print(np.shape(spectrogram))\n    \n    # wav, sample_rate = librosa.load(file_name, duration=5, sr=16000, offset=0 )\n    # wav = wav / np.average(wav)\n    # # spectrogram = np.abs(np.fft.fftn(wav))\n    # # spectrogram *= 1. / np.max(spectrogram)\n    # spectrogram = spectrogram[0:int(len(spectrogram)/2)]\n    # x_shape = np.shape(spectrogram)[0]\n    # x = np.arange(1, x_shape+1)\n    # spectrogram = spectrogram * np.exp(-(2000 / x)**5)\n    # spectrogram = np.append(spectrogram, np.flip(spectrogram))\n    # wav = np.fft.ifftn(spectrogram)\n    \n    # axs[i,3].plot(wav)\n    #spectrogram = get_spectrogram(wav)\n    #axs[i,3].imshow(spectrogram, vmin=0, vmax=0.1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#np.shape(np.reshape(spectrogram, (10, 4000)))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.shape(spectrogram)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nn_grid = 3\n\ndf_rinduc = df[df[\"bird_type\"] == \"jabwar\"]\ndf_rinduc = df_rinduc.reset_index(drop=True)\n\nfig, axs = plt.subplots(n_grid, 2, figsize=(8,15))\nfor i in range(n_grid):\n    file_name = df_rinduc[\"files\"][i]\n    print(file_name)\n    wav, sample_rate = librosa.load(file_name, duration=5, sr=16000, offset=0 )\n    axs[i,0].plot(wav)\n    spectrogram = get_spectrogram(wav)\n    axs[i,1].imshow(spectrogram, vmin=0, vmax=0.01)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\n# Loop through each audio file\n# For a current file, calculate the spectrogram, and kill of the low frequncy components\n# Save that numpy file to some folder. Remember the file path for the dataframe\n# If the maximum height is below 0.6, classify this as \"nocall\". Otherwise, keep the label as before. \n# Save the labels in the csv file and upload to the kaggle. \n# implement the xgboost  for this !!\n\ndef load_feature_file(file_path, offset):\n    # Load the wav file\n    wav, sample_rate = librosa.load(file_path, duration=5, sr=16000, offset=offset)\n    # Make average equal to zero to avoid sporious k=0 FFT contributions\n    wav = wav - np.average(wav)\n    # Take only 5 second intervals. If the file is less then 5 seconds, pad with zeros.\n    num_data = int(16000 * 5)\n    zero_padding = np.zeros(num_data - np.shape(wav)[0])\n    wav = np.concatenate([zero_padding, wav])\n    # Take the fft and normalize it\n    spectrogram = np.abs(np.fft.fftn(wav))\n    spectrogram *= 1. / np.max(spectrogram)\n    spectrogram = spectrogram[0:int(len(spectrogram)/2)]\n    # Calculate the moving average since the FFTs have a lot of noise\n    spectrogram = moving_average(spectrogram, 400)\n    x_shape = np.shape(spectrogram)[0]\n    x = np.arange(1, x_shape+1)\n    # Apply np.exp(-(2000 / x)**5) on the FFT, to remove noise around x=0 (low frequencies)\n    spectrogram = spectrogram * np.exp(-(2000 / x)**5)\n    # Take each 100 data point, since the FFTs have a lot of data\n    spectrogram = spectrogram[::100]\n    return spectrogram\n\nfiles_new = []\nbird_type_new_label = []\nroot_folder = \"/mnt/disks/birdclef-2022/clean_ffts\"\n\nn_procs = 16\n\ndef process_file(i_offset, file_path, bird_type):\n    offset = i_offset * 5\n    feature = load_feature_file(file_path, offset)\n    new_label = bird_type\n    if np.max(feature) < 0.6:\n        new_label = \"nocall\"\n    \n    bird_folder = os.path.join(root_folder, new_label)\n    \n    file_name = file_path.split(\"/\")[-1].split(\".\")[0] + f\"_{i_offset}\"\n    if new_label == \"nocall\":\n        file_path.replace(bird_type, \"nocall\")\n    file_path_new = os.path.join(bird_folder, file_name)\n    np.save(file_path_new, feature)\n\n\nbird_types = list(set(list(df[\"bird_type\"])))\nindx = 0\nfor bird_type in bird_types:\n    print(f\"Analyzing bird {bird_type}\")\n    df_bird = df[df[\"bird_type\"] == bird_type]\n\n    bird_folder = os.path.join(root_folder, bird_type)\n    if os.path.exists(bird_folder) is False:\n        os.mkdir(bird_folder)\n    for file_path, duration_sec in zip(list(df_bird[\"files\"]), list(df_bird[\"duration_sec\"])):\n        n_procs = multiprocessing.cpu_count()\n        res = Parallel(n_jobs=n_procs)(delayed(process_file)(\n                                            i_offset,\n                                            file_path,\n                                            bird_type\n                                            ) for i_offset in range(int(duration_sec / 5)))   \n                                ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_with_silent = pd.DataFrame({\n    \"file_name\" : file_path_new,\n    \"bird_type\" : bird_type_new_label,\n                               })\ndf_with_silent","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df[\"bird_type\"] == \"fragul\"]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}