{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"print(\"\\n... IMPORTS STARTING ...\\n\")\n\n\n# print(\"\\n... PIP/APT INSTALLS AND DOWNLOADS/ZIP STARTING ...\")\n# print(\"... PIP/APT INSTALLS COMPLETE ...\\n\")\n\nprint(\"\\n\\tVERSION INFORMATION\")\n# Machine Learning and Data Science Imports\nimport tensorflow as tf; print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\");\nimport tensorflow_addons as tfa; print(f\"\\t\\t– TENSORFLOW ADDONS VERSION: {tfa.__version__}\");\nimport tensorflow_io as tfio; print(f\"\\t\\t– TENSORFLOW I/O VERSION: {tfio.__version__}\");\nimport pandas as pd; pd.options.mode.chained_assignment = None;\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\");\nfrom sklearn.preprocessing import RobustScaler, PolynomialFeatures\nfrom pandarallel import pandarallel; pandarallel.initialize();\nfrom sklearn.model_selection import GroupKFold, StratifiedKFold\n\n# Competition Specific (AUDIO)\nimport librosa; import librosa.display\nimport panel as pn; pn.extension()\nimport soundfile as sf\nfrom  soundfile import SoundFile\nimport IPython.display as ipd\n\n# Built In Imports\nfrom kaggle_datasets import KaggleDatasets\nfrom collections import Counter\nfrom datetime import datetime\nfrom glob import glob\nimport warnings\nimport requests\nimport hashlib\nimport imageio\nimport IPython\nimport sklearn\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport os\nimport gc\nimport re\n\n# Visualization Imports\nfrom matplotlib.colors import ListedColormap\nimport matplotlib.patches as patches\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm; tqdm.pandas();\nimport plotly.express as px\nimport seaborn as sns\nfrom PIL import Image, ImageEnhance\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nfrom matplotlib import animation, rc; rc('animation', html='jshtml')\nimport plotly\nimport PIL\nimport cv2\n\nimport plotly.io as pio\nprint(pio.renderers)\n\ndef seed_it_all(seed=7):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\n    \nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-10T21:58:16.189397Z","iopub.execute_input":"2022-03-10T21:58:16.189753Z","iopub.status.idle":"2022-03-10T21:58:28.091851Z","shell.execute_reply.started":"2022-03-10T21:58:16.189657Z","shell.execute_reply":"2022-03-10T21:58:28.090859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_audio_info(filepath):\n    \"\"\"\n    Get some properties from  an audio file\n    \n    SOURCE: https://www.kaggle.com/kneroma/birdclef-mels-computer-public\n    \"\"\"\n    with SoundFile(filepath) as f:\n        sr = f.samplerate\n        frames = f.frames\n        duration = float(frames)/sr\n    return {\"frames\": frames, \"sr\": sr, \"duration\": duration}\n\ndef add_audio_to_df(row):\n    for k,v in get_audio_info(row[\"f_path\"]).items():\n        row[k] = v\n    return row","metadata":{"execution":{"iopub.status.busy":"2022-03-10T21:58:28.093881Z","iopub.execute_input":"2022-03-10T21:58:28.094113Z","iopub.status.idle":"2022-03-10T21:58:28.101024Z","shell.execute_reply.started":"2022-03-10T21:58:28.094086Z","shell.execute_reply":"2022-03-10T21:58:28.100077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 65-80 seconds usually\nt1 = time.time()\ntrain_df = pd.read_csv(\"../input/birdclef-2022/train_metadata.csv\")\ntrain_df[\"f_path\"] = \"/kaggle/input/birdclef-2022/train_audio/\"+train_df[\"filename\"]\ntrain_df = train_df.parallel_apply(add_audio_to_df, axis=1)\nprint(time.time()-t1)","metadata":{"execution":{"iopub.status.busy":"2022-03-10T21:58:28.102406Z","iopub.execute_input":"2022-03-10T21:58:28.102798Z","iopub.status.idle":"2022-03-10T21:59:44.415767Z","shell.execute_reply.started":"2022-03-10T21:58:28.102754Z","shell.execute_reply":"2022-03-10T21:59:44.414561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sample Rate\nSR = 32_000\n\n# Window Size (in seconds)\nWINDOW = 7\n\n# Window Size (in samples)\nWINDOW_SR = WINDOW*SR\n\n# Step/Overlap Size (in seconds)\nSTEP = WINDOW*(2/3)\nOVERLAP = WINDOW*(1/3)\n\n# Step/Overlap Size (in samples)(needs to be an int for indexing)\nSTEP_SR = int(round(STEP*SR))\nOVERLAP_SR = int(round(OVERLAP*SR))\n\n# Discard Threshold Time (in seconds)\nDISCARD = 6\n\n# Discard Threshold Time (in samples)\nDISCARD_SR = int(DISCARD*SR)\n\nOUT_DIR = \"/kaggle/working/train_segments\"\nos.makedirs(OUT_DIR, exist_ok=True)\n\ndemo_sr = SR\ndemo_path = train_df.iloc[12].f_path\nfor demo_audio in sf.blocks(demo_path, blocksize=WINDOW_SR, overlap=OVERLAP_SR, dtype=\"float32\", fill_value=0.0):\n    print(f\"\\n\\n\\nAUDIO SHAPE         : {demo_audio.shape}\")\n    print(f\"APPROX # PAD VALUES : {np.count_nonzero(demo_audio==0.0)}\\n\")\n    display(ipd.Audio(demo_audio, rate=demo_sr))\n    plt.plot(demo_audio)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-10T21:59:44.418814Z","iopub.execute_input":"2022-03-10T21:59:44.419146Z","iopub.status.idle":"2022-03-10T21:59:45.979407Z","shell.execute_reply.started":"2022-03-10T21:59:44.419107Z","shell.execute_reply":"2022-03-10T21:59:45.978422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"segment_kwargs = dict(\n    sample_rate=SR, \n    window_size=WINDOW_SR, \n    step_size=STEP_SR,\n    overlap_size=OVERLAP_SR, \n    discard_size=DISCARD_SR, \n    output_dir=OUT_DIR,\n    _fill_value=0.0, \n    _format=\"OGG\"\n)\n\ndef segment_audio(audio_path, sample_rate, window_size, step_size, overlap_size, discard_size, output_dir, _fill_value=0.0, _format=\"OGG\"):\n    \"\"\"\n    \n    TBD\n    \n    Args:\n        TBD\n    \n    Returns:\n        TBD\n    \n    \"\"\"\n    # Prepare\n    example_id = audio_path.rsplit(\"/\", 1)[-1][:-4]\n    save_dir = os.path.join(output_dir, example_id)\n    if not os.path.isdir(save_dir): os.makedirs(save_dir, exist_ok=True)\n        \n    # Process\n    for i, audio_block in enumerate(sf.blocks(audio_path, window_size, overlap_size, dtype=\"float32\", fill_value=_fill_value)):\n        \n        # Discard if not the first clip and less than the discard amount of the clip is padding\n        if i==0 or np.count_nonzero(audio_block==0.0)<discard_size:\n            \n            # Get start and end slice locations in samples\n            _start, _end = step_size*i, (step_size*i)+step_size\n            \n            # Save the file\n            sf.write(os.path.join(save_dir, f\"{_start:>09}_{_end:>09}.ogg\"), audio_block, samplerate=sample_rate, format=_format)","metadata":{"execution":{"iopub.status.busy":"2022-03-10T21:59:45.980919Z","iopub.execute_input":"2022-03-10T21:59:45.98125Z","iopub.status.idle":"2022-03-10T21:59:45.992332Z","shell.execute_reply.started":"2022-03-10T21:59:45.981208Z","shell.execute_reply":"2022-03-10T21:59:45.99148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"FIRST CLIP\")\nsegment_audio(train_df.iloc[0].f_path, **segment_kwargs)\ndisplay(ipd.Audio(\"./train_segments/XC125458/000000000_000149333.ogg\"))\ndisplay(ipd.Audio(\"./train_segments/XC125458/000149333_000298666.ogg\"))\n\nprint(\"CROPPED CLIP\")\nsegment_audio(train_df.iloc[4].f_path, **segment_kwargs)\ndisplay(ipd.Audio(\"./train_segments/XC207431/000000000_000149333.ogg\"))","metadata":{"execution":{"iopub.status.busy":"2022-03-10T21:59:45.993917Z","iopub.execute_input":"2022-03-10T21:59:45.994171Z","iopub.status.idle":"2022-03-10T21:59:46.281152Z","shell.execute_reply.started":"2022-03-10T21:59:45.994143Z","shell.execute_reply":"2022-03-10T21:59:46.280078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 23 seconds for 24 examples with progress_apply (~1 ex/s)\n# 12 seconds for 24 examples with parallel apply (~0.5 ex/s)(~50% reduction in time taken)\n# Assuming 0.5 ex/s, 14852 examples will take 7426 seconds or ~2 hours\n\nt1 = time.time()\ntrain_df.f_path.progress_apply(lambda x: segment_audio(x, **segment_kwargs))\nprint(time.time()-t1)","metadata":{"execution":{"iopub.status.busy":"2022-03-10T21:59:46.353263Z","iopub.execute_input":"2022-03-10T21:59:46.353653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**How To Read .ogg Files Into tf.data.Dataset**","metadata":{}},{"cell_type":"code","source":"def power_to_db(S, amin=1e-10, top_db=80.0):\n    \"\"\"Convert a power-spectrogram (magnitude squared) to decibel (dB) units.\n    Computes the scaling ``10 * log10(S / max(S))`` in a numerically\n    stable way.\n    Based on:\n    https://librosa.github.io/librosa/generated/librosa.core.power_to_db.html\n    \"\"\"\n    def _tf_log10(x):\n        numerator = tf.math.log(x)\n        denominator = tf.math.log(tf.constant(10, dtype=numerator.dtype))\n        return numerator / denominator\n    \n    # Scale magnitude relative to maximum value in S. Zeros in the output \n    # correspond to positions where S == ref.\n    ref = tf.reduce_max(S)\n\n    log_spec = 10.0 * _tf_log10(tf.maximum(amin, S))\n    log_spec -= 10.0 * _tf_log10(tf.maximum(amin, ref))\n\n    log_spec = tf.maximum(log_spec, tf.reduce_max(log_spec) - top_db)\n\n    return log_spec\n\n\ndef tf_mono_to_color(X, eps=1e-6, mean=None, std=None):\n    \"\"\" https://www.kaggle.com/kneroma/birdclef-mels-computer-public \"\"\"\n    mean = mean or tf.math.reduce_mean(X)\n    std = std or tf.math.reduce_std(X)\n    \n    X = (X-mean)/(std+eps)\n    _min, _max = tf.math.reduce_min(X), tf.math.reduce_max(X)\n\n    if (_max - _min) > eps:\n        V = tf.keras.backend.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = tf.cast(V, tf.uint8)\n    else:\n        V = tf.zeros_like(X, dtype=tf.uint8)\n    return V","metadata":{"execution":{"iopub.status.busy":"2022-03-08T23:29:08.920898Z","iopub.execute_input":"2022-03-08T23:29:08.921421Z","iopub.status.idle":"2022-03-08T23:29:08.932762Z","shell.execute_reply.started":"2022-03-08T23:29:08.921387Z","shell.execute_reply":"2022-03-08T23:29:08.931716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_ogg_files = tf.io.gfile.glob('./train_segments/**/*.ogg')\n\naudio_ds = tf.data.Dataset.from_tensor_slices(all_ogg_files)\nprint(next(iter(audio_ds)))\naudio_ds = audio_ds.map(lambda x: tfio.audio.decode_vorbis(tf.io.read_file(x), shape=(224000,-1))[:, 0])\naudio_ds = audio_ds.map(lambda x: tfio.audio.spectrogram(x, nfft=3200, window=3200, stride=800))\naudio_ds = audio_ds.map(lambda x: tfio.audio.melscale(x, rate=32000, mels=128, fmin=0, fmax=16000))\naudio_ds = audio_ds.map(lambda x: tf.transpose(power_to_db(x), perm=(1,0))).map(tf_mono_to_color)\naudio_ds","metadata":{"execution":{"iopub.status.busy":"2022-03-08T23:29:14.590226Z","iopub.execute_input":"2022-03-08T23:29:14.590499Z","iopub.status.idle":"2022-03-08T23:29:14.903308Z","shell.execute_reply.started":"2022-03-08T23:29:14.590468Z","shell.execute_reply":"2022-03-08T23:29:14.902334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(next(iter(audio_ds)))\nplt.show()\nplt.imshow(ogg_path_to_melspec('./train_segments/XC174953/000000000_000149333.ogg')[0])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T23:29:36.058176Z","iopub.execute_input":"2022-03-08T23:29:36.058489Z","iopub.status.idle":"2022-03-08T23:29:41.569111Z","shell.execute_reply.started":"2022-03-08T23:29:36.058453Z","shell.execute_reply":"2022-03-08T23:29:41.568454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mono_to_color(X, eps=1e-6, mean=None, std=None):\n    \"\"\" https://www.kaggle.com/kneroma/birdclef-mels-computer-public \"\"\"\n    mean = mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n    \n    _min, _max = X.min(), X.max()\n\n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n\n    return V\n\ndef crop_or_pad(y, length, is_train=True, start=None):\n    \"\"\" https://www.kaggle.com/kneroma/birdclef-mels-computer-public \"\"\"\n    if len(y) < length:\n        y = np.concatenate([y, np.zeros(length - len(y))])\n        \n        n_repeats = length // len(y)\n        epsilon = length % len(y)\n        \n        y = np.concatenate([y]*n_repeats + [y[:epsilon]])\n        \n    elif len(y) > length:\n        if not is_train:\n            start = start or 0\n        else:\n            start = start or np.random.randint(len(y) - length)\n\n        y = y[start:start + length]\n\n    return y\n\ndef ogg_path_to_melspec(path, duration=7, n_mels=128, sr=32_000, fmin=0, fmax=None, resample=True, resample_method=\"kaiser_fast\", step_size=None, n_fft=None, hop_length=None):\n    \"\"\" TBD - KKILLER INSPIRED\"\"\"\n    def __get_melspec(audio):\n        melspec = librosa.feature.melspectrogram(y=audio, sr=sr, n_mels=n_mels, fmin=fmin, fmax=fmax, n_fft=n_fft, hop_length=hop_length)\n        melspec = librosa.power_to_db(melspec).astype(np.float32)\n        return mono_to_color(melspec)\n    \n    # Initialize And Preparation\n    step_size = step_size or int(duration*sr*0.666)\n    n_fft = n_fft or sr//10\n    hop_length = hop_length or sr//40\n    fmax = fmax or sr//2\n    duration*=sr\n    \n    # Load Audio From Path, Fix Dual Channel & Resample If Necessary (Shouldn't Be)\n    audio, orig_sr = sf.read(path, dtype=\"float32\")\n    if len(audio.shape)>1: audio=audio[:, 0]\n    if resample and orig_sr!=sr:\n        audio = librosa.resample(audio, orig_sr=orig_sr, target_sr=sr, res_type=resample_method)\n    \n    # Split audio clip into equal segements of specified duration\n    audio_segments = [audio[i:i+duration] for i in range(0, max(1, len(audio)-duration+1), step_size)]\n    \n    # Fix last audio segment\n    audio_segments[-1] = crop_or_pad(audio_segments[-1] , length=duration)\n    \n    # Generate the individual segment mel spectrograms\n    images = np.stack([__get_melspec(audio_segment) for audio_segment in tqdm(audio_segments, total=len(audio_segments))])\n    \n    return images\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-03-08T23:26:25.859621Z","iopub.execute_input":"2022-03-08T23:26:25.859925Z","iopub.status.idle":"2022-03-08T23:26:26.240567Z","shell.execute_reply.started":"2022-03-08T23:26:25.859891Z","shell.execute_reply":"2022-03-08T23:26:26.239903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%timeit\n\ndemo_audio, demo_sr = sf.read(demo_path, dtype=\"float32\", start=0, stop=WINDOW_SR)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T16:41:57.528955Z","iopub.execute_input":"2022-03-08T16:41:57.530111Z","iopub.status.idle":"2022-03-08T16:42:05.493751Z","shell.execute_reply.started":"2022-03-08T16:41:57.53005Z","shell.execute_reply":"2022-03-08T16:42:05.492675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%timeit\n\ndemo_audio, demo_sr = sf.read(demo_path, dtype=\"float32\")","metadata":{"execution":{"iopub.status.busy":"2022-03-08T16:42:05.49589Z","iopub.execute_input":"2022-03-08T16:42:05.496242Z","iopub.status.idle":"2022-03-08T16:42:16.342025Z","shell.execute_reply.started":"2022-03-08T16:42:05.496192Z","shell.execute_reply":"2022-03-08T16:42:16.339657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%timeit\n\nfor demo_audio in sf.blocks(demo_path, blocksize=WINDOW_SR, overlap=WINDOW_SR-STEP_SR, dtype=\"float32\", fill_value=0.0):\n    pass\n    # print(demo_audio.shape)\n    #     display(ipd.Audio(demo_audio, rate=SR))\n    #     plt.plot(demo_audio)\n    #     plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%timeit\n\ndemo_audio, demo_sr = sf.read(demo_path, dtype=\"float32\", start=0, stop=WINDOW_SR, fill_value=0.0)\ndemo_audio, demo_sr = sf.read(demo_path, dtype=\"float32\", start=STEP_SR, stop=STEP_SR+WINDOW_SR, fill_value=0.0)","metadata":{},"execution_count":null,"outputs":[]}]}