{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# install fastkaggle if not available\ntry: import fastkaggle\nexcept ModuleNotFoundError:\n    !pip install -Uq fastkaggle\n\nfrom fastkaggle import *","metadata":{"execution":{"iopub.status.busy":"2023-04-29T12:11:14.087608Z","iopub.execute_input":"2023-04-29T12:11:14.088039Z","iopub.status.idle":"2023-04-29T12:11:31.130019Z","shell.execute_reply.started":"2023-04-29T12:11:14.088000Z","shell.execute_reply":"2023-04-29T12:11:31.128816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import multiprocessing\nfrom fastai.vision.all import *\nimport librosa as lb\nfrom IPython.display import Audio\nfrom sklearn.model_selection import train_test_split\nimport soundfile as sf\nfrom  soundfile import SoundFile\nfrom tqdm.notebook import tqdm\nimport joblib\nimport matplotlib.pyplot as plt\nfrom joblib import Parallel, delayed","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-29T12:11:55.217152Z","iopub.execute_input":"2023-04-29T12:11:55.218160Z","iopub.status.idle":"2023-04-29T12:12:00.439707Z","shell.execute_reply.started":"2023-04-29T12:11:55.218104Z","shell.execute_reply":"2023-04-29T12:12:00.438129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAKE_DATASET = False\ncomp = 'birdclef-2023'\npath = setup_comp(comp, install='fastai \"timm>=0.6.2.dev0\"')\npath","metadata":{"execution":{"iopub.status.busy":"2023-04-29T12:12:14.602469Z","iopub.execute_input":"2023-04-29T12:12:14.603004Z","iopub.status.idle":"2023-04-29T12:12:31.189010Z","shell.execute_reply.started":"2023-04-29T12:12:14.602953Z","shell.execute_reply":"2023-04-29T12:12:31.187919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_meta_data_file_path = path/'train_metadata.csv'\ntrain_audio_path = path/'train_audio'\ntrain_img_path = Path('/kaggle/working/train_images')\ntrain_bagwea1_audio_path = train_audio_path/'bagwea1'\n\n# params\nsr = 32000\nn_mels = 128\nfmin = 0\nfmax = None or sr // 2\nduration = 5\naudio_length = duration * sr\nstep = None or audio_length\nres_type=\"kaiser_fast\"\nresample=True","metadata":{"execution":{"iopub.status.busy":"2023-04-29T12:12:31.194140Z","iopub.execute_input":"2023-04-29T12:12:31.196578Z","iopub.status.idle":"2023-04-29T12:12:31.205933Z","shell.execute_reply.started":"2023-04-29T12:12:31.196535Z","shell.execute_reply":"2023-04-29T12:12:31.204263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mkdir(train_img_path, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T12:12:44.239212Z","iopub.execute_input":"2023-04-29T12:12:44.239670Z","iopub.status.idle":"2023-04-29T12:12:44.248335Z","shell.execute_reply.started":"2023-04-29T12:12:44.239619Z","shell.execute_reply":"2023-04-29T12:12:44.247272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_audio_info(filepath):\n    with SoundFile(filepath) as f:\n        sr = f.samplerate\n        frames = f.frames\n        duration = float(frames)/sr\n    return {\"frames\": frames, \"sr\": sr, \"duration\": duration}\n\n\ndef compute_melspec(y, sr, n_mels, fmin, fmax):\n    melspec = lb.feature.melspectrogram(y=y, sr=sr, n_mels=n_mels, fmin=fmin, fmax=fmax)\n    melspec = lb.power_to_db(melspec).astype(np.float32)\n    return melspec\n\n\ndef mono_to_color(X, eps=1e-6, mean=None, std=None):\n    mean = mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n    \n    _min, _max = X.min(), X.max()\n\n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n\n    return V\n\n\ndef crop_or_pad(y, length, is_train=True, start=None):\n    if len(y) < length:\n        y = np.concatenate([y, np.zeros(length - len(y))])\n        \n        n_repeats = length // len(y)\n        epsilon = length % len(y)\n        \n        y = np.concatenate([y]*n_repeats + [y[:epsilon]])\n        \n    elif len(y) > length:\n        if not is_train:\n            start = start or 0\n        else:\n            start = start or np.random.randint(len(y) - length)\n\n        y = y[start:start + length]\n\n    return y\n\n\ndef audio_to_image(audio, sr, n_mels, fmin, fmax):\n    melspec = compute_melspec(audio, sr, n_mels, fmin, fmax) \n    image = mono_to_color(melspec)\n    return image\n\n\ndef create_train_imgs(bird, train_img_path, train_audio_path, audio_length, sr, resample, res_type, n_mels, fmin, fmax, step):\n    mkdir(train_img_path / f'{bird.stem}', exist_ok=True)\n\n    for aud in bird.ls().sorted():\n        audio, orig_sr = sf.read(aud, dtype=\"float32\")\n\n        if resample and orig_sr != sr:\n            audio = lb.resample(audio, orig_sr, sr, res_type=res_type)\n\n        audios = [audio[i: i + audio_length] for i in range(0, max(1, len(audio) - audio_length + 1), step)]\n        audios[-1] = crop_or_pad(audios[-1], length=audio_length)\n\n        for i, audio in enumerate(audios):\n            img = audio_to_image(audio=audio, sr=sr, n_mels=n_mels, fmin=fmin, fmax=fmax)\n            img = Image.fromarray(img)\n            img.save(train_img_path/f'{bird.stem}/{aud.stem}_{5*(i+1)}.png')\n\n            \ndef parallel_train_process(train_audio_path, train_img_path, audio_length, sr, resample, res_type, n_mels, fmin, fmax, step, n_jobs=-1):\n    birds = train_audio_path.ls().sorted()\n    Parallel(n_jobs=n_jobs)(delayed(create_train_imgs)(bird, train_img_path, train_audio_path, audio_length, sr, resample, res_type, n_mels, fmin, fmax, step) for bird in tqdm(birds))","metadata":{"execution":{"iopub.status.busy":"2023-04-29T12:12:45.943070Z","iopub.execute_input":"2023-04-29T12:12:45.943516Z","iopub.status.idle":"2023-04-29T12:12:45.969104Z","shell.execute_reply.started":"2023-04-29T12:12:45.943478Z","shell.execute_reply":"2023-04-29T12:12:45.967985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path.ls()","metadata":{"execution":{"iopub.status.busy":"2023-04-29T12:12:46.845676Z","iopub.execute_input":"2023-04-29T12:12:46.846256Z","iopub.status.idle":"2023-04-29T12:12:46.856864Z","shell.execute_reply.started":"2023-04-29T12:12:46.846207Z","shell.execute_reply":"2023-04-29T12:12:46.855555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_audio_path.ls()","metadata":{"execution":{"iopub.status.busy":"2023-04-29T12:12:47.407387Z","iopub.execute_input":"2023-04-29T12:12:47.407899Z","iopub.status.idle":"2023-04-29T12:12:47.442266Z","shell.execute_reply.started":"2023-04-29T12:12:47.407854Z","shell.execute_reply":"2023-04-29T12:12:47.441208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pick a single audio\naud_bagwea1 = train_bagwea1_audio_path.ls()[0]\naud_bagwea1","metadata":{"execution":{"iopub.status.busy":"2023-04-29T12:12:47.930727Z","iopub.execute_input":"2023-04-29T12:12:47.931222Z","iopub.status.idle":"2023-04-29T12:12:47.944184Z","shell.execute_reply.started":"2023-04-29T12:12:47.931180Z","shell.execute_reply":"2023-04-29T12:12:47.942963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_bagwea1, orig_sr = sf.read(aud_bagwea1, dtype=\"float32\")\n\nif resample and orig_sr != sr:\n    audio_bagwea1 = lb.resample(audio_bagwea1, orig_sr, sr, res_type=res_type)\nimg_bagwea1 = audio_to_image(audio_bagwea1, sr, n_mels, fmin, fmax)\nshow_image(img_bagwea1, figsize=(15, 15), title=f'{train_bagwea1_audio_path.stem}');","metadata":{"execution":{"iopub.status.busy":"2023-04-29T12:12:48.731163Z","iopub.execute_input":"2023-04-29T12:12:48.731603Z","iopub.status.idle":"2023-04-29T12:13:02.045594Z","shell.execute_reply.started":"2023-04-29T12:12:48.731562Z","shell.execute_reply":"2023-04-29T12:13:02.042811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parallel_train_process(train_audio_path, train_img_path, audio_length, sr, resample, res_type, n_mels, fmin, fmax, step)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T12:13:02.047784Z","iopub.execute_input":"2023-04-29T12:13:02.048463Z","iopub.status.idle":"2023-04-29T12:45:31.654438Z","shell.execute_reply.started":"2023-04-29T12:13:02.048423Z","shell.execute_reply":"2023-04-29T12:45:31.653277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if MAKE_DATASET:\n    import os\n    from kaggle_secrets import UserSecretsClient\n\n    secrets = UserSecretsClient()\n    os.environ['KAGGLE_USERNAME'] = secrets.get_secret('KAGGLE_USERNAME')\n    os.environ['KAGGLE_KEY'] = secrets.get_secret('KAGGLE_KEY')\n\n    mk_dataset('/kaggle/train_images', 'spectrograms-birdclef-2023', force=True, upload=True)\n    ! cat /kaggle/train_images/dataset-metadata.json","metadata":{"execution":{"iopub.status.busy":"2023-04-29T12:45:31.656168Z","iopub.execute_input":"2023-04-29T12:45:31.657081Z","iopub.status.idle":"2023-04-29T12:45:31.666352Z","shell.execute_reply.started":"2023-04-29T12:45:31.657045Z","shell.execute_reply":"2023-04-29T12:45:31.665202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}