{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport librosa as lb\nimport librosa.display as lbd\nimport soundfile as sf\nfrom soundfile import SoundFile\nimport pandas as pd\nfrom IPython.display import Audio\nfrom pathlib import Path\n\nfrom matplotlib import pyplot as plt\n\nfrom tqdm.notebook import tqdm\nimport joblib, json, re\n\nfrom sklearn.model_selection import StratifiedKFold\ntqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:49.393521Z","iopub.execute_input":"2023-06-04T05:32:49.394160Z","iopub.status.idle":"2023-06-04T05:32:50.878150Z","shell.execute_reply.started":"2023-06-04T05:32:49.394127Z","shell.execute_reply":"2023-06-04T05:32:50.877065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/birdclef-2023/train_metadata.csv')\ndf['secondary'] = df['secondary_labels'].apply(lambda x: re.findall(r\"'(\\w+)'\",x))\ndf['len_sec_labels'] = df['secondary_labels'].map(len)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:50.883960Z","iopub.execute_input":"2023-06-04T05:32:50.886429Z","iopub.status.idle":"2023-06-04T05:32:51.077188Z","shell.execute_reply.started":"2023-06-04T05:32:50.886393Z","shell.execute_reply":"2023-06-04T05:32:51.076179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df.len_sec_labels>0].sample(3)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:55.113226Z","iopub.execute_input":"2023-06-04T05:32:55.113587Z","iopub.status.idle":"2023-06-04T05:32:55.146881Z","shell.execute_reply.started":"2023-06-04T05:32:55.113557Z","shell.execute_reply":"2023-06-04T05:32:55.145871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.primary_label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:55.493146Z","iopub.execute_input":"2023-06-04T05:32:55.493489Z","iopub.status.idle":"2023-06-04T05:32:55.508439Z","shell.execute_reply.started":"2023-06-04T05:32:55.493461Z","shell.execute_reply":"2023-06-04T05:32:55.507481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport pandas as pd\n\ndef birds_stratified_split(df, target_col, test_size=0.2):\n    class_counts = df[target_col].value_counts()\n    low_count_classes = class_counts[class_counts < 2].index.tolist() #Birds with single counts\n    \n    df['train'] = df[target_col].isin(low_count_classes)\n    \n    train_df, val_df = train_test_split(df[~df['train']], test_size= test_size, stratify = df[~df['train']][target_col], random_state = 42)\n    \n    train_df= pd.concat([train_df, df[df['train']]], axis =0).reset_index(drop=True)\n    \n    # remove the valid column\n    train_df.drop('train', axis= 1, inplace= True)\n    val_df.drop('train', axis =1, inplace=True)\n    \n    return train_df, val_df","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:55.722389Z","iopub.execute_input":"2023-06-04T05:32:55.722755Z","iopub.status.idle":"2023-06-04T05:32:55.730198Z","shell.execute_reply.started":"2023-06-04T05:32:55.722725Z","shell.execute_reply":"2023-06-04T05:32:55.729257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, valid_df = birds_stratified_split(df, 'primary_label', 0.2)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:56.029111Z","iopub.execute_input":"2023-06-04T05:32:56.029448Z","iopub.status.idle":"2023-06-04T05:32:56.088036Z","shell.execute_reply.started":"2023-06-04T05:32:56.029420Z","shell.execute_reply":"2023-06-04T05:32:56.087159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.primary_label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:56.314052Z","iopub.execute_input":"2023-06-04T05:32:56.314406Z","iopub.status.idle":"2023-06-04T05:32:56.327106Z","shell.execute_reply.started":"2023-06-04T05:32:56.314376Z","shell.execute_reply":"2023-06-04T05:32:56.325564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.primary_label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:56.515470Z","iopub.execute_input":"2023-06-04T05:32:56.516458Z","iopub.status.idle":"2023-06-04T05:32:56.529102Z","shell.execute_reply.started":"2023-06-04T05:32:56.516424Z","shell.execute_reply":"2023-06-04T05:32:56.528176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_df.primary_label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:56.621517Z","iopub.execute_input":"2023-06-04T05:32:56.622370Z","iopub.status.idle":"2023-06-04T05:32:56.631059Z","shell.execute_reply.started":"2023-06-04T05:32:56.622333Z","shell.execute_reply":"2023-06-04T05:32:56.630105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config: \n    sampling_rate = 32000\n    duration = 5\n    fmin = 0\n    fmax = None\n    audio_path = Path('../input/birdclef-2023/train_audio')\n    out_dir_train = Path('/specs/train')\n    \n    out_dir_valid = Path('/specs/valid')","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:56.814166Z","iopub.execute_input":"2023-06-04T05:32:56.814866Z","iopub.status.idle":"2023-06-04T05:32:56.820697Z","shell.execute_reply.started":"2023-06-04T05:32:56.814819Z","shell.execute_reply":"2023-06-04T05:32:56.819852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Config.out_dir_train.mkdir(exist_ok =True, parents= True)\nConfig.out_dir_valid.mkdir(exist_ok =True, parents= True)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:57.037101Z","iopub.execute_input":"2023-06-04T05:32:57.037754Z","iopub.status.idle":"2023-06-04T05:32:57.043855Z","shell.execute_reply.started":"2023-06-04T05:32:57.037721Z","shell.execute_reply":"2023-06-04T05:32:57.042926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_audio_info(filepath):\n    # get some properties from an audio file\n    with SoundFile(filepath) as f:\n        sr = f.samplerate\n        frames = f.frames\n        duration = float(frames)/sr\n    return {\n        'frames':frames,\n        'sr' : sr,\n        'duration': duration\n    }","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:57.190251Z","iopub.execute_input":"2023-06-04T05:32:57.190633Z","iopub.status.idle":"2023-06-04T05:32:57.196026Z","shell.execute_reply.started":"2023-06-04T05:32:57.190592Z","shell.execute_reply":"2023-06-04T05:32:57.195132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_path_df(df):\n    df['path'] = [str(Config.audio_path/filename) for filename in df.filename]\n    df= df.reset_index(drop=True)\n    pool = joblib.Parallel(2)\n    mapper = joblib.delayed(get_audio_info)\n    tasks = [mapper(filepath) for filepath in df.path]\n    df2 = pd.DataFrame(pool(tqdm(tasks))).reset_index(drop=True)\n    df = pd.concat([df, df2],axis=1).reset_index(drop=True)\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:57.561283Z","iopub.execute_input":"2023-06-04T05:32:57.561998Z","iopub.status.idle":"2023-06-04T05:32:57.569127Z","shell.execute_reply.started":"2023-06-04T05:32:57.561960Z","shell.execute_reply":"2023-06-04T05:32:57.567934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:57.751598Z","iopub.execute_input":"2023-06-04T05:32:57.752066Z","iopub.status.idle":"2023-06-04T05:32:57.758044Z","shell.execute_reply.started":"2023-06-04T05:32:57.752033Z","shell.execute_reply":"2023-06-04T05:32:57.757110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df= add_path_df(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:32:57.999408Z","iopub.execute_input":"2023-06-04T05:32:58.000247Z","iopub.status.idle":"2023-06-04T05:34:21.319767Z","shell.execute_reply.started":"2023-06-04T05:32:58.000207Z","shell.execute_reply":"2023-06-04T05:34:21.318681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_df= add_path_df(valid_df)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:34:21.321969Z","iopub.execute_input":"2023-06-04T05:34:21.322266Z","iopub.status.idle":"2023-06-04T05:34:41.825324Z","shell.execute_reply.started":"2023-06-04T05:34:21.322240Z","shell.execute_reply":"2023-06-04T05:34:41.824430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['duration'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:34:41.826890Z","iopub.execute_input":"2023-06-04T05:34:41.827546Z","iopub.status.idle":"2023-06-04T05:34:41.840757Z","shell.execute_reply.started":"2023-06-04T05:34:41.827512Z","shell.execute_reply":"2023-06-04T05:34:41.839775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compute_melspec(y, sr, n_mels, fmin, fmax):\n    melspec = lb.feature.melspectrogram(\n    y=y,\n    sr=sr,\n    n_mels= n_mels,\n    fmin=fmin,\n    fmax = fmax\n    )\n    melspec = lb.power_to_db(melspec).astype(np.float32)\n    return melspec","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:34:41.844094Z","iopub.execute_input":"2023-06-04T05:34:41.844367Z","iopub.status.idle":"2023-06-04T05:34:41.849205Z","shell.execute_reply.started":"2023-06-04T05:34:41.844344Z","shell.execute_reply":"2023-06-04T05:34:41.848282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mono_to_color(X, eps= 1e-6, mean=None, std=None):\n    mean= mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n    \n    _min, _max = X.min(), X.max()\n    \n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n\n    return V\n\ndef crop_or_pad(y, length, is_train=True, start=None):\n    if len(y) < length:\n        y = np.concatenate([y, np.zeros(length - len(y))])\n        \n        n_repeats = length // len(y)\n        epsilon = length % len(y)\n        \n        y = np.concatenate([y]*n_repeats + [y[:epsilon]])\n        \n    elif len(y) > length:\n        if not is_train:\n            start = start or 0\n        else:\n            start = start or np.random.randint(len(y) - length)\n\n        y = y[start:start + length]\n\n    return y","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:34:41.850795Z","iopub.execute_input":"2023-06-04T05:34:41.851456Z","iopub.status.idle":"2023-06-04T05:34:41.862130Z","shell.execute_reply.started":"2023-06-04T05:34:41.851425Z","shell.execute_reply":"2023-06-04T05:34:41.861141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AudioToImage:\n    def __init__(self, sr=Config.sampling_rate, n_mels= 128, fmin= Config.fmin, fmax=Config.fmax, duration = Config.duration, step=None, res_type=\"kaiser_fast\", resample= True, train=True):\n        \n        self.sr = sr\n        self.n_mels = n_mels\n        self.fmin = fmin\n        self.fmax = fmax or sr//2\n        self.duration = duration\n        self.audio_length = self.duration* self.sr\n        self.step = step or self.audio_length\n        self.res_type = res_type\n        self.resample = resample\n        self.train = train\n    \n    def audio_to_image(self, audio):\n        melspec= compute_melspec(audio, self.sr, self.n_mels, self.fmin, self.fmax)\n        image = mono_to_color(melspec)\n        return image\n    \n    def __call__(self, row, save=True):\n        \n        audio, orig_sr = sf.read(row.path, dtype=\"float32\")\n\n        if self.resample and orig_sr != self.sr:\n            audio = lb.resample(audio, orig_sr, self.sr, res_type=self.res_type)\n\n        audios = [audio[i:i+self.audio_length] for i in range(0, max(1, len(audio) - self.audio_length + 1), self.step)]\n        audios[-1] = crop_or_pad(audios[-1] , length=self.audio_length)\n        images = [self.audio_to_image(audio) for audio in audios]\n        images = np.stack(images)\n\n        if save:\n            if self.train:\n                path = Config.out_dir_train/f\"{row.filename}.npy\"\n            else:\n                path = Config.out_dir_valid/f\"{row.filename}.npy\"\n\n            path.parent.mkdir(exist_ok=True, parents=True)\n            np.save(str(path), images)\n        else:\n            return  row.filename, images\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:34:41.863467Z","iopub.execute_input":"2023-06-04T05:34:41.864085Z","iopub.status.idle":"2023-06-04T05:34:41.877190Z","shell.execute_reply.started":"2023-06-04T05:34:41.864056Z","shell.execute_reply":"2023-06-04T05:34:41.876255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:34:41.879926Z","iopub.execute_input":"2023-06-04T05:34:41.880300Z","iopub.status.idle":"2023-06-04T05:34:41.889460Z","shell.execute_reply.started":"2023-06-04T05:34:41.880270Z","shell.execute_reply":"2023-06-04T05:34:41.888674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_audios_as_images(df, train=True):\n    pool = joblib.Parallel(2)\n    converter = AudioToImage(step=int(Config.duration * 0.666 * Config.sampling_rate), train=train)\n    mapper= joblib.delayed(converter)\n    tasks = [mapper(row) for row in df.itertuples(False)]\n    pool(tqdm(tasks))","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:34:41.890815Z","iopub.execute_input":"2023-06-04T05:34:41.891483Z","iopub.status.idle":"2023-06-04T05:34:41.899577Z","shell.execute_reply.started":"2023-06-04T05:34:41.891454Z","shell.execute_reply":"2023-06-04T05:34:41.898482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_audios_as_images(train_df, train=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T05:34:41.900958Z","iopub.execute_input":"2023-06-04T05:34:41.901794Z","iopub.status.idle":"2023-06-04T06:06:53.354078Z","shell.execute_reply.started":"2023-06-04T05:34:41.901758Z","shell.execute_reply":"2023-06-04T06:06:53.353128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_audios_as_images(valid_df, train=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-04T06:06:53.357018Z","iopub.execute_input":"2023-06-04T06:06:53.357521Z","iopub.status.idle":"2023-06-04T06:15:05.370347Z","shell.execute_reply.started":"2023-06-04T06:06:53.357484Z","shell.execute_reply":"2023-06-04T06:15:05.369407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}