{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":44224,"databundleVersionId":5188730,"sourceType":"competition"}],"dockerImageVersionId":30407,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Notbeook to generate Mel Specs for BirdClef 2023 Competition\n#### Let's Think of a Better Split Method","metadata":{"papermill":{"duration":0.018531,"end_time":"2021-04-27T16:16:17.440560","exception":false,"start_time":"2021-04-27T16:16:17.422029","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import numpy as np\nimport librosa as lb\nimport librosa.display as lbd\nimport soundfile as sf\nfrom  soundfile import SoundFile\nimport pandas as pd\nfrom  IPython.display import Audio\nfrom pathlib import Path\n\nfrom matplotlib import pyplot as plt\n\nfrom tqdm.notebook import tqdm\nimport joblib, json, re\n\nfrom  sklearn.model_selection  import StratifiedKFold\ntqdm.pandas()","metadata":{"id":"2dt7oG43VAqc","papermill":{"duration":2.637633,"end_time":"2021-04-27T16:16:20.164867","exception":false,"start_time":"2021-04-27T16:16:17.527234","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-11T19:29:27.304937Z","iopub.execute_input":"2023-03-11T19:29:27.305406Z","iopub.status.idle":"2023-03-11T19:29:27.313651Z","shell.execute_reply.started":"2023-03-11T19:29:27.305363Z","shell.execute_reply":"2023-03-11T19:29:27.312248Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('../input/birdclef-2023/train_metadata.csv')\ndf['secondary_labels'] = df['secondary_labels'].apply(lambda x: re.findall(r\"'(\\w+)'\", x))\ndf['len_sec_labels'] = df['secondary_labels'].map(len)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:29:27.478621Z","iopub.execute_input":"2023-03-11T19:29:27.481420Z","iopub.status.idle":"2023-03-11T19:29:27.619537Z","shell.execute_reply.started":"2023-03-11T19:29:27.481361Z","shell.execute_reply":"2023-03-11T19:29:27.617834Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[df.len_sec_labels>0].sample(3)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:29:27.679984Z","iopub.execute_input":"2023-03-11T19:29:27.680388Z","iopub.status.idle":"2023-03-11T19:29:27.713241Z","shell.execute_reply.started":"2023-03-11T19:29:27.680354Z","shell.execute_reply":"2023-03-11T19:29:27.712132Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.primary_label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:29:28.917567Z","iopub.execute_input":"2023-03-11T19:29:28.917943Z","iopub.status.idle":"2023-03-11T19:29:28.931917Z","shell.execute_reply.started":"2023-03-11T19:29:28.917911Z","shell.execute_reply":"2023-03-11T19:29:28.931024Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Something has to be done for Birds with <= 1 samples.","metadata":{}},{"cell_type":"markdown","source":"## Also the fact that we have to perform inference in 2 hours w/ CPU, I think best solution is just to have single split rather than using multiple folds.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport pandas as pd\n\ndef birds_stratified_split(df, target_col, test_size=0.2):\n    class_counts = df[target_col].value_counts()\n    low_count_classes = class_counts[class_counts < 2].index.tolist() ### Birds with single counts\n\n    df['train'] = df[target_col].isin(low_count_classes)\n\n    train_df, val_df = train_test_split(df[~df['train']], test_size=test_size, stratify=df[~df['train']][target_col], random_state=42)\n\n    train_df = pd.concat([train_df, df[df['train']]], axis=0).reset_index(drop=True)\n\n    # Remove the 'valid' column\n    train_df.drop('train', axis=1, inplace=True)\n    val_df.drop('train', axis=1, inplace=True)\n\n    return train_df, val_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:29:29.226601Z","iopub.execute_input":"2023-03-11T19:29:29.227753Z","iopub.status.idle":"2023-03-11T19:29:29.234693Z","shell.execute_reply.started":"2023-03-11T19:29:29.227698Z","shell.execute_reply":"2023-03-11T19:29:29.233791Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df, valid_df = birds_stratified_split(df, 'primary_label', 0.2)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:29:29.644077Z","iopub.execute_input":"2023-03-11T19:29:29.644783Z","iopub.status.idle":"2023-03-11T19:29:29.690912Z","shell.execute_reply.started":"2023-03-11T19:29:29.644734Z","shell.execute_reply":"2023-03-11T19:29:29.689264Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.primary_label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:29:30.676332Z","iopub.execute_input":"2023-03-11T19:29:30.676690Z","iopub.status.idle":"2023-03-11T19:29:30.686035Z","shell.execute_reply.started":"2023-03-11T19:29:30.676657Z","shell.execute_reply":"2023-03-11T19:29:30.685265Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.primary_label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:29:30.851024Z","iopub.execute_input":"2023-03-11T19:29:30.852146Z","iopub.status.idle":"2023-03-11T19:29:30.862462Z","shell.execute_reply.started":"2023-03-11T19:29:30.852098Z","shell.execute_reply":"2023-03-11T19:29:30.861304Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"valid_df.primary_label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:29:31.588444Z","iopub.execute_input":"2023-03-11T19:29:31.588801Z","iopub.status.idle":"2023-03-11T19:29:31.599234Z","shell.execute_reply.started":"2023-03-11T19:29:31.588764Z","shell.execute_reply":"2023-03-11T19:29:31.598285Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    sampling_rate = 32000\n    duration = 5 \n    fmin = 0\n    fmax = None\n    audios_path = Path(\"/kaggle/input/birdclef-2023/train_audio\")\n    out_dir_train = Path(\"specs/train\") \n    \n    out_dir_valid = Path(\"specs/valid\") \n","metadata":{"papermill":{"duration":0.027531,"end_time":"2021-04-27T16:16:20.271347","exception":false,"start_time":"2021-04-27T16:16:20.243816","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-11T19:29:32.122901Z","iopub.execute_input":"2023-03-11T19:29:32.123570Z","iopub.status.idle":"2023-03-11T19:29:32.128697Z","shell.execute_reply.started":"2023-03-11T19:29:32.123535Z","shell.execute_reply":"2023-03-11T19:29:32.127724Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Config.out_dir_train.mkdir(exist_ok=True, parents=True)\nConfig.out_dir_valid.mkdir(exist_ok=True, parents=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:29:32.468269Z","iopub.execute_input":"2023-03-11T19:29:32.468621Z","iopub.status.idle":"2023-03-11T19:29:32.475110Z","shell.execute_reply.started":"2023-03-11T19:29:32.468591Z","shell.execute_reply":"2023-03-11T19:29:32.473579Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_audio_info(filepath):\n    \"\"\"Get some properties from  an audio file\"\"\"\n    with SoundFile(filepath) as f:\n        sr = f.samplerate\n        frames = f.frames\n        duration = float(frames)/sr\n    return {\"frames\": frames, \"sr\": sr, \"duration\": duration}","metadata":{"papermill":{"duration":0.026875,"end_time":"2021-04-27T16:16:20.351194","exception":false,"start_time":"2021-04-27T16:16:20.324319","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-11T19:29:33.039561Z","iopub.execute_input":"2023-03-11T19:29:33.039900Z","iopub.status.idle":"2023-03-11T19:29:33.045013Z","shell.execute_reply.started":"2023-03-11T19:29:33.039871Z","shell.execute_reply":"2023-03-11T19:29:33.044101Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def add_path_df(df):\n    \n    df[\"path\"] = [str(Config.audios_path/filename) for filename in df.filename]\n    df = df.reset_index(drop=True)\n    pool = joblib.Parallel(2)\n    mapper = joblib.delayed(get_audio_info)\n    tasks = [mapper(filepath) for filepath in df.path]\n    df2 =  pd.DataFrame(pool(tqdm(tasks))).reset_index(drop=True)\n    df = pd.concat([df,df2], axis=1).reset_index(drop=True)\n\n    return df","metadata":{"id":"Kmh6xx5_NCjJ","outputId":"ad61f09f-6f0e-4204-c658-21112e051785","papermill":{"duration":0.031496,"end_time":"2021-04-27T16:16:20.401055","exception":false,"start_time":"2021-04-27T16:16:20.369559","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-11T19:29:34.244901Z","iopub.execute_input":"2023-03-11T19:29:34.245279Z","iopub.status.idle":"2023-03-11T19:29:34.253031Z","shell.execute_reply.started":"2023-03-11T19:29:34.245249Z","shell.execute_reply":"2023-03-11T19:29:34.251598Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tqdm.pandas()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:29:34.674989Z","iopub.execute_input":"2023-03-11T19:29:34.675377Z","iopub.status.idle":"2023-03-11T19:29:34.680554Z","shell.execute_reply.started":"2023-03-11T19:29:34.675345Z","shell.execute_reply":"2023-03-11T19:29:34.679388Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = add_path_df(train_df)","metadata":{"papermill":{"duration":0.017674,"end_time":"2021-04-27T16:16:20.436943","exception":false,"start_time":"2021-04-27T16:16:20.419269","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-11T19:29:35.298345Z","iopub.execute_input":"2023-03-11T19:29:35.299285Z","iopub.status.idle":"2023-03-11T19:30:51.763088Z","shell.execute_reply.started":"2023-03-11T19:29:35.299250Z","shell.execute_reply":"2023-03-11T19:30:51.761767Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"valid_df = add_path_df(valid_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:30:51.765780Z","iopub.execute_input":"2023-03-11T19:30:51.766154Z","iopub.status.idle":"2023-03-11T19:31:10.706013Z","shell.execute_reply.started":"2023-03-11T19:30:51.766123Z","shell.execute_reply":"2023-03-11T19:31:10.705038Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.to_csv('train.csv',index=False)\nvalid_df.to_csv('valid.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T19:32:04.193086Z","iopub.execute_input":"2023-03-11T19:32:04.193484Z","iopub.status.idle":"2023-03-11T19:32:04.353322Z","shell.execute_reply.started":"2023-03-11T19:32:04.193453Z","shell.execute_reply":"2023-03-11T19:32:04.352411Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[\"duration\"].describe()","metadata":{"papermill":{"duration":0.257618,"end_time":"2021-04-27T16:18:04.144335","exception":false,"start_time":"2021-04-27T16:18:03.886717","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-11T16:56:32.414546Z","iopub.execute_input":"2023-03-11T16:56:32.415212Z","iopub.status.idle":"2023-03-11T16:56:32.427404Z","shell.execute_reply.started":"2023-03-11T16:56:32.415175Z","shell.execute_reply":"2023-03-11T16:56:32.426232Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def compute_melspec(y, sr, n_mels, fmin, fmax):\n    \"\"\"\n    Computes a mel-spectrogram and puts it at decibel scale\n    Arguments:\n        y {np array} -- signal\n        params {AudioParams} -- Parameters to use for the spectrogram. Expected to have the attributes sr, n_mels, f_min, f_max\n    Returns:\n        np array -- Mel-spectrogram\n    \"\"\"\n    melspec = lb.feature.melspectrogram(\n        y=y, sr=sr, n_mels=n_mels, fmin=fmin, fmax=fmax,\n    )\n\n    melspec = lb.power_to_db(melspec).astype(np.float32)\n    return melspec","metadata":{"execution":{"iopub.status.busy":"2023-03-11T16:56:32.429165Z","iopub.execute_input":"2023-03-11T16:56:32.429884Z","iopub.status.idle":"2023-03-11T16:56:32.435154Z","shell.execute_reply.started":"2023-03-11T16:56:32.429848Z","shell.execute_reply":"2023-03-11T16:56:32.434139Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def mono_to_color(X, eps=1e-6, mean=None, std=None):\n    mean = mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n    \n    _min, _max = X.min(), X.max()\n\n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n\n    return V\n\ndef crop_or_pad(y, length, is_train=True, start=None):\n    if len(y) < length:\n        y = np.concatenate([y, np.zeros(length - len(y))])\n        \n        n_repeats = length // len(y)\n        epsilon = length % len(y)\n        \n        y = np.concatenate([y]*n_repeats + [y[:epsilon]])\n        \n    elif len(y) > length:\n        if not is_train:\n            start = start or 0\n        else:\n            start = start or np.random.randint(len(y) - length)\n\n        y = y[start:start + length]\n\n    return y","metadata":{"id":"-Nlw4E5UVAqi","papermill":{"duration":0.036932,"end_time":"2021-04-27T16:18:04.507902","exception":false,"start_time":"2021-04-27T16:18:04.470970","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-11T16:56:32.436772Z","iopub.execute_input":"2023-03-11T16:56:32.437285Z","iopub.status.idle":"2023-03-11T16:56:32.448069Z","shell.execute_reply.started":"2023-03-11T16:56:32.437250Z","shell.execute_reply":"2023-03-11T16:56:32.447067Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AudioToImage:\n    def __init__(self, sr=Config.sampling_rate, n_mels=128, fmin=Config.fmin, fmax=Config.fmax, duration=Config.duration, step=None, res_type=\"kaiser_fast\", resample=True, train = True):\n\n        self.sr = sr\n        self.n_mels = n_mels\n        self.fmin = fmin\n        self.fmax = fmax or self.sr//2\n\n        self.duration = duration\n        self.audio_length = self.duration*self.sr\n        self.step = step or self.audio_length\n        \n        self.res_type = res_type\n        self.resample = resample\n\n        self.train = train\n    def audio_to_image(self, audio):\n        melspec = compute_melspec(audio, self.sr, self.n_mels, self.fmin, self.fmax ) \n        image = mono_to_color(melspec)\n#         compute_melspec(y, sr, n_mels, fmin, fmax)\n        return image\n\n    def __call__(self, row, save=True):\n\n      audio, orig_sr = sf.read(row.path, dtype=\"float32\")\n\n      if self.resample and orig_sr != self.sr:\n        audio = lb.resample(audio, orig_sr, self.sr, res_type=self.res_type)\n        \n      audios = [audio[i:i+self.audio_length] for i in range(0, max(1, len(audio) - self.audio_length + 1), self.step)]\n      audios[-1] = crop_or_pad(audios[-1] , length=self.audio_length)\n      images = [self.audio_to_image(audio) for audio in audios]\n      images = np.stack(images)\n        \n      if save:\n        if self.train:\n            path = Config.out_dir_train/f\"{row.filename}.npy\"\n        else:\n            path = Config.out_dir_valid/f\"{row.filename}.npy\"\n            \n        path.parent.mkdir(exist_ok=True, parents=True)\n        np.save(str(path), images)\n      else:\n        return  row.filename, images","metadata":{"papermill":{"duration":0.039542,"end_time":"2021-04-27T16:18:04.570629","exception":false,"start_time":"2021-04-27T16:18:04.531087","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-11T16:56:32.449684Z","iopub.execute_input":"2023-03-11T16:56:32.450349Z","iopub.status.idle":"2023-03-11T16:56:32.464927Z","shell.execute_reply.started":"2023-03-11T16:56:32.450301Z","shell.execute_reply":"2023-03-11T16:56:32.464080Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T16:56:32.466483Z","iopub.execute_input":"2023-03-11T16:56:32.467187Z","iopub.status.idle":"2023-03-11T16:56:32.480139Z","shell.execute_reply.started":"2023-03-11T16:56:32.467127Z","shell.execute_reply":"2023-03-11T16:56:32.479068Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_audios_as_images(df, train = True):\n    pool = joblib.Parallel(2)\n    \n    converter = AudioToImage(step=int(Config.duration*0.666*Config.sampling_rate),train=train)\n    mapper = joblib.delayed(converter)\n    tasks = [mapper(row) for row in df.itertuples(False)]\n    pool(tqdm(tasks))","metadata":{"papermill":{"duration":0.032362,"end_time":"2021-04-27T16:18:04.626619","exception":false,"start_time":"2021-04-27T16:18:04.594257","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-11T16:56:32.481937Z","iopub.execute_input":"2023-03-11T16:56:32.482398Z","iopub.status.idle":"2023-03-11T16:56:32.489931Z","shell.execute_reply.started":"2023-03-11T16:56:32.482352Z","shell.execute_reply":"2023-03-11T16:56:32.489013Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"get_audios_as_images(train_df, train = True)\n","metadata":{"papermill":{"duration":4117.309995,"end_time":"2021-04-27T17:26:41.959921","exception":false,"start_time":"2021-04-27T16:18:04.649926","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-11T16:56:32.491613Z","iopub.execute_input":"2023-03-11T16:56:32.492232Z","iopub.status.idle":"2023-03-11T17:46:24.116092Z","shell.execute_reply.started":"2023-03-11T16:56:32.492197Z","shell.execute_reply":"2023-03-11T17:46:24.113697Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"get_audios_as_images(valid_df, train = False)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:24.119811Z","iopub.execute_input":"2023-03-11T17:46:24.121148Z","iopub.status.idle":"2023-03-11T17:59:00.753417Z","shell.execute_reply.started":"2023-03-11T17:46:24.121076Z","shell.execute_reply":"2023-03-11T17:59:00.751498Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#","metadata":{"papermill":{"duration":0.027772,"end_time":"2021-04-27T17:26:42.374518","exception":false,"start_time":"2021-04-27T17:26:42.346746","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-11T18:02:35.428219Z","iopub.execute_input":"2023-03-11T18:02:35.429291Z","iopub.status.idle":"2023-03-11T18:02:35.435696Z","shell.execute_reply.started":"2023-03-11T18:02:35.429228Z","shell.execute_reply":"2023-03-11T18:02:35.434458Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}