{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd, numpy as np, librosa as lb\nfrom pathlib import Path\nfrom IPython.display import Audio\nimport warnings\nwarnings.filterwarnings(\"ignore\") # Filter annoying librosa warnings","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DATA_ROOT = Path(\"../input/birdsong-recognition\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# The dataset","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"class BirdDataset:\n    \"\"\"Fastly load and sample the audio file in order to get same wave size for batch items.\n    \n    Parameters:\n    ----------\n    sr: int\n        The sample rate, defaults to librosa's 22050 Hz.\n        \n    nseconds: int\n        Targetted duration in seconds. The wave will right-padded if it lasts less than `nseconds`.\n        This is useful when batching.\n    \"\"\"\n    def __init__(self, sr = 22050, nseconds=5):\n        self.sr = sr\n        self.nseconds = nseconds\n        self.df = pd.read_csv(DATA_ROOT/\"train.csv\")\n        self.df.sort_values([\"ebird_code\", \"filename\"], inplace=True)\n        self.df.reset_index(drop=True, inplace=True)\n        \n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, i):\n        \"\"\"Load the ith wave file.\"\"\"\n        x = self.load(self.ith_file(i))\n        \n        return  self.sample(x),BIRDS_MAP[self.df.loc[i, \"ebird_code\"]]\n    \n    \n    def ith_file(self, i):\n        row = self.df.loc[i]\n        filename = \"{}/{}\".format(row[\"ebird_code\"], row[\"filename\"])\n        return filename\n    \n    def load(self, filename, res_type = 'kaiser_best'):\n        \"\"\"Load the wave file by name.\"\"\"\n        filename = DATA_ROOT/\"train_audio\"/filename\n        y, _ = lb.load(filename.as_posix(), sr = self.sr, res_type=res_type)\n        return y\n    \n    def display(self, audio):\n        return Audio(self.load(audio) if isinstance(audio, str) else audio, rate=self.sr)\n    \n    \n    def sample(self, x):\n        \"\"\"Sample the wave file in order to make it last exactly `self.nseconds`.\n        The wave will be right-padded if it's shorter.\n        \"\"\"\n        max_frames = self.nseconds*self.sr\n        nframes = len(x)\n        if max_frames < nframes:\n            offset = np.random.choice(nframes - max_frames)\n            x = x[offset:offset + max_frames]\n        elif max_frames>nframes:\n            x = np.concatenate([np.concatenate([x]*(max_frames//nframes)), x[-max_frames%nframes:]])\n        return x\n    \n    \n    def sample_on_load(self,filename, duration, res_type = 'kaiser_best'):\n        \"\"\"Fastly and directly sample the wave file on load time in order to make it last \n        exactly `self.nseconds`. The wave will be right-padded if it's shorter.\n        \"\"\"\n        target_duration = self.nseconds\n        filename = DATA_ROOT/\"train_audio\"/filename\n        \n        if duration > target_duration:\n            offset = np.random.choice(duration - target_duration)\n            x, sr = lb.load(filename, offset=offset, duration=target_duration, res_type= res_type, sr= self.sr)\n        else:\n            x, sr = lb.load(filename, sr=self.sr)\n            nframes = len(x)\n            target_frames = self.nseconds*self.sr\n            x = np.concatenate([np.concatenate([x]*(target_frames//nframes)), x[-target_frames%nframes:]])\n        return x","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":false},"cell_type":"markdown","source":"# Benchmark","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"bds = BirdDataset(nseconds=5) # Sample 5 seconds  lasting audio from wave files","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<h2 style=\"background-color:gray; color:white\"> A half-minute long wave file </h2>","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 105\nrow = bds.df.loc[i]\nbird_file = \"{}/{}\".format(row.ebird_code, row.filename)\nduration = row.duration\nbird_file,duration","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n\nx = bds.sample(bds.load(bird_file))\nprint(x.shape)\nbds.display(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n\nx = bds.sample_on_load(bird_file, duration)\nprint(x.shape)\nbds.display(x)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"> Sampling on laod is **4** times faster for a  **30 s** long wave file.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<h2 style=\"background-color:gray; color:white\"> A one-minute long wave file </h2>","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 8390\nrow = bds.df.loc[i]\nbird_file = \"{}/{}\".format(row.ebird_code, row.filename)\nduration = row.duration\nbird_file,duration","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n\nx = bds.sample(bds.load(bird_file))\nprint(x.shape)\nbds.display(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n\nx = bds.sample_on_load(bird_file, duration)\nprint(x.shape)\nbds.display(x)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"> Sampling on laod is **7** times faster for a  **60 s** long wave file. Indeed, longer the wave and higher the gain !","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<h2 style=\"background-color:gray; color:white\"> A 10 minutes long wave file </h2>","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 1341\nrow = bds.df.loc[i]\nbird_file = \"{}/{}\".format(row.ebird_code, row.filename)\nduration = row.duration\nbird_file,duration","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n\nx = bds.sample(bds.load(bird_file))\nprint(x.shape)\nbds.display(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n\nx = bds.sample_on_load(bird_file, duration)\nprint(x.shape)\nbds.display(x)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"> Sampling on laod is about **25** times faster for a  **10  minutes** long wave file. As there are some waves longer than 30 minutes, sampling on load is a really good hack.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<h2 style=\"text-align:center\">kkiller</h2>","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}