{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        break\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_path = os.path.join('/', 'kaggle', 'input', 'birdsong-recognition')\nfnames = os.listdir(data_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fpaths = [os.path.join(data_path, fname) for fname in os.listdir(data_path)]\nfname_dict = dict(zip(fnames, fpaths)) # convenient mapping between filenames and absolute path","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data = pd.read_csv(fname_dict['train.csv'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"distilled_data = train_data[(train_data.rating >= 4.0)\n                            & (train_data.secondary_labels == \"[]\")\n                            & (train_data.background.isna())\n                            & (train_data.type == \"song\")]\nprint(f\"size full dataset: {len(train_data)}\")\nprint(f\"size distilled dataset: {len(distilled_data)}, keeping {round(len(distilled_data)/len(train_data)*100, 2)} %\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"distilled_data.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"---","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import librosa\nimport warnings","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from typing import NamedTuple, List","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class Spectrogram(NamedTuple):\n    ebird_code: str\n    mel: np.array\n    fpath: str","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class BirdSpecs(NamedTuple):\n    ebird_code: str\n    spectrograms: List[Spectrogram]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def generate_mel_spectrogram(fpath):\n    with warnings.catch_warnings():\n        warnings.simplefilter('ignore')\n        y, sr = librosa.load(fpath)\n        return librosa.feature.melspectrogram(y=y, sr=sr)\n    \n    return None","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_audio_path = os.path.join(data_path, 'train_audio')\ndef get_fpath(ebird_code, filename):\n    return os.path.join(train_audio_path, ebird_code, filename)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"birds = dict()\nfor ix, row in distilled_data.reset_index().iterrows():\n    print(f\"Row: {ix}\", end=\"\\r\")\n    if row.ebird_code in birds:\n        bird = birds[row.ebird_code] \n    else:\n        bird = BirdSpecs(ebird_code=row.ebird_code,\n                         spectrograms=list())\n        \n    fpath = get_fpath(row.ebird_code, row.filename)\n    try:\n        mel = generate_mel_spectrogram(fpath)\n    \n    except BaseException:\n        mel = None\n    spec = Spectrogram(ebird_code=bird.ebird_code,\n                       mel=mel,\n                       fpath=fpath)\n    bird.spectrograms.append(spec)\n    birds[row.ebird_code] = bird","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"birds","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}