{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.patches as mpl_patches\nimport matplotlib.pylab as pylab\nparams = {'legend.fontsize': 'x-large',\n          'figure.figsize': (15, 8),\n         'axes.labelsize': 'x-large',\n         'axes.titlesize':'x-large',\n         'xtick.labelsize':'x-large',\n         'ytick.labelsize':'x-large',\n         'text.color' : \"green\",\n          \n         }\npylab.rcParams.update(params)\n\nimport torch\nimport torchaudio\nfrom torch.utils.data.dataset import Dataset\nfrom torch.utils.data.dataloader import DataLoader\nfrom torchvision.models import resnet34\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport librosa\nfrom tqdm.notebook import tqdm\n\nfrom colorama import Fore, Back, Style\ny_ = Fore.YELLOW\nr_ = Fore.RED\ng_ = Fore.GREEN\nb_ = Fore.BLUE\nm_ = Fore.MAGENTA\nc_ = Fore.CYAN\nsr_ = Style.RESET_ALL\n\nimport warnings\nwarnings.filterwarnings('ignore')\nplt.style.use('fivethirtyeight')\nimport IPython.display as ipd\nfrom IPython.display import display, HTML","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_path = Path(\"../input/rfcx-species-audio-detection/train/\")\ntest_path = Path(\"../input/rfcx-species-audio-detection/test/\")\ntrain_recs = list(train_path.glob(\"*.flac\"))\ntest_recs = list(test_path.glob(\"*.flac\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/rfcx-species-audio-detection/train_tp.csv')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# rec_dict = {}\n\n# for spid in train[\"species_id\"].unique().tolist():\n#     rec_id = train.loc[np.where(train[\"species_id\"] == spid)][\"recording_id\"].tolist()[0]\n# #     print(f\"{spid} : {rec_id}\")\n#     rec_dict[spid] = rec_id\n#     fig, ax = plt.subplots()\n#     rec_path = Path(train_path) / f\"{rec_id}.flac\"\n#     wf, sr = torchaudio.load(rec_path)\n#     labels = []\n#     handles = [mpl_patches.Rectangle((0, 0), 1, 1, fc=\"white\", ec=\"white\", lw=0, alpha=0)] * 4 \n#     labels.append(f\"label = {spid}\")\n#     labels.append(f\"min of waveform: {np.around(wf.min(), 5)}\")\n#     labels.append(f\"max of waveform: {np.around(wf.max(), 5)}\")\n#     labels.append(f\"mean of waveform: {np.around(wf.mean(), 5)}\")\n#     ax.legend(handles, labels, loc='best', fontsize='large', \n#           fancybox=True, framealpha=0.7, \n#           handlelength=0, handletextpad=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# display(HTML(f'<span style=\"color:green\"> <h3>Sound of each species </h3></span>'))\n# def get_species_audio():\n#     \"\"\"\n#     get audio image for each of the species\n#     \"\"\"\n#     for spid in train[\"species_id\"].unique().tolist():\n#         rec_id = train.loc[np.where(train[\"species_id\"] == spid)][\"recording_id\"].tolist()[0]\n#     #     print(f\"{spid} : {rec_id}\")\n#         rec_dict[spid] = rec_id\n#         rec_path = Path(train_path) / f\"{rec_id}.flac\"\n#         wf, sr = torchaudio.load(rec_path)\n#         display(HTML(f'<span style=\"color:green\"> species id: {spid} </span>'))\n#         ipd.display(ipd.Audio(data=wf, rate=sr))\n# get_species_audio()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# def get_recording():\n#     ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from pathlib import PurePath\nfrom pydub import AudioSegment","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from pathlib import PurePath\nfrom pydub import AudioSegment\nimport os\nos.mkdir('train_wav_format') \nos.mkdir('test_wav_format') \n\n# file_path = PurePath(\"/kaggle/input/rfcx-species-audio-detection/test/73022e436.flac\")\n\n# flac_tmp_audio_data = AudioSegment.from_file(file_path, file_path.suffix[1:])\n\n# flac_tmp_audio_data.export('train_wav/'+file_path.name.replace(file_path.suffix, \"\") + \".wav\", format=\"wav\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(test_recs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_files = list(train.recording_id.unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train_files)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#get wav files from flac files\nfor file_path in test_recs:\n    flac_tmp_audio_data = AudioSegment.from_file(file_path, file_path.suffix[1:])\n\n    flac_tmp_audio_data.export('test_wav_format/'+file_path.name.replace(file_path.suffix, \"\") + \".wav\", format=\"wav\")\n    \n    \n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in train_files:\n    i=i+'.flac'\n    file_path = Path(train_path,i)\n    flac_tmp_audio_data = AudioSegment.from_file(file_path, file_path.suffix[1:])\n\n    flac_tmp_audio_data.export('train_wav_format/'+file_path.name.replace(file_path.suffix, \"\") + \".wav\", format=\"wav\")\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}