{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Nemo for speaker recognition \nfind nemo here - https://docs.nvidia.com/deeplearning/nemo/user-guide/docs/en/stable/index.html\n\ntryin nemo by nvedia to predict the animal species nemo has pretrained models and is easy to transfer learn things. in this notebook I am trying to use nemo for detect the animal species mostly nemo has good performance In asr."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-output":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.patches as mpl_patches\nimport matplotlib.pylab as pylab\nparams = {'legend.fontsize': 'x-large',\n          'figure.figsize': (15, 8),\n         'axes.labelsize': 'x-large',\n         'axes.titlesize':'x-large',\n         'xtick.labelsize':'x-large',\n         'ytick.labelsize':'x-large',\n         'text.color' : \"green\",\n          \n         }\npylab.rcParams.update(params)\n\nimport torch\nimport torchaudio\nfrom torch.utils.data.dataset import Dataset\nfrom torch.utils.data.dataloader import DataLoader\nfrom torchvision.models import resnet34\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport librosa\nfrom tqdm.notebook import tqdm\n\nfrom colorama import Fore, Back, Style\ny_ = Fore.YELLOW\nr_ = Fore.RED\ng_ = Fore.GREEN\nb_ = Fore.BLUE\nm_ = Fore.MAGENTA\nc_ = Fore.CYAN\nsr_ = Style.RESET_ALL\n\nimport warnings\nwarnings.filterwarnings('ignore')\nplt.style.use('fivethirtyeight')\nimport IPython.display as ipd\nfrom IPython.display import display, HTML","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_path = Path(\"../input/rainforest-wav-format/train_wav_format/\")\ntest_path = Path(\"../input/rainforest-wav-format/test_wav_format/\")\ntrain_recs = list(train_path.glob(\"*.wav\"))\ntest_recs = list(test_path.glob(\"*.wav\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/rfcx-species-audio-detection/train_tp.csv')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# lets make a manifest for nemo training\n# {\"audio_filepath\": \"/content/data/an4/wav/an4_clstk/mjda/cen6-mjda-b.wav\", \"duration\": 2.0, \"label\": \"mjda\"}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import librosa\nimport IPython.display as ipd\n\n# Load and listen to a random audio file\nexample_file = '../input/rainforest-wav-format/train_wav_format/0099c367b.wav'\naudio, sample_rate = librosa.load(example_file)\n#lets hear the audio\nipd.Audio(example_file, rate=sample_rate)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# train test split\nfrom sklearn.model_selection import train_test_split\nX = train.drop(['species_id'] , axis = 1)\ny = train['species_id'] \nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42 , stratify = y)\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_manifest_data = X_train.copy()\ntrain_manifest_data['species_id'] = y_train\n\ntest_manifest_data = X_test.copy()\ntest_manifest_data['species_id'] = y_test\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_manifest_data[['recording_id' , 'species_id']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''train manifest lets first pass the whole audio recording \nwhich doesnt seem right but we can improve once we setup the whole pipeline'''\nfrom scipy.io import wavfile\nimport json\nmanifest_path = 'train_manifest.json'\nos.mkdir('train_trim') \nos.mkdir('test_trim') \n\nwith open(manifest_path, 'w') as fout:\n    for i,r,s,tmax,tmin in train_manifest_data[['recording_id' , 'species_id' ,'t_max','t_min']].itertuples():\n        audio_path = str(train_path)+'/'+r+'.wav'\n        \n        sampleRate, waveData = wavfile.read( audio_path )\n        startSample = int( tmin * sampleRate )\n        endSample = int( tmax * sampleRate )\n        wavfile.write( 'train_trim/'+r+\"_\"+str(i)+'.wav' , sampleRate, waveData[startSample:endSample])\n        # converting to 16khz from 48khz\n        y, sr = torchaudio.load('train_trim/'+r+\"_\"+str(i)+'.wav')\n        if sr != 16000:\n            resampler = torchaudio.transforms.Resample(sr, 16000)\n            y_resampled = resampler(y)\n        torchaudio.save('train_trim/'+r+\"_\"+str(i)+'.wav' ,y_resampled,16000)\n\n        duration = librosa.core.get_duration(filename='train_trim/'+r+\"_\"+str(i)+'.wav' )\n        metadata = {\n                    \"audio_filepath\": 'train_trim/'+r+\"_\"+str(i)+'.wav',\n                    \"duration\": duration,\n                    \"label\": s\n                }\n        json.dump(metadata, fout)\n        fout.write('\\n')\n    \n    \n    \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# test manifest\nmanifest_path = 'test_manifest.json'\n\nwith open(manifest_path, 'w') as fout:\n    for i,r,s,tmax,tmin in test_manifest_data[['recording_id' , 'species_id' ,'t_max','t_min']].itertuples():\n        audio_path = str(train_path)+'/'+r+'.wav'\n        \n        sampleRate, waveData = wavfile.read( audio_path )\n        startSample = int( tmin * sampleRate )\n        endSample = int( tmax * sampleRate )\n        wavfile.write( 'test_trim/'+r+\"_\"+str(i)+'.wav' , sampleRate, waveData[startSample:endSample])\n        # converting to 16khz from 48khz\n        y, sr = torchaudio.load('test_trim/'+r+\"_\"+str(i)+'.wav')\n        if sr != 16000:\n            resampler = torchaudio.transforms.Resample(sr, 16000)\n            y_resampled = resampler(y)\n        torchaudio.save('test_trim/'+r+\"_\"+str(i)+'.wav' ,y_resampled,16000)\n\n        duration = librosa.core.get_duration(filename='test_trim/'+r+\"_\"+str(i)+'.wav' )\n        metadata = {\n                    \"audio_filepath\": 'test_trim/'+r+\"_\"+str(i)+'.wav',\n                    \"duration\": duration,\n                    \"label\": s\n                }\n        json.dump(metadata, fout)\n        fout.write('\\n')\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# setup nemo\nBRANCH = 'r1.0.0b3'\n!python -m pip install git+https://github.com/NVIDIA/NeMo.git@$BRANCH#egg=nemo_toolkit[asr]\nNEMO_ROOT = os.getcwd()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import nemo\n# NeMo's ASR collection - this collections contains complete ASR models and\n# building blocks (modules) for ASR\nimport nemo.collections.asr as nemo_asr\nfrom omegaconf import OmegaConf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# This line will print the entire config of sample SpeakerNet model\n\n!mkdir conf \n!wget -P conf https://raw.githubusercontent.com/NVIDIA/NeMo/$BRANCH/examples/speaker_recognition/conf/SpeakerNet_recognition_3x2x512.yaml\nMODEL_CONFIG = os.path.join(NEMO_ROOT,'conf/SpeakerNet_recognition_3x2x512.yaml')\nconfig = OmegaConf.load(MODEL_CONFIG)\nprint(OmegaConf.to_yaml(config))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# setup manifest\ntrain_manifest = 'train_manifest.json'\nvalidation_manifest = 'test_manifest.json'\ntest_manifest = 'test_manifest.json'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"config.model.train_ds.manifest_filepath = train_manifest\nconfig.model.validation_ds.manifest_filepath = validation_manifest\nconfig.model.test_ds.manifest_filepath = validation_manifest","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"config.model.decoder.num_classes = 24","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import torch\nimport pytorch_lightning as pl","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Trainer config - \\n\")\nprint(OmegaConf.to_yaml(config.trainer))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"torch.cuda.is_available()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let us modify some trainer configs for this demo\n# Checks if we have GPU available and uses it\ncuda = 1 if torch.cuda.is_available() else 0\nconfig.trainer.gpus = cuda\n\n\nconfig.trainer.max_epochs = 15\n\n# Remove distributed training flags\nconfig.trainer.accelerator = None","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainer = pl.Trainer(**config.trainer)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from nemo.utils.exp_manager import exp_manager\nlog_dir = exp_manager(trainer, config.get(\"exp_manager\", None))\n# The log_dir provides a path to the current logging directory for easy access\nprint(log_dir)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"speaker_model = nemo_asr.models.EncDecSpeakerLabelModel(cfg=config.model, trainer=trainer)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainer.fit(speaker_model)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"validation loss is all over the place lets try to fix it"},{"metadata":{"trusted":true},"cell_type":"code","source":"# lets optimize it\ntrainer.test(speaker_model, ckpt_path=None)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let us list all the checkpoints we have\nfrom glob import glob\ncheckpoint_dir = os.path.join(log_dir, 'checkpoints')\ncheckpoint_paths = list(glob(os.path.join(checkpoint_dir, \"*.ckpt\")))\ncheckpoint_paths","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"final_checkpoint = list(filter(lambda x: \"-last.ckpt\" in x, checkpoint_paths))[0]\nprint(final_checkpoint)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Restoring from a PyTorch Lightning checkpoint\nrestored_model = nemo_asr.models.EncDecSpeakerLabelModel.load_from_checkpoint(final_checkpoint)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# restored_model.forward(input_signal = 'test_trim/11bafff5d_74.wav',input_signal_length = 2.352 )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# restored_model.forward(input_signal =  y ,input_signal_length = (2.352) )\n# (input_signal = 'test_trim/11bafff5d_74.wav',input_signal_length= 2.352)\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# lets get the nemo file\nrestored_model.save_to(os.path.join(log_dir, '..',\"SpeakerNet.nemo\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"verification_model = nemo_asr.models.ExtractSpeakerEmbeddingsModel.restore_from(os.path.join(log_dir, '..', 'SpeakerNet.nemo'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Now, we need to pass the necessary manifest_filepath and params to set up the data loader for extracting embeddings\n# lets check first five cases for demo\n!head -5 {validation_manifest} > embeddings_manifest.json\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"config.model.train_ds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_config = OmegaConf.create(dict(\n    manifest_filepath = os.path.join(NEMO_ROOT,'embeddings_manifest.json'),\n    sample_rate = 16000,\n    labels = None,\n    batch_size = 1,\n    shuffle = False,\n    time_length = 8,\n    embedding_dir='./'\n))\nprint(OmegaConf.to_yaml(test_config))\nverification_model.setup_test_data(test_config)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainer = pl.Trainer(gpus=cuda,accelerator=None)\ntrainer.test(verification_model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ls embeddings/\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}