{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport pickle \nimport tensorflow as tf \nimport librosa\n\nfrom tensorflow_addons.metrics import F1Score","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:47:29.862992Z","iopub.execute_input":"2021-06-01T07:47:29.863567Z","iopub.status.idle":"2021-06-01T07:47:37.706383Z","shell.execute_reply.started":"2021-06-01T07:47:29.863486Z","shell.execute_reply":"2021-06-01T07:47:37.705363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Important Params","metadata":{}},{"cell_type":"code","source":"RANDOM_SEED = 1337\nSAMPLE_RATE = 32000\nSIGNAL_LENGTH = 5 # seconds\nSPEC_SHAPE = (48, 128) # height x width\nFMIN = 500\nFMAX = 12500\n","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:47:37.707947Z","iopub.execute_input":"2021-06-01T07:47:37.708197Z","iopub.status.idle":"2021-06-01T07:47:37.712567Z","shell.execute_reply.started":"2021-06-01T07:47:37.708167Z","shell.execute_reply":"2021-06-01T07:47:37.711674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test data ","metadata":{}},{"cell_type":"code","source":"test_dir='../input/birdclef-2021/test_soundscapes'\ntest=pd.read_csv('../input/birdclef-2021/test.csv')\nsample_sub=pd.read_csv('../input/birdclef-2021/sample_submission.csv')\n\ntest","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:47:37.714584Z","iopub.execute_input":"2021-06-01T07:47:37.714837Z","iopub.status.idle":"2021-06-01T07:47:37.766524Z","shell.execute_reply.started":"2021-06-01T07:47:37.714811Z","shell.execute_reply":"2021-06-01T07:47:37.765585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:47:37.767794Z","iopub.execute_input":"2021-06-01T07:47:37.768074Z","iopub.status.idle":"2021-06-01T07:47:37.77822Z","shell.execute_reply.started":"2021-06-01T07:47:37.768047Z","shell.execute_reply":"2021-06-01T07:47:37.777045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#loading presaved pickle\ndef load_pickle(path):\n    with open(path,'rb') as f:\n        file=pickle.load(f)\n        \n    return file\n","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:47:37.779613Z","iopub.execute_input":"2021-06-01T07:47:37.780067Z","iopub.status.idle":"2021-06-01T07:47:37.786054Z","shell.execute_reply.started":"2021-06-01T07:47:37.780027Z","shell.execute_reply":"2021-06-01T07:47:37.785287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Labels from model training**","metadata":{}},{"cell_type":"code","source":"#labels:\n\nLABELS=load_pickle('../input/birdclef2021-model-training/LABELS.pkl')\n\nf1_score=F1Score(num_classes=len(LABELS),average='macro',name='f1_score')","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:47:37.787316Z","iopub.execute_input":"2021-06-01T07:47:37.787683Z","iopub.status.idle":"2021-06-01T07:47:37.842035Z","shell.execute_reply.started":"2021-06-01T07:47:37.787646Z","shell.execute_reply":"2021-06-01T07:47:37.841235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading pretrained models","metadata":{}},{"cell_type":"code","source":"#loading pretrained models:\n\nmodel1=tf.keras.models.load_model('../input/birdclef2021-model-training/best_model.h5')\nmodel2=tf.keras.models.load_model('../input/birdclef2021-model-training/best_model2.h5')\n","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:47:37.844095Z","iopub.execute_input":"2021-06-01T07:47:37.84435Z","iopub.status.idle":"2021-06-01T07:47:38.644806Z","shell.execute_reply.started":"2021-06-01T07:47:37.844326Z","shell.execute_reply":"2021-06-01T07:47:38.644044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Test audio data**","metadata":{}},{"cell_type":"code","source":"def list_files(path):\n    '''get test sound files'''\n    return [os.path.join(path, f) for f in os.listdir(path) if f.rsplit('.', 1)[-1] in ['ogg']]\n\ntest_audio=list_files(test_dir)\n\n# test files are hidden,  hence checking on train_soundscapes\nif len(test_audio) == 0:\n    test_audio = list_files('../input/birdclef-2021/train_soundscapes')\n    \nprint('{} FILES IN TEST SET.'.format(len(test_audio)))","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:47:38.646718Z","iopub.execute_input":"2021-06-01T07:47:38.647237Z","iopub.status.idle":"2021-06-01T07:47:38.665311Z","shell.execute_reply.started":"2021-06-01T07:47:38.647196Z","shell.execute_reply":"2021-06-01T07:47:38.664305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction","metadata":{}},{"cell_type":"code","source":"def predict(threshold):\n    row_id=[]\n    preds=[]\n    \n    for file_path in test_audio[:2]:\n        # Open it with librosa\n        sig, rate = librosa.load(file_path, sr=SAMPLE_RATE)\n\n        sig_splits = []\n        for i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n            split = sig[i:i + int(SIGNAL_LENGTH * SAMPLE_RATE)]\n\n            # End of signal?\n            if len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE):\n                break\n\n            sig_splits.append(split)\n\n        seconds= 0\n        for chunk in sig_splits:\n\n            # Keep track of the end time of each chunk\n            seconds += 5\n\n            # Get the spectrogram\n            hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n            mel_spec = librosa.feature.melspectrogram(y=chunk, \n                                                      sr=SAMPLE_RATE, \n                                                      n_fft=1024, \n                                                      hop_length=hop_length, \n                                                      n_mels=SPEC_SHAPE[0], \n                                                      fmin=FMIN, \n                                                      fmax=FMAX)\n\n            mel_spec = librosa.power_to_db(mel_spec, ref=np.max) \n\n            # Normalize to match the value range we used during training.\n            # That's something you should always double check!\n            mel_spec -= mel_spec.min()\n            mel_spec /= mel_spec.max()\n\n            # Add channel axis to 2D array\n            mel_spec = np.expand_dims(mel_spec, -1)\n\n            # Add new dimension for batch size\n            mel_spec = np.expand_dims(mel_spec, 0)\n\n            # Predict\n            p = 0.5*model1.predict(mel_spec)[0] + 0.5* model2.predict(mel_spec)[0]\n\n            # Get highest scoring species\n            idx = p.argmax()\n            species = LABELS[idx]\n            score = p[idx]\n\n            # Prepare submission entry\n            row_id.append(file_path.split(os.sep)[-1].rsplit('_', 1)[0] + \n                                  '_' + str(seconds))    \n\n            # Decide if it's a \"nocall\" or a species by applying a threshold\n            if score > threshold:\n                preds.append(species)\n            else:\n                preds.append('nocall')\n                \n    result=pd.DataFrame({'row_id': row_id, 'birds': preds})\n\n    return result","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:47:38.666619Z","iopub.execute_input":"2021-06-01T07:47:38.666863Z","iopub.status.idle":"2021-06-01T07:47:38.676956Z","shell.execute_reply.started":"2021-06-01T07:47:38.666839Z","shell.execute_reply":"2021-06-01T07:47:38.6759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=predict(0.6)\n\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:47:38.678096Z","iopub.execute_input":"2021-06-01T07:47:38.678347Z","iopub.status.idle":"2021-06-01T07:47:40.334818Z","shell.execute_reply.started":"2021-06-01T07:47:38.678323Z","shell.execute_reply":"2021-06-01T07:47:40.332326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-01T07:47:40.33568Z","iopub.status.idle":"2021-06-01T07:47:40.33604Z"},"trusted":true},"execution_count":null,"outputs":[]}]}