{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Baseline BirdCLEF Notebook 2022\n\n### This notebook is a basic random coinflip baseline. ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import os\nimport json\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport random\n\n# First, load list of audio files. We could use 'test.csv' as well,\n# but for now, let's stick with parsing the test_soundscape folder.\ntest_audio_dir = '../input/birdclef-2022/test_soundscapes/'\nfile_list = [f.split('.')[0] for f in sorted(os.listdir(test_audio_dir))]\n\n# At the moment, there should only be a single soundscape visible.\n# During the submission re-run, all other hidden soundscapes\n# will be visible too and can be processed by your notebook.\nprint('Number of test soundscapes:', len(file_list))","metadata":{"execution":{"iopub.status.busy":"2022-03-23T22:57:56.581338Z","iopub.execute_input":"2022-03-23T22:57:56.581615Z","iopub.status.idle":"2022-03-23T22:57:56.590635Z","shell.execute_reply.started":"2022-03-23T22:57:56.58158Z","shell.execute_reply":"2022-03-23T22:57:56.589657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load scored birds\nwith open('../input/birdclef-2022/scored_birds.json') as sbfile:\n    scored_birds = json.load(sbfile)","metadata":{"execution":{"iopub.status.busy":"2022-03-23T22:57:57.269956Z","iopub.execute_input":"2022-03-23T22:57:57.270655Z","iopub.status.idle":"2022-03-23T22:57:57.275588Z","shell.execute_reply.started":"2022-03-23T22:57:57.270619Z","shell.execute_reply":"2022-03-23T22:57:57.274771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This is where we will store our results\npred = {'row_id': [], 'target': []}\n\n# Process audio files and make predictions\nfor afile in file_list:\n    \n    # Complete file path\n    path = test_audio_dir + afile + '.ogg'\n    \n    # Open file with librosa and split signal into 5-second chunks\n    # sig, rate = librosa.load(path)\n    # ...\n    \n    # Let's assume we have a list of 12 audio chunks (1min / 5s == 12 segments)\n    chunks = [[] for i in range(12)]\n    \n    # Make prediction for each chunk\n    # Each scored bird gets a random value in our case\n    # since we don't actually have a model\n    for i in range(len(chunks)):        \n        chunk_end_time = (i + 1) * 5\n        for bird in scored_birds:\n            \n            # This is our random prediction score for this bird\n            score = np.random.uniform()\n            \n            # Assemble the row_id which we need to do for each scored bird\n            row_id = afile + '_' + bird + '_' + str(chunk_end_time)\n            \n            # Put the result into our prediction dict and\n            # apply a \"confidence\" threshold of 0.5\n            pred['row_id'].append(row_id)\n            pred['target'].append(True if score > 0.5 else False)","metadata":{"execution":{"iopub.status.busy":"2022-03-23T22:57:57.701439Z","iopub.execute_input":"2022-03-23T22:57:57.701721Z","iopub.status.idle":"2022-03-23T22:57:57.712557Z","shell.execute_reply.started":"2022-03-23T22:57:57.701692Z","shell.execute_reply":"2022-03-23T22:57:57.71117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make a new data frame and look at some results        \nresults = pd.DataFrame(pred, columns = ['row_id', 'target'])\n\n# Quick sanity check\nprint(results.head())  ","metadata":{"execution":{"iopub.status.busy":"2022-03-23T22:57:58.165249Z","iopub.execute_input":"2022-03-23T22:57:58.16557Z","iopub.status.idle":"2022-03-23T22:57:58.175524Z","shell.execute_reply.started":"2022-03-23T22:57:58.165537Z","shell.execute_reply":"2022-03-23T22:57:58.174687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## List of options to choose from for target variable\ntargets = [True, False]\n\n## Choose randomly from list\nfor target in results['target']:\n    target = random.choice(targets)","metadata":{"execution":{"iopub.status.busy":"2022-03-23T22:58:05.334441Z","iopub.execute_input":"2022-03-23T22:58:05.334967Z","iopub.status.idle":"2022-03-23T22:58:05.340978Z","shell.execute_reply.started":"2022-03-23T22:58:05.334918Z","shell.execute_reply":"2022-03-23T22:58:05.340147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert our results to csv\nresults.to_csv(\"submission.csv\", index=False)   ","metadata":{"execution":{"iopub.status.busy":"2022-03-23T22:58:05.996288Z","iopub.execute_input":"2022-03-23T22:58:05.996598Z","iopub.status.idle":"2022-03-23T22:58:06.003258Z","shell.execute_reply.started":"2022-03-23T22:58:05.996563Z","shell.execute_reply":"2022-03-23T22:58:06.00246Z"},"trusted":true},"execution_count":null,"outputs":[]}]}