{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Example Submission of All False [LB 0.48]\n* Out of curiousity - guessing all false gives LB 0.48. Only reason it's not lower is because the submission actually takes predictions on all 21 bird species for each time chunk\n* Our models will have to beat this at least.\n* see https://www.kaggle.com/stefankahl/how-to-submit-to-birdclef-2022/notebook @stefankahl code\n\n## Other Notebooks in this series:\n### 1. [Tutorial: BirdCLEF Visualizations with Plotly](https://www.kaggle.com/alexteboul/tutorial-birdclef-visualizations-with-plotly)\n### 2. [Tutorial: Play Bird Audio on Map with Folium](https://www.kaggle.com/alexteboul/tutorial-play-bird-audio-on-map-with-folium)","metadata":{}},{"cell_type":"code","source":"import os\nimport json\nimport numpy as np\nimport pandas as pd\nimport librosa\n\n# list of audio files\ntest_audio_dir = '../input/birdclef-2022/test_soundscapes/'\nfile_list = [f.split('.')[0] for f in sorted(os.listdir(test_audio_dir))]\nprint(file_list)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-16T17:13:01.842273Z","iopub.execute_input":"2022-02-16T17:13:01.84281Z","iopub.status.idle":"2022-02-16T17:13:01.850109Z","shell.execute_reply.started":"2022-02-16T17:13:01.842773Z","shell.execute_reply":"2022-02-16T17:13:01.849278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#These are the birds to score\nwith open('../input/birdclef-2022/scored_birds.json') as sbfile:\n    scored_birds = json.load(sbfile)\n\n#Alternatively:\nbirdslist = ['akiapo', 'aniani', 'apapan', 'barpet', 'crehon', 'elepai', 'ercfra', 'hawama', 'hawcre', 'hawgoo', 'hawhaw', 'hawpet1', 'houfin', 'iiwi', 'jabwar', 'maupar', 'omao', 'puaioh', 'skylar', 'warwhe1', 'yefcan']","metadata":{"execution":{"iopub.status.busy":"2022-02-16T17:23:52.660415Z","iopub.execute_input":"2022-02-16T17:23:52.660736Z","iopub.status.idle":"2022-02-16T17:23:52.67371Z","shell.execute_reply.started":"2022-02-16T17:23:52.660702Z","shell.execute_reply":"2022-02-16T17:23:52.67297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(scored_birds))","metadata":{"execution":{"iopub.status.busy":"2022-02-16T17:23:53.185198Z","iopub.execute_input":"2022-02-16T17:23:53.18551Z","iopub.status.idle":"2022-02-16T17:23:53.190408Z","shell.execute_reply.started":"2022-02-16T17:23:53.185479Z","shell.execute_reply":"2022-02-16T17:23:53.189296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(scored_birds)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T17:23:54.40533Z","iopub.execute_input":"2022-02-16T17:23:54.406185Z","iopub.status.idle":"2022-02-16T17:23:54.410824Z","shell.execute_reply.started":"2022-02-16T17:23:54.406138Z","shell.execute_reply":"2022-02-16T17:23:54.410155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#code from https://www.kaggle.com/stefankahl/how-to-submit-to-birdclef-2022/notebook @stefankahl\n# This is where we will store our results\npred = {'row_id': [], 'target': []}\n\n# Process audio files and make predictions\nfor afile in file_list:\n    \n    # Complete file path\n    path = test_audio_dir + afile + '.ogg'\n    \n    # Let's assume we have a list of 12 audio chunks (1min / 5s == 12 segments)\n    chunks = [[] for i in range(12)]\n    \n    # Make prediction for each chunk\n    # Each scored bird gets a random value in our case\n    # since we don't actually have a model\n    for i in range(len(chunks)):        \n        chunk_end_time = (i + 1) * 5\n        for bird in scored_birds:\n            \n            # This is our random prediction score for this bird\n            score = np.random.uniform()\n            \n            # Assemble the row_id which we need to do for each scored bird\n            row_id = afile + '_' + bird + '_' + str(chunk_end_time)\n            \n            # Put the result into our prediction dict and\n            # apply a \"confidence\" threshold of 0.5 (not this time - make em all false)\n            pred['row_id'].append(row_id)\n            #pred['target'].append(True if score > 0.5 else False)\n            pred['target'].append(False)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T17:25:47.126622Z","iopub.execute_input":"2022-02-16T17:25:47.126901Z","iopub.status.idle":"2022-02-16T17:25:47.13486Z","shell.execute_reply.started":"2022-02-16T17:25:47.126871Z","shell.execute_reply":"2022-02-16T17:25:47.134052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make a new data frame and look at some results        \nresults = pd.DataFrame(pred, columns = ['row_id', 'target'])\n\n#size of submission\nprint(results.shape)\n\n# Quick sanity check\nprint(results.head(22)) \n    \n# Convert our results to csv\nresults.to_csv(\"submission.csv\", index=False)  ","metadata":{"execution":{"iopub.status.busy":"2022-02-16T17:37:02.548642Z","iopub.execute_input":"2022-02-16T17:37:02.549435Z","iopub.status.idle":"2022-02-16T17:37:02.56264Z","shell.execute_reply.started":"2022-02-16T17:37:02.549378Z","shell.execute_reply":"2022-02-16T17:37:02.561416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So the naive (wrong) method of just guessing all false gives an LB score of 0.48.","metadata":{}}]}