{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import librosa\nimport pandas as pd\nimport numpy as np","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Random sample from trainset"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import IPython.display as ipd\nipd.Audio(\"../input/rfcx-species-audio-detection/train/1535d0c9b.flac\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Random sample from testset"},{"metadata":{"trusted":true},"cell_type":"code","source":"import IPython.display as ipd\nipd.Audio(\"../input/rfcx-species-audio-detection/test/01812f522.flac\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tp = pd.read_csv(\"../input/rfcx-species-audio-detection/train_tp.csv\")\ntrain_tp","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tp.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_fp = pd.read_csv(\"../input/rfcx-species-audio-detection/train_fp.csv\")\ntrain_fp","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_fp.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Number of species in total"},{"metadata":{"trusted":true},"cell_type":"code","source":"(len(train_tp[\"species_id\"].unique()),len(train_fp[\"species_id\"].unique()))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Check entry types in train_tp and train_fp"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check if there are more than 1 entries in audiofile found\ncount_multiple = 0\n# Check if there are more than 1 species in the audiofile detected\ncount_different_spec = 0\nfor recording_id in train_tp[\"recording_id\"]:\n    if len(train_tp[train_tp[\"recording_id\"] == recording_id]) > 1:\n        count_multiple = count_multiple + 1\n    if len(set(train_tp[train_tp[\"recording_id\"] == recording_id][\"species_id\"])) > 1:\n        count_different_spec = count_different_spec + 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"(count_multiple/len(train_tp),count_different_spec/len(train_tp), count_different_spec/count_multiple, 1-(count_multiple/len(train_tp)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check if there are more than 1 entries in audiofile found\ncount_multiple = 0\n# Check if there are more than 1 species in the audiofile detected\ncount_different_spec = 0\nfor recording_id in train_fp[\"recording_id\"]:\n    if len(train_fp[train_fp[\"recording_id\"] == recording_id]) > 1:\n        count_multiple = count_multiple + 1\n    if len(set(train_fp[train_fp[\"recording_id\"] == recording_id][\"species_id\"])) > 1:\n        count_different_spec = count_different_spec + 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"(count_multiple/len(train_fp),count_different_spec/len(train_fp), count_different_spec/count_multiple, 1-(count_multiple/len(train_fp)))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The hardest part is obvious, and the numbers give evidence for that: \n* Recognizing multiple species within an audiofile "},{"metadata":{},"cell_type":"markdown","source":"How many recordings are in train_tp and train_fp"},{"metadata":{"trusted":true},"cell_type":"code","source":"recordings = []\nfor recording_id in train_tp[\"recording_id\"]:\n    if len(train_fp[train_fp[\"recording_id\"] == recording_id]) > 0:\n        recordings.append(recording_id)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(recordings)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"recordings[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tp[train_tp[\"recording_id\"] == recordings[3]]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_fp[train_fp[\"recording_id\"] == recordings[3]]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"How many samples are completly correctly recognized and therefore not in train_fp"},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train_tp[\"recording_id\"].unique()) - len(set(recordings))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tp_complete = list(set(list(train_tp[\"recording_id\"])) - set(recordings))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Create DataFrame with only true positives which are only in train_tp and not in train_fp"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tp_comp = pd.DataFrame()\nfor rec_id in train_tp_complete:\n    chunk = train_tp[train_tp[\"recording_id\"] == rec_id]\n    train_tp_comp = pd.concat([train_tp_comp, chunk])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tp_comp","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check if there are more than 1 entries in audiofile found\ncount_multiple = 0\n# Check if there are more than 1 species in the audiofile detected\ncount_different_spec = 0\nfor recording_id in train_tp_comp[\"recording_id\"]:\n    if len(train_tp_comp[train_tp_comp[\"recording_id\"] == recording_id]) > 1:\n        count_multiple = count_multiple + 1\n    if len(set(train_tp_comp[train_fp[\"recording_id\"] == recording_id][\"species_id\"])) > 1:\n        count_different_spec = count_different_spec + 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"(count_multiple/len(train_fp),count_different_spec/len(train_fp), count_different_spec/count_multiple, 1-(count_multiple/len(train_fp)))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Almost all of them have only 1 entry and none of them have multiple species"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}