{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":170575100,"sourceType":"kernelVersion"},{"sourceId":175551649,"sourceType":"kernelVersion"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#!/usr/bin/python3\n\nimport asyncio\nimport json\nimport os\nimport shutil\nimport sys\nimport time\nfrom urllib import request, error\n\nimport aiofiles\nimport aiohttp\n\n# Uses JSON metadata files to generate a list of recording URLs\ndef list_urls(path):\n    url_list = []\n    page = 1\n\n    # Initial opening of JSON to retrieve amount of pages and recordings\n    with open(path + '/page' + str(page) + '.json', 'r') as jsonfile:\n        data = jsonfile.read()\n        jsonfile.close()\n    data = json.loads(data)\n    page_num = data['numPages']\n    recordings_num = int(data['numRecordings'])\n\n    # Clear may not be required if setting to None, included for redundancy\n    data.clear()\n    data = None\n\n    # Set the first element to the number of recordings\n    url_list.append(recordings_num)\n\n    # Second element will be a list of tuples with (name, track_id, file url)\n    url_list.append(list())\n\n    # Read each metadata file and extract information into list as a tuple\n    while page < page_num + 1:\n        with open(path + '/page' + str(page) + '.json', 'r') as jsonfile:\n            data = jsonfile.read()\n            jsonfile.close()\n        data = json.loads(data)\n\n        # Extract the number of recordings in the opened metadata file\n        rec_length = len(data['recordings'])\n\n        # Parse through the opened data and add it to the URL list\n        for i in range(0, rec_length):\n            name = (data['recordings'][i]['en']).replace(' ', '')\n            track_id = data['recordings'][i]['id']\n            track_url = data['recordings'][i]['file']\n            track_format = os.path.splitext(data['recordings'][i]['file-name'])[-1]\n            track_info = (name, track_id, track_url, track_format)\n            url_list[1].append(track_info)\n        page += 1\n    return url_list\n\n\n# Client that processes the list of track information concurrently\ndef chunked_http_client(num_chunks):\n\n    # Semaphore used to limit the number of requests with num_chunks\n    semaphore = asyncio.Semaphore(num_chunks)\n\n    # Processes a tuple from the url_list using the aiohttp client_session\n    async def http_get(track_tuple, client_session):\n\n        # Work with semaphore located outside the function\n        nonlocal semaphore\n        async with semaphore:\n\n            # Pull relevant info from tuple\n            name = str(track_tuple[0])\n            track_id = str(track_tuple[1])\n            url = track_tuple[2]\n            track_format = track_tuple[3]\n\n            # Set up the paths required for saving the audio file\n            folder_path = 'dataset/audio/' + name + '/'\n            file_path = folder_path + track_id + track_format\n\n            # Create an audio folder for the species if it does not exist\n            if not os.path.exists(folder_path):\n                print(\"Creating recording folder at \" + str(folder_path))\n                os.makedirs(folder_path)\n\n            # If the file exists in the directory, we will skip it\n            if os.path.exists(file_path):\n                print(track_id + track_format + \" is already present. Skipping...\")\n                return\n\n            # Use the aiohttp client to retrieve the audio file asynchronously\n            async with client_session.get(url) as response:\n                if response.status == 200:\n                    f = await aiofiles.open((file_path), mode='wb')\n                    await f.write(await response.content.read())\n                    await f.close()\n                elif response.status == 503:\n                    print(\"Error 503 occurred when downloading \" + track_id\n                          + track_format + \". Please try using a lower value for \"\n                          \"num_chunks. Consult the README for more \"\n                          \"information.\")\n                else:\n                    print(\"Error \" + str(response.status) + \" occurred \"\n                          \"when downloading \" + track_id + track_format + \".\")\n\n    return http_get\n\n\n# Retrieves metadata and recordings for a given set of input param\nasync def download(url_list, num_chunks=4):\n    # Setup the aiohttp client with the desired semaphore limit\n    http_client = chunked_http_client(num_chunks)\n    async with aiohttp.ClientSession() as client_session:\n\n        # Collect tasks and await futures to ensure concurrent processing\n        tasks = [http_client(track_tuple, client_session) for track_tuple in\n                 url_list]\n        for future in asyncio.as_completed(tasks):\n            data = await future\n    print(\"Download complete.\")\n\n# Accepts command line input to determine function to execute\ndef main():\n    start = time.time()\n    asyncio.run(download(url_list))\n    end = time.time()\n    print(\"Duration: \" + str(int(end - start)) + \"s\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-04T02:55:52.387380Z","iopub.execute_input":"2024-05-04T02:55:52.387943Z","iopub.status.idle":"2024-05-04T02:55:52.518667Z","shell.execute_reply.started":"2024-05-04T02:55:52.387913Z","shell.execute_reply":"2024-05-04T02:55:52.517565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\npd.options.display.max_rows = 200\n\nretrival_candidates = pd.read_parquet(\"/kaggle/input/birdclef2024-exam-noobjbirds/extract.pqt\")\nretrival_candidates[\"track_format\"] = retrival_candidates[\"file-name\"].apply(lambda x: \".\" + x.split(\".\")[-1].lower())\nretrival_candidates = retrival_candidates[retrival_candidates.track_format.isin([\".mp3\",\".wav\"])].reset_index(drop=True)\nurl_list = [tuple(x) for x in retrival_candidates[[\"SPECIES_CODE\", \"id\", \"file\", \"track_format\"]].values]","metadata":{"execution":{"iopub.status.busy":"2024-05-04T02:56:31.563873Z","iopub.execute_input":"2024-05-04T02:56:31.564441Z","iopub.status.idle":"2024-05-04T02:56:31.661919Z","shell.execute_reply.started":"2024-05-04T02:56:31.564407Z","shell.execute_reply":"2024-05-04T02:56:31.661120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"await download(url_list)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T02:56:49.903582Z","iopub.execute_input":"2024-05-04T02:56:49.904172Z","iopub.status.idle":"2024-05-04T02:56:54.564018Z","shell.execute_reply.started":"2024-05-04T02:56:49.904139Z","shell.execute_reply":"2024-05-04T02:56:54.561996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:13:00.612640Z","iopub.execute_input":"2024-04-11T14:13:00.613025Z","iopub.status.idle":"2024-04-11T14:13:02.953936Z","shell.execute_reply.started":"2024-04-11T14:13:00.612996Z","shell.execute_reply":"2024-04-11T14:13:02.952694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:01:15.858969Z","iopub.execute_input":"2024-04-11T14:01:15.859385Z","iopub.status.idle":"2024-04-11T14:01:15.907682Z","shell.execute_reply.started":"2024-04-11T14:01:15.859355Z","shell.execute_reply":"2024-04-11T14:01:15.906487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}