{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8659080,"sourceType":"datasetVersion","datasetId":5187581},{"sourceId":182631439,"sourceType":"kernelVersion"}],"dockerImageVersionId":30732,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!python -m pip install --no-index --find-links=../input/birdnet-offlain -r ../input/birdnet-offlain/requirements.txt","metadata":{"execution":{"iopub.status.busy":"2024-06-10T19:20:32.270628Z","iopub.execute_input":"2024-06-10T19:20:32.271288Z","iopub.status.idle":"2024-06-10T19:20:43.988594Z","shell.execute_reply.started":"2024-06-10T19:20:32.271254Z","shell.execute_reply":"2024-06-10T19:20:43.987453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if not filename.endswith(\"ogg\"):\n        #if True:\n            print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-10T19:20:43.991095Z","iopub.execute_input":"2024-06-10T19:20:43.991442Z","iopub.status.idle":"2024-06-10T19:20:45.340402Z","shell.execute_reply.started":"2024-06-10T19:20:43.991412Z","shell.execute_reply":"2024-06-10T19:20:45.339069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from birdnetlib import Recording\nfrom birdnetlib import analyzer\n\ndir = \"/kaggle/input/birdnet-svc/\"\n\nanalyzer.MODEL_PATH = os.path.join(dir, \"BirdNET_GLOBAL_6K_V2.4_Model_FP32.tflite\")\nanalyzer.LABEL_PATH = os.path.join(dir, \"BirdNET_GLOBAL_6K_V2.4_Labels.txt\")","metadata":{"execution":{"iopub.status.busy":"2024-06-10T19:20:45.342057Z","iopub.execute_input":"2024-06-10T19:20:45.342545Z","iopub.status.idle":"2024-06-10T19:20:45.349254Z","shell.execute_reply.started":"2024-06-10T19:20:45.342509Z","shell.execute_reply":"2024-06-10T19:20:45.347632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport glob\n\nKAGGLE = True\ncfg = {\"dir\" : \"/kaggle/input/birdclef-2024/\"}\n\nif KAGGLE == True:\n    test_audio_dir = os.path.join(cfg[\"dir\"] , \"test_soundscapes/\")\n    file_list = glob.glob(test_audio_dir+\"*.ogg\")\n    file_list = sorted(file_list)\n\nif len(file_list) == 0 or KAGGLE == False:\n    test_audio_dir = os.path.join(cfg[\"dir\"], \"unlabeled_soundscapes/\")\n    file_list = glob.glob(test_audio_dir+\"*.ogg\")\n    file_list = sorted(file_list)[:10]","metadata":{"execution":{"iopub.status.busy":"2024-06-10T19:20:45.351301Z","iopub.execute_input":"2024-06-10T19:20:45.351752Z","iopub.status.idle":"2024-06-10T19:20:45.391897Z","shell.execute_reply.started":"2024-06-10T19:20:45.351719Z","shell.execute_reply":"2024-06-10T19:20:45.390819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_every_second_embed(recording):\n    embedding = list()\n    for i, chunk in enumerate(recording.embeddings):\n        if i % 2 == 0:\n            prev_chunk = np.array(chunk[\"embeddings\"])\n        else:\n            embedding.append((prev_chunk + np.array(chunk[\"embeddings\"]))/2)\n    else:\n        if i % 2 == 0:\n            embedding.append(np.array(chunk[\"embeddings\"]))\n    embedding = np.array(embedding)\n    return embedding\n\n\ndef add_embeds_to_df(file_list, save=True):\n    row_ids = list()\n    embedds = list()\n    analyze = analyzer.Analyzer()\n\n    for i, filename_ in enumerate(file_list):\n        print (i)\n\n        print (filename_)\n        recording = Recording(\n            analyze,\n            os.path.join(filename_),\n            min_conf=0.01,\n            overlap=0)\n        recording.sample_secs = 2.5\n        recording.extract_embeddings()\n        embedding = get_every_second_embed(recording)\n        #if not np.any(np.isnan(embedding.mean(axis=0))):\n        if embedding.size != 0:\n            embedds.append(embedding)\n        else:\n            print (\"HERE NAN\")\n            embedds.append(np.zeros(1024))\n\n        shortname = filename_.split(\"/\")[-1][:-4]\n        row_ids += [shortname + \"_\" + str((i+1) * 5) for i in range(len(embedding))]\n        #if i >= 100:\n        #    break\n\n    #return embedds, row_ids\n\n    output = pd.DataFrame(np.vstack(embedds), index=row_ids)\n    if save:\n        output.to_csv(\"embedds.csv\")\n\n    return output","metadata":{"execution":{"iopub.status.busy":"2024-06-10T19:20:45.394947Z","iopub.execute_input":"2024-06-10T19:20:45.395691Z","iopub.status.idle":"2024-06-10T19:20:45.408190Z","shell.execute_reply.started":"2024-06-10T19:20:45.395654Z","shell.execute_reply":"2024-06-10T19:20:45.407227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from concurrent.futures import ProcessPoolExecutor\n\ndef add_embeds_single_df(file_list, save=False):\n    return add_embeds_to_df([file_list], save=save)\n\ndef parallel_process_files(file_paths):\n    # Use ProcessPoolExecutor to parallelize file processing\n    with ProcessPoolExecutor() as executor:\n        results = list(executor.map(add_embeds_single_df, file_paths))\n    \n    # Concatenate all the individual DataFrames into one\n    final_df = pd.concat(results, ignore_index=False)\n    return final_df\n\nembedds = parallel_process_files(file_list)\n\nembedds.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-10T19:20:45.409629Z","iopub.execute_input":"2024-06-10T19:20:45.410615Z","iopub.status.idle":"2024-06-10T19:20:47.343065Z","shell.execute_reply.started":"2024-06-10T19:20:45.410585Z","shell.execute_reply":"2024-06-10T19:20:47.341422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\nfrom sklearn import svm\n\ndir = \"/kaggle/input/birdnet-svc\"\n\nmodel = joblib.load(os.path.join(dir, \"SVC.pkl\"))\n\nmodel","metadata":{"execution":{"iopub.status.busy":"2024-06-10T19:41:41.836039Z","iopub.execute_input":"2024-06-10T19:41:41.836476Z","iopub.status.idle":"2024-06-10T19:41:45.935300Z","shell.execute_reply.started":"2024-06-10T19:41:41.836428Z","shell.execute_reply":"2024-06-10T19:41:45.934315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict_proba(embedds)\n\nprint (preds.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-10T19:46:39.795114Z","iopub.execute_input":"2024-06-10T19:46:39.795526Z","iopub.status.idle":"2024-06-10T19:46:42.712742Z","shell.execute_reply.started":"2024-06-10T19:46:39.795488Z","shell.execute_reply":"2024-06-10T19:46:42.711423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/birdclef-2024/sample_submission.csv\", index_col=\"row_id\")\n\ncolumns = sample.columns","metadata":{"execution":{"iopub.status.busy":"2024-06-10T19:49:34.471508Z","iopub.execute_input":"2024-06-10T19:49:34.472330Z","iopub.status.idle":"2024-06-10T19:49:34.485367Z","shell.execute_reply.started":"2024-06-10T19:49:34.472292Z","shell.execute_reply":"2024-06-10T19:49:34.484227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(preds, columns=columns, index=embedds.index)\n\nsubmission.to_csv('submission.csv',index=False)\n\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-10T19:54:09.749724Z","iopub.execute_input":"2024-06-10T19:54:09.750702Z","iopub.status.idle":"2024-06-10T19:54:09.863270Z","shell.execute_reply.started":"2024-06-10T19:54:09.750661Z","shell.execute_reply":"2024-06-10T19:54:09.862305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}