{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8382706,"sourceType":"datasetVersion","datasetId":4985205},{"sourceId":172204890,"sourceType":"kernelVersion"},{"sourceId":180994946,"sourceType":"kernelVersion"}],"dockerImageVersionId":30407,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Locale Filtered Training Set","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\n#let a loose bounding box of the western ghats be the 4 points: \n#NW: 21.109472, 72.302583 \n#NE: 21.109472, 78.439028\n#SW: 7.58875, 72.302583\n#SE: 7.58875, 78.439028\n\n#no logic for crossing 180+-deg needed\ndef isInBox(lat, long):    \n    top = 21.109472\n    bottom = 7.58875\n    left = 72.302583\n    right = 78.439028\n\n    return (lat <= top) & (lat >= bottom) & (long <= right) & (long >= left) \n\ndef runTests():\n    western_ghats_lat = 10.169722\n    western_ghats_long = 77.061111\n    print(\"Testing bounding box. Should be true, is: \" + str(isInBox(western_ghats_lat, western_ghats_long)))\n    print(\"Testing bounding box. Should be false, is: \" + str(isInBox(0, 0)))\n\n#runTests()\n\n#select from the dataframe only entries that are inside this box\n#returns about 2500 entries\nfiltered_df = df.loc[(isInBox(df['latitude'], df['longitude']))]\n\n#birds not in box\nexcluded_set = pd.concat([df,filtered_df]).drop_duplicates(keep=False)\n\nfiltered_df","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:24:32.598003Z","iopub.execute_input":"2024-06-08T09:24:32.598532Z","iopub.status.idle":"2024-06-08T09:24:32.968741Z","shell.execute_reply.started":"2024-06-08T09:24:32.598485Z","shell.execute_reply":"2024-06-08T09:24:32.966932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\nmapbox_access_token = 'pk.eyJ1IjoiZ2FiZWRsIiwiYSI6ImNrdDJrc25saTBxYnAyd3BrOHZ2OHk4cHcifQ.zZGVJxA2XLRCUquiqGkcEg'\npx.set_mapbox_access_token(mapbox_access_token)\n\nfig = px.scatter_mapbox(filtered_df, lat=\"latitude\", lon=\"longitude\", zoom=4)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:24:32.971105Z","iopub.execute_input":"2024-06-08T09:24:32.971565Z","iopub.status.idle":"2024-06-08T09:24:37.964806Z","shell.execute_reply.started":"2024-06-08T09:24:32.971521Z","shell.execute_reply":"2024-06-08T09:24:37.963309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Excluded Set","metadata":{}},{"cell_type":"code","source":"import plotly.express as px\nmapbox_access_token = 'pk.eyJ1IjoiZ2FiZWRsIiwiYSI6ImNrdDJrc25saTBxYnAyd3BrOHZ2OHk4cHcifQ.zZGVJxA2XLRCUquiqGkcEg'\npx.set_mapbox_access_token(mapbox_access_token)\n\nfig2 = px.scatter_mapbox(excluded_set, lat=\"latitude\", lon=\"longitude\", zoom=4)\nfig2.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:24:37.966566Z","iopub.execute_input":"2024-06-08T09:24:37.967046Z","iopub.status.idle":"2024-06-08T09:24:38.061459Z","shell.execute_reply.started":"2024-06-08T09:24:37.966992Z","shell.execute_reply":"2024-06-08T09:24:38.060047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Locale Filtered Dataset using Geometry","metadata":{}},{"cell_type":"code","source":"\nimport pandas as pd\nfrom shapely.geometry import Polygon\nfrom shapely.geometry import Point\n\nlats = [21, 18.22, 12.83, 7.42, 13.06, 20.42]\nlongs = [73.32, 72.95, 74.51, 77.58, 78.55, 74.07]\n\npolygon_geom = Polygon(zip(longs, lats))\ndf['in_western_ghats'] = \"\"\nin_western_ghats_column = []\nsource = []\nfor lon,lat in zip(df.longitude, df.latitude):\n    pt = Point(lon, lat)\n    intersects = pt.within(polygon_geom)\n    in_western_ghats_column.append(intersects)\n    source.append(\"training_data\")\n\ndf['in_western_ghats'] = in_western_ghats_column\ndf['source'] = source\nfiltered_df = df[df.in_western_ghats == True]\nfiltered_df","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:24:38.064517Z","iopub.execute_input":"2024-06-08T09:24:38.065416Z","iopub.status.idle":"2024-06-08T09:24:38.912806Z","shell.execute_reply.started":"2024-06-08T09:24:38.065367Z","shell.execute_reply":"2024-06-08T09:24:38.911487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# In Western Ghats / Outside visual","metadata":{}},{"cell_type":"code","source":"import plotly.express as px\nmapbox_access_token = 'pk.eyJ1IjoiZ2FiZWRsIiwiYSI6ImNrdDJrc25saTBxYnAyd3BrOHZ2OHk4cHcifQ.zZGVJxA2XLRCUquiqGkcEg'\npx.set_mapbox_access_token(mapbox_access_token)\nin_western_ghats_list = ['red', 'blue']\nfig = px.scatter_mapbox(df, lat=\"latitude\", lon=\"longitude\", color=df['in_western_ghats'],\n    color_discrete_sequence=in_western_ghats_list)\nfig.update_layout(\n    mapbox = {\n        'style': \"open-street-map\",\n        'center': {'lon': 75.16, 'lat': 14.45 },\n        'zoom': 3},\n    showlegend = False)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:24:38.914501Z","iopub.execute_input":"2024-06-08T09:24:38.915022Z","iopub.status.idle":"2024-06-08T09:24:39.033847Z","shell.execute_reply.started":"2024-06-08T09:24:38.914979Z","shell.execute_reply":"2024-06-08T09:24:39.032346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bounds","metadata":{}},{"cell_type":"code","source":"import plotly.graph_objects as go\nfig_boundary = go.Figure(go.Scattermapbox(\n    fill = \"toself\",\n    lon = longs, lat = lats,\n    marker = { 'size': 10, 'color': \"orange\" }))\nfig_boundary.update_layout(\n    mapbox = {\n        'style': \"open-street-map\",\n        'center': {'lon': 75.16, 'lat': 14.45 },\n        'zoom': 4},\n    showlegend = False)\nfig_boundary.show()\n\nfig_bird_plot = px.scatter_mapbox(filtered_df, lat=\"latitude\", lon=\"longitude\", zoom=4)\nfig_bird_plot.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:24:39.035878Z","iopub.execute_input":"2024-06-08T09:24:39.036333Z","iopub.status.idle":"2024-06-08T09:24:39.131196Z","shell.execute_reply.started":"2024-06-08T09:24:39.036291Z","shell.execute_reply":"2024-06-08T09:24:39.129564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Color coded top 5","metadata":{}},{"cell_type":"code","source":"print(filtered_df['common_name'].value_counts()[0:5])\nfiltered_df[\"color\"] = None\ncolor_column = []\nfor name in filtered_df['common_name']:\n    if name == 'White-cheeked Barbet':\n        color_column.append('#ff0000')\n    elif name == 'Indian Scimitar-Babbler':\n        color_column.append('#00ff00')\n    elif name == 'Black-hooded Oriole':\n        color_column.append('#0000ff')\n    elif name == 'Gray Junglefowl':\n        color_column.append('#ffff00')\n    elif name == 'Asian Koel':\n        color_column.append('#ff00ff')\n    else:\n        color_column.append('#000000')\n\nfiltered_df[\"color\"] = color_column\n\nfig_bird_plot = px.scatter_mapbox(filtered_df, lat=\"latitude\", lon=\"longitude\", height=1000, color=filtered_df.color,hover_name=filtered_df.common_name, zoom=4)\nfig_bird_plot.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:24:39.132953Z","iopub.execute_input":"2024-06-08T09:24:39.133394Z","iopub.status.idle":"2024-06-08T09:24:39.259733Z","shell.execute_reply.started":"2024-06-08T09:24:39.133352Z","shell.execute_reply":"2024-06-08T09:24:39.258348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('./training_data_with_locale_presence.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:24:39.261596Z","iopub.execute_input":"2024-06-08T09:24:39.262004Z","iopub.status.idle":"2024-06-08T09:24:39.657831Z","shell.execute_reply.started":"2024-06-08T09:24:39.261963Z","shell.execute_reply":"2024-06-08T09:24:39.656390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\n\nembeddings = torch.load(\"/kaggle/input/bc24-google-bird-model-embeddings-predict-score/embeddings.pt\")\ndf['index'] = df.index\nfilename = df.filename[0]\nfilename, embeddings[filename].shape\nfor x in embeddings[filename][0]:\n    print(x)","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:24:39.659417Z","iopub.execute_input":"2024-06-08T09:24:39.659825Z","iopub.status.idle":"2024-06-08T09:25:11.049831Z","shell.execute_reply.started":"2024-06-08T09:24:39.659785Z","shell.execute_reply":"2024-06-08T09:25:11.048215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Audio\nimport torchaudio\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nimport numpy as np\n\nfirst5sec_embeddings = np.stack([embeddings[filename][0] for filename in df.filename])\nfirst5sec_embeddings.shape\n\nSAMPLE_RATE = 32_000\nTRAIN_PATH = Path('/kaggle/input/birdclef-2024/train_audio/')\n\ncompute_melspec = torchaudio.transforms.MelSpectrogram(\n    sample_rate=SAMPLE_RATE,\n    n_mels=128,\n    n_fft=2048, \n    hop_length=512,\n    f_min=0,\n    f_max=SAMPLE_RATE // 2,\n)\n\npower_to_db = torchaudio.transforms.AmplitudeToDB(\n    stype=\"power\",\n    top_db=80.0,\n)\n\ndef show_bird(index, start=0):\n    audio = torchaudio.load(TRAIN_PATH / df.filename[index], start, start+32_000*5)[0][0]\n    display(Audio(audio, rate=SAMPLE_RATE))\n    plt.figure(figsize=(12, 2.5))\n    plt.subplot(121)\n    plt.plot(audio)\n    plt.gca().get_xaxis().set_visible(False)\n    plt.subplot(122)\n    plt.imshow(power_to_db(compute_melspec(audio)))\n    plt.show()\n    return df.iloc[index]\n\n\nshow_bird(0)","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:25:11.056181Z","iopub.execute_input":"2024-06-08T09:25:11.058073Z","iopub.status.idle":"2024-06-08T09:25:15.243616Z","shell.execute_reply.started":"2024-06-08T09:25:11.058017Z","shell.execute_reply":"2024-06-08T09:25:15.242218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Compute 3D umap","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfirst5sec_embeddings = np.stack([embeddings[filename][0] for filename in df.filename])\nfirst5sec_embeddings.shape\n\n\ndef compute_umap(df, embeddings):\n    import umap\n    \n    reducer = umap.UMAP(\n        random_state=42,\n        n_components=3\n    )\n    umap_embedding = reducer.fit_transform(embeddings)\n    df['umap_x'] = umap_embedding[:, 0]\n    df['umap_y'] = umap_embedding[:, 1]\n    df['umap_z'] = umap_embedding[:, 2]\n    \nprint('computing umap...')\ncompute_umap(df, first5sec_embeddings)\nprint('done!')","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:25:15.245570Z","iopub.execute_input":"2024-06-08T09:25:15.245984Z","iopub.status.idle":"2024-06-08T09:26:46.901289Z","shell.execute_reply.started":"2024-06-08T09:25:15.245944Z","shell.execute_reply":"2024-06-08T09:26:46.899761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize Embeddings 3D","metadata":{}},{"cell_type":"code","source":"print(\"Now plotting embeddings for all records\")\nfig = px.scatter_3d (\n    data_frame=df, \n    x='umap_x', \n    y='umap_y',\n    z='umap_z', \n    color='primary_label',\n    hover_data=['index'],\n    width=None, \n    height=800,\n)\nfig.show()\n\nprint(\"Now plotting embeddings for records in western ghats\")\nfig2 = px.scatter_3d(\n    data_frame=df[df.in_western_ghats], \n    x='umap_x', \n    y='umap_y',\n    z='umap_z', \n    color='primary_label',\n    hover_data=['index'],\n    width=None, \n    height=800,\n)\nfig2.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:26:46.903314Z","iopub.execute_input":"2024-06-08T09:26:46.904455Z","iopub.status.idle":"2024-06-08T09:26:49.086669Z","shell.execute_reply.started":"2024-06-08T09:26:46.904408Z","shell.execute_reply":"2024-06-08T09:26:49.085226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Compute 2D umap","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfirst5sec_embeddings = np.stack([embeddings[filename][0] for filename in df.filename])\nfirst5sec_embeddings.shape\n\n\ndef compute_umap(df, embeddings):\n    import umap\n    \n    reducer = umap.UMAP(\n        random_state=42,\n        n_components=2\n    )\n    umap_embedding = reducer.fit_transform(embeddings)\n    df['umap_x'] = umap_embedding[:, 0]\n    df['umap_y'] = umap_embedding[:, 1]\n    \nprint('computing 2D umap...')\ncompute_umap(df, first5sec_embeddings)\nprint('done!')","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:26:49.088967Z","iopub.execute_input":"2024-06-08T09:26:49.089495Z","iopub.status.idle":"2024-06-08T09:27:16.390350Z","shell.execute_reply.started":"2024-06-08T09:26:49.089433Z","shell.execute_reply":"2024-06-08T09:27:16.388588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize Embeddings 2D","metadata":{}},{"cell_type":"code","source":"print(\"Now plotting embeddings for all records\")\n\nfig = px.scatter(\n    data_frame=df, \n    x='umap_x', \n    y='umap_y',\n    color='primary_label',\n    hover_data=['index'],\n    width=1000, \n    height=1000,\n)\nfig.show()\n\nprint(\"Now plotting embeddings for records in western ghats\")\nfig2 = px.scatter(\n    data_frame=df[df.in_western_ghats], \n    x='umap_x', \n    y='umap_y',\n    color='primary_label',\n    hover_data=['index'],\n    width=1000, \n    height=1000,\n)\nfig2.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:27:16.392182Z","iopub.execute_input":"2024-06-08T09:27:16.392676Z","iopub.status.idle":"2024-06-08T09:27:18.532629Z","shell.execute_reply.started":"2024-06-08T09:27:16.392623Z","shell.execute_reply":"2024-06-08T09:27:18.531108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\n\nfig = px.scatter(\n    data_frame=df, \n    x='umap_x', \n    y='umap_y',\n    color='primary_label',\n    hover_data=['index'],\n    width=1000, \n    height=1000,\n)\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:27:18.534789Z","iopub.execute_input":"2024-06-08T09:27:18.535356Z","iopub.status.idle":"2024-06-08T09:27:19.668164Z","shell.execute_reply.started":"2024-06-08T09:27:18.535299Z","shell.execute_reply":"2024-06-08T09:27:19.666622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Overlay Embeddings - Inside Western Ghats and Outside","metadata":{}},{"cell_type":"code","source":"fig = px.scatter(\n    data_frame=df, \n    x='umap_x', \n    y='umap_y',\n    color='in_western_ghats',\n    hover_data=['index'],\n    width=1000, \n    height=1000,\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:27:19.670121Z","iopub.execute_input":"2024-06-08T09:27:19.671185Z","iopub.status.idle":"2024-06-08T09:27:19.791964Z","shell.execute_reply.started":"2024-06-08T09:27:19.671115Z","shell.execute_reply":"2024-06-08T09:27:19.790564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport numpy as np\n\nimport pandas as pd\nimport os\n\nfile_list = []\nin_western_ghats_list = []\nsource_list = []\nfor file in os.listdir('/kaggle/input/birdclef-2024/unlabeled_soundscapes/'):\n    file_list.append(\"/kaggle/input/birdclef-2024/unlabeled_soundscapes/\" + file)\n    in_western_ghats_list.append(\"Unknown\")\n    source_list.append(\"unlabeled_soundscapes\")\n\nnew_df = pd.DataFrame({'filename': file_list, \"in_western_ghats\": in_western_ghats_list, \"source\": source_list})\nall_df = pd.concat([new_df,df])\nunlabeled_embeddings = torch.load(\"/kaggle/input/embeddings/embeddings_unlabeled_soundscapes.pt\")\nall_embeddings = unlabeled_embeddings\nall_embeddings.update(embeddings)\n\n#all_embeddings[filename][0] for filename in all_df.filename:\n\nstacked_embeddings = np.stack([all_embeddings[filename][0] for filename in all_df.filename])\nprint(all_embeddings[filename][0])\n#for x in all_embeddings[filename][0]:\n#    print(x)\nprint(stacked_embeddings.shape)\n\n\ndef compute_umap(all_df, embeddings_to_transform):\n    import umap\n    \n    reducer = umap.UMAP(\n        random_state=42,\n        n_components=2\n    )\n    umap_embedding = reducer.fit_transform(embeddings_to_transform)\n    all_df['umap_x'] = umap_embedding[:, 0]\n    all_df['umap_y'] = umap_embedding[:, 1]\n    \nprint('computing 2D umap...')\ncompute_umap(all_df, stacked_embeddings)\nprint('done!')\n\n#print(all_df)\nall_df.to_csv('umap_all.csv')","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:27:19.793975Z","iopub.execute_input":"2024-06-08T09:27:19.794407Z","iopub.status.idle":"2024-06-08T09:29:04.168685Z","shell.execute_reply.started":"2024-06-08T09:27:19.794360Z","shell.execute_reply":"2024-06-08T09:29:04.166439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualization - All Embeddings","metadata":{}},{"cell_type":"code","source":"fig = px.scatter(\n    data_frame=all_df, \n    x='umap_x', \n    y='umap_y',\n    color='in_western_ghats',\n    width=1000, \n    height=1000,\n    custom_data=['filename', 'primary_label', 'common_name','in_western_ghats']\n)\nfig.update_traces(\n    hovertemplate=\"<br>\".join([\n        \"X: %{x}\",\n        \"Y: %{y}\",\n        \"filename: %{customdata[0]}\",\n        \"primary_label: %{customdata[1]}\",\n        \"common_name: %{customdata[2]}\",\n        \"in_western_ghats: %{customdata[3]}\",\n    ])\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:29:04.172318Z","iopub.execute_input":"2024-06-08T09:29:04.172941Z","iopub.status.idle":"2024-06-08T09:29:06.075242Z","shell.execute_reply.started":"2024-06-08T09:29:04.172876Z","shell.execute_reply":"2024-06-08T09:29:06.073192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"Process free field embeddings data and join metadata","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport sys\n\nfree_field_file_list = []\nfree_field_embeddings_list = []\nfree_field_source_list = []\n# traverse root directory, and list directories as dirs and files as files\nfor root, dirs, files in os.walk(\"/kaggle/input/bc24-run-gbm-on-freefield1010/freefield1010/\"):\n    path = root.split(os.sep)\n    for file in files:\n        if \"embeddings\" in file:\n            #to simplify merge, match the exact format of the values in freefield metadata filepath column\n            # '/kaggle/input/bc24-run-gbm-on-freefield1010/freefield1010/07/07/2421_embeddings.npy',\n            # filepath column example value 07/07/2421\n            full_file_path = (root + \"/\" + file)\n            start_index = 58\n            end_index = full_file_path.index(\"_\")\n            filepath = full_file_path[start_index:end_index]\n            #print(\"filepath: \" + filepath)\n            free_field_file_list.append(filepath)\n            free_field_embeddings_list.append(np.load(root + \"/\" + file))\n            free_field_source_list.append(\"freefield\") \n\nfree_field_embeddings_with_filename_df = pd.DataFrame({'filename': free_field_file_list, \"embeddings\": free_field_embeddings_list, 'source': free_field_source_list})\nfree_field_embeddings_df = pd.DataFrame({\"embeddings\": free_field_embeddings_list})\n\nfree_field_metadata_df = pd.read_csv(\"/kaggle/input/bc24-run-gbm-on-freefield1010/freefield1010.csv\")\nfree_field_metadata_df.rename(columns={'filepath': 'filename'}, inplace=True)\n\n\nfree_field_metadata_embeddings_df = pd.merge(free_field_embeddings_with_filename_df, free_field_metadata_df, on='filename', how='inner')\nfree_field_metadata_embeddings_df.to_csv(\"free_field_metadata_and_embeddings.csv\", encoding='utf-8', index=False)\nembeddingsList = {}\n\nfor filepath in free_field_embeddings_with_filename_df.filename:\n    #print(filepath)\n    filename = \"/kaggle/input/bc24-run-gbm-on-freefield1010/freefield1010/\"+filepath+\"_embeddings.npy\"\n    embeddingsList[filename] = np.load(filename)\n\n\nnew_all_df = pd.concat([free_field_metadata_embeddings_df,all_df])\nnew_all_df.to_csv(\"new_all_embeddings.csv\", encoding='utf-8', index=False)\n\nall_embeddings_filename_list = []\nall_embeddings_first_5_second_embedding_list = []\nkeys = list(all_embeddings)\nfor key in keys:\n    all_embeddings_filename_list.append(key)\n    all_embeddings_first_5_second_embedding_list.append(all_embeddings[key][0])\n\nall_2024_embeddings_df = pd.DataFrame({'filename': all_embeddings_filename_list, \"embeddings\": all_embeddings_first_5_second_embedding_list})\n\nall_2024_embeddings_df.to_csv(\"2024_embeddings_first_5_seconds.csv\", encoding='utf-8', index=False)\nall_df.to_csv(\"2024_metadata.csv\", encoding='utf-8', index=False)\n\n\n# I don't know why, but only after saving it and reading it back could I merge these two dfs. Something to do with the column types?\ndf_1 = pd.read_csv('/kaggle/working/2024_embeddings_first_5_seconds.csv')\ndf_2 = pd.read_csv('/kaggle/working/2024_metadata.csv')\nmetadata_and_embeddings_2024 = df_1.merge(df_2, left_on='filename', right_on='filename')\nmetadata_and_embeddings_2024.to_csv(\"metadata_and_embeddings_2024_5_s.csv\", encoding='utf-8', index=False)\n\n\ndf_3 = pd.read_csv('/kaggle/working/free_field_metadata_and_embeddings.csv')\ndf_4 = pd.read_csv('/kaggle/working/metadata_and_embeddings_2024_5_s.csv')\nmetadata_and_embeddings_2024_and_free_field_df = pd.concat([df_3, df_4])\nmetadata_and_embeddings_2024_and_free_field_df.to_csv(\"metadata_and_embeddings_2024_free_field_df.csv\", encoding='utf-8', index=False)\nprint(\"done\")","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:29:06.077894Z","iopub.execute_input":"2024-06-08T09:29:06.078353Z","iopub.status.idle":"2024-06-08T09:30:32.375596Z","shell.execute_reply.started":"2024-06-08T09:29:06.078305Z","shell.execute_reply":"2024-06-08T09:30:32.373961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compute Embeddings - 2024 Training Data, Unlabeled Soundscapes, Free Field 1010 ","metadata":{}},{"cell_type":"code","source":"metadata_and_embeddings_2024_free_field_df = pd.read_csv('metadata_and_embeddings_2024_free_field_df.csv')\n#for x in all_embeddings_first_5_second_embedding_list[0]:\n#    print(x)\nlen(all_embeddings_first_5_second_embedding_list)\nlen(free_field_embeddings_list[0])\n#for x in free_field_embeddings_list[0]:\n#    print(x)\n#dict all_embeddings_filename_list + all_embeddings_first_5_second_embedding_list\n\ncomplete_file_name_list = all_embeddings_filename_list + free_field_file_list\ncomplete_first_5_s_embedding_list = all_embeddings_first_5_second_embedding_list + free_field_embeddings_list\n#print(len(complete_file_name_list))\n#print(len(complete_first_5_s_embedding_list))\n\ncomplete_dict = dict(zip(complete_file_name_list, complete_first_5_s_embedding_list))\n#for filename in metadata_and_embeddings_2024_free_field_df.filename:\n#    print(complete_dict[filename])\nstacked_embeddings_complete = np.stack([complete_dict[filename] for filename in metadata_and_embeddings_2024_free_field_df.filename])\nprint(stacked_embeddings_complete.shape)\n\ndef compute_umap(metadata_and_embeddings_2024_and_free_field_df, embeddings_to_transform):\n    import umap\n    \n    reducer = umap.UMAP(\n        random_state=42,\n        n_components=2\n    )\n    umap_embedding = reducer.fit_transform(embeddings_to_transform)\n    metadata_and_embeddings_2024_and_free_field_df['umap_x'] = umap_embedding[:, 0]\n    metadata_and_embeddings_2024_and_free_field_df['umap_y'] = umap_embedding[:, 1]\n    \nprint('computing 2D umap...')\ncompute_umap(metadata_and_embeddings_2024_and_free_field_df, stacked_embeddings_complete)\nmetadata_and_embeddings_2024_and_free_field_df.to_csv(\"/kaggle/working/metadata_and_embeddings_2024_free_field_df_umap.csv\", encoding='utf-8', index=False)\nprint('done!')","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:30:32.377586Z","iopub.execute_input":"2024-06-08T09:30:32.378041Z","iopub.status.idle":"2024-06-08T09:31:21.842825Z","shell.execute_reply.started":"2024-06-08T09:30:32.377998Z","shell.execute_reply":"2024-06-08T09:31:21.840979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize Embeddings - 2024 Training Data, Unlabeled Soundscapes, Free Field 1010 ","metadata":{}},{"cell_type":"code","source":"fig = px.scatter(\n    data_frame=metadata_and_embeddings_2024_and_free_field_df, \n    x='umap_x', \n    y='umap_y',\n    color='source',\n    width=1000, \n    height=1000,\n    custom_data=['filename', 'primary_label', 'common_name','in_western_ghats', 'tags']\n)\nfig.update_traces(\n    marker=dict(size=2),\n    hovertemplate=\"<br>\".join([\n        \"X: %{x}\",\n        \"Y: %{y}\",\n        \"filename: %{customdata[0]}\",\n        \"primary_label: %{customdata[1]}\",\n        \"common_name: %{customdata[2]}\",\n        \"in_western_ghats: %{customdata[3]}\",\n        \"tags: %{customdata[4]}\",\n    ])\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:31:21.844609Z","iopub.execute_input":"2024-06-08T09:31:21.845018Z","iopub.status.idle":"2024-06-08T09:31:23.329227Z","shell.execute_reply.started":"2024-06-08T09:31:21.844975Z","shell.execute_reply":"2024-06-08T09:31:23.326831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Recording Investigations","metadata":{}},{"cell_type":"code","source":"filenames = [\"/kaggle/input/birdclef-2024/unlabeled_soundscapes/658548753.ogg\",\n            \"/kaggle/input/birdclef-2024/unlabeled_soundscapes/1895975685.ogg\",\n             \"/kaggle/input/birdclef-2024/train_audio/whbtre1/XC665707.ogg\",\n             \"/kaggle/input/birdclef-2024/unlabeled_soundscapes/978841862.ogg\",\n             \"/kaggle/input/birdclef-2024/unlabeled_soundscapes/1530466624.ogg\",\n             \"/kaggle/input/birdclef-2024/unlabeled_soundscapes/2000870529.ogg\",\n             \"/kaggle/input/birdclef-2024/train_audio/plapri1/XC508287.ogg\",\n             \"/kaggle/input/birdclef-2024/train_audio/rorpar/XC780690.ogg\",\n             \"/kaggle/input/birdclef-2024/unlabeled_soundscapes/1207666989.ogg\",\n             \"/kaggle/input/birdclef-2024/unlabeled_soundscapes/974332807.ogg\",\n             \"/kaggle/input/birdclef-2024/train_audio/bkskit1/XC198489.ogg\",\n             \"/kaggle/input/birdclef-2024/train_audio/nutman/XC204627.ogg\",\n            ]\nnotes = [\"top left blob unlabeled soundscape: clicking, crickets, insects\",\n         \"top left blob unlabeled soundscape: clicking, crickets, insects\",\n         \"top left blob training data: overwhelmingly crickets, but some other weird instrument sound too\",\n        \n         \"top center blob unlabeled soundscape: insects, possibly train horn in distance\",\n         \n         \"bottom center stray diagonal unlabeled soundscape: static\",\n         \"bottom center stray diagonal unlabeled soundscape: static\",\n         \"bottom center stray diagonal training data: valid bird, but repeats organically like static\",\n         \"far left island training data: bird repeats like consistent quarter notes\",\n         \"top center near core blob: sounds like a bird with long call, but overlaid with insects, ambient people noise\",\n        \"top center near core  blob: birds, but really loud ambiance\",\n        'lower right intersection of green/blue: bird, mostly backed by insects', \"lower right intersection of green/blue: sounds like multiple birds, noisy\"]\n# need to import freefield sounds, but the tags match to the nearby noisy data\nfor i in range(0,len(filenames)):\n    print(notes[i])\n    audio = torchaudio.load(filenames[i], 0, 0+32_000*5)[0][0]\n    display(Audio(audio, rate=SAMPLE_RATE))\n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:31:23.331390Z","iopub.execute_input":"2024-06-08T09:31:23.331786Z","iopub.status.idle":"2024-06-08T09:31:23.787093Z","shell.execute_reply.started":"2024-06-08T09:31:23.331748Z","shell.execute_reply":"2024-06-08T09:31:23.785599Z"},"trusted":true},"execution_count":null,"outputs":[]}]}