{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":309742,"sourceType":"modelInstanceVersion","modelInstanceId":262853,"modelId":283974},{"sourceId":322384,"sourceType":"modelInstanceVersion","modelInstanceId":271686,"modelId":292676}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from tensorflow.keras.models import load_model\nimport os\nimport pandas as pd\nimport librosa\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom tqdm.notebook import tqdm\nimport numpy as np\ntqdm.pandas()\nfrom pathlib import Path\n\n\nmodel = load_model('/kaggle/input/bird_classification_sk/keras/default/1/bird_classification_nw.h5')\n\nIDX_TO_LABEL = sorted(pd.read_csv('/kaggle/input/birdclef-2025/train.csv').primary_label.unique())\n\ndf_taxonomy = pd.read_csv('/kaggle/input/birdclef-2025/taxonomy.csv')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-17T17:15:33.004395Z","iopub.execute_input":"2025-04-17T17:15:33.004722Z","iopub.status.idle":"2025-04-17T17:15:33.224292Z","shell.execute_reply.started":"2025-04-17T17:15:33.004699Z","shell.execute_reply":"2025-04-17T17:15:33.223543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T17:15:33.381755Z","iopub.execute_input":"2025-04-17T17:15:33.382176Z","iopub.status.idle":"2025-04-17T17:15:33.402905Z","shell.execute_reply.started":"2025-04-17T17:15:33.382142Z","shell.execute_reply":"2025-04-17T17:15:33.401932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# SAMPLING_RATE = 32000\n\n# test_audio_dir = '../input/birdclef-2025/test_soundscapes'\n# test_soundscape_path = '/kaggle/input/birdclef-2025/test_soundscapes/'\n# test_soundscapes = [os.path.join(test_soundscape_path, afile) for afile in sorted(os.listdir(test_soundscape_path)) if afile.endswith('.ogg')]\n# # test_audio_dir = '/kaggle/input/birdclef-2025/train_soundscapes/'\n# file_list = [f for f in sorted(os.listdir(test_audio_dir))]\n# ogg_files = [file.split('.')[0] for file in file_list if file.endswith('.ogg')]\n# # file_list = file_list[:10]\n# debug = False\n# print(len(test_soundscapes))\n# if len(ogg_files) == 0:\n#     debug = True\n#     debug_st_num = 1\n#     debug_num = 1\n#     test_audio_dir = '/kaggle/input/birdclef-2025/train_soundscapes/'\n#     file_list = [f for f in sorted(os.listdir(test_audio_dir))]\n#     file_list = [file.split('.')[0] for file in file_list if file.endswith('.ogg')]\n#     ogg_files = file_list[debug_st_num:debug_st_num+debug_num]\n# test_data = []\n\n# from sklearn.preprocessing import LabelEncoder\n\n# df = pd.read_csv(\"/kaggle/input/birdclef-2025/train.csv\")\n\n# label_encoder = LabelEncoder()\n# y = label_encoder.fit_transform(df['primary_label'])\n# actual_class_names = label_encoder.classes_\n\n# print(ogg_files)\n\n\n# for file in ogg_files:\n#     print(file)\n#     try:\n#         audio,_ = librosa.load(f\"{test_audio_dir}{file}.ogg\", duration=10)\n#         mfccs = np.mean(librosa.feature.mfcc(y=audio, sr=SAMPLING_RATE, n_mfcc=40).T,axis=0)\n#         mfccs_1 = [0]*40\n#     except Exception(e):\n#         mfccs_1 = [0]*40\n    \n#     test_data.append(mfccs_1)\n#     # predictions = model_1.predict(mfccs)4\n\n# X_test = np.array(test_data)\n# print(X_test.shape)\n\n# # X_test = X_test.reshape(X_test.shape[0], -1)\n# # try:\n# probabilities = model.predict(X_test)\n\n\n# # Create a DataFrame with file names and predicted probabilities for each class\n# df_predictions = pd.DataFrame(probabilities, columns=actual_class_names)\n\n# df_predictions['row_id'] = ogg_files\n\n# df_predictions = df_predictions[['row_id'] + [col for col in df_predictions.columns if col != 'row_id']]\n\n# print(\"Saving Submission File\")\n# df_predictions.to_csv(\"/kaggle/working/submission.csv\", index=False)\n# df_predictions.to_csv(\"submission.csv\", index=False)\n# print(\"Submission file saved successfully\")\n# # except Exception as e:\n# #     print(e)\n# #     print(\"No Test sound files found\")\n# #     columns = [\"row_id\"] + actual_class_names.tolist()\n# #     df_predictions = pd.DataFrame(columns=columns)\n    \n# #     df_predictions.to_csv(\"/kaggle/working/submission.csv\", index=False)\n# #     df_predictions.to_csv(\"submission.csv\", index=False)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T06:48:54.536523Z","iopub.execute_input":"2025-04-14T06:48:54.536842Z","iopub.status.idle":"2025-04-14T06:48:54.541313Z","shell.execute_reply.started":"2025-04-14T06:48:54.536819Z","shell.execute_reply":"2025-04-14T06:48:54.540194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SAMPLING_RATE = 32000\ntest_audio_dir = '/kaggle/input/birdclef-2025/test_soundscapes/'\nfile_list = [f for f in sorted(os.listdir(test_audio_dir))]\nogg_files = [file.split('.')[0] for file in file_list if file.endswith('.ogg')]\ntest_soundscape_path = '/kaggle/input/birdclef-2025/test_soundscapes/'\ntest_soundscapes = [os.path.join(test_soundscape_path, afile) for afile in sorted(os.listdir(test_soundscape_path)) if afile.endswith('.ogg')]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T17:15:43.694323Z","iopub.execute_input":"2025-04-17T17:15:43.694600Z","iopub.status.idle":"2025-04-17T17:15:43.705900Z","shell.execute_reply.started":"2025-04-17T17:15:43.694579Z","shell.execute_reply":"2025-04-17T17:15:43.705078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.preprocessing import LabelEncoder\n\n# df = pd.read_csv(\"/kaggle/input/birdclef-2025/train.csv\")\n\n# label_encoder = LabelEncoder()\n# y = label_encoder.fit_transform(df['primary_label'])\n# actual_class_names = label_encoder.classes_\n\n# # test_data = [[0]*40]\n# # X_test = np.array(test_data)\n# # probabilities = [[1]*206]\n\n# probabilities = []\n# # test_soundscapes = test_soundscapes[1:10]\n# for file in test_soundscapes:\n#     audio,_ = librosa.load(file, duration=10)\n#     print(file)\n#     # data = [0.81]*206\n#     data = np.random.rand(206)\n#     probabilities.append(data)\n\n\n# df_predictions = pd.DataFrame(probabilities, columns=actual_class_names)\n# df_predictions['row_id'] = ogg_files\n# df_predictions = df_predictions[['row_id'] + [col for col in df_predictions.columns if col != 'row_id']]\n\n# print(\"Saving Submission File\")\n# # df_predictions.to_csv(\"submission.csv\", index=False)\n\n# df_predictions.to_csv('submission.csv', index=False)\n# df_predictions.head()\n\n# test_data = []\n# if len(test_soundscapes) !=0:\n#     for file in test_soundscapes:\n#         mfccs = [0]*40\n#         test_data.append(mfccs)\n# else:\n#     test_data = [[0]*40]\n# X_test = np.array(test_data)\n\n# probabilities = model.predict(X_test)\n\n# df_predictions = pd.DataFrame(probabilities, columns=actual_class_names)\n# if len(test_soundscapes) !=0: \n#     df_predictions['row_id'] = ogg_files\n# else:\n#     df_predictions['row_id'] = ['test']\n\n# df_predictions = df_predictions[['row_id'] + [col for col in df_predictions.columns if col != 'row_id']]\n\n# print(\"Saving Submission File\")\n# df_predictions.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:03:27.071731Z","iopub.execute_input":"2025-04-14T07:03:27.072047Z","iopub.status.idle":"2025-04-14T07:03:27.183981Z","shell.execute_reply.started":"2025-04-14T07:03:27.072024Z","shell.execute_reply":"2025-04-14T07:03:27.183061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_labels = sorted(os.listdir('/kaggle/input/birdclef-2025/train_audio/'))\npredictions = pd.DataFrame(columns=['row_id'] + class_labels)\ntest_soundscapes = test_soundscapes\nfor soundscape in test_soundscapes:\n    # Load audio\n    sig, rate = librosa.load(path=soundscape, sr=None)\n\n    # Split into 5-second chunks\n    chunks = []\n    for i in range(0, len(sig), rate*5):\n        chunk = sig[i:i+rate*5]\n        chunks.append(chunk)\n     \n    # Make predictions for each chunk\n    for i, chunk in enumerate(chunks):\n        try:\n            # Get row id  (soundscape id + end time of 5s chunk)      \n            row_id = os.path.basename(soundscape).split('.')[0] + f'_{i * 5 + 5}'\n            # mfccs = np.mean(librosa.feature.mfcc(y=chunk, sr=SAMPLING_RATE, n_mfcc=40).T,axis=0)\n            # Make prediction (let's use random scores for now)\n            # scores = model.predict...\n            # scores = np.random.rand(len(class_labels))\n            scores = model.predict(np.array(mfccs))\n        except:\n            scores = np.random.rand(len(class_labels))\n        \n        # Append to predictions as new row\n        new_row = pd.DataFrame([[row_id] + list(scores)], columns=['row_id'] + class_labels)\n        predictions = pd.concat([predictions, new_row], axis=0, ignore_index=True)\n        \n# Save prediction as csv\npredictions.to_csv('submission.csv', index=False)\npredictions.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T17:15:47.131970Z","iopub.execute_input":"2025-04-17T17:15:47.132336Z","iopub.status.idle":"2025-04-17T17:15:47.163199Z","shell.execute_reply.started":"2025-04-17T17:15:47.132309Z","shell.execute_reply":"2025-04-17T17:15:47.162233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T17:15:49.591153Z","iopub.execute_input":"2025-04-17T17:15:49.591465Z","iopub.status.idle":"2025-04-17T17:15:49.596882Z","shell.execute_reply.started":"2025-04-17T17:15:49.591439Z","shell.execute_reply":"2025-04-17T17:15:49.595881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}