{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Import Libraries**","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\nimport soundfile as sf\nfrom sklearn.metrics import roc_auc_score, roc_curve\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.model_selection import train_test_split\nimport xgboost as xgb \nimport lightgbm as lgb  \n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T14:44:47.497828Z","iopub.execute_input":"2025-04-05T14:44:47.498310Z","iopub.status.idle":"2025-04-05T14:44:47.504634Z","shell.execute_reply.started":"2025-04-05T14:44:47.498261Z","shell.execute_reply":"2025-04-05T14:44:47.503416Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Load Data**","metadata":{}},{"cell_type":"code","source":"# Define paths\nINPUT_PATH = '/kaggle/input/birdclef-2025/'\nTRAIN_AUDIO_PATH = os.path.join(INPUT_PATH, 'train_audio')\nTEST_SOUNDSCAPES_PATH = os.path.join(INPUT_PATH, 'test_soundscapes')\nTRAIN_SOUNDSCAPES_PATH = os.path.join(INPUT_PATH, 'train_soundscapes')\n\n# Load data\ntaxonomy = pd.read_csv(os.path.join(INPUT_PATH, 'taxonomy.csv'))\ntrain_meta = pd.read_csv(os.path.join(INPUT_PATH, 'train.csv'))\nsample_submission = pd.read_csv(os.path.join(INPUT_PATH, 'sample_submission.csv'))\nrecording_locations = pd.read_csv(os.path.join(INPUT_PATH, 'recording_location.txt'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T14:44:47.506068Z","iopub.execute_input":"2025-04-05T14:44:47.506405Z","iopub.status.idle":"2025-04-05T14:44:47.671694Z","shell.execute_reply.started":"2025-04-05T14:44:47.506380Z","shell.execute_reply":"2025-04-05T14:44:47.670559Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Data Preprocessing**","metadata":{}},{"cell_type":"code","source":"# Data Preprocessing\ndef preprocess_train_meta(df):\n    \"\"\"Preprocesses the training metadata.\"\"\"\n    df['secondary_labels'] = df['secondary_labels'].apply(lambda x: re.findall(r\"'(\\w+)'\", x))\n    df['len_sec_labels'] = df['secondary_labels'].map(len)\n    df['file_path'] = df.apply(lambda row: os.path.join(TRAIN_AUDIO_PATH, row['filename']), axis=1)\n    return df\n\ntrain_meta = preprocess_train_meta(train_meta)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T14:44:47.673556Z","iopub.execute_input":"2025-04-05T14:44:47.673863Z","iopub.status.idle":"2025-04-05T14:44:47.932257Z","shell.execute_reply.started":"2025-04-05T14:44:47.673840Z","shell.execute_reply":"2025-04-05T14:44:47.931332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print train_meta shape\nprint(\"Train Meta Shape:\", train_meta.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T14:44:47.933500Z","iopub.execute_input":"2025-04-05T14:44:47.933789Z","iopub.status.idle":"2025-04-05T14:44:47.939018Z","shell.execute_reply.started":"2025-04-05T14:44:47.933765Z","shell.execute_reply":"2025-04-05T14:44:47.937740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print train_meta head with gradient background\nprint(\"Train Meta Head:\")\ndisplay(train_meta.head().style.background_gradient(cmap='YlOrBr'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T14:44:47.940384Z","iopub.execute_input":"2025-04-05T14:44:47.940671Z","iopub.status.idle":"2025-04-05T14:44:47.976976Z","shell.execute_reply.started":"2025-04-05T14:44:47.940647Z","shell.execute_reply":"2025-04-05T14:44:47.975861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print taxonomy head with gradient background\nprint(\"Taxonomy Head:\")\ndisplay(taxonomy.head().style.background_gradient(cmap='plasma'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T14:44:47.978143Z","iopub.execute_input":"2025-04-05T14:44:47.978613Z","iopub.status.idle":"2025-04-05T14:44:48.006603Z","shell.execute_reply.started":"2025-04-05T14:44:47.978511Z","shell.execute_reply":"2025-04-05T14:44:48.005386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print recording_locations head with gradient background\nprint(\"Recording Locations Head:\")\ndisplay(recording_locations.head().style.background_gradient(cmap='plasma'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T14:44:55.944516Z","iopub.execute_input":"2025-04-05T14:44:55.944904Z","iopub.status.idle":"2025-04-05T14:44:55.955582Z","shell.execute_reply.started":"2025-04-05T14:44:55.944876Z","shell.execute_reply":"2025-04-05T14:44:55.954358Z"}},"outputs":[],"execution_count":null}]}