{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19596,"databundleVersionId":1292430,"sourceType":"competition"},{"sourceId":1264575,"sourceType":"datasetVersion","datasetId":725893},{"sourceId":1262046,"sourceType":"datasetVersion","datasetId":726424}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-15T18:19:01.992812Z","iopub.execute_input":"2024-01-15T18:19:01.993329Z","iopub.status.idle":"2024-01-15T18:19:02.584081Z","shell.execute_reply.started":"2024-01-15T18:19:01.993292Z","shell.execute_reply":"2024-01-15T18:19:02.583103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport audioread\nimport logging\nimport os\nimport random\nimport time\nimport warnings\n\nimport librosa\nimport numpy as np\nimport pandas as pd\nimport soundfile as sf\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.utils.data as data\n\nfrom contextlib import contextmanager\nfrom pathlib import Path\nfrom typing import Optional\n\nfrom fastprogress import progress_bar\nfrom sklearn.metrics import f1_score\nfrom torchvision import models","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:02.586346Z","iopub.execute_input":"2024-01-15T18:19:02.587148Z","iopub.status.idle":"2024-01-15T18:19:02.595052Z","shell.execute_reply.started":"2024-01-15T18:19:02.587104Z","shell.execute_reply":"2024-01-15T18:19:02.593934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# # **EDA**","metadata":{}},{"cell_type":"code","source":"df=pd.read_csv('/kaggle/input/birdsong-recognition/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:02.596356Z","iopub.execute_input":"2024-01-15T18:19:02.597272Z","iopub.status.idle":"2024-01-15T18:19:03.007269Z","shell.execute_reply.started":"2024-01-15T18:19:02.597233Z","shell.execute_reply":"2024-01-15T18:19:03.005550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.010367Z","iopub.execute_input":"2024-01-15T18:19:03.010993Z","iopub.status.idle":"2024-01-15T18:19:03.045621Z","shell.execute_reply.started":"2024-01-15T18:19:03.010960Z","shell.execute_reply":"2024-01-15T18:19:03.044260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.047478Z","iopub.execute_input":"2024-01-15T18:19:03.047956Z","iopub.status.idle":"2024-01-15T18:19:03.056238Z","shell.execute_reply.started":"2024-01-15T18:19:03.047911Z","shell.execute_reply":"2024-01-15T18:19:03.054779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns',None)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.058608Z","iopub.execute_input":"2024-01-15T18:19:03.059094Z","iopub.status.idle":"2024-01-15T18:19:03.066245Z","shell.execute_reply.started":"2024-01-15T18:19:03.059051Z","shell.execute_reply":"2024-01-15T18:19:03.064622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.expand_frame_repr', False)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.068302Z","iopub.execute_input":"2024-01-15T18:19:03.069559Z","iopub.status.idle":"2024-01-15T18:19:03.079806Z","shell.execute_reply.started":"2024-01-15T18:19:03.069501Z","shell.execute_reply":"2024-01-15T18:19:03.078291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.081498Z","iopub.execute_input":"2024-01-15T18:19:03.082000Z","iopub.status.idle":"2024-01-15T18:19:03.135299Z","shell.execute_reply.started":"2024-01-15T18:19:03.081956Z","shell.execute_reply":"2024-01-15T18:19:03.133644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.137339Z","iopub.execute_input":"2024-01-15T18:19:03.137831Z","iopub.status.idle":"2024-01-15T18:19:03.236451Z","shell.execute_reply.started":"2024-01-15T18:19:03.137785Z","shell.execute_reply":"2024-01-15T18:19:03.235150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.columns[df.isnull().any()])","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.242433Z","iopub.execute_input":"2024-01-15T18:19:03.242850Z","iopub.status.idle":"2024-01-15T18:19:03.324672Z","shell.execute_reply.started":"2024-01-15T18:19:03.242816Z","shell.execute_reply":"2024-01-15T18:19:03.323401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['rating'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.326810Z","iopub.execute_input":"2024-01-15T18:19:03.327341Z","iopub.status.idle":"2024-01-15T18:19:03.338501Z","shell.execute_reply.started":"2024-01-15T18:19:03.327298Z","shell.execute_reply":"2024-01-15T18:19:03.337418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.playback_used.unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.339980Z","iopub.execute_input":"2024-01-15T18:19:03.340419Z","iopub.status.idle":"2024-01-15T18:19:03.351255Z","shell.execute_reply.started":"2024-01-15T18:19:03.340386Z","shell.execute_reply":"2024-01-15T18:19:03.350254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.playback_used.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.352626Z","iopub.execute_input":"2024-01-15T18:19:03.353273Z","iopub.status.idle":"2024-01-15T18:19:03.368681Z","shell.execute_reply.started":"2024-01-15T18:19:03.353208Z","shell.execute_reply":"2024-01-15T18:19:03.367250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.playback_used.fillna('no',inplace = True)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.369952Z","iopub.execute_input":"2024-01-15T18:19:03.370304Z","iopub.status.idle":"2024-01-15T18:19:03.383680Z","shell.execute_reply.started":"2024-01-15T18:19:03.370266Z","shell.execute_reply":"2024-01-15T18:19:03.382142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.playback_used.unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.385026Z","iopub.execute_input":"2024-01-15T18:19:03.385799Z","iopub.status.idle":"2024-01-15T18:19:03.397919Z","shell.execute_reply.started":"2024-01-15T18:19:03.385764Z","shell.execute_reply":"2024-01-15T18:19:03.396808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.ebird_code.unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.399622Z","iopub.execute_input":"2024-01-15T18:19:03.400517Z","iopub.status.idle":"2024-01-15T18:19:03.412124Z","shell.execute_reply.started":"2024-01-15T18:19:03.400443Z","shell.execute_reply":"2024-01-15T18:19:03.410908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.ebird_code.nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.413999Z","iopub.execute_input":"2024-01-15T18:19:03.414625Z","iopub.status.idle":"2024-01-15T18:19:03.423817Z","shell.execute_reply.started":"2024-01-15T18:19:03.414591Z","shell.execute_reply":"2024-01-15T18:19:03.422368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.channels.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.425723Z","iopub.execute_input":"2024-01-15T18:19:03.426336Z","iopub.status.idle":"2024-01-15T18:19:03.441925Z","shell.execute_reply.started":"2024-01-15T18:19:03.426301Z","shell.execute_reply":"2024-01-15T18:19:03.440441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.channels.unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.443318Z","iopub.execute_input":"2024-01-15T18:19:03.444357Z","iopub.status.idle":"2024-01-15T18:19:03.453686Z","shell.execute_reply.started":"2024-01-15T18:19:03.444318Z","shell.execute_reply":"2024-01-15T18:19:03.452619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.date.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.455176Z","iopub.execute_input":"2024-01-15T18:19:03.455961Z","iopub.status.idle":"2024-01-15T18:19:03.476622Z","shell.execute_reply.started":"2024-01-15T18:19:03.455914Z","shell.execute_reply":"2024-01-15T18:19:03.474883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[['year', 'month', 'day']] = df['date'].str.split('-', expand=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.478872Z","iopub.execute_input":"2024-01-15T18:19:03.479290Z","iopub.status.idle":"2024-01-15T18:19:03.543671Z","shell.execute_reply.started":"2024-01-15T18:19:03.479254Z","shell.execute_reply":"2024-01-15T18:19:03.542561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.545284Z","iopub.execute_input":"2024-01-15T18:19:03.546096Z","iopub.status.idle":"2024-01-15T18:19:03.595590Z","shell.execute_reply.started":"2024-01-15T18:19:03.546063Z","shell.execute_reply":"2024-01-15T18:19:03.594455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df = df.drop(['year', 'day'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.597213Z","iopub.execute_input":"2024-01-15T18:19:03.597563Z","iopub.status.idle":"2024-01-15T18:19:03.601497Z","shell.execute_reply.started":"2024-01-15T18:19:03.597533Z","shell.execute_reply":"2024-01-15T18:19:03.600570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.602558Z","iopub.execute_input":"2024-01-15T18:19:03.602878Z","iopub.status.idle":"2024-01-15T18:19:03.655443Z","shell.execute_reply.started":"2024-01-15T18:19:03.602849Z","shell.execute_reply":"2024-01-15T18:19:03.654237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.month.unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.657037Z","iopub.execute_input":"2024-01-15T18:19:03.658066Z","iopub.status.idle":"2024-01-15T18:19:03.669968Z","shell.execute_reply.started":"2024-01-15T18:19:03.657994Z","shell.execute_reply":"2024-01-15T18:19:03.668598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['month'] = df['month'].replace('00', '05')","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.671365Z","iopub.execute_input":"2024-01-15T18:19:03.671899Z","iopub.status.idle":"2024-01-15T18:19:03.686677Z","shell.execute_reply.started":"2024-01-15T18:19:03.671867Z","shell.execute_reply":"2024-01-15T18:19:03.685446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.pitch.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.688262Z","iopub.execute_input":"2024-01-15T18:19:03.689136Z","iopub.status.idle":"2024-01-15T18:19:03.706393Z","shell.execute_reply.started":"2024-01-15T18:19:03.689101Z","shell.execute_reply":"2024-01-15T18:19:03.705265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.pitch.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.714460Z","iopub.execute_input":"2024-01-15T18:19:03.715426Z","iopub.status.idle":"2024-01-15T18:19:03.728164Z","shell.execute_reply.started":"2024-01-15T18:19:03.715388Z","shell.execute_reply":"2024-01-15T18:19:03.726805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.speed.value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.730434Z","iopub.execute_input":"2024-01-15T18:19:03.730867Z","iopub.status.idle":"2024-01-15T18:19:03.744537Z","shell.execute_reply.started":"2024-01-15T18:19:03.730834Z","shell.execute_reply":"2024-01-15T18:19:03.743181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.species.value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.745973Z","iopub.execute_input":"2024-01-15T18:19:03.746389Z","iopub.status.idle":"2024-01-15T18:19:03.762694Z","shell.execute_reply.started":"2024-01-15T18:19:03.746356Z","shell.execute_reply":"2024-01-15T18:19:03.761356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.bird_seen.value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.766721Z","iopub.execute_input":"2024-01-15T18:19:03.767673Z","iopub.status.idle":"2024-01-15T18:19:03.780441Z","shell.execute_reply.started":"2024-01-15T18:19:03.767609Z","shell.execute_reply":"2024-01-15T18:19:03.779618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.bird_seen.fillna('yes', inplace=True)\ndf.bird_seen.unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.781753Z","iopub.execute_input":"2024-01-15T18:19:03.782090Z","iopub.status.idle":"2024-01-15T18:19:03.795598Z","shell.execute_reply.started":"2024-01-15T18:19:03.782060Z","shell.execute_reply":"2024-01-15T18:19:03.794247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.sampling_rate.unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.796958Z","iopub.execute_input":"2024-01-15T18:19:03.797900Z","iopub.status.idle":"2024-01-15T18:19:03.807546Z","shell.execute_reply.started":"2024-01-15T18:19:03.797866Z","shell.execute_reply":"2024-01-15T18:19:03.806279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop('sampling_rate', axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.808968Z","iopub.execute_input":"2024-01-15T18:19:03.809364Z","iopub.status.idle":"2024-01-15T18:19:03.827895Z","shell.execute_reply.started":"2024-01-15T18:19:03.809333Z","shell.execute_reply":"2024-01-15T18:19:03.826923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.829093Z","iopub.execute_input":"2024-01-15T18:19:03.830146Z","iopub.status.idle":"2024-01-15T18:19:03.878575Z","shell.execute_reply.started":"2024-01-15T18:19:03.830104Z","shell.execute_reply":"2024-01-15T18:19:03.877353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" # df = df.drop('type', axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.880144Z","iopub.execute_input":"2024-01-15T18:19:03.880771Z","iopub.status.idle":"2024-01-15T18:19:03.886947Z","shell.execute_reply.started":"2024-01-15T18:19:03.880735Z","shell.execute_reply":"2024-01-15T18:19:03.885405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.elevation.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.889168Z","iopub.execute_input":"2024-01-15T18:19:03.889624Z","iopub.status.idle":"2024-01-15T18:19:03.909217Z","shell.execute_reply.started":"2024-01-15T18:19:03.889589Z","shell.execute_reply":"2024-01-15T18:19:03.908339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['elevation'] = df['elevation'].str.replace(' m','', regex=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.910408Z","iopub.execute_input":"2024-01-15T18:19:03.911617Z","iopub.status.idle":"2024-01-15T18:19:03.947022Z","shell.execute_reply.started":"2024-01-15T18:19:03.911580Z","shell.execute_reply":"2024-01-15T18:19:03.945796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values_to_replace = ['?', '0', '-', 'Unknown', '', '??']\ndf['elevation'] = df['elevation'].replace(values_to_replace, np.nan)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.948694Z","iopub.execute_input":"2024-01-15T18:19:03.949415Z","iopub.status.idle":"2024-01-15T18:19:03.972948Z","shell.execute_reply.started":"2024-01-15T18:19:03.949379Z","shell.execute_reply":"2024-01-15T18:19:03.971564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['elevation'] = df['elevation'].str.replace('m', '')","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.974874Z","iopub.execute_input":"2024-01-15T18:19:03.975378Z","iopub.status.idle":"2024-01-15T18:19:03.997467Z","shell.execute_reply.started":"2024-01-15T18:19:03.975341Z","shell.execute_reply":"2024-01-15T18:19:03.995530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.elevation.unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:03.999319Z","iopub.execute_input":"2024-01-15T18:19:04.000715Z","iopub.status.idle":"2024-01-15T18:19:04.019703Z","shell.execute_reply.started":"2024-01-15T18:19:04.000671Z","shell.execute_reply":"2024-01-15T18:19:04.018107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"PERFECT","metadata":{}},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:04.021603Z","iopub.execute_input":"2024-01-15T18:19:04.021986Z","iopub.status.idle":"2024-01-15T18:19:04.071443Z","shell.execute_reply.started":"2024-01-15T18:19:04.021955Z","shell.execute_reply":"2024-01-15T18:19:04.070275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns_to_drop = ['description', 'bitrate_of_mp3', 'volume', 'background', 'xc_id', 'url', 'author', 'primary_label', 'recordist']\ndf = df.drop(columns=columns_to_drop)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:04.072867Z","iopub.execute_input":"2024-01-15T18:19:04.073223Z","iopub.status.idle":"2024-01-15T18:19:04.089579Z","shell.execute_reply.started":"2024-01-15T18:19:04.073167Z","shell.execute_reply":"2024-01-15T18:19:04.088295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:04.091680Z","iopub.execute_input":"2024-01-15T18:19:04.092107Z","iopub.status.idle":"2024-01-15T18:19:04.177109Z","shell.execute_reply.started":"2024-01-15T18:19:04.092034Z","shell.execute_reply":"2024-01-15T18:19:04.175640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:04.178783Z","iopub.execute_input":"2024-01-15T18:19:04.179178Z","iopub.status.idle":"2024-01-15T18:19:04.187084Z","shell.execute_reply.started":"2024-01-15T18:19:04.179144Z","shell.execute_reply":"2024-01-15T18:19:04.185782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['length'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:04.189441Z","iopub.execute_input":"2024-01-15T18:19:04.190405Z","iopub.status.idle":"2024-01-15T18:19:04.202408Z","shell.execute_reply.started":"2024-01-15T18:19:04.190365Z","shell.execute_reply":"2024-01-15T18:19:04.201322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['min_value'] = df['length'].str.extract(r'(\\d+)')\ndf['min_value'] = pd.to_numeric(df['min_value'], errors='coerce')\ndf['max_value'] = df['length'].str.extract(r'-(\\d+)')\ndf['max_value'] = pd.to_numeric(df['max_value'], errors='coerce')\ndf['max_value'] = df['max_value'].fillna(3)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:04.203915Z","iopub.execute_input":"2024-01-15T18:19:04.204267Z","iopub.status.idle":"2024-01-15T18:19:04.301268Z","shell.execute_reply.started":"2024-01-15T18:19:04.204236Z","shell.execute_reply":"2024-01-15T18:19:04.300125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['min_value'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:04.302947Z","iopub.execute_input":"2024-01-15T18:19:04.303310Z","iopub.status.idle":"2024-01-15T18:19:04.312904Z","shell.execute_reply.started":"2024-01-15T18:19:04.303280Z","shell.execute_reply":"2024-01-15T18:19:04.311584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['min_value'] = df['min_value'].fillna(0)\n\n# Display value counts for 'min_value'\nprint(df['min_value'].value_counts())\n","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:04.314603Z","iopub.execute_input":"2024-01-15T18:19:04.315048Z","iopub.status.idle":"2024-01-15T18:19:04.328889Z","shell.execute_reply.started":"2024-01-15T18:19:04.315006Z","shell.execute_reply":"2024-01-15T18:19:04.327318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nplt.figure(figsize=(15, 5))\n\nplt.subplot(1, 3, 1)\nsns.histplot(df['rating'], bins=5, kde=True)\nplt.title('Distribution of Ratings')\n\nplt.subplot(1, 3, 2)\ndf['playback_used'].value_counts().plot(kind='bar')\nplt.title('Playback Used')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:04.330783Z","iopub.execute_input":"2024-01-15T18:19:04.331326Z","iopub.status.idle":"2024-01-15T18:19:05.036917Z","shell.execute_reply.started":"2024-01-15T18:19:04.331275Z","shell.execute_reply":"2024-01-15T18:19:05.035723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_month_counts = df.groupby(['month', 'species']).size().reset_index(name='Count')\n\ntop_birds = bird_month_counts.groupby('species')['Count'].sum().nlargest(5).index\n\ntop_bird_month_counts = bird_month_counts[bird_month_counts['species'].isin(top_birds)]\n\nplt.figure(figsize=(14, 8))\nsns.lineplot(x='month', y='Count', hue='species', data=top_bird_month_counts, marker='o', palette='muted')\nplt.xlabel('Month')\nplt.ylabel('Count')\nplt.title('Top 5 Bird Sightings Over Months')\nplt.legend(title='Species', bbox_to_anchor=(1, 1))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:05.038450Z","iopub.execute_input":"2024-01-15T18:19:05.039640Z","iopub.status.idle":"2024-01-15T18:19:05.502442Z","shell.execute_reply.started":"2024-01-15T18:19:05.039587Z","shell.execute_reply":"2024-01-15T18:19:05.501011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Assuming you have a DataFrame named 'your_dataframe' instead of 'data'\nplt.figure(figsize=(6, 6))\nsns.countplot(x='pitch', data=df)\nplt.xticks(rotation=90)\nplt.title('Plot of Pitch')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:05.504301Z","iopub.execute_input":"2024-01-15T18:19:05.505257Z","iopub.status.idle":"2024-01-15T18:19:05.769940Z","shell.execute_reply.started":"2024-01-15T18:19:05.505202Z","shell.execute_reply":"2024-01-15T18:19:05.768690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (16,8))\n\nsns.countplot( x='rating', data = df, order = df['rating'].value_counts().index, palette = 'viridis')\nplt.title(\"Ratings distribution\")\nplt.xlabel(\"Ratings\")\nplt.ylabel(\"Count\")\nplt.xticks(rotation = 45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:05.771677Z","iopub.execute_input":"2024-01-15T18:19:05.772143Z","iopub.status.idle":"2024-01-15T18:19:06.083307Z","shell.execute_reply.started":"2024-01-15T18:19:05.772098Z","shell.execute_reply":"2024-01-15T18:19:06.082024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values_count = df['number_of_notes'].value_counts()\n\ncustom_colors = [(239/255, 71/255, 111/255, 1),\n                 (247/255,140/255,107/255, 1),\n                 (255/255, 209/255, 102//255, 1),\n                 (6/255, 214/255, 160/255, 1),\n                 (17/255, 138/255, 178/255, 1)]\nplt.figure(figsize=(6,6))\nplt.pie(values_count, labels= None, autopct='%1.1f%%', startangle=90, colors=custom_colors, wedgeprops=dict(width=0.3))\n\nplt.gca().set_facecolor('#07384C')\n\nplt.title(f'Distribution of number of notes', fontsize=15)\n\nplt.legend(values_count.index, title='categories', loc='upper right',bbox_to_anchor=(1,0,0.5,1))\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:06.084915Z","iopub.execute_input":"2024-01-15T18:19:06.085294Z","iopub.status.idle":"2024-01-15T18:19:06.363229Z","shell.execute_reply.started":"2024-01-15T18:19:06.085261Z","shell.execute_reply":"2024-01-15T18:19:06.362165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values_count = df['bird_seen'].value_counts()\n\ncustom_colors = [(239/255, 71/255, 111/255, 1),\n                 (247/255,140/255,107/255, 1),\n                 (255/255, 209/255, 102//255, 1),\n                 (6/255, 214/255, 160/255, 1),\n                 (17/255, 138/255, 178/255, 1)]\nplt.figure(figsize=(6,6))\nplt.pie(values_count, labels= None, autopct='%1.1f%%', startangle=90, colors=custom_colors, wedgeprops=dict(width=0.3))\n\nplt.gca().set_facecolor('#07384C')\n\nplt.title(f'Distribution of bird seen', fontsize=15)\n\nplt.legend(values_count.index, title='categories', loc='upper right',bbox_to_anchor=(1,0,0.5,1))\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:06.365912Z","iopub.execute_input":"2024-01-15T18:19:06.366932Z","iopub.status.idle":"2024-01-15T18:19:06.644995Z","shell.execute_reply.started":"2024-01-15T18:19:06.366832Z","shell.execute_reply":"2024-01-15T18:19:06.643445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.year.value_counts().nlargest(5)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:06.647217Z","iopub.execute_input":"2024-01-15T18:19:06.648235Z","iopub.status.idle":"2024-01-15T18:19:06.664986Z","shell.execute_reply.started":"2024-01-15T18:19:06.648156Z","shell.execute_reply":"2024-01-15T18:19:06.664050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values_count = df['year'].value_counts().nlargest(5)\n\ncustom_colors = [(239/255, 71/255, 111/255, 1),\n                 (247/255,140/255,107/255, 1),\n                 (255/255, 209/255, 102//255, 1),\n                 (6/255, 214/255, 160/255, 1),\n                 (17/255, 138/255, 178/255, 1)]\nplt.figure(figsize=(6,6))\nplt.pie(values_count, labels= None, autopct='%1.1f%%', startangle=90, colors=custom_colors, wedgeprops=dict(width=0.3))\n\nplt.gca().set_facecolor('#07384C')\n\nplt.title(f'Distribution of year', fontsize=15)\n\nplt.legend(values_count.index, title='categories', loc='upper right',bbox_to_anchor=(1,0,0.5,1))\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:06.666726Z","iopub.execute_input":"2024-01-15T18:19:06.667546Z","iopub.status.idle":"2024-01-15T18:19:06.963767Z","shell.execute_reply.started":"2024-01-15T18:19:06.667502Z","shell.execute_reply":"2024-01-15T18:19:06.962563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values_count = df['time'].value_counts().nlargest(10)\n\ncustom_colors = [(239/255, 71/255, 111/255, 1),\n                 (247/255,140/255,107/255, 1),\n                 (255/255, 209/255, 102//255, 1),\n                 (6/255, 214/255, 160/255, 1),\n                 (17/255, 138/255, 178/255, 1)]\nplt.figure(figsize=(6,6))\nplt.pie(values_count, labels= None, autopct='%1.1f%%', startangle=90, colors=custom_colors, wedgeprops=dict(width=0.3))\n\nplt.gca().set_facecolor('#07384C')\n\nplt.title(f'Distribution of time of bird seen', fontsize=15)\n\nplt.legend(values_count.index, title='categories', loc='upper right',bbox_to_anchor=(1,0,0.5,1))\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:06.965447Z","iopub.execute_input":"2024-01-15T18:19:06.966524Z","iopub.status.idle":"2024-01-15T18:19:07.366511Z","shell.execute_reply.started":"2024-01-15T18:19:06.966479Z","shell.execute_reply":"2024-01-15T18:19:07.365237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# # **FEATURE ENGINEERING**","metadata":{}},{"cell_type":"code","source":"import librosa\nimport librosa.display\nimport soundfile as sf\nimport matplotlib.pyplot as plt\nimport IPython.display as ipd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:07.368080Z","iopub.execute_input":"2024-01-15T18:19:07.368468Z","iopub.status.idle":"2024-01-15T18:19:07.375102Z","shell.execute_reply.started":"2024-01-15T18:19:07.368436Z","shell.execute_reply":"2024-01-15T18:19:07.373729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio=[]\nsample=[]\n#aldfly\na1_1,sr1_1 = librosa.load(\"/kaggle/input/birdsong-recognition/train_audio/aldfly/XC135454.mp3\")\na1_2,sr1_2 = librosa.load(\"/kaggle/input/birdsong-recognition/train_audio/aldfly/XC135455.mp3\")\naudio.append(a1_1)\nsample.append(sr1_1)\naudio.append(a1_2)\nsample.append(sr1_2)\n\n#ameavo\na1_3,sr1_3 = librosa.load(\"/kaggle/input/birdsong-recognition/train_audio/ameavo/XC133080.mp3\")\na1_4,sr1_4 = librosa.load(\"/kaggle/input/birdsong-recognition/train_audio/ameavo/XC139829.mp3\")\naudio.append(a1_3)\nsample.append(sr1_3)\naudio.append(a1_4)\nsample.append(sr1_4)\n#barswa\na1_5,sr1_5 = librosa.load(\"/kaggle/input/birdsong-recognition/train_audio/barswa/XC134349.mp3\")\na2_1,sr2_1 = librosa.load(\"/kaggle/input/birdsong-recognition/train_audio/barswa/XC139171.mp3\")\naudio.append(a1_5)\nsample.append(sr1_5)\naudio.append(a2_1)\nsample.append(sr2_1)\n#balori\na2_2,sr2_2 = librosa.load(\"/kaggle/input/birdsong-recognition/train_audio/balori/XC101614.mp3\")\na2_3,sr2_3 = librosa.load(\"/kaggle/input/birdsong-recognition/train_audio/balori/XC11476.mp3\")\naudio.append(a2_2)\nsample.append(sr2_2)\naudio.append(a2_3)\nsample.append(sr2_3)\n#ampeip\na2_4,sr2_4 = librosa.load(\"/kaggle/input/birdsong-recognition/train_audio/amepip/XC111043.mp3\")\na2_5,sr2_5 = librosa.load(\"/kaggle/input/birdsong-recognition/train_audio/amepip/XC113721.mp3\")\naudio.append(a2_4)\nsample.append(sr2_4)\naudio.append(a2_5)\nsample.append(sr2_5)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:07.377032Z","iopub.execute_input":"2024-01-15T18:19:07.377437Z","iopub.status.idle":"2024-01-15T18:19:08.048051Z","shell.execute_reply.started":"2024-01-15T18:19:07.377404Z","shell.execute_reply":"2024-01-15T18:19:08.046614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display audio for each pair of a and sr\nfor a, sr in zip(audio, sample):\n    ipd.display(ipd.Audio(data=a, rate=sr))","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:08.049374Z","iopub.execute_input":"2024-01-15T18:19:08.049707Z","iopub.status.idle":"2024-01-15T18:19:08.393039Z","shell.execute_reply.started":"2024-01-15T18:19:08.049679Z","shell.execute_reply":"2024-01-15T18:19:08.392095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport librosa.display\nimport IPython.display as ipd\nimport matplotlib.pyplot as plt\n\n# List of bird names, audio, and sample rate\nbird_data = [\n    {\"name\": \"ALDFLY\", \"audio\": audio[0], \"sr\": sample[0]},\n    {\"name\": \"ALDFLY\", \"audio\": audio[1], \"sr\": sample[1]},\n    {\"name\": \"AMEAVO\", \"audio\": audio[2], \"sr\": sample[2]},\n    {\"name\": \"AMEAVO\", \"audio\": audio[3], \"sr\": sample[3]},\n    {\"name\": \"BARSWA\", \"audio\": audio[4], \"sr\": sample[4]},\n    {\"name\": \"BARSWA\", \"audio\": audio[5], \"sr\": sample[5]},\n    {\"name\": \"BALORI\", \"audio\": audio[6], \"sr\": sample[6]},\n    {\"name\": \"BALORI\", \"audio\": audio[7], \"sr\": sample[7]},\n    {\"name\": \"AMEPIP\", \"audio\": audio[8], \"sr\": sample[8]},\n    {\"name\": \"AMEPIP\", \"audio\": audio[9], \"sr\": sample[9]}\n]\n\nplt.figure(figsize=(20, 12))\n\nfor i, bird_info in enumerate(bird_data, start=1):\n    plt.subplot(2, 5, i)\n    librosa.display.waveshow(bird_info[\"audio\"], sr=bird_info[\"sr\"])\n    plt.title(bird_info[\"name\"])\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:08.394315Z","iopub.execute_input":"2024-01-15T18:19:08.395056Z","iopub.status.idle":"2024-01-15T18:19:21.730710Z","shell.execute_reply.started":"2024-01-15T18:19:08.395021Z","shell.execute_reply":"2024-01-15T18:19:21.729294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,30))\nspec1 = librosa.amplitude_to_db(np.abs(librosa.stft(a1_1)), ref=np.max)\nplt.subplot(5,2,1)\nlibrosa.display.specshow(spec1, sr=sr1_1, x_axis='time', y_axis='log')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"ALDFLY\")\n\nspec1 = librosa.amplitude_to_db(np.abs(librosa.stft(a1_2)), ref=np.max)\nplt.subplot(5,2,2)\nlibrosa.display.specshow(spec1, sr=sr1_2, x_axis='time', y_axis='log')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"ALDFLY\")\n\nspec1 = librosa.amplitude_to_db(np.abs(librosa.stft(a1_3)), ref=np.max)\nplt.subplot(5,2,3)\nlibrosa.display.specshow(spec1, sr=sr1_3, x_axis='time', y_axis='log')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"AMEAVO\")\n\nspec1 = librosa.amplitude_to_db(np.abs(librosa.stft(a1_4)), ref=np.max)\nplt.subplot(5,2,4)\nlibrosa.display.specshow(spec1, sr=sr1_4, x_axis='time', y_axis='log')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"AMEAVO\")\n\nspec1 = librosa.amplitude_to_db(np.abs(librosa.stft(a1_5)), ref=np.max)\nplt.subplot(5,2,5)\nlibrosa.display.specshow(spec1, sr=sr1_5, x_axis='time', y_axis='log')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"BARSWA\")\n\nspec2 = librosa.amplitude_to_db(np.abs(librosa.stft(a2_1)), ref=np.max)\nplt.subplot(5,2,6)\nlibrosa.display.specshow(spec2, sr=sr2_1, x_axis='time', y_axis='log')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"BARSWA\")\n\n\nspec2 = librosa.amplitude_to_db(np.abs(librosa.stft(a2_2)), ref=np.max)\nplt.subplot(5,2,7)\nlibrosa.display.specshow(spec2, sr=sr2_2, x_axis='time', y_axis='log')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"BALORI\")\n\n\nspec2 = librosa.amplitude_to_db(np.abs(librosa.stft(a2_3)), ref=np.max)\nplt.subplot(5,2,8)\nlibrosa.display.specshow(spec2, sr=sr2_3, x_axis='time', y_axis='log')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"BALORI\")\n\n\nspec2 = librosa.amplitude_to_db(np.abs(librosa.stft(a2_4)), ref=np.max)\nplt.subplot(5,2,9)\nlibrosa.display.specshow(spec2, sr=sr2_4, x_axis='time', y_axis='log')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"AMEPIP\")\n\n\nspec2 = librosa.amplitude_to_db(np.abs(librosa.stft(a2_5)), ref=np.max)\nplt.subplot(5,2,10)\nlibrosa.display.specshow(spec2, sr=sr2_5, x_axis='time', y_axis='log')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"AMEPIP\")","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:21.732334Z","iopub.execute_input":"2024-01-15T18:19:21.732701Z","iopub.status.idle":"2024-01-15T18:19:43.363612Z","shell.execute_reply.started":"2024-01-15T18:19:21.732668Z","shell.execute_reply":"2024-01-15T18:19:43.361744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,15))\n\nmfcc1 = librosa.feature.mfcc(y=a1_1, sr=sr1_1)\nplt.subplot(3, 2, 1)\nlibrosa.display.specshow(mfcc1)\nplt.title('MFCC for a1_1')\nplt.ylabel('MFCC')\nplt.xlabel('Time')\nplt.colorbar()\n\nmfcc2 = librosa.feature.mfcc(y=a1_2, sr=sr1_2)\nplt.subplot(3, 2, 2)\nlibrosa.display.specshow(mfcc2)\nplt.title('MFCC for a1_2')\nplt.ylabel('MFCC')\nplt.xlabel('Time')\nplt.colorbar()\n\nmfcc3 = librosa.feature.mfcc(y=a2_1, sr=sr2_1)\nplt.subplot(3, 2, 3)\nlibrosa.display.specshow(mfcc2)\nplt.title('MFCC for a2_1')\nplt.ylabel('MFCC')\nplt.xlabel('Time')\nplt.colorbar()\n\nmfcc4 = librosa.feature.mfcc(y=a2_2, sr=sr2_2)\nplt.subplot(3, 2, 4)\nlibrosa.display.specshow(mfcc4)\nplt.title('MFCC for a2_2')\nplt.ylabel('MFCC')\nplt.xlabel('Time')\nplt.colorbar()\n\nmfcc5 = librosa.feature.mfcc(y=a1_3, sr=sr1_3)\nplt.subplot(3, 2, 5)\nlibrosa.display.specshow(mfcc5)\nplt.title('MFCC for a1_3')\nplt.ylabel('MFCC')\nplt.xlabel('Time')\nplt.colorbar()\n\nmfcc6 = librosa.feature.mfcc(y=a2_3, sr=sr2_3)\nplt.subplot(3, 2, 6)\nlibrosa.display.specshow(mfcc6)\nplt.title('MFCC for a2_3')\nplt.ylabel('MFCC')\nplt.xlabel('Time')\nplt.colorbar()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:43.373736Z","iopub.execute_input":"2024-01-15T18:19:43.374132Z","iopub.status.idle":"2024-01-15T18:19:45.354387Z","shell.execute_reply.started":"2024-01-15T18:19:43.374102Z","shell.execute_reply":"2024-01-15T18:19:45.353124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 20))\n\nplt.subplot(6, 1, 1)\nfor i in range(mfcc1.shape[0]):\n    plt.plot(mfcc1[i], label=f'MFCC {i+1}')\nplt.title('MFCC for a1_1')\n\n\nplt.subplot(6, 1, 2)\nfor i in range(mfcc2.shape[0]):\n    plt.plot(mfcc2[i], label=f'MFCC {i+1}')\nplt.title('MFCC for a1_2')\n\n\nplt.subplot(6, 1, 3)\nfor i in range(mfcc5.shape[0]):\n    plt.plot(mfcc5[i], label=f'MFCC {i+1}')\nplt.title('MFCC for a1_3')\n\n\n\nplt.subplot(6, 1, 4)\nfor i in range(mfcc3.shape[0]):\n    plt.plot(mfcc3[i], label=f'MFCC {i+1}')\nplt.title('MFCC for a2_1')\n\n\nplt.subplot(6, 1, 5)\nfor i in range(mfcc4.shape[0]):\n    plt.plot(mfcc4[i], label=f'MFCC {i+1}')\nplt.title('MFCC for a2_2')\n\n\nplt.subplot(6, 1, 6)\nfor i in range(mfcc6.shape[0]):\n    plt.plot(mfcc6[i], label=f'MFCC {i+1}')\nplt.title('MFCC for a2_3')\n\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:45.355732Z","iopub.execute_input":"2024-01-15T18:19:45.356103Z","iopub.status.idle":"2024-01-15T18:19:47.357657Z","shell.execute_reply.started":"2024-01-15T18:19:45.356071Z","shell.execute_reply":"2024-01-15T18:19:47.356559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n0 = 1\nn1 = 10\nplt.figure(figsize=(20, 30))\nplt.subplot(5,2,1)\nzcr7_1 = librosa.feature.zero_crossing_rate(a1_1)\nzcr7_2 = librosa.feature.zero_crossing_rate(a1_2)\nplt.plot(a1_1[n0:n1], label='aldfly')\nplt.plot(a1_2[n0:n1], label='aldfly')\nplt.title(\"ZCR Comparision\")\nplt.legend()\nplt.grid()\n\n\nplt.subplot(5,2,2)\nzcr7_1 = librosa.feature.zero_crossing_rate(a1_3)\nzcr7_2 = librosa.feature.zero_crossing_rate(a1_4)\nplt.plot(a1_3[n0:n1], label='ameavo')\nplt.plot(a1_4[n0:n1], label='ameavo')\nplt.title(\"ZCR Comparision\")\nplt.legend()\nplt.grid()\n\nplt.subplot(5,2,3)\nzcr7_1 = librosa.feature.zero_crossing_rate(a1_5)\nzcr7_2 = librosa.feature.zero_crossing_rate(a2_1)\nplt.plot(a1_5[n0:n1], label='barswa')\nplt.plot(a2_1[n0:n1], label='barswa')\nplt.title(\"ZCR Comparision\")\nplt.legend()\nplt.grid()\n\nplt.subplot(5,2,4)\nzcr7_1 = librosa.feature.zero_crossing_rate(a2_2)\nzcr7_2 = librosa.feature.zero_crossing_rate(a2_3)\nplt.plot(a2_2[n0:n1], label='balori')\nplt.plot(a2_3[n0:n1], label='balori')\nplt.title(\"ZCR Comparision\")\nplt.legend()\nplt.grid()\n\nplt.subplot(5,2,5)\nzcr7_1 = librosa.feature.zero_crossing_rate(a1_1)\nzcr7_2 = librosa.feature.zero_crossing_rate(a1_2)\nplt.plot(a2_4[n0:n1], label='amepip')\nplt.plot(a2_5[n0:n1], label='amepip')\nplt.title(\"ZCR Comparision\")\nplt.legend()\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:47.359576Z","iopub.execute_input":"2024-01-15T18:19:47.360600Z","iopub.status.idle":"2024-01-15T18:19:49.315680Z","shell.execute_reply.started":"2024-01-15T18:19:47.360550Z","shell.execute_reply":"2024-01-15T18:19:49.314522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# # **MODEL**","metadata":{}},{"cell_type":"code","source":"import cv2\nimport audioread\nimport logging\nimport os\nimport random\nimport time\nimport warnings\n\nimport numpy as np\nimport pandas as pd\nimport soundfile as sf\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.utils.data as data\n\nfrom contextlib import contextmanager\nfrom pathlib import Path\nfrom typing import Optional\n\nfrom fastprogress import progress_bar\nfrom sklearn.metrics import f1_score\nfrom torchvision import models","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:49.317224Z","iopub.execute_input":"2024-01-15T18:19:49.318474Z","iopub.status.idle":"2024-01-15T18:19:49.326146Z","shell.execute_reply.started":"2024-01-15T18:19:49.318436Z","shell.execute_reply":"2024-01-15T18:19:49.324575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**UTILITIES**","metadata":{}},{"cell_type":"code","source":"def set_seed(seed: int = 42):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)  # type: ignore\n    torch.backends.cudnn.deterministic = True  # type: ignore\n    torch.backends.cudnn.benchmark = True  # type: ignore\n    \n    \ndef get_logger(out_file=None):\n    logger = logging.getLogger()\n    formatter = logging.Formatter(\"%(asctime)s - %(levelname)s - %(message)s\")\n    logger.handlers = []\n    logger.setLevel(logging.INFO)\n\n    handler = logging.StreamHandler()\n    handler.setFormatter(formatter)\n    handler.setLevel(logging.INFO)\n    logger.addHandler(handler)\n\n    if out_file is not None:\n        fh = logging.FileHandler(out_file)\n        fh.setFormatter(formatter)\n        fh.setLevel(logging.INFO)\n        logger.addHandler(fh)\n    logger.info(\"logger set up\")\n    return logger\n    \n    \n@contextmanager\ndef timer(name: str, logger: Optional[logging.Logger] = None):\n    t0 = time.time()\n    msg = f\"[{name}] start\"\n    if logger is None:\n        print(msg)\n    else:\n        logger.info(msg)\n    yield\n\n    msg = f\"[{name}] done in {time.time() - t0:.2f} s\"\n    if logger is None:\n        print(msg)\n    else:\n        logger.info(msg)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:49.327691Z","iopub.execute_input":"2024-01-15T18:19:49.328135Z","iopub.status.idle":"2024-01-15T18:19:49.342147Z","shell.execute_reply.started":"2024-01-15T18:19:49.328101Z","shell.execute_reply":"2024-01-15T18:19:49.341261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logger = get_logger(\"main.log\")\nset_seed(1213)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:49.343829Z","iopub.execute_input":"2024-01-15T18:19:49.344607Z","iopub.status.idle":"2024-01-15T18:19:49.363496Z","shell.execute_reply.started":"2024-01-15T18:19:49.344551Z","shell.execute_reply":"2024-01-15T18:19:49.362222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**DATA LOADING**","metadata":{}},{"cell_type":"code","source":"TARGET_SR = 32000\ntest=pd.read_csv(\"/kaggle/input/birdcall-check/test.csv\")\ntest_audio = \"/kaggle/input/birdcall-check/test_audio\"\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:22:49.682997Z","iopub.execute_input":"2024-01-15T18:22:49.683490Z","iopub.status.idle":"2024-01-15T18:22:49.702368Z","shell.execute_reply.started":"2024-01-15T18:22:49.683454Z","shell.execute_reply":"2024-01-15T18:22:49.701242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/birdsong-recognition/sample_submission.csv\")\nsub.to_csv(\"submission.csv\", index=False)  # this will be overwritten if everything goes well","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:49.389112Z","iopub.execute_input":"2024-01-15T18:19:49.389810Z","iopub.status.idle":"2024-01-15T18:19:49.399086Z","shell.execute_reply.started":"2024-01-15T18:19:49.389772Z","shell.execute_reply":"2024-01-15T18:19:49.398070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ResNet(nn.Module):\n    def __init__(self, base_model_name: str, pretrained=False,\n                 num_classes=264):\n        super().__init__()\n        base_model = models.__getattribute__(base_model_name)(\n            pretrained=pretrained)\n        layers = list(base_model.children())[:-2]\n        layers.append(nn.AdaptiveMaxPool2d(1))\n        self.encoder = nn.Sequential(*layers)\n\n        in_features = base_model.fc.in_features\n\n        self.classifier = nn.Sequential(\n            nn.Linear(in_features, 1024), nn.ReLU(), nn.Dropout(p=0.2),\n            nn.Linear(1024, 1024), nn.ReLU(), nn.Dropout(p=0.2),\n            nn.Linear(1024, num_classes))\n\n    def forward(self, x):\n        batch_size = x.size(0)\n        x = self.encoder(x).view(batch_size, -1)\n        x = self.classifier(x)\n        multiclass_proba = F.softmax(x, dim=1)\n        multilabel_proba = F.sigmoid(x)\n        return {\n            \"logits\": x,\n            \"multiclass_proba\": multiclass_proba,\n            \"multilabel_proba\": multilabel_proba\n        }","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:49.401598Z","iopub.execute_input":"2024-01-15T18:19:49.402464Z","iopub.status.idle":"2024-01-15T18:19:49.414858Z","shell.execute_reply.started":"2024-01-15T18:19:49.402417Z","shell.execute_reply":"2024-01-15T18:19:49.413849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**PARAMETER**","metadata":{}},{"cell_type":"code","source":"model_config = {\n    \"base_model_name\": \"resnet50\",\n    \"pretrained\": False,\n    \"num_classes\": 264\n}\n\nmelspectrogram_parameters = {\n    \"n_mels\": 128,\n    \"fmin\": 20,\n    \"fmax\": 16000\n}\n\nweights_path = \"/kaggle/input/birdcall-resnet50-init-weights/best.pth\"","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:49.416689Z","iopub.execute_input":"2024-01-15T18:19:49.417517Z","iopub.status.idle":"2024-01-15T18:19:49.432768Z","shell.execute_reply.started":"2024-01-15T18:19:49.417473Z","shell.execute_reply":"2024-01-15T18:19:49.431292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\ndf = pd.read_csv(\"/kaggle/input/birdsong-recognition/train.csv\")\n\nunique_bird_names = df.ebird_code.unique()\nlabel_encoder = LabelEncoder()\nencoded_labels = label_encoder.fit_transform(unique_bird_names)\nBIRD_CODE = dict(zip(unique_bird_names, encoded_labels))\n\n#for bird_name, label in BIRD_CODE.items():\n#    print(f\"{bird_name}:{label}\")\n\nINV_BIRD_CODE = {v: k for k, v in BIRD_CODE.items()}\n#for bird_name, label in INV_BIRD_CODE.items():\n#    print(f\"{bird_name}:{label}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:49.434864Z","iopub.execute_input":"2024-01-15T18:19:49.435770Z","iopub.status.idle":"2024-01-15T18:19:49.835215Z","shell.execute_reply.started":"2024-01-15T18:19:49.435730Z","shell.execute_reply":"2024-01-15T18:19:49.833713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**DEFINE DATASET**","metadata":{}},{"cell_type":"code","source":"def mono_to_color(X:np.ndarray,mean=None,std=None,norm_max=None,norm_min= None,eps=1e-6):\n    X=np.stack([X,X,X],axis=-1)\n    \n    mean = mean or X.mean()\n    X=X-mean\n    std=std or X.std()\n    Xstd= X/(std+eps)\n    \n    _min,_max= Xstd.min(),Xstd.max()\n    norm_max= norm_max or _max\n    norm_min= norm_min or _min\n    \n    if(_max - _min)>eps:\n        V=Xstd\n        V[V<norm_min]=norm_min\n        V[V>norm_max]=norm_max\n        \n        V=255*(V-norm_min)/(norm_max - norm_min)\n        V=V.astype(np.uint8)\n    else:\n        V=np.zeroes_like(Xstd, dtype=np.uint8)\n    return V\nclass TestDataset(data.Dataset):\n    def __init__(self,df:pd.DataFrame, clip:np.array,img_size=224,melspectrogram_parameters={}):\n        self.df=df\n        self.clip=clip\n        self.img_size= img_size\n        self.melspectrogram_parameters = melspectrogram_parameters\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx: int):\n        SR = 32000\n\n        sample = self.df.loc[idx, :]  # return row\n        site = sample.site\n        row_id = sample.row_id\n\n        if site == \"site_3\":\n            y = self.clip.astype(np.float32)\n            len_y = len(y)\n            start = 0\n            end = SR * 5\n\n            images = []\n            while len_y > start:\n                y_batch = y[start:end].astype(np.float32)\n\n                if len(y_batch) != (SR * 5):\n                    break\n                start = end\n                end = end + SR * 5\n\n                melspec = librosa.feature.melspectrogram(y=y_batch, sr=SR, **self.melspectrogram_parameters)\n                melspec = librosa.power_to_db(melspec).astype(np.float32)\n\n                image = mono_to_color(melspec)\n\n                height, width, _ =image.shape\n\n                image = cv2.resize(image, (int(width * self.img_size / height), self.img_size))\n\n                image = np.moveaxis(image, 2, 0)  # color channel axis to the first dimension\n\n                image = (image / 255.0).astype(np.float32)\n                images.append(image)\n\n            images = np.asarray(images)\n\n            return images, row_id, site\n\n        else:\n\n            end_seconds = int(sample.seconds)\n            start_seconds = int(end_seconds - 5)\n            start_index = SR * start_seconds\n            end_index = SR * end_seconds\n\n            y = self.clip[start_index:end_index].astype(np.float32)\n\n            melspec = librosa.feature.melspectrogram(y=y, sr=SR, **self.melspectrogram_parameters)\n\n            melspec = librosa.power_to_db(melspec).astype(np.float32)\n\n            image = mono_to_color(melspec)\n            height, width, _ =image.shape\n            image = cv2.resize(image,(int(width * self.img_size / height),self.img_size))\n            image = np.moveaxis(image, 2, 0)\n            image = (image / 255.0).astype(np.float32)\n            \n            return image, row_id, site\n        ","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:49.837175Z","iopub.execute_input":"2024-01-15T18:19:49.837607Z","iopub.status.idle":"2024-01-15T18:19:49.861069Z","shell.execute_reply.started":"2024-01-15T18:19:49.837571Z","shell.execute_reply":"2024-01-15T18:19:49.859931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**PREDICTION LOOP**","metadata":{}},{"cell_type":"code","source":"def get_model(config: dict, weights_path: str):\n    model = ResNet(**config)\n\n    # Load the checkpoint and map the tensors to the CPU if CUDA is not available\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    checkpoint = torch.load(weights_path, map_location=device)\n\n    # Load the model's learned parameters\n    model.load_state_dict(checkpoint[\"model_state_dict\"])\n\n    # Move the model to the specified device\n    model.to(device)\n\n    # Set the model to evaluation mode\n    model.eval()\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:49.862361Z","iopub.execute_input":"2024-01-15T18:19:49.862685Z","iopub.status.idle":"2024-01-15T18:19:49.877219Z","shell.execute_reply.started":"2024-01-15T18:19:49.862656Z","shell.execute_reply":"2024-01-15T18:19:49.875739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prediction_for_clip(test_df: pd.DataFrame,\n                        clip: np.ndarray,\n                        model: ResNet,\n                        mel_params: dict,\n                        threshold=0.5):\n\n\n    dataset = TestDataset(df=test_df,\n                          clip=clip,\n                          img_size=224,\n                          melspectrogram_parameters=mel_params)\n\n    loader = data.DataLoader(dataset, batch_size=1, shuffle=False)\n\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n    model.eval()\n\n    prediction_dict = {}\n\n    for image, row_id, site in progress_bar(loader):\n        site = site[0]\n        row_id = row_id[0]\n        if site in {\"site_1\", \"site_2\"}:\n            image = image.to(device)\n\n            with torch.no_grad():\n                prediction = model(image)\n                proba = prediction[\"multilabel_proba\"].detach().cpu().numpy().reshape(-1)\n                \n            events = proba >= threshold\n            labels = np.argwhere(events).reshape(-1).tolist()\n            \n        else:\n    # to avoid prediction on large batch\n            image = image.squeeze(0)\n            batch_size = 16\n            whole_size = image.size(0)\n\n            if whole_size % batch_size == 0:\n                n_iter = whole_size // batch_size\n            else:\n                n_iter = whole_size // batch_size + 1\n\n            all_events = set()\n\n            for batch_i in range(n_iter):\n                batch = image[batch_i * batch_size: (batch_i + 1) * batch_size]\n\n                if batch.ndim == 3:\n                    batch = batch.unsqueeze(0)\n                    \n                batch = batch.to(device)\n                with torch.no_grad():\n                    prediction = model(batch)\n\n                    proba = prediction[\"multilabel_proba\"].detach().cpu().numpy()\n                events = proba >= threshold\n\n                for i in range(len(events)):\n                    event = events[i, :]\n                    labels = np.argwhere(event).reshape(-1).tolist()\n\n                    for label in labels:\n                        all_events.add(label)\n\n            labels=list(all_events)\n            \n        if len(labels)==0:\n            prediction_dict[row_id]=\"nocall\"\n        else:\n            labels_str_list=list(map(lambda x: INV_BIRD_CODE[x], labels))\n            label_string=\" \".join(labels_str_list)\n            prediction_dict[row_id]= label_string\n            \n    return prediction_dict","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:49.879158Z","iopub.execute_input":"2024-01-15T18:19:49.879844Z","iopub.status.idle":"2024-01-15T18:19:49.898457Z","shell.execute_reply.started":"2024-01-15T18:19:49.879809Z","shell.execute_reply":"2024-01-15T18:19:49.897011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prediction(test_df: pd.DataFrame,\n               test_audio: Path,\n               model_config: dict,\n               mel_params: dict,\n               weights_path: str,\n               threshold=0.5):\n    model = get_model(model_config, weights_path)\n    unique_audio_id = test_df.audio_id.unique()\n\n    warnings.filterwarnings(\"ignore\")\n    prediction_dfs = []\n    for audio_id in unique_audio_id:\n        with timer(f\"Loading {audio_id}\", logger):\n            clip, _ = librosa.load(test_audio / (audio_id + \".mp3\"),\n                                   sr=TARGET_SR,\n                                   mono=True)\n        \n        test_df_for_audio_id = test_df.query(\n            f\"audio_id == '{audio_id}'\").reset_index(drop=True)\n        with timer(f\"Prediction on {audio_id}\", logger):\n            prediction_dict = prediction_for_clip(test_df_for_audio_id,\n                                                  clip=clip,\n                                                  model=model,\n                                                  mel_params=mel_params,\n                                                  threshold=threshold)\n        row_id = list(prediction_dict.keys())\n        birds = list(prediction_dict.values())\n        prediction_df = pd.DataFrame({\n            \"row_id\": row_id,\n            \"birds\": birds\n        })\n        prediction_dfs.append(prediction_df)\n    \n    prediction_df = pd.concat(prediction_dfs, axis=0, sort=False).reset_index(drop=True)\n    return prediction_df","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:49.900691Z","iopub.execute_input":"2024-01-15T18:19:49.901279Z","iopub.status.idle":"2024-01-15T18:19:49.916036Z","shell.execute_reply.started":"2024-01-15T18:19:49.901111Z","shell.execute_reply":"2024-01-15T18:19:49.914970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prediction(test_df: pd.DataFrame,\n              test_audio: Path,\n              model_config: dict,\n              mel_params: dict,\n              weights_path: str,\n              threshold=0.5):\n    \n    model = get_model(model_config, weights_path)\n    unique_audio_id = test_df.audio_id.unique()\n\n    warnings.filterwarnings(\"ignore\")\n\n    prediction_dfs = []\n\n    for audio_id in unique_audio_id:\n        with timer(f\"Loading {audio_id}\", logger):\n            clip, _ = librosa.load(Path(test_audio) / (audio_id + \".mp3\"),\n                       sr=TARGET_SR,\n                       mono=True,\n                       res_type=\"scipy\")\n            \n        test_df_for_audio_id = test_df.query(\n            f\"audio_id == '{audio_id}'\").reset_index(drop=True)\n        with timer(f\"Prediction on {audio_id}\", logger):\n            prediction_dict = prediction_for_clip(test_df_for_audio_id,\n                                                  clip=clip,\n                                                  model=model,\n                                                  mel_params=mel_params,\n                                                  threshold=threshold)\n        row_id = list(prediction_dict.keys())\n        birds = list(prediction_dict.values())\n\n        prediction_df = pd.DataFrame({\n           \"row_id\": row_id,\n           \"birds\": birds\n        })\n\n        prediction_dfs.append(prediction_df)\n\n    prediction_df = pd.concat(prediction_dfs, axis=0, sort=False).reset_index(drop=True)\n\n    return prediction_df","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:49.917427Z","iopub.execute_input":"2024-01-15T18:19:49.918714Z","iopub.status.idle":"2024-01-15T18:19:49.933635Z","shell.execute_reply.started":"2024-01-15T18:19:49.918668Z","shell.execute_reply":"2024-01-15T18:19:49.932526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BIRD_CODE = {\n    'aldfly': 0, 'ameavo': 1, 'amebit': 2, 'amecro': 3, 'amegfi': 4,\n    'amekes': 5, 'amepip': 6, 'amered': 7, 'amerob': 8, 'amewig': 9,\n    'amewoo': 10, 'amtspa': 11, 'annhum': 12, 'astfly': 13, 'baisan': 14,\n    'baleag': 15, 'balori': 16, 'banswa': 17, 'barswa': 18, 'bawwar': 19,\n    'belkin1': 20, 'belspa2': 21, 'bewwre': 22, 'bkbcuc': 23, 'bkbmag1': 24,\n    'bkbwar': 25, 'bkcchi': 26, 'bkchum': 27, 'bkhgro': 28, 'bkpwar': 29,\n    'bktspa': 30, 'blkpho': 31, 'blugrb1': 32, 'blujay': 33, 'bnhcow': 34,\n    'boboli': 35, 'bongul': 36, 'brdowl': 37, 'brebla': 38, 'brespa': 39,\n    'brncre': 40, 'brnthr': 41, 'brthum': 42, 'brwhaw': 43, 'btbwar': 44,\n    'btnwar': 45, 'btywar': 46, 'buffle': 47, 'buggna': 48, 'buhvir': 49,\n    'bulori': 50, 'bushti': 51, 'buwtea': 52, 'buwwar': 53, 'cacwre': 54,\n    'calgul': 55, 'calqua': 56, 'camwar': 57, 'cangoo': 58, 'canwar': 59,\n    'canwre': 60, 'carwre': 61, 'casfin': 62, 'caster1': 63, 'casvir': 64,\n    'cedwax': 65, 'chispa': 66, 'chiswi': 67, 'chswar': 68, 'chukar': 69,\n    'clanut': 70, 'cliswa': 71, 'comgol': 72, 'comgra': 73, 'comloo': 74,\n    'commer': 75, 'comnig': 76, 'comrav': 77, 'comred': 78, 'comter': 79,\n    'comyel': 80, 'coohaw': 81, 'coshum': 82, 'cowscj1': 83, 'daejun': 84,\n    'doccor': 85, 'dowwoo': 86, 'dusfly': 87, 'eargre': 88, 'easblu': 89,\n    'easkin': 90, 'easmea': 91, 'easpho': 92, 'eastow': 93, 'eawpew': 94,\n    'eucdov': 95, 'eursta': 96, 'evegro': 97, 'fiespa': 98, 'fiscro': 99,\n    'foxspa': 100, 'gadwal': 101, 'gcrfin': 102, 'gnttow': 103, 'gnwtea': 104,\n    'gockin': 105, 'gocspa': 106, 'goleag': 107, 'grbher3': 108, 'grcfly': 109,\n    'greegr': 110, 'greroa': 111, 'greyel': 112, 'grhowl': 113, 'grnher': 114,\n    'grtgra': 115, 'grycat': 116, 'gryfly': 117, 'haiwoo': 118, 'hamfly': 119,\n    'hergul': 120, 'herthr': 121, 'hoomer': 122, 'hoowar': 123, 'horgre': 124,\n    'horlar': 125, 'houfin': 126, 'houspa': 127, 'houwre': 128, 'indbun': 129,\n    'juntit1': 130, 'killde': 131, 'labwoo': 132, 'larspa': 133, 'lazbun': 134,\n    'leabit': 135, 'leafly': 136, 'leasan': 137, 'lecthr': 138, 'lesgol': 139,\n    'lesnig': 140, 'lesyel': 141, 'lewwoo': 142, 'linspa': 143, 'lobcur': 144,\n    'lobdow': 145, 'logshr': 146, 'lotduc': 147, 'louwat': 148, 'macwar': 149,\n    'magwar': 150, 'mallar3': 151, 'marwre': 152, 'merlin': 153, 'moublu': 154,\n    'mouchi': 155, 'moudov': 156, 'norcar': 157, 'norfli': 158, 'norhar2': 159,\n    'normoc': 160, 'norpar': 161, 'norpin': 162, 'norsho': 163, 'norwat': 164,\n    'nrwswa': 165, 'nutwoo': 166, 'olsfly': 167, 'orcwar': 168, 'osprey': 169,\n    'ovenbi1': 170, 'palwar': 171, 'pasfly': 172, 'pecsan': 173, 'perfal': 174,\n    'phaino': 175, 'pibgre': 176, 'pilwoo': 177, 'pingro': 178, 'pinjay': 179,\n    'pinsis': 180, 'pinwar': 181, 'plsvir': 182, 'prawar': 183, 'purfin': 184,\n    'pygnut': 185, 'rebmer': 186, 'rebnut': 187, 'rebsap': 188, 'rebwoo': 189,\n    'redcro': 190, 'redhea': 191, 'reevir1': 192, 'renpha': 193, 'reshaw': 194,\n    'rethaw': 195, 'rewbla': 196, 'ribgul': 197, 'rinduc': 198, 'robgro': 199,\n    'rocpig': 200, 'rocwre': 201, 'rthhum': 202, 'ruckin': 203, 'rudduc': 204,\n    'rufgro': 205, 'rufhum': 206, 'rusbla': 207, 'sagspa1': 208, 'sagthr': 209,\n    'savspa': 210, 'saypho': 211, 'scatan': 212, 'scoori': 213, 'semplo': 214,\n    'semsan': 215, 'sheowl': 216, 'shshaw': 217, 'snobun': 218, 'snogoo': 219,\n    'solsan': 220, 'sonspa': 221, 'sora': 222, 'sposan': 223, 'spotow': 224,\n    'stejay': 225, 'swahaw': 226, 'swaspa': 227, 'swathr': 228, 'treswa': 229,\n    'truswa': 230, 'tuftit': 231, 'tunswa': 232, 'veery': 233, 'vesspa': 234,\n    'vigswa': 235, 'warvir': 236, 'wesblu': 237, 'wesgre': 238, 'weskin': 239,\n    'wesmea': 240, 'wessan': 241, 'westan': 242, 'wewpew': 243, 'whbnut': 244,\n    'whcspa': 245, 'whfibi': 246, 'whtspa': 247, 'whtswi': 248, 'wilfly': 249,\n    'wilsni1': 250, 'wiltur': 251, 'winwre3': 252, 'wlswar': 253, 'wooduc': 254,\n    'wooscj2': 255, 'woothr': 256, 'y00475': 257, 'yebfly': 258, 'yebsap': 259,\n    'yehbla': 260, 'yelwar': 261, 'yerwar': 262, 'yetvir': 263\n}\n\nINV_BIRD_CODE = {v: k for k, v in BIRD_CODE.items()}\n","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:49.935394Z","iopub.execute_input":"2024-01-15T18:19:49.936304Z","iopub.status.idle":"2024-01-15T18:19:49.965856Z","shell.execute_reply.started":"2024-01-15T18:19:49.936265Z","shell.execute_reply":"2024-01-15T18:19:49.965015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**PREDICTION**","metadata":{}},{"cell_type":"code","source":"submission = prediction(test_df=test,\n                        test_audio=test_audio,\n                        model_config=model_config,\n                        mel_params=melspectrogram_parameters,\n                        weights_path=weights_path,\n                        threshold=0.8)\n\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:19:49.967242Z","iopub.execute_input":"2024-01-15T18:19:49.968400Z","iopub.status.idle":"2024-01-15T18:20:47.775653Z","shell.execute_reply.started":"2024-01-15T18:19:49.968354Z","shell.execute_reply":"2024-01-15T18:20:47.774512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-01-15T18:20:47.776979Z","iopub.execute_input":"2024-01-15T18:20:47.778113Z","iopub.status.idle":"2024-01-15T18:20:47.793828Z","shell.execute_reply.started":"2024-01-15T18:20:47.778063Z","shell.execute_reply":"2024-01-15T18:20:47.792324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}