{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":1071249,"sourceType":"datasetVersion","datasetId":595272}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"text-align: center;\">\n    <img src=\"https://storage.googleapis.com/kaggle-media/competitions/BirdCLEF/sholicola-ian_lockwood_inset.jpeg\" width=\"400\" height=\"300\">\n</div>\n\n<h1><center>🦉BirdCLEF 2024 - Bird call/song recognition🦉</center></h1>\n","metadata":{}},{"cell_type":"markdown","source":"# **Importing all necessary libraries**","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport regex as re\nimport os\n\nimport librosa        # analyzing and processing audio and music signals. It provides tools for feature extraction, time-frequency analysis, and audio playback. \nimport IPython.display as ipd \nimport librosa.display as lid\n\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nfrom plotly.subplots import make_subplots\nimport plotly.express as px\n\nfrom PIL import Image   # Support for opening, manipulating, and saving many different image file formats\nimport os\nfrom sklearn.utils import shuffle\nimport plotly.graph_objects as go\n\ncmap = mpl.colormaps.get_cmap('coolwarm')","metadata":{"execution":{"iopub.status.busy":"2024-05-15T05:31:07.218488Z","iopub.execute_input":"2024-05-15T05:31:07.219356Z","iopub.status.idle":"2024-05-15T05:31:07.760534Z","shell.execute_reply.started":"2024-05-15T05:31:07.219323Z","shell.execute_reply":"2024-05-15T05:31:07.75971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-05-15T04:13:30.02538Z","iopub.execute_input":"2024-05-15T04:13:30.026039Z","iopub.status.idle":"2024-05-15T04:13:30.032307Z","shell.execute_reply.started":"2024-05-15T04:13:30.025984Z","shell.execute_reply":"2024-05-15T04:13:30.031061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **EDA**\n","metadata":{}},{"cell_type":"markdown","source":"# The .csv file 📁\n\n> 📌**Note**:\n* `train.csv` contains information about the audio files available in `train_audio`. It contains 24459 datapoints in 12 unique columns.\n\n<div class=\"alert alert-block alert-info\">\n<b>Note:</b> The TRAIN data has 1 labeled bird species per recording (primary_label). However, in nature usually you can hear tens (even hundreds) of birds in one go, so in TEST set we need to predict 0, 1 or multiple species for one recording. Because of this, in TRAIN we have <code>species</code> column - or <code>primary_label</code> - (main bird) or <code>secondarylabel</code> \n</div>","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/birdclef-2024/train_metadata.csv\")\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:30.033801Z","iopub.execute_input":"2024-05-15T04:13:30.034445Z","iopub.status.idle":"2024-05-15T04:13:30.262494Z","shell.execute_reply.started":"2024-05-15T04:13:30.034328Z","shell.execute_reply":"2024-05-15T04:13:30.261352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The 12 unique columns are:\n* primary_label : species of the bird ,\n* secondary_labels : background/other species heard in the respective audio file,\n* type : song or call,\n* latitude ,longitude : of the audio recording ,\n* scientific_name ,common_name ,\n* author : person who recorded,\n* license ,\n* rating ,\n* url : audio from the website xeno_canto ,\n* filename: audio file in train_audio dir*","metadata":{}},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:30.263904Z","iopub.execute_input":"2024-05-15T04:13:30.264388Z","iopub.status.idle":"2024-05-15T04:13:30.271515Z","shell.execute_reply.started":"2024-05-15T04:13:30.264288Z","shell.execute_reply":"2024-05-15T04:13:30.270281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Primary label exploration**","metadata":{}},{"cell_type":"code","source":"data['primary_label'].value_counts()\n\n# There are 182 unique species across the dataset","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:30.274473Z","iopub.execute_input":"2024-05-15T04:13:30.274885Z","iopub.status.idle":"2024-05-15T04:13:30.295758Z","shell.execute_reply.started":"2024-05-15T04:13:30.274855Z","shell.execute_reply":"2024-05-15T04:13:30.294445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = data['primary_label'].nunique()\nprint(\"Number of species: \", labels)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:30.297381Z","iopub.execute_input":"2024-05-15T04:13:30.297822Z","iopub.status.idle":"2024-05-15T04:13:30.308508Z","shell.execute_reply.started":"2024-05-15T04:13:30.297781Z","shell.execute_reply":"2024-05-15T04:13:30.307057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the number of samples for each species \nspecies = data['primary_label'].value_counts()\nfig = go.Figure(data=[go.Bar(y=species.values, x=species.index)],\n               layout=go.Layout(margin=go.layout.Margin(l=0, r=0, b=10, t=50)))\n# number of recordings for each species in the training metadata\nfig.update_layout(title='Number of traning samples per species')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:30.310168Z","iopub.execute_input":"2024-05-15T04:13:30.31099Z","iopub.status.idle":"2024-05-15T04:13:30.808977Z","shell.execute_reply.started":"2024-05-15T04:13:30.310951Z","shell.execute_reply":"2024-05-15T04:13:30.808156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Average species in the metadata\nspecies = data['primary_label'].value_counts()\nsumall = 0\nfor i in species.values:\n    sumall += i\nsumall/182","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:30.810164Z","iopub.execute_input":"2024-05-15T04:13:30.810867Z","iopub.status.idle":"2024-05-15T04:13:30.822801Z","shell.execute_reply.started":"2024-05-15T04:13:30.810836Z","shell.execute_reply":"2024-05-15T04:13:30.821883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Secondary label exploration**","metadata":{}},{"cell_type":"code","source":"# These secondary label attributes pertain to the background bird calls within the audio data. While not as prominent as the primary call, these secondary calls originate from distinct species.\ndata['secondary_labels'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:30.823954Z","iopub.execute_input":"2024-05-15T04:13:30.824474Z","iopub.status.idle":"2024-05-15T04:13:30.845679Z","shell.execute_reply.started":"2024-05-15T04:13:30.824445Z","shell.execute_reply":"2024-05-15T04:13:30.844267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the number of calls in secondary label of each sample and adding a new column to keep track of it in the data frame\ndata['secondary_labels'] = data['secondary_labels'].astype(str)\ndata['secondary_labels'] = data['secondary_labels'].apply(lambda x: re.findall(r\"'(\\w+)'\", x))\ndata['len_sec_labels'] = data['secondary_labels'].map(len)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:30.847365Z","iopub.execute_input":"2024-05-15T04:13:30.848052Z","iopub.status.idle":"2024-05-15T04:13:31.217849Z","shell.execute_reply.started":"2024-05-15T04:13:30.848009Z","shell.execute_reply":"2024-05-15T04:13:31.216699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[data.len_sec_labels>3].sample(3)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:31.219739Z","iopub.execute_input":"2024-05-15T04:13:31.22018Z","iopub.status.idle":"2024-05-15T04:13:31.250512Z","shell.execute_reply.started":"2024-05-15T04:13:31.22014Z","shell.execute_reply":"2024-05-15T04:13:31.24942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### **This is a configuration class CFG that holds various parameters and settings for a machine learning model,  designed specifically for BirdCLEF 24 dataset.**","metadata":{}},{"cell_type":"code","source":"class CFG:\n    seed = 42\n        # Input image size and batch size\n    img_size = [128, 384]\n#     batch_size = 64\n    \n    # Audio duration, sample rate, and length\n    duration = 15 # second\n    sample_rate = 32000\n    audio_len = duration*sample_rate\n    \n    # STFT parameters\n    nfft = 2028\n    window = 2048\n    hop_length = audio_len // (img_size[1] - 1)\n    fmin = 20\n    fmax = 16000\n    \n    # Data augmentation parameters\n    augment=True\n    \n    # Class Labels for BirdCLEF 24\n    class_names = sorted(os.listdir('/kaggle/input/birdclef-2024/train_audio/')) # Lists all the files and directories in the train_audio directory of the BirdCLEF 24 dataset and it is sorted alphabetically\n    num_classes = len(class_names) #   Determines the number of classes by counting the elements in the class_names list\n    class_labels = list(range(num_classes))  #  Generating a list of numerical labels ranging from 0 to num_classes - 1.\n    label2name = dict(zip(class_labels, class_names)) # mapping numerical labels to class names\n    name2label = {v:k for k,v in label2name.items()}  # mapping class names to numerical labels.\n ","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:31.252434Z","iopub.execute_input":"2024-05-15T04:13:31.252924Z","iopub.status.idle":"2024-05-15T04:13:31.274708Z","shell.execute_reply.started":"2024-05-15T04:13:31.252883Z","shell.execute_reply":"2024-05-15T04:13:31.273534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Adding the corresponding audio file's path for every sample and creating a new column in the df**","metadata":{}},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/birdclef-2024'\ndata = pd.read_csv(f'{BASE_PATH}/train_metadata.csv')\ndata['filepath'] = BASE_PATH + '/train_audio/' + data.filename \ndata['target'] = data.primary_label.map(CFG.name2label)\ndata['filename'] = data.filepath.map(lambda x: x.split('/')[-1])\ndata['xc_id'] = data.filepath.map(lambda x: x.split('/')[-1].split('.')[0])\n\n# Display rwos\nprint(data.filepath[0])\nprint(data.target[24458]) # class index starts from 0 -181 (total of 182 classes)\nprint(data.filename[0])\nprint(data.xc_id[0])\n\ndata.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:31.276429Z","iopub.execute_input":"2024-05-15T04:13:31.277433Z","iopub.status.idle":"2024-05-15T04:13:31.496216Z","shell.execute_reply.started":"2024-05-15T04:13:31.277392Z","shell.execute_reply":"2024-05-15T04:13:31.49476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_audio(filepath):\n    audio, sr = librosa.load(filepath)\n    return audio, sr\n\ndef get_spectrogram(audio):\n    spec = librosa.feature.melspectrogram(y=audio, \n                                   sr=CFG.sample_rate, \n                                   n_mels=256,\n                                   n_fft=2048,\n                                   hop_length=512,\n                                   fmax=CFG.fmax,\n                                   fmin=CFG.fmin,\n                                   )\n    spec = librosa.power_to_db(spec, ref=1.0)\n    min_ = spec.min()\n    max_ = spec.max()\n    if max_ != min_:\n        spec = (spec - min_)/(max_ - min_)\n    return spec\n\ndef display_audio(row):\n    # Caption for viz\n    caption = f'Id: {row.filename} | Name: {row.common_name} | Sci.Name: {row.scientific_name} | Rating: {row.rating}'\n    # Read audio file\n    audio, sr = load_audio(row.filepath)\n    # Keep fixed length audio\n    audio = audio[:CFG.audio_len]\n    # Spectrogram from audio\n    spec = get_spectrogram(audio)\n    # Display audio\n    print(\"# Audio:\")\n    display(ipd.Audio(audio, rate=CFG.sample_rate))\n    print('# Visualization:')\n    fig, ax = plt.subplots(2, 1, figsize=(12, 2*3), sharex=True, tight_layout=True)\n    fig.suptitle(caption)\n    # Waveplot\n    lid.waveshow(audio,\n                 sr=CFG.sample_rate,\n                 ax=ax[0],\n                 color= cmap(0.1))\n    # Specplot\n    lid.specshow(spec, \n                 sr = CFG.sample_rate, \n                 hop_length=512,\n                 n_fft=2048,\n                 fmin=CFG.fmin,\n                 fmax=CFG.fmax,\n                 x_axis = 'time', \n                 y_axis = 'mel',\n                 cmap = 'coolwarm',\n                 ax=ax[1])\n    ax[0].set_xlabel('');\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:31.501778Z","iopub.execute_input":"2024-05-15T04:13:31.502167Z","iopub.status.idle":"2024-05-15T04:13:31.51665Z","shell.execute_reply.started":"2024-05-15T04:13:31.502136Z","shell.execute_reply":"2024-05-15T04:13:31.515403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row = data.iloc[35]\n\n# Display audio\ndisplay_audio(row)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:31.518123Z","iopub.execute_input":"2024-05-15T04:13:31.518671Z","iopub.status.idle":"2024-05-15T04:13:47.591604Z","shell.execute_reply.started":"2024-05-15T04:13:31.518444Z","shell.execute_reply":"2024-05-15T04:13:47.590194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row = data.iloc[120]\n\n# Display audio\ndisplay_audio(row)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:47.593038Z","iopub.execute_input":"2024-05-15T04:13:47.593686Z","iopub.status.idle":"2024-05-15T04:13:49.419428Z","shell.execute_reply.started":"2024-05-15T04:13:47.593647Z","shell.execute_reply":"2024-05-15T04:13:49.418232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row = data.iloc[183]\n\n# Display audio\ndisplay_audio(row)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:49.420647Z","iopub.execute_input":"2024-05-15T04:13:49.420948Z","iopub.status.idle":"2024-05-15T04:13:51.542866Z","shell.execute_reply.started":"2024-05-15T04:13:49.420922Z","shell.execute_reply":"2024-05-15T04:13:51.542037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = data.filepath[10] \npath","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:51.544284Z","iopub.execute_input":"2024-05-15T04:13:51.545095Z","iopub.status.idle":"2024-05-15T04:13:51.552089Z","shell.execute_reply.started":"2024-05-15T04:13:51.545062Z","shell.execute_reply":"2024-05-15T04:13:51.550744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"signal, sr = librosa.load(path)\nprint('Signal:', signal)\nprint('Shape of signal:', np.shape(signal))\nprint('Sampling Rate (KHz):', sr)\nprint('Duration of audio:', np.shape(signal)[0]/sr)\nplt.plot(signal)\nplt.show()\nipd.Audio(path)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:51.553396Z","iopub.execute_input":"2024-05-15T04:13:51.553793Z","iopub.status.idle":"2024-05-15T04:13:51.913809Z","shell.execute_reply.started":"2024-05-15T04:13:51.553754Z","shell.execute_reply":"2024-05-15T04:13:51.91296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_dist(feature):\n    feature_distribution = data[feature].value_counts().sort_index(ascending=False)\n    colors = ['#2E8B57', '#3CB371', '#58D68D', '#74F9A4', '#9EF8C0',\n              '#C7F8DB', '#E6F5D0', '#F9E79F', '#F7C677', '#F7A54A', '#F58231']\n    rating_colors = {rating: color for rating, color in zip(feature_distribution.index, colors)}\n    # Bar plot\n    bar_data = go.Bar(x=feature_distribution.values, y=feature_distribution.index,\n                      orientation='h', marker=dict(color=[rating_colors[rating] for rating in feature_distribution.index]), showlegend=False, name='')\n    # Donut chart\n    labels = feature_distribution.index.tolist()\n    values = feature_distribution.values.tolist()\n    total_count = sum(values)\n    percentages = [(value / total_count) * 100 for value in values]\n    donut_data = go.Pie(labels=labels, values=values, marker=dict(colors=[rating_colors[rating] for rating in feature_distribution.index], line=dict(color='#ffffff', width=0.5)), hole=0.55,\n                        showlegend=False, name='', textinfo='label+percent', hoverinfo='label+percent', textposition='inside')\n    fig = make_subplots(rows=1, cols=2, specs=[[{\"type\": \"bar\"}, {\"type\": \"pie\"}]])\n    fig.add_trace(bar_data, row=1, col=1)\n    fig.add_trace(donut_data, row=1, col=2)\n    fig.update_layout(title=f'{feature} Distribution', plot_bgcolor='#1C1D20', paper_bgcolor='#1C1D20',\n                      title_font=dict(size=20, family=\"Lato, sans-serif\"), font=dict(color='#E1B12D'))\n    fig.show()\nplot_dist(\"rating\")","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:15:28.328231Z","iopub.execute_input":"2024-05-15T04:15:28.328711Z","iopub.status.idle":"2024-05-15T04:15:28.651623Z","shell.execute_reply.started":"2024-05-15T04:15:28.328677Z","shell.execute_reply":"2024-05-15T04:15:28.650455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets plot for western ghat\n# boundaries for the Western Ghats - as per Google search\nlower_latitude = 8\nupper_latitude = 22\nlower_longitude = 73\nupper_longitude = 79\nfiltered_data = data[(data['latitude'] >= lower_latitude) & \n                    (data['latitude'] <= upper_latitude) &\n                    (data['longitude'] >= lower_longitude) &\n                    (data['longitude'] <= upper_longitude)]\nfig = px.scatter_mapbox(filtered_data, lat=\"latitude\", lon=\"longitude\", color=\"common_name\",zoom=5, center={\"lat\": (lower_latitude + upper_latitude) / 2, \"lon\": (lower_longitude + upper_longitude) / 2})\nfig.update_layout( title=\"Distribution of Bird Species in the Western Ghats\", plot_bgcolor=None, paper_bgcolor=None,title_font=dict(size=20, family=\"Lato, sans-serif\", color='#E1B12D'),  font=dict(color='#E1B12D'),mapbox_style=\"carto-positron\", margin=dict(t=50, r=50, b=50, l=50), hovermode='closest',\nshowlegend=True, legend=dict(title='Bird Species'), legend_title_font=dict(color='#E1B12D'))\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:31:44.79859Z","iopub.execute_input":"2024-05-15T04:31:44.798992Z","iopub.status.idle":"2024-05-15T04:31:45.400064Z","shell.execute_reply.started":"2024-05-15T04:31:44.798961Z","shell.execute_reply":"2024-05-15T04:31:45.398842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport pandas as pd\nfrom IPython.display import Audio    \ncolor_pal = plt.rcParams[\"axes.prop_cycle\"].by_key()[\"color\"]\n\naudio_1, sr_1= librosa.load(\"/kaggle/input/birdclef-2024/train_audio/asbfly/XC134896.ogg\", sr=32000)\nprint(audio_1.shape, sr_1)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T05:34:03.760593Z","iopub.execute_input":"2024-05-15T05:34:03.761007Z","iopub.status.idle":"2024-05-15T05:34:03.799632Z","shell.execute_reply.started":"2024-05-15T05:34:03.760978Z","shell.execute_reply":"2024-05-15T05:34:03.798504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(audio_1).plot(figsize=(10, 5),\n                          lw=1,\n                          title='Raw Audio',\n                         color=color_pal[0])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-15T05:31:16.440012Z","iopub.execute_input":"2024-05-15T05:31:16.440378Z","iopub.status.idle":"2024-05-15T05:31:16.850142Z","shell.execute_reply.started":"2024-05-15T05:31:16.44035Z","shell.execute_reply":"2024-05-15T05:31:16.849131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Audio(data=audio_1, rate=sr_1)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T05:34:06.941435Z","iopub.execute_input":"2024-05-15T05:34:06.941823Z","iopub.status.idle":"2024-05-15T05:34:07.004602Z","shell.execute_reply.started":"2024-05-15T05:34:06.941791Z","shell.execute_reply":"2024-05-15T05:34:07.003256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# UP UNTIL HERE","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport pandas as pd\n\ndef birds_stratified_split(df, target_col, test_size=0.2):\n    class_counts = df[target_col].value_counts()\n    low_count_classes = class_counts[class_counts < 2].index.tolist() ### Birds with single counts\n\n    df['train'] = df[target_col].isin(low_count_classes)\n\n    train_df, val_df = train_test_split(df[~df['train']], test_size=test_size, stratify=df[~df['train']][target_col], random_state=42)\n\n    train_df = pd.concat([train_df, df[df['train']]], axis=0).reset_index(drop=True)\n\n    # Remove the 'valid' column\n    train_df.drop('train', axis=1, inplace=True)\n    val_df.drop('train', axis=1, inplace=True)\n\n    return train_df, val_df","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.080587Z","iopub.status.idle":"2024-05-15T04:13:52.08098Z","shell.execute_reply.started":"2024-05-15T04:13:52.080789Z","shell.execute_reply":"2024-05-15T04:13:52.080805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data, valid_data = birds_stratified_split(data, 'primary_label', 0.2)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.084089Z","iopub.status.idle":"2024-05-15T04:13:52.08505Z","shell.execute_reply.started":"2024-05-15T04:13:52.084743Z","shell.execute_reply":"2024-05-15T04:13:52.084769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.086665Z","iopub.status.idle":"2024-05-15T04:13:52.087431Z","shell.execute_reply.started":"2024-05-15T04:13:52.087104Z","shell.execute_reply":"2024-05-15T04:13:52.087129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.primary_label.value_counts().tail(60)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.089062Z","iopub.status.idle":"2024-05-15T04:13:52.090243Z","shell.execute_reply.started":"2024-05-15T04:13:52.089951Z","shell.execute_reply":"2024-05-15T04:13:52.089975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.primary_label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.091775Z","iopub.status.idle":"2024-05-15T04:13:52.09238Z","shell.execute_reply.started":"2024-05-15T04:13:52.092089Z","shell.execute_reply":"2024-05-15T04:13:52.092112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_data.primary_label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.094193Z","iopub.status.idle":"2024-05-15T04:13:52.094874Z","shell.execute_reply.started":"2024-05-15T04:13:52.094545Z","shell.execute_reply":"2024-05-15T04:13:52.094567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Code adapted from: \n# https://www.kaggle.com/frlemarchand/bird-song-classification-using-an-efficientnet\n# Make sure to check out the entire notebook.\n\n# Load metadata file\ntrain = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv',)\n\n# Limit the number of training samples and classes\n# First, only use high quality samples\ntrain = train.query('rating>=3')\n\n# Second, assume that birds with the most training samples are also the most common\n# A species needs at least 200 recordings with a rating above 4 to be considered common\nbirds_count = {}\nfor bird_species, count in zip(train.primary_label.unique(), \n                               train.groupby('primary_label')['primary_label'].count().values):\n    birds_count[bird_species] = count\nmost_represented_birds = [key for key,value in birds_count.items() if value >= 100] \n\nTRAIN = train.query('primary_label in @most_represented_birds')\nLABELS = sorted(TRAIN.primary_label.unique())\n\n# Let's see how many species and samples we have left\nprint('NUMBER OF SPECIES IN TRAIN DATA:', len(LABELS))\nprint('NUMBER OF SAMPLES IN TRAIN DATA:', len(TRAIN))\nprint('LABELS:', most_represented_birds)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.096182Z","iopub.status.idle":"2024-05-15T04:13:52.096596Z","shell.execute_reply.started":"2024-05-15T04:13:52.096402Z","shell.execute_reply":"2024-05-15T04:13:52.096419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.097617Z","iopub.status.idle":"2024-05-15T04:13:52.098003Z","shell.execute_reply.started":"2024-05-15T04:13:52.097807Z","shell.execute_reply":"2024-05-15T04:13:52.097823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN['primary_label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.099573Z","iopub.status.idle":"2024-05-15T04:13:52.099955Z","shell.execute_reply.started":"2024-05-15T04:13:52.099766Z","shell.execute_reply":"2024-05-15T04:13:52.099783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_samp = TRAIN.sample(10)\ndata_samp.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.101974Z","iopub.status.idle":"2024-05-15T04:13:52.102412Z","shell.execute_reply.started":"2024-05-15T04:13:52.102181Z","shell.execute_reply":"2024-05-15T04:13:52.102198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom os import path\n\nimport numpy as np\nfrom numpy.linalg import norm\n\nimport pandas as pd\n\n# Signal processing\nfrom scipy.io import wavfile\nfrom scipy.interpolate import interp1d\nfrom scipy import signal\nfrom scipy.signal import periodogram\nfrom scipy.signal import find_peaks\n\n# Audio processing\nfrom pydub import AudioSegment\n\n# MFCC\nimport librosa, librosa.display\n\n# Wavelet \nimport pywt\nfrom pywt import wavedec\n\n# Visualization\nimport matplotlib.pyplot as plt\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\nimport IPython\n\n# to display full dataframe information\npd.set_option('display.max_colwidth', None)\n\nfrom PIL import Image\n\nfrom mlxtend.preprocessing import minmax_scaling\n\nimport pickle\n\nfrom sklearn.decomposition import PCA\nfrom sklearn.cross_decomposition import CCA","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.104147Z","iopub.status.idle":"2024-05-15T04:13:52.105998Z","shell.execute_reply.started":"2024-05-15T04:13:52.105798Z","shell.execute_reply":"2024-05-15T04:13:52.105817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def catstrvec_2_catnumvec(strcats, df_col):\n    act_val = []\n    for i in df_col.to_numpy():\n        xy, x_ind, y_ind = np.intersect1d(strcats, i, return_indices=True)\n        act_val.append(x_ind[0])\n    \n    return act_val\n\ndef convert_mp3_to_sig(audio_filepath):\n    # files                                                                         \n    src = audio_filepath\n    dst = \"test.wav\"\n\n    # convert wav to mp3                                                            \n    sound = AudioSegment.from_mp3(src)\n    sound.export(dst, format=\"wav\")\n    \n    return dst\n\n# Load a wav file\ndef get_wav_info(wav_file):\n    rate, data = wavfile.read(wav_file)\n    return rate, data\n\n# Calculate and plot spectrogram for a wav audio file\n# Binning the time-series and calculating the periodogram per time bin\n\ndef graph_spectrogram(wav_file):\n    rate, data = get_wav_info(wav_file)\n    # print('size of extracted data from sound file: ', data.shape)\n    nfft = 100 # Length of each window segment - frequency data will be binned by nfft/2 \n    fs = 8000 # Sampling frequencies\n    noverlap = 0 #120 # Overlap between windows\n    nchannels = data.ndim\n    # print('nchannels: ', nchannels)\n    \n    # https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.spectrogram.html\n    \n    # OR\n    \n    # pxx = periodogram per frequency (nfft/2 = num_of_freqs, bins=num_of_bins)\n    # freqs = freqencies that correspond to each magnitude\n    # bins = number of bins to group time-series\n    # im = the image of the axis\n    if nchannels == 1:\n        pxx, freqs, bins, im = plt.specgram(data, nfft, fs, noverlap=noverlap)\n    elif nchannels == 2:\n        pxx, freqs, bins, im = plt.specgram(data[:,0], nfft, fs, noverlap=noverlap)\n        \n    # print('number of bins: ', bins.shape)\n    # print('shape of pxx: ', pxx.shape)\n    # print('shape of freqs: ', freqs.shape)\n    \n    plotORnot = 1\n    if plotORnot == 1:\n        plt.ylabel(\"Frequency\")\n        plt.xlabel(\"Number of time windows\")\n        plt.title(\"Spectrogram\")\n        plt.show()\n\n    return data, pxx, freqs\n\ndef make_a_properlist(vec):\n    \n    out = []\n    for i in range(len(vec)):\n        out = out + [np.ravel(vec[i])]\n        \n    if is_empty(out) == False:\n        vecout = np.concatenate(out).ravel().tolist()\n    else:\n        vecout = list(np.ravel(out))\n    \n    return vecout\n\ndef is_empty(vec):\n    vec = np.array(vec)\n    if vec.shape[0] == 0:\n        out = True\n    else:\n        out = False\n        \n    return out\n\ndef linear_intercurrentpt_makeshortSIGlong_interp1d(shortSIG, longSIG):\n\n    x = np.linspace(shortSIG[0], len(shortSIG), num=len(shortSIG), endpoint=True)\n    y = shortSIG\n    # print('x : ', x)\n\n\n    # -------------\n    kind = 'linear'\n    # kind : Specifies the kind of interpolation as a string or as an integer specifying the order of the spline interpolator to use. The string has to be one of ‘linear’, ‘nearest’, ‘nearest-up’, ‘zero’, ‘slinear’, ‘quadratic’, ‘cubic’, ‘previous’, or ‘next’. ‘zero’, ‘slinear’, ‘quadratic’ and ‘cubic’ refer to a spline interpolation of zeroth, first, second or third order; ‘previous’ and ‘next’ simply return the previous or next value of the point; ‘nearest-up’ and ‘nearest’ differ when interpolating half-integers (e.g. 0.5, 1.5) in that ‘nearest-up’ rounds up and ‘nearest’ rounds down. Default is ‘linear’.\n\n    if kind == 'linear':\n        f = interp1d(x, y)\n    elif kind == 'cubic':\n        f = interp1d(x, y, kind='cubic')\n    # -------------\n\n    xnew = np.linspace(shortSIG[0], len(shortSIG), num=len(longSIG), endpoint=True)\n    # print('xnew : ', xnew)\n\n    siglong = f(xnew)\n\n    return siglong\n\ndef scale_feature_data(feat, plotORnot, type_scale):\n    \n    columns = ['0']\n    dat = pd.DataFrame(data=feat, columns=columns)\n    \n    # which type of scaling\n    if type_scale == 'minmax':\n        # Values from 0 to 1\n        scaled_data0 = minmax_scaling(dat, columns=columns)\n        scaled_data = list(scaled_data0.to_numpy().ravel())\n        \n    elif type_scale == 'normalization':\n        # normalization : same as mlxtend - Values from 0 to 1\n        scaled_data = []\n        for q in range(len(feat)):\n            scaled_data.append( (feat[q] - np.min(feat))/(np.max(feat) - np.min(feat)) )\n    \n    elif type_scale == 'pos_normalization':\n        # positive normalization : same as mlxtend - Values from 0 to 1\n        shift_up = [i - np.min(feat) for i in feat]\n        scaled_data = [q/np.max(shift_up) for q in shift_up]\n    \n    elif type_scale == 'standardization':\n        # standardization : values are not restricted to a range, but scaled appropreately\n        scaled_data = [(q - np.mean(feat))/np.std(feat) for q in feat]\n\n    return scaled_data\n\n# level : the number of levels to decompose the time signal, le nombre des marquers par signale\ndef tsig_2_discrete_wavelet_transform(sig, waveletname, level, plotORnot):\n\n    # On peut calculater dans deux façons: 0) dwt en boucle and then idwt, 1) wavedec et waverec\n    # Mais le deux ne donnent pas le meme reponses, wavedec et waverec semble plus raisonable.\n    coeff = wavedec(sig, waveletname, level)\n\n    if plotORnot == 1:\n        fig, axx = plt.subplots(nrows=len(coeff), ncols=1, figsize=(5,5))\n        axx[0].set_title(\"coef\")  # Pas certain si c'est coef0 ou coef1\n        for k in range(len(coeff)):\n            axx[k].plot(coeff[k], 'r') # output of the low pass filter (averaging filter) of the DWT\n        plt.tight_layout()\n        plt.show()\n\n    return coeff\n\ndef tsig_2_spectrogram(sig, fs, nfft, noverlap, img_dim, plotORnot):\n\n    # -----------------------------------\n    fig,ax = plt.subplots(1)\n    fig.subplots_adjust(left=0,right=1,bottom=0,top=1)\n    ax.axis('off')\n    # spectrum2D array : Columns are the periodograms of successive segments.\n    # freqs1-D array : The frequencies corresponding to the rows in spectrum.\n    # t1-D array : The times corresponding to midpoints of segments (i.e., the columns in spectrum).\n    # imAxesImage : The image created by imshow containing the spectrogram.\n    pxx, freqs, bins, img = ax.specgram(sig, nfft, fs, noverlap=noverlap)\n    ax.axis('off')\n\n    # -----------------------------------\n\n    my_image = 'temp.png'\n    fig.savefig(my_image)\n    fname = os.path.abspath(os.getcwd()) + \"/\" +  my_image\n    \n    # Convert image to an array:\n    # Read image \n    img = Image.open(fname)         # PIL: img is not in array form, it is a PIL.PngImagePlugin.PngImageFile \n\n    # -----------------------------------\n    \n    # Resize sectrogram image\n    image = imgORmat_resize_imgORmat_CNN(img_dim, data_in=img, inpt='img3D', outpt='mat2D', norm='non', thresh='non')\n    \n    # -----------------------------------\n    \n    # Flatten image into a vector\n    img_flatten = np.reshape(np.ravel(image), (img_dim*img_dim, ), order='F')\n    # print('img_flatten.shape: ', img_flatten.shape)\n\n    # -----------------------------------\n\n    return img_flatten\nprint(\"skdlchn\")\n\ndef tsig_2_continuous_wavelet_transform(sig, fs, scales, waveletname, img_dim, plotORnot):\n    \n    # e.g.:\n    # scales = np.arange(1, 128)\n    # waveletname = 'mexh'\n    # sig = data2[0:100]\n    # print('len(sig): ', len(sig))\n    \n    dt = 1/fs\n    # print('dt : ', dt)\n\n    coefficients, frequencies = pywt.cwt(sig, scales, waveletname, dt)\n    coefficients = np.array(coefficients)\n    ylen, xlen = coefficients.shape\n\n    # Time by frequency plot of cwt : then flatten and use as a feature\n    stop_val = len(sig)/fs\n    x = np.arange(0, stop_val, dt)  # time\n    y = frequencies # frequency \n    X, Y = np.meshgrid(x, y)\n    Z = coefficients\n\n    # Each cwt versus frequencies\n    # for i in range(len(coefficients)):\n    #     plt.plot(frequencies, coefficients[i])\n\n    #  \n    fig = plt.figure()\n    ax = plt.axes() # creates a 3D axis by using the keyword projection='3d'\n    ax.contourf(X, Y, Z, xlen, cmap=plt.cm.seismic) # contour fill\n    ax.axis('off')\n\n    # -----------------------------------\n\n    my_image = 'temp.png'\n    fig.savefig(my_image)\n    fname = os.path.abspath(os.getcwd()) + \"/\" +  my_image\n\n    # Convert image to an array:\n    # Read image \n    img = Image.open(fname)  \n    \n    # -----------------------------------\n    \n    # Resize sectrogram image\n    image = imgORmat_resize_imgORmat_CNN(img_dim, data_in=img, inpt='img3D', outpt='mat2D', norm='non', thresh='non')\n    \n    # -----------------------------------\n    \n    # Flatten image into a vector\n    img_flatten = np.reshape(np.ravel(image), (img_dim*img_dim, ), order='F')\n    # print('img_flatten.shape: ', img_flatten.shape)\n    \n    return img_flatten\n\ndef resize_img(img, img_dim):\n    if type(img) == 'numpy.ndarray':\n        # img is an array, retuns an image object\n        rgb_image = Image.fromarray(img , 'RGB')\n    else:\n        # img is an image object, returns an image object\n        try:\n            rgb_image = img.convert('RGB')\n        except AttributeError:\n            rgb_image = Image.fromarray(img , 'RGB')\n\n    # Resize image into a 64, 64, 3\n    new_h, new_w = int(img_dim), int(img_dim)\n    img3 = rgb_image.resize((new_w, new_h), Image.ANTIALIAS)\n    w_resized, h_resized = img3.size[0], img3.size[1]\n    return img3\n\ndef convert_img_a_mat(img, outpt):\n    mat = np.array(img)  # Convert image to an array\n    if outpt == 'mat2D':\n        # Transformer l'image de 3D à 2D\n        # Convert image back to a 2D array\n        matout = np.mean(mat, axis=2)\n    elif outpt == 'img3D': # techniquement c'est un image parce qu'il y a trois RGB channels \n        matout = mat\n    return matout\n\ndef norm_mat(mat2Dor3D, norm):\n    if norm == 'zero2one':\n        # Normalizer l'image entre 0 et 1\n        norout = mat2Dor3D/255\n    elif norm == 'negone2posone':\n        # Normalize the images to [-1, 1]\n        norout = (mat2Dor3D - 127.5) / 127.5\n    elif norm == 'non':\n        norout = mat2Dor3D\n    return norout\n\ndef threshold_mat(mat2D, thresh):\n    # Threshold image\n    val = 255/2\n    if thresh == 'zero_moins_que_val':\n        row, col = mat2D.shape\n        mat_thresh = mat2D\n        min_val = np.min(mat_thresh)\n        for i in range(row):\n            for j in range(col):\n                if mat_thresh[i,j] < val:\n                    mat_thresh[i,j] = min_val\n    elif thresh == 'non':\n        mat_thresh = mat2D\n    return mat_thresh\n\ndef imgORmat_resize_imgORmat_CNN(img_dim, data_in, inpt='img3D', outpt='mat2D', norm='non', thresh='non'):\n    if inpt == 'img3D' and outpt=='mat2D':\n        img = resize_img(data_in, img_dim)\n        img3D = convert_img_a_mat(img, outpt)\n        out = norm_mat(img3D, norm)\n    elif inpt == 'mat2D' and outpt=='mat2D':\n        data_in = np.array(data_in)\n        img = Image.fromarray(data_in , 'L')\n        img = resize_img(img, img_dim)\n        mat2D = convert_img_a_mat(img, outpt)\n        out = norm_mat(mat2D, norm)\n    elif inpt == 'mat2D' and outpt=='img3D':\n        data_in = np.array(data_in)\n        img = Image.fromarray(data_in , 'L')\n        img = resize_img(img, img_dim)\n        img3D = convert_img_a_mat(img, outpt)\n        out = norm_mat(img3D, norm)\n    elif inpt == 'img3D' and outpt=='img3D':\n        img = resize_img(data_in, img_dim)\n        img3D = convert_img_a_mat(img, outpt)\n        out = norm_mat(img3D, norm)\n\n    return out\n\ndef save_dat_pickle(outSIG, file_name=\"outSIG.pkl\"):\n    # Save data matrices to file\n    open_file = open(file_name, \"wb\")\n    pickle.dump(outSIG, open_file)\n    open_file.close()\n\ndef load_dat_pickle(file_name=\"outSIG.pkl\"):\n    open_file = open(file_name, \"rb\")\n    dataout = pickle.load(open_file)\n    open_file.close()\n    return dataout\n\ndef binarize_Y1Dvec_2_Ybin(Y):\n    \n    # Transform a 1D Y vector (n_samples by 1) to a Y_bin (n_samples by n_classes) vector\n\n    # Ensure vector is of integers\n    Y = [int(i) for i in Y]\n\n    # Number of samples\n    m_examples = len(Y)\n\n    # Number of classes\n    temp = np.unique(Y)\n    unique_classes = [int(i) for i in temp]\n    # print('unique_classes : ', unique_classes)\n\n    whichone = 2\n    # Binarize the output\n    if whichone == 0:\n        from sklearn.preprocessing import label_binarize\n        Y_bin = label_binarize(Y, classes=unique_classes)  # seems to work now\n\n    elif whichone == 1:\n        from sklearn import preprocessing\n        lb = preprocessing.LabelBinarizer()\n        Y_bin = lb.fit_transform(Y)  # seems to work now\n        \n    elif whichone == 2:\n        # By hand\n        Y_bin = np.zeros((m_examples, len(unique_classes)))\n        for i in range(0, m_examples):\n            if Y[i] == unique_classes[0]:\n                Y_bin[i,0] = 1\n            elif Y[i] == unique_classes[1]:\n                Y_bin[i,1] = 1\n            elif Y[i] == unique_classes[2]:\n                Y_bin[i,2] = 1\n            elif Y[i] == unique_classes[3]:\n                Y_bin[i,3] = 1\n            elif Y[i] == unique_classes[4]:\n                Y_bin[i,4] = 1\n            elif Y[i] == unique_classes[5]:\n                Y_bin[i,5] = 1\n            elif Y[i] == unique_classes[6]:\n                Y_bin[i,6] = 1\n            elif Y[i] == unique_classes[7]:\n                Y_bin[i,7] = 1\n            elif Y[i] == unique_classes[8]:\n                Y_bin[i,8] = 1\n            elif Y[i] == unique_classes[9]:\n                Y_bin[i,9] = 1\n            elif Y[i] == unique_classes[10]:\n                Y_bin[i,10] = 1\n                \n    print('shape of Y_bin : ', Y_bin.shape)\n\n    return Y_bin, unique_classes\n\ndef binarize_audio_signal(wind, data2):\n    # wind = 1000\n    a = int(np.floor(len(data2)/wind))\n    vals = np.arange(0, a*wind, wind)\n    sig = [np.max(data2[i:(i+wind)])*np.ones((wind)) for i in vals]    \n    sig = make_a_properlist(sig)\n    \n    return sig\n\ndef temporal_repeating_signatures(data2, timesteps, plotORnot):\n    \n    # ------------------------------------\n\n    # choose wind such that sig is timesteps long\n    size_of_mat = timesteps/2\n    wind = int(len(data2)/size_of_mat)\n    sig = binarize_audio_signal(wind, data2)\n\n    min_border = np.mean(sig) + 1*np.std(sig)\n\n    # On a besoin d'avoir au moins de 2 peaks\n    peaks = []\n    while len(peaks) < 2:\n        # peaks, properties = find_peaks(sig, height=(min_border, np.max(sig)), prominence=100)\n        peaks, properties = find_peaks(sig, height=(min_border, np.max(sig)))\n        min_border = min_border - 10\n    vv = [sig[i] for i in peaks]\n\n    if plotORnot == 1:\n        plt.plot(sig)\n        plt.plot(peaks, vv, 'r*')\n        plt.ylabel(\"Amplitude\")\n        plt.xlabel(\"Data points\")\n        plt.title(\"Bird call : binarized data2\")\n        plt.show()\n\n    # ------------------------------------\n\n    # Calculate the difference in peaks and find the difference that repeats the most\n    # This repeating difference is the ideal window to cut the data for bird call temporal signatures\n    peakdiff = [peaks[i+1]-peaks[i] for i in range(len(peaks)-1)]\n    z = sorted(peakdiff)\n    o = [len(list(str(z[i]))) for i in range(len(z))]\n\n    from collections import Counter\n    c = Counter(o)\n    mc = c.most_common()[0][0]\n\n    for i in range(len(z)):\n        if len(list(str(z[i]))) == mc:\n            wind = z[i]\n\n    # ------------------------------------\n\n    # Cut the data and compare each binarized piece, using a metric\n    # To see which pieces are similar\n    a = int(np.floor(len(sig)/wind))\n    vals = np.arange(0, a*wind, wind)\n\n    dic = {}\n    dic2 = {}\n    for i in vals:\n        out = []\n        out2 = []\n\n        A = sig[i:(i+wind)]\n        A = [int(r) for r in A]\n        for j in vals:\n            # import required libraries\n            B = sig[j:(j+wind)]\n            B = [int(r) for r in B]\n\n            # cosine similarity\n            #cosine = np.dot(A,B)/(norm(A)*norm(B))\n            #out.append(cosine)\n\n            # absolute error\n            err = [np.abs(A[q]-B[q]) for q in range(len(A))]\n            out.append(err)\n\n            # height value\n            height = np.max(A)\n            out2.append(height)\n\n        dic[i] = out\n        dic2[i] = out2\n\n    # ------------------------------------\n\n    # tot = np.array(tot)\n    # tot.shape\n    # import seaborn as sns\n    # sns.heatmap(data=tot, annot=True)\n\n    # ------------------------------------\n\n    # Take the mean across similarity measures for each temporal piece\n    zz = [np.mean(dic[vals[i]]) for i in range(len(dic))]\n    zz2 = [np.mean(dic2[vals[i]]) for i in range(len(dic2))]\n\n    # ------------------------------------\n\n    # Find the temporal piece that are most similar to all other pieces\n    select_crit = 2\n\n    if select_crit == 0:\n        # No height restriction\n        # ind_piece = np.argmax(zz)  # cosine similarity\n        ind_piece = np.argmin(zz)  # absolute error\n    elif select_crit == 1:\n        # With height restriction\n        # use height as threshold for zz\n        marg = 1*np.std(sig)\n        thresh = np.max(zz2) - marg\n        print('thresh : ', thresh)\n        # OU\n        # thresh = np.min(zz2)\n        store = [zz for i in range(len(zz)) if zz2[i] > thresh]\n\n        # the select index using metric (cosine, absolute error)\n        # ind_piece = np.argmax(store)  # cosine similarity\n        ind_piece = np.argmin(store)  # absolute error\n    elif select_crit == 2:\n        # Rien a faire avec metrics, choisi par rapport la hauteur\n        ind_piece = np.argmax(zz2)\n    \n    # Plot selected temporal piece, with data\n    st = vals[ind_piece]\n    endd = st+wind\n    sig2 = sig[st:endd]\n\n    if plotORnot == 1:\n        plt.plot(sig)\n        plt.plot(np.arange(st,endd), sig2)\n        plt.ylabel(\"Amplitude\")\n        plt.xlabel(\"Data points\")\n        plt.title(\"Non-downsampled : Bird call and call signature\")\n        plt.show()\n\n    # ------------------------------------\n\n    # Downsampling\n    v = downsample_sig(sig2, timesteps)\n\n    # ------------------------------------\n        \n    return v\n\ndef downsample_sig(sig, timesteps):\n\n    # Downsampling\n    a = int(np.floor(len(sig)/timesteps))\n    vals = np.arange(0, a*timesteps, a)\n    v = [sig[i:(i+a)][0] for i in vals] # duh, pick the first point from each window, that is why [0] is there\n    \n    return v\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.107709Z","iopub.status.idle":"2024-05-15T04:13:52.108558Z","shell.execute_reply.started":"2024-05-15T04:13:52.108346Z","shell.execute_reply":"2024-05-15T04:13:52.108369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strcats = data.primary_label\nprint('Activities include: ', strcats)\nnum_of_birds = len(strcats)\nprint('Number of birds: ', num_of_birds)\n\ny = catstrvec_2_catnumvec(strcats, data.primary_label)\ny = np.array(y)\nprint('shape of y: ', y.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.110673Z","iopub.status.idle":"2024-05-15T04:13:52.111077Z","shell.execute_reply.started":"2024-05-15T04:13:52.110881Z","shell.execute_reply":"2024-05-15T04:13:52.110901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/birdclef-2024/train_audio/\"\n# Get file names \nfiles = os.listdir(path)\nall_fn = []\nfor f in files:\n    fn = f.partition('.')[0]\n    \n    if fn[0:2] != '__':\n        all_fn.append(fn)\nprint('FILE NAMES: ', all_fn)\nprint('length of all_fn: ', len(all_fn))","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.112171Z","iopub.status.idle":"2024-05-15T04:13:52.112579Z","shell.execute_reply.started":"2024-05-15T04:13:52.112391Z","shell.execute_reply":"2024-05-15T04:13:52.112408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_mp3_to_sig(audio_filepath):\n    # files                                                                         \n    src = audio_filepath\n    dst = \"test.wav\"\n\n    # convert wav to mp3                                                            \n    sound = AudioSegment.from_ogg(src)\n    sound.export(dst, format=\"wav\")\n    \n    return dst\n# Load a wav file\ndef get_wav_info(wav_file):\n    rate, data = wavfile.read(wav_file)\n    return rate, data\n# Calculate and plot spectrogram for a wav audio file\n# Binning the time-series and calculating the periodogram per time bin\n\ndef graph_spectrogram(wav_file):\n    rate, data = get_wav_info(wav_file)\n    # print('size of extracted data from sound file: ', data.shape)\n    nfft = 100 # Length of each window segment - frequency data will be binned by nfft/2 \n    fs = 8000 # Sampling frequencies\n    noverlap = 0 #120 # Overlap between windows\n    nchannels = data.ndim\n    # print('nchannels: ', nchannels)\n    \n    # https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.spectrogram.html\n    \n    # OR\n    \n    # pxx = periodogram per frequency (nfft/2 = num_of_freqs, bins=num_of_bins)\n    # freqs = freqencies that correspond to each magnitude\n    # bins = number of bins to group time-series\n    # im = the image of the axis\n    if nchannels == 1:\n        pxx, freqs, bins, im = plt.specgram(data, nfft, fs, noverlap=noverlap)\n    elif nchannels == 2:\n        pxx, freqs, bins, im = plt.specgram(data[:,0], nfft, fs, noverlap=noverlap)\n        \n    # print('number of bins: ', bins.shape)\n    # print('shape of pxx: ', pxx.shape)\n    # print('shape of freqs: ', freqs.shape)\n    \n    plotORnot = 0\n    if plotORnot == 1:\n        plt.ylabel(\"Frequency\")\n        plt.xlabel(\"Number of time windows\")\n        plt.title(\"Spectrogram\")\n        plt.show()\n\n    return data, pxx, freqs\ndef make_a_properlist(vec):\n    \n    out = []\n    for i in range(len(vec)):\n        out = out + [np.ravel(vec[i])]\n        \n    if is_empty(out) == False:\n        vecout = np.concatenate(out).ravel().tolist()\n    else:\n        vecout = list(np.ravel(out))\n    \n    return vecout\n\ndef is_empty(vec):\n    vec = np.array(vec)\n    if vec.shape[0] == 0:\n        out = True\n    else:\n        out = False\n        \n    return out\ndef linear_intercurrentpt_makeshortSIGlong_interp1d(shortSIG, longSIG):\n\n    x = np.linspace(shortSIG[0], len(shortSIG), num=len(shortSIG), endpoint=True)\n    y = shortSIG\n    # print('x : ', x)\n\n\n    # -------------\n    kind = 'linear'\n    # kind : Specifies the kind of interpolation as a string or as an integer specifying the order of the spline interpolator to use. The string has to be one of ‘linear’, ‘nearest’, ‘nearest-up’, ‘zero’, ‘slinear’, ‘quadratic’, ‘cubic’, ‘previous’, or ‘next’. ‘zero’, ‘slinear’, ‘quadratic’ and ‘cubic’ refer to a spline interpolation of zeroth, first, second or third order; ‘previous’ and ‘next’ simply return the previous or next value of the point; ‘nearest-up’ and ‘nearest’ differ when interpolating half-integers (e.g. 0.5, 1.5) in that ‘nearest-up’ rounds up and ‘nearest’ rounds down. Default is ‘linear’.\n\n    if kind == 'linear':\n        f = interp1d(x, y)\n    elif kind == 'cubic':\n        f = interp1d(x, y, kind='cubic')\n    # -------------\n\n    xnew = np.linspace(shortSIG[0], len(shortSIG), num=len(longSIG), endpoint=True)\n    # print('xnew : ', xnew)\n\n    siglong = f(xnew)\n\n    return siglong\ndef scale_feature_data(feat, plotORnot, type_scale):\n    \n    columns = ['0']\n    dat = pd.DataFrame(data=feat, columns=columns)\n    \n    # which type of scaling\n    if type_scale == 'minmax':\n        # Values from 0 to 1\n        scaled_data0 = minmax_scaling(dat, columns=columns)\n        scaled_data = list(scaled_data0.to_numpy().ravel())\n        \n    elif type_scale == 'normalization':\n        # normalization : same as mlxtend - Values from 0 to 1\n        scaled_data = []\n        for q in range(len(feat)):\n            scaled_data.append( (feat[q] - np.min(feat))/(np.max(feat) - np.min(feat)) )\n    \n    elif type_scale == 'pos_normalization':\n        # positive normalization : same as mlxtend - Values from 0 to 1\n        shift_up = [i - np.min(feat) for i in feat]\n        scaled_data = [q/np.max(shift_up) for q in shift_up]\n    \n    elif type_scale == 'standardization':\n        # standardization : values are not restricted to a range, but scaled appropreately\n        scaled_data = [(q - np.mean(feat))/np.std(feat) for q in feat]\n\n    return scaled_data\n# level : the number of levels to decompose the time signal, le nombre des marquers par signale\ndef tsig_2_discrete_wavelet_transform(sig, waveletname, level, plotORnot):\n\n    # On peut calculater dans deux façons: 0) dwt en boucle and then idwt, 1) wavedec et waverec\n    # Mais le deux ne donnent pas le meme reponses, wavedec et waverec semble plus raisonable.\n    coeff = wavedec(sig, waveletname, level)\n\n    if plotORnot == 1:\n        fig, axx = plt.subplots(nrows=len(coeff), ncols=1, figsize=(5,5))\n        axx[0].set_title(\"coef\")  # Pas certain si c'est coef0 ou coef1\n        for k in range(len(coeff)):\n            axx[k].plot(coeff[k], 'r') # output of the low pass filter (averaging filter) of the DWT\n        plt.tight_layout()\n        plt.show()\n\n    return coeff\ndef tsig_2_spectrogram(sig, fs, nfft, noverlap, img_dim, plotORnot):\n\n    # -----------------------------------\n    fig,ax = plt.subplots(1)\n    fig.subplots_adjust(left=0,right=1,bottom=0,top=1)\n    ax.axis('off')\n    # spectrum2D array : Columns are the periodograms of successive segments.\n    # freqs1-D array : The frequencies corresponding to the rows in spectrum.\n    # t1-D array : The times corresponding to midpoints of segments (i.e., the columns in spectrum).\n    # imAxesImage : The image created by imshow containing the spectrogram.\n    pxx, freqs, bins, img = ax.specgram(sig, nfft, fs, noverlap=noverlap)\n    ax.axis('off')\n\n    # -----------------------------------\n\n    my_image = 'temp.png'\n    fig.savefig(my_image)\n    fname = os.path.abspath(os.getcwd()) + \"/\" +  my_image\n    \n    # Convert image to an array:\n    # Read image \n    img = Image.open(fname)         # PIL: img is not in array form, it is a PIL.PngImagePlugin.PngImageFile \n\n    # -----------------------------------\n    \n    # Resize sectrogram image\n    image = imgORmat_resize_imgORmat_CNN(img_dim, data_in=img, inpt='img3D', outpt='mat2D', norm='non', thresh='non')\n    \n    # -----------------------------------\n    \n    # Flatten image into a vector\n    img_flatten = np.reshape(np.ravel(image), (img_dim*img_dim, ), order='F')\n    # print('img_flatten.shape: ', img_flatten.shape)\n\n    # -----------------------------------\n\n    return img_flatten\ndef tsig_2_continuous_wavelet_transform(sig, fs, scales, waveletname, img_dim, plotORnot):\n    \n    # e.g.:\n    # scales = np.arange(1, 128)\n    # waveletname = 'mexh'\n    # sig = data2[0:100]\n    # print('len(sig): ', len(sig))\n    \n    dt = 1/fs\n    # print('dt : ', dt)\n\n    coefficients, frequencies = pywt.cwt(sig, scales, waveletname, dt)\n    coefficients = np.array(coefficients)\n    ylen, xlen = coefficients.shape\n\n    # Time by frequency plot of cwt : then flatten and use as a feature\n    stop_val = len(sig)/fs\n    x = np.arange(0, stop_val, dt)  # time\n    y = frequencies # frequency \n    X, Y = np.meshgrid(x, y)\n    Z = coefficients\n\n    # Each cwt versus frequencies\n    # for i in range(len(coefficients)):\n    #     plt.plot(frequencies, coefficients[i])\n\n    #  \n    fig = plt.figure()\n    ax = plt.axes() # creates a 3D axis by using the keyword projection='3d'\n    ax.contourf(X, Y, Z, xlen, cmap=plt.cm.seismic) # contour fill\n    ax.axis('off')\n\n    # -----------------------------------\n\n    my_image = 'temp.png'\n    fig.savefig(my_image)\n    fname = os.path.abspath(os.getcwd()) + \"/\" +  my_image\n\n    # Convert image to an array:\n    # Read image \n    img = Image.open(fname)  \n    \n    # -----------------------------------\n    \n    # Resize sectrogram image\n    image = imgORmat_resize_imgORmat_CNN(img_dim, data_in=img, inpt='img3D', outpt='mat2D', norm='non', thresh='non')\n    \n    # -----------------------------------\n    \n    # Flatten image into a vector\n    img_flatten = np.reshape(np.ravel(image), (img_dim*img_dim, ), order='F')\n    # print('img_flatten.shape: ', img_flatten.shape)\n    \n    return img_flatten\ndef resize_img(img, img_dim):\n    if type(img) == 'numpy.ndarray':\n        # img is an array, retuns an image object\n        rgb_image = Image.fromarray(img , 'RGB')\n    else:\n        # img is an image object, returns an image object\n        try:\n            rgb_image = img.convert('RGB')\n        except AttributeError:\n            rgb_image = Image.fromarray(img , 'RGB')\n\n    # Resize image into a 64, 64, 3\n    new_h, new_w = int(img_dim), int(img_dim)\n    img3 = rgb_image.resize((new_w, new_h), Image.ANTIALIAS)\n    w_resized, h_resized = img3.size[0], img3.size[1]\n    return img3\n\ndef convert_img_a_mat(img, outpt):\n    mat = np.array(img)  # Convert image to an array\n    if outpt == 'mat2D':\n        # Transformer l'image de 3D à 2D\n        # Convert image back to a 2D array\n        matout = np.mean(mat, axis=2)\n    elif outpt == 'img3D': # techniquement c'est un image parce qu'il y a trois RGB channels \n        matout = mat\n    return matout\n\ndef norm_mat(mat2Dor3D, norm):\n    if norm == 'zero2one':\n        # Normalizer l'image entre 0 et 1\n        norout = mat2Dor3D/255\n    elif norm == 'negone2posone':\n        # Normalize the images to [-1, 1]\n        norout = (mat2Dor3D - 127.5) / 127.5\n    elif norm == 'non':\n        norout = mat2Dor3D\n    return norout\n\ndef threshold_mat(mat2D, thresh):\n    # Threshold image\n    val = 255/2\n    if thresh == 'zero_moins_que_val':\n        row, col = mat2D.shape\n        mat_thresh = mat2D\n        min_val = np.min(mat_thresh)\n        for i in range(row):\n            for j in range(col):\n                if mat_thresh[i,j] < val:\n                    mat_thresh[i,j] = min_val\n    elif thresh == 'non':\n        mat_thresh = mat2D\n    return mat_thresh\n\ndef imgORmat_resize_imgORmat_CNN(img_dim, data_in, inpt='img3D', outpt='mat2D', norm='non', thresh='non'):\n    if inpt == 'img3D' and outpt=='mat2D':\n        img = resize_img(data_in, img_dim)\n        img3D = convert_img_a_mat(img, outpt)\n        out = norm_mat(img3D, norm)\n    elif inpt == 'mat2D' and outpt=='mat2D':\n        data_in = np.array(data_in)\n        img = Image.fromarray(data_in , 'L')\n        img = resize_img(img, img_dim)\n        mat2D = convert_img_a_mat(img, outpt)\n        out = norm_mat(mat2D, norm)\n    elif inpt == 'mat2D' and outpt=='img3D':\n        data_in = np.array(data_in)\n        img = Image.fromarray(data_in , 'L')\n        img = resize_img(img, img_dim)\n        img3D = convert_img_a_mat(img, outpt)\n        out = norm_mat(img3D, norm)\n    elif inpt == 'img3D' and outpt=='img3D':\n        img = resize_img(data_in, img_dim)\n        img3D = convert_img_a_mat(img, outpt)\n        out = norm_mat(img3D, norm)\n\n    return out\ndef save_dat_pickle(outSIG, file_name=\"outSIG.pkl\"):\n    # Save data matrices to file\n    open_file = open(file_name, \"wb\")\n    pickle.dump(outSIG, open_file)\n    open_file.close()\ndef load_dat_pickle(file_name=\"outSIG.pkl\"):\n    open_file = open(file_name, \"rb\")\n    dataout = pickle.load(open_file)\n    open_file.close()\n    return dataout\ndef binarize_Y1Dvec_2_Ybin(Y):\n    \n    # Transform a 1D Y vector (n_samples by 1) to a Y_bin (n_samples by n_classes) vector\n\n    # Ensure vector is of integers\n    Y = [int(i) for i in Y]\n\n    # Number of samples\n    m_examples = len(Y)\n\n    # Number of classes\n    temp = np.unique(Y)\n    unique_classes = [int(i) for i in temp]\n    # print('unique_classes : ', unique_classes)\n\n    whichone = 2\n    # Binarize the output\n    if whichone == 0:\n        from sklearn.preprocessing import label_binarize\n        Y_bin = label_binarize(Y, classes=unique_classes)  # seems to work now\n\n    elif whichone == 1:\n        from sklearn import preprocessing\n        lb = preprocessing.LabelBinarizer()\n        Y_bin = lb.fit_transform(Y)  # seems to work now\n        \n    elif whichone == 2:\n        # By hand\n        Y_bin = np.zeros((m_examples, len(unique_classes)))\n        for i in range(0, m_examples):\n            if Y[i] == unique_classes[0]:\n                Y_bin[i,0] = 1\n            elif Y[i] == unique_classes[1]:\n                Y_bin[i,1] = 1\n            elif Y[i] == unique_classes[2]:\n                Y_bin[i,2] = 1\n            elif Y[i] == unique_classes[3]:\n                Y_bin[i,3] = 1\n            elif Y[i] == unique_classes[4]:\n                Y_bin[i,4] = 1\n            elif Y[i] == unique_classes[5]:\n                Y_bin[i,5] = 1\n            elif Y[i] == unique_classes[6]:\n                Y_bin[i,6] = 1\n            elif Y[i] == unique_classes[7]:\n                Y_bin[i,7] = 1\n            elif Y[i] == unique_classes[8]:\n                Y_bin[i,8] = 1\n            elif Y[i] == unique_classes[9]:\n                Y_bin[i,9] = 1\n            elif Y[i] == unique_classes[10]:\n                Y_bin[i,10] = 1\n                \n    print('shape of Y_bin : ', Y_bin.shape)\n\n    return Y_bin, unique_classes\ndef binarize_audio_signal(wind, data2):\n    # wind = 1000\n    a = int(np.floor(len(data2)/wind))\n    vals = np.arange(0, a*wind, wind)\n    sig = [np.max(data2[i:(i+wind)])*np.ones((wind)) for i in vals]    \n    sig = make_a_properlist(sig)\n    \n    return sig\ndef temporal_repeating_signatures(data2, timesteps, plotORnot):\n    \n    # ------------------------------------\n\n    # choose wind such that sig is timesteps long\n    size_of_mat = timesteps/2\n    wind = int(len(data2)/size_of_mat)\n    sig = binarize_audio_signal(wind, data2)\n\n    min_border = np.mean(sig) + 1*np.std(sig)\n\n    # On a besoin d'avoir au moins de 2 peaks\n    peaks = []\n    while len(peaks) < 2:\n        # peaks, properties = find_peaks(sig, height=(min_border, np.max(sig)), prominence=100)\n        peaks, properties = find_peaks(sig, height=(min_border, np.max(sig)))\n        min_border = min_border - 10\n    vv = [sig[i] for i in peaks]\n\n    if plotORnot == 1:\n        plt.plot(sig)\n        plt.plot(peaks, vv, 'r*')\n        plt.ylabel(\"Amplitude\")\n        plt.xlabel(\"Data points\")\n        plt.title(\"Bird call : binarized data2\")\n        plt.show()\n\n    # ------------------------------------\n\n    # Calculate the difference in peaks and find the difference that repeats the most\n    # This repeating difference is the ideal window to cut the data for bird call temporal signatures\n    peakdiff = [peaks[i+1]-peaks[i] for i in range(len(peaks)-1)]\n    z = sorted(peakdiff)\n    o = [len(list(str(z[i]))) for i in range(len(z))]\n\n    from collections import Counter\n    c = Counter(o)\n    mc = c.most_common()[0][0]\n\n    for i in range(len(z)):\n        if len(list(str(z[i]))) == mc:\n            wind = z[i]\n\n    # ------------------------------------\n\n    # Cut the data and compare each binarized piece, using a metric\n    # To see which pieces are similar\n    a = int(np.floor(len(sig)/wind))\n    vals = np.arange(0, a*wind, wind)\n\n    dic = {}\n    dic2 = {}\n    for i in vals:\n        out = []\n        out2 = []\n\n        A = sig[i:(i+wind)]\n        A = [int(r) for r in A]\n        for j in vals:\n            # import required libraries\n            B = sig[j:(j+wind)]\n            B = [int(r) for r in B]\n\n            # cosine similarity\n            #cosine = np.dot(A,B)/(norm(A)*norm(B))\n            #out.append(cosine)\n\n            # absolute error\n            err = [np.abs(A[q]-B[q]) for q in range(len(A))]\n            out.append(err)\n\n            # height value\n            height = np.max(A)\n            out2.append(height)\n\n        dic[i] = out\n        dic2[i] = out2\n\n    # ------------------------------------\n\n    # tot = np.array(tot)\n    # tot.shape\n    # import seaborn as sns\n    # sns.heatmap(data=tot, annot=True)\n\n    # ------------------------------------\n\n    # Take the mean across similarity measures for each temporal piece\n    zz = [np.mean(dic[vals[i]]) for i in range(len(dic))]\n    zz2 = [np.mean(dic2[vals[i]]) for i in range(len(dic2))]\n\n    # ------------------------------------\n\n    # Find the temporal piece that are most similar to all other pieces\n    select_crit = 2\n\n    if select_crit == 0:\n        # No height restriction\n        # ind_piece = np.argmax(zz)  # cosine similarity\n        ind_piece = np.argmin(zz)  # absolute error\n    elif select_crit == 1:\n        # With height restriction\n        # use height as threshold for zz\n        marg = 1*np.std(sig)\n        thresh = np.max(zz2) - marg\n        print('thresh : ', thresh)\n        # OU\n        # thresh = np.min(zz2)\n        store = [zz for i in range(len(zz)) if zz2[i] > thresh]\n\n        # the select index using metric (cosine, absolute error)\n        # ind_piece = np.argmax(store)  # cosine similarity\n        ind_piece = np.argmin(store)  # absolute error\n    elif select_crit == 2:\n        # Rien a faire avec metrics, choisi par rapport la hauteur\n        ind_piece = np.argmax(zz2)\n    \n    # Plot selected temporal piece, with data\n    st = vals[ind_piece]\n    endd = st+wind\n    sig2 = sig[st:endd]\n\n    if plotORnot == 1:\n        plt.plot(sig)\n        plt.plot(np.arange(st,endd), sig2)\n        plt.ylabel(\"Amplitude\")\n        plt.xlabel(\"Data points\")\n        plt.title(\"Non-downsampled : Bird call and call signature\")\n        plt.show()\n\n    # ------------------------------------\n\n    # Downsampling\n    v = downsample_sig(sig2, timesteps)\n\n    # ------------------------------------\n        \n    return v\ndef downsample_sig(sig, timesteps):\n\n    # Downsampling\n    a = int(np.floor(len(sig)/timesteps))\n    vals = np.arange(0, a*timesteps, a)\n    v = [sig[i:(i+a)][0] for i in vals] # duh, pick the first point from each window, that is why [0] is there\n    \n    return v","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.113856Z","iopub.status.idle":"2024-05-15T04:13:52.114235Z","shell.execute_reply.started":"2024-05-15T04:13:52.114051Z","shell.execute_reply":"2024-05-15T04:13:52.114068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# son_des_oiseaux2_4\ntry_some = 1 # 100, 300, 500\nsamps = np.random.permutation(try_some)\nsamps\n\nX = pd.DataFrame()\n\ntimesteps = 500  # 50, 100, 150 (best), 200\nlongsig = np.arange(timesteps)\n\nplotORnot = 0\n\nfor bird in samps:\n\n    audio_filepath = path + all_fn[bird] + '.ogg'\n    print(audio_filepath)\n    # ------------------------------------\n\n    # Do steps 1 and 2 to get the rate\n    # 1) Converter mp3 à signale\n    dst = convert_mp3_to_sig(audio_filepath)\n\n    # 2) Obtenir le fréquence d'échantillonnage\n    rate, data = get_wav_info(dst)\n\n    del dst, data\n\n    # ------------------------------------\n\n    # 3) Downsample such that the audio is not distorted\n    sound = AudioSegment.from_file(audio_filepath, format='ogg', frame_rate=rate)\n    fs = 10000\n    sound = sound.set_frame_rate(fs)\n    data2 = sound.get_array_of_samples()\n\n    if plotORnot == 1:\n        # Confirmer signale\n        t = np.linspace(0, len(data2) / fs, len(data2))\n        plt.plot(data2)  # freqs is 101 long\n        plt.ylabel(\"Amplitude\")\n        plt.xlabel(\"Data points\")\n        plt.title(\"Bird call : data2\")\n        plt.show()\n\n    del sound\n\n    # ------------------------------------\n\n    # Time-series downsampled\n    useit = 1\n    if useit == 1:\n        timsig_down = downsample_sig(data2, timesteps)\n\n        if plotORnot == 1:\n            plt.plot(timsig_down)\n            plt.ylabel(\"Amplitude\")\n            plt.xlabel(\"Data points\")\n            plt.title(\"Downsampled : data2\")\n            plt.show()\n\n    # ------------------------------------\n\n    # Choose wind such that sig is timesteps long\n    useit = 1\n    if useit == 1:\n        plotORnot = 1\n        timeseries_repeat = temporal_repeating_signatures(data2, timesteps, plotORnot)\n\n    # If the mp3 has all other sounds and one or two bird calls, it is likely to find a signature\n    # for the other sounds because it partitions the data and looks for similar chunks of data.\n    # If you have only bird calls, it will pick up a bird call signature.\n\n    # ------------------------------------\n\n    useit = 1\n    if useit == 1:\n        # Get the first 3principle componets : Principle Component Analysis (PCA)\n        X_dat = [data2, data2]\n        X_dat = np.reshape(X_dat, (len(data2), 2))\n        feat_PCA = PCA(n_components=2).fit_transform(X_dat)  # shape (len(data2), n_components)\n\n        cols = feat_PCA.shape[1]\n        pca_out = []\n        for i in range(cols):\n            sig = feat_PCA[:,i]\n            pca_out.append(downsample_sig(sig, timesteps))\n\n        if plotORnot == 1:\n            plt.plot(data2)\n            plt.ylabel(\"Magnitude\")\n            plt.xlabel(\"Data points\")\n            plt.title(\"data2\")\n            plt.show()\n\n            plt.plot(feat_PCA[:])\n            plt.ylabel(\"Magnitude\")\n            plt.xlabel(\"Data points\")\n            plt.title(\"Non-downsampled PCA\")\n            plt.show()\n\n            plt.plot(pca_out[0])\n            plt.plot(pca_out[1])\n            plt.ylabel(\"Magnitude\")\n            plt.xlabel(\"Data points\")\n            plt.title(\"Downsampled PCA\")\n            plt.show()\n\n    # ------------------------------------\n\n    useit = 0\n    if useit == 1:\n        # ....................\n        # Binned periodogram\n        # ....................\n        # Plot directly the periodogram of the signal (Magnitude vs Frequency) : Unmanageable length = 149401\n        detrend='linear'\n        freqencies, spectrum = periodogram(data2, fs=fs, detrend=detrend, window=\"boxcar\", scaling='spectrum',)\n\n        if plotORnot == 1:\n            plt.plot(freqencies, spectrum)\n            plt.ylabel(\"Magnitude\")\n            plt.xlabel(\"Frequency\")\n            plt.title(\"Periodogram\")\n            plt.show()\n\n        grp = 550\n        spec_bin = [sum(spectrum[i:i+grp]) for i in range(0,len(spectrum), grp)]\n        freq_bin = [freqencies[i] for i in range(0,len(freqencies), grp)]\n\n        while len(spec_bin) > timesteps:\n            grp = grp+5\n            spec_bin = [sum(spectrum[i:i+grp]) for i in range(0,len(spectrum), grp)]\n            freq_bin = [freqencies[i] for i in range(0,len(freqencies), grp)]\n\n        if plotORnot == 1:\n            plt.plot(freq_bin, spec_bin)\n            plt.ylabel(\"Magnitude\")\n            plt.xlabel(\"Frequency\")\n            plt.title(\"Periodogram\")\n            plt.show()\n\n        del spectrum\n\n    # ------------------------------------\n\n    useit = 0\n    if useit == 1:\n        # ....................\n        # MEL frequency\n        # ....................\n        data2 = np.array(data2)\n        data2 = np.float32(data2)\n        mfcc = librosa.feature.mfcc(data2)\n\n        mfcc_mean = mfcc.mean(axis=1)  # Manageable length = 20\n\n    # ------------------------------------\n\n    useit = 0\n    if useit == 1:\n        # ....................\n        # Discrete time wavelets\n        # ....................\n        level = 5\n        coeff = tsig_2_discrete_wavelet_transform(data2, 'sym5', level, plotORnot)  # Manageable length = 16\n\n        # Not sure why it gives more than level+1 coefficents\n        # Only take level+1\n        coeff = coeff[0:level+1]\n\n        # Ensure that all wavelets are no longer than timesteps\n        coeff_cut = [coeff[i] if len(coeff[i]) <= timesteps else coeff[i][0:timesteps] for i in range(len(coeff))]\n\n    # ------------------------------------\n\n    useit = 0\n    if useit == 1:\n        # Selectionez le taille d'image pour le STFFT and CTFFT par rapport le taille des autres marquers \n        img_dim = int(np.sqrt(timesteps))\n\n        # flattened short-time fourier transform : spectrogram\n        nfft = 100 # Length of each window segment - frequency data will be binned by nfft/2 \n        fs = 8000 # Sampling frequencies\n        noverlap = 0 #120 # Overlap between windows - this give data consistency\n        stfft = tsig_2_spectrogram(data2, fs, nfft, noverlap, img_dim, plotORnot) # Manageable length = img_dim*img_dim\n\n    # ------------------------------------\n\n    useit = 0\n    if useit == 1:\n        # Take a fixed small amount of the signal, for slower processing features\n        part = 1000\n        sig3 = data2[0:part]\n        scales = np.arange(1, 128)\n        waveletname = 'mexh'\n\n        # Manageable length = img_dim*img_dim\n        ctfft = tsig_2_continuous_wavelet_transform(sig3, fs, scales, waveletname, img_dim, plotORnot)\n\n    # ------------------------------------\n\n    # Not all the columns are the same length but we just need to store them somewhere together\n\n    # Time features :\n    col0 = pd.Series(timsig_down)\n    col1 = pd.Series(timeseries_repeat)\n    cols2 = pd.DataFrame(pca_out).T\n\n    # Frequency features :\n    #col3 = pd.Series(spec_bin)\n    # col4 = pd.Series(mfcc_mean)\n    # cols5 = pd.DataFrame(coeff_cut).T\n    #col6 = pd.Series(stfft) \n    #col7 = pd.Series(ctfft)\n\n    # df = pd.concat([col0, col1, col2, cols3, col4, col5], axis=1)\n    # df = pd.concat([col1, col2, cols3], axis=1)\n    # df = pd.concat([col0, cols01, col2, cols3], axis=1)\n    df = pd.concat([col0, col1, cols2], axis=1)\n\n    # ------------------------------------\n\n    # Interpolate toutes les vectors\n    for i in range(len(df.columns)):\n        shortsig = df.iloc[:,i].dropna().to_numpy()\n        intrp = linear_intercurrentpt_makeshortSIGlong_interp1d(shortsig, longsig)\n        # ------------------------------------\n\n        # Scale toutes les vectors\n        type_scale = 'standardization'  # \n        df.iloc[:,i] = scale_feature_data(intrp, plotORnot, type_scale)\n\n    X = pd.concat([X, df], axis=0)\n\n\nX = X.to_numpy()\n\nsave_dat_pickle(X, file_name=\"X.pkl\")\nsave_dat_pickle(samps, file_name=\"samps.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.116154Z","iopub.status.idle":"2024-05-15T04:13:52.116557Z","shell.execute_reply.started":"2024-05-15T04:13:52.116367Z","shell.execute_reply":"2024-05-15T04:13:52.116384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Create Temp Folder\nfrom pathlib import Path\nTMP_DIR = Path('/kaggle/working/temp')\nTMP_DIR.mkdir(exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.118112Z","iopub.status.idle":"2024-05-15T04:13:52.12001Z","shell.execute_reply.started":"2024-05-15T04:13:52.119774Z","shell.execute_reply":"2024-05-15T04:13:52.119793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load audio and split into chunks and gen mel spect \ninp_path = '/kaggle/input/birdclef-2024/train_audio/'\nout_path = '/kaggle/working/temp/'\n\ndef load_audio(filename):\n    # 1) Load file\n    signal, sr = librosa.load(inp_path+filename)\n    soundscape = filename.split(\"/\")[-1].split(\".\")[0]\n    paths = []\n    \n    # 2) Splitting into chunks of 5s\n    splits = []\n    for i in range(0, len(signal), 5*sr):\n        split = signal[i:i + 5*sr]\n        if len(split) < 5*sr:\n            break    \n        splits.append(split)\n    \n    # 3) Generating a mel spectrogram for each 5s chunk\n    for i, chunk in enumerate(splits):\n        SPEC_SHAPE = (48, 150) \n        n_fft=2048\n        hop_length = int(5 * sr / (SPEC_SHAPE[1] - 1))\n        mel_spec = librosa.feature.melspectrogram(y=chunk, sr=sr, n_fft=n_fft, hop_length=hop_length, \n                                                  n_mels=SPEC_SHAPE[0], fmin=500, fmax=12500)\n\n        mel_spec = librosa.power_to_db(mel_spec, ref=np.max) \n        mel_spec -= mel_spec.min()\n        mel_spec /= mel_spec.max()\n        plt.figure(figsize=(1, 1))\n        librosa.display.specshow(mel_spec, sr=sr, hop_length=hop_length, x_axis='time', y_axis='mel')\n        plt.axis('off')\n        \n        # 4) Saving audios in the form of filename_{endtime}.jpg\n        path = f\"{out_path}{soundscape}_{(i+1)*5}\"\n        plt.savefig(path, bbox_inches='tight', pad_inches=0)\n        paths.append(path+\".png\")\n        plt.close()\n    return paths","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.121152Z","iopub.status.idle":"2024-05-15T04:13:52.122035Z","shell.execute_reply.started":"2024-05-15T04:13:52.121821Z","shell.execute_reply":"2024-05-15T04:13:52.121839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samples = []\nfor i, row in data_samp.iterrows():\n    samples += load_audio(row.filename)\nsamples = shuffle(samples)\nsamples","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.123751Z","iopub.status.idle":"2024-05-15T04:13:52.124314Z","shell.execute_reply.started":"2024-05-15T04:13:52.123999Z","shell.execute_reply":"2024-05-15T04:13:52.124021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 7))\nfor i in range(12):\n    spec = Image.open(samples[i])\n    plt.subplot(3, 4, i + 1)\n    plt.title(samples[i].split(os.sep)[-1])\n    plt.imshow(spec, origin='lower')\n    plt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.125645Z","iopub.status.idle":"2024-05-15T04:13:52.126186Z","shell.execute_reply.started":"2024-05-15T04:13:52.125897Z","shell.execute_reply":"2024-05-15T04:13:52.125921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Exploring the audio data","metadata":{}},{"cell_type":"code","source":"from IPython.display import Audio    \npath = '/kaggle/input/birdclef-2024/train_audio/' + data.iloc[5, -1]\nsignal, sr = librosa.load(path)\npath","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.127646Z","iopub.status.idle":"2024-05-15T04:13:52.128167Z","shell.execute_reply.started":"2024-05-15T04:13:52.12789Z","shell.execute_reply":"2024-05-15T04:13:52.127911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y, sr = librosa.load(path)\n\nprint('y:', y, '\\n')\nprint('y shape:', np.shape(y), '\\n')\nprint('Sample Rate (KHz):', sr, '\\n')\n\n# Verify length of the audio\nprint('Check Len of Audio:', np.shape(y)[0]/sr)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.129233Z","iopub.status.idle":"2024-05-15T04:13:52.129755Z","shell.execute_reply.started":"2024-05-15T04:13:52.129495Z","shell.execute_reply":"2024-05-15T04:13:52.129517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Loading Audio Files","metadata":{}},{"cell_type":"code","source":"\npath = '/kaggle/input/birdclef-2024/train_audio/' + data.iloc[200, -1]","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.131236Z","iopub.status.idle":"2024-05-15T04:13:52.131764Z","shell.execute_reply.started":"2024-05-15T04:13:52.131502Z","shell.execute_reply":"2024-05-15T04:13:52.131524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.iloc[170, -1], path","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.133099Z","iopub.status.idle":"2024-05-15T04:13:52.133633Z","shell.execute_reply.started":"2024-05-15T04:13:52.133369Z","shell.execute_reply":"2024-05-15T04:13:52.133391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"signal, sr = librosa.load(path)\nprint('Signal:', signal)\nprint('Shape of signal:', np.shape(signal))\nprint('Sampling Rate (KHz):', sr)\nprint('Duration of audio:', np.shape(signal)[0]/sr)\nplt.plot(signal)\nplt.show()\nipd.Audio(path)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.135021Z","iopub.status.idle":"2024-05-15T04:13:52.135837Z","shell.execute_reply.started":"2024-05-15T04:13:52.135281Z","shell.execute_reply":"2024-05-15T04:13:52.135319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Mel Spectrogram\ndata_samp = data.sample(10)\ndata_samp.head()\n# Create Temp Folder\nfrom pathlib import Path## Splitting data \nTMP_DIR = Path('/kaggle/working/temp')\nTMP_DIR.mkdir(exist_ok=True)\n# load audio and split into chunks and gen mel spect \ninp_path = '/kaggle/input/birdclef-2024/train_audio/'\nout_path = '/kaggle/working/temp/'\n\ndef load_audio(filename):\n    # 1) Load file\n    signal, sr = librosa.load(inp_path+filename)\n    soundscape = filename.split(\"/\")[-1].split(\".\")[0]\n    paths = []\n    \n    # 2) Splitting into chunks of 5s\n    splits = []\n    for i in range(0, len(signal), 5*sr):\n        split = signal[i:i + 5*sr]\n        if len(split) < 5*sr:\n            break    \n        splits.append(split)\n    \n    # 3) Generating a mel spectrogram for each 5s chunk\n    for i, chunk in enumerate(splits):\n        SPEC_SHAPE = (48, 150) \n        n_fft=2048\n        hop_length = int(5 * sr / (SPEC_SHAPE[1] - 1))\n        mel_spec = librosa.feature.melspectrogram(y=chunk, sr=sr, n_fft=n_fft, hop_length=hop_length, \n                                                  n_mels=SPEC_SHAPE[0], fmin=500, fmax=12500)\n\n        mel_spec = librosa.power_to_db(mel_spec, ref=np.max) \n        mel_spec -= mel_spec.min()\n        mel_spec /= mel_spec.max()\n        plt.figure(figsize=(1, 1))\n        librosa.display.specshow(mel_spec, sr=sr, hop_length=hop_length, x_axis='time', y_axis='mel')\n        plt.axis('off')\n        \n        # 4) Saving audios in the form of filename_{endtime}.jpg\n        path = f\"{out_path}{soundscape}_{(i+1)*5}\"\n        plt.savefig(path, bbox_inches='tight', pad_inches=0)\n        paths.append(path+\".png\")\n        plt.close()\n    return paths\n    \nsamples = []\nfor i, row in data_samp.iterrows():\n    samples += load_audio(row.filename)\nsamples = shuffle(samples)\nprint(samples)\nplt.figure(figsize=(15, 7))\nfor i in range(12):\n    spec = Image.open(samples[i])\n    plt.subplot(3, 4, i + 1)\n    plt.title(samples[i].split(os.sep)[-1])\n    plt.imshow(spec, origin='lower')\n    plt.tight_layout()\n### Splitting and storing spectrograms??\n### How to store along w labels??","metadata":{"execution":{"iopub.status.busy":"2024-05-15T04:13:52.138144Z","iopub.status.idle":"2024-05-15T04:13:52.138684Z","shell.execute_reply.started":"2024-05-15T04:13:52.138411Z","shell.execute_reply":"2024-05-15T04:13:52.138435Z"},"trusted":true},"execution_count":null,"outputs":[]}]}