{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-17T13:33:40.843587Z","iopub.execute_input":"2025-04-17T13:33:40.843916Z","iopub.status.idle":"2025-04-17T13:33:42.875455Z","shell.execute_reply.started":"2025-04-17T13:33:40.843891Z","shell.execute_reply":"2025-04-17T13:33:42.873482Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport numpy as np\nfrom tqdm import tqdm\ntqdm.pandas()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T18:35:37.429266Z","iopub.execute_input":"2025-04-28T18:35:37.429599Z","iopub.status.idle":"2025-04-28T18:35:37.435622Z","shell.execute_reply.started":"2025-04-28T18:35:37.429575Z","shell.execute_reply":"2025-04-28T18:35:37.433894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Define path to dataset\nbase_path = '/kaggle/input/birdclef-2025'\naudio_path = 'train_audio/'\n\nprint(os.listdir(base_path))  # to see available files and folders\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T18:09:23.720960Z","iopub.execute_input":"2025-04-28T18:09:23.721455Z","iopub.status.idle":"2025-04-28T18:09:23.730762Z","shell.execute_reply.started":"2025-04-28T18:09:23.721421Z","shell.execute_reply":"2025-04-28T18:09:23.729645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta= pd.read_csv(os.path.join(base_path, 'train.csv'))\nmeta['id'] = range(1, len(meta) + 1)\nmeta.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T18:09:26.364381Z","iopub.execute_input":"2025-04-28T18:09:26.364758Z","iopub.status.idle":"2025-04-28T18:09:26.888865Z","shell.execute_reply.started":"2025-04-28T18:09:26.364730Z","shell.execute_reply":"2025-04-28T18:09:26.887804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = meta[['id' , 'filename' , 'primary_label']]\n\ndf['filename'] = df['filename'].apply(lambda x: os.path.join(base_path, audio_path, x))\nfile_path = df['filename'][0]  # now this is a correct full path","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T19:05:37.347262Z","iopub.execute_input":"2025-04-28T19:05:37.347613Z","iopub.status.idle":"2025-04-28T19:05:37.404108Z","shell.execute_reply.started":"2025-04-28T19:05:37.347581Z","shell.execute_reply":"2025-04-28T19:05:37.403049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#sampling\ndf =df.head(500)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T19:05:42.551589Z","iopub.execute_input":"2025-04-28T19:05:42.551933Z","iopub.status.idle":"2025-04-28T19:05:42.556877Z","shell.execute_reply.started":"2025-04-28T19:05:42.551906Z","shell.execute_reply":"2025-04-28T19:05:42.555808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# load audio\n\ndef load_audio_from_filename(filename):\n    try:\n        y, sr = librosa.load(filename, sr=None)\n        return y, sr\n    except Exception as e:\n        print(f\"Error loading file {filename}: {e}\")\n        return None, None\n# Apply and split the results\ndf[['waveform', 'sample_rate']] = df['filename'].progress_apply(\n    lambda filename: pd.Series(load_audio_from_filename(filename))\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T19:05:46.948092Z","iopub.execute_input":"2025-04-28T19:05:46.948398Z","iopub.status.idle":"2025-04-28T19:06:13.892501Z","shell.execute_reply.started":"2025-04-28T19:05:46.948376Z","shell.execute_reply":"2025-04-28T19:06:13.891527Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T18:49:26.959547Z","iopub.execute_input":"2025-04-28T18:49:26.959841Z","iopub.status.idle":"2025-04-28T18:49:26.972652Z","shell.execute_reply.started":"2025-04-28T18:49:26.959818Z","shell.execute_reply":"2025-04-28T18:49:26.971429Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_mel_spectrogram(y, sr):\n    if y is not None and sr is not None:\n        try:\n            mel = librosa.feature.melspectrogram(y=y, sr=sr, n_mels=128)\n            mel_db = librosa.power_to_db(mel, ref=np.max)\n            return mel_db\n        except Exception as e:\n            print(f\"Error processing Mel Spectrogram: {e}\")\n            return None\n    else:\n        return None\n        \ndef extract_mfcc(y, sr):\n    if y is not None and sr is not None:\n        try:\n            mfcc_feat = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=13)\n            return mfcc_feat\n        except Exception as e:\n            print(f\"Error processing MFCC: {e}\")\n            return None\n    else:\n        return None\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T18:49:26.995780Z","iopub.execute_input":"2025-04-28T18:49:26.996073Z","iopub.status.idle":"2025-04-28T18:49:27.006754Z","shell.execute_reply.started":"2025-04-28T18:49:26.996049Z","shell.execute_reply":"2025-04-28T18:49:27.005734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extracting spectrogram\ndf['mel_spectrogram'] = df.progress_apply(lambda row: extract_mel_spectrogram(row['waveform'], row['sample_rate']), axis=1)\ndf['mfcc'] = df.progress_apply(lambda row: extract_mfcc(row['waveform'], row['sample_rate']), axis=1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T19:06:13.894229Z","iopub.execute_input":"2025-04-28T19:06:13.894585Z","iopub.status.idle":"2025-04-28T19:07:33.277390Z","shell.execute_reply.started":"2025-04-28T19:06:13.894555Z","shell.execute_reply":"2025-04-28T19:07:33.276396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T18:50:46.495362Z","iopub.execute_input":"2025-04-28T18:50:46.495698Z","iopub.status.idle":"2025-04-28T18:50:46.649516Z","shell.execute_reply.started":"2025-04-28T18:50:46.495670Z","shell.execute_reply":"2025-04-28T18:50:46.648773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to calculate the mean of Mel spectrogram\ndef calculate_mel_mean(mel_spectrogram):\n    return np.mean(mel_spectrogram, axis=1)\n\n# Function to calculate the variance of Mel spectrogram\ndef calculate_mel_var(mel_spectrogram):\n    return np.var(mel_spectrogram, axis=1)\n\n# Function to calculate the median of Mel spectrogram\ndef calculate_mel_median(mel_spectrogram):\n    return np.median(mel_spectrogram, axis=1)\n\n# Function to calculate the max of Mel spectrogram\ndef calculate_mel_max(mel_spectrogram):\n    return np.max(mel_spectrogram, axis=1)\n\n# Function to calculate the min of Mel spectrogram\ndef calculate_mel_min(mel_spectrogram):\n    return np.min(mel_spectrogram, axis=1)\n\n# Function to calculate the range (max - min) of Mel spectrogram\ndef calculate_mel_range(mel_spectrogram):\n    mel_max = np.max(mel_spectrogram, axis=1)\n    mel_min = np.min(mel_spectrogram, axis=1)\n    return mel_max - mel_min\n\n# Now, assuming the 'mel_spectrogram' column exists in the dataframe 'df', \n# we will apply each of these functions to the dataframe.\nfeatures_columns = ['mel_mean', 'mel_var', 'mel_median', 'mel_max', 'mel_min', 'mel_range']\n\ndf['mel_mean'] = df['mel_spectrogram'].progress_apply(calculate_mel_mean)\ndf['mel_var'] = df['mel_spectrogram'].progress_apply(calculate_mel_var)\n# df['mel_median'] = df['mel_spectrogram'].progress_apply(calculate_mel_median)\n# df['mel_max'] = df['mel_spectrogram'].progress_apply(calculate_mel_max)\n# df['mel_min'] = df['mel_spectrogram'].progress_apply(calculate_mel_min)\n# df['mel_range'] = df['mel_spectrogram'].progress_apply(calculate_mel_range)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T19:10:09.203563Z","iopub.execute_input":"2025-04-28T19:10:09.203919Z","iopub.status.idle":"2025-04-28T19:10:09.786281Z","shell.execute_reply.started":"2025-04-28T19:10:09.203884Z","shell.execute_reply":"2025-04-28T19:10:09.785156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T19:10:13.547303Z","iopub.execute_input":"2025-04-28T19:10:13.547652Z","iopub.status.idle":"2025-04-28T19:10:13.706950Z","shell.execute_reply.started":"2025-04-28T19:10:13.547629Z","shell.execute_reply":"2025-04-28T19:10:13.705985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Function to expand the feature column into separate bin columns\ndef expand_feature_column(df, feature_name):\n    # Check the first row to confirm the format (ensure it's a numpy array)\n    first_row = df[feature_name].iloc[0]\n    if not isinstance(first_row, np.ndarray):\n        raise ValueError(f\"Values in column '{feature_name}' are not numpy arrays.\")\n    \n    # Get the number of bins (assuming each row contains the same number of bins)\n    n_bins = len(first_row)  # The length of the array in the first row\n\n    # Create a list of new columns for the bins\n    bin_columns = []\n    for bin_idx in range(n_bins):\n        # Create a new column for each bin (e.g., mel_mean_bin_1, mel_mean_bin_2, ...)\n        bin_columns.append(\n            df[feature_name].apply(lambda x: x[bin_idx] if isinstance(x, np.ndarray) else None)\n        )\n        \n    # Combine the list of bin columns into a DataFrame\n    bin_df = pd.concat(bin_columns, axis=1)\n    \n    # Rename the new columns appropriately (e.g., mel_mean_bin_1, mel_mean_bin_2, ...)\n    bin_df.columns = [f'{feature_name}_bin_{bin_idx + 1}' for bin_idx in range(n_bins)]\n    \n\n    # Drop the original feature column (optional)\n    #df.drop(columns=[feature_name], inplace=True)\n    \n    return bin_df\n\nfeatures_columns = ['mel_mean', 'mel_var']\n\ndf_expanded = df\nfeatures = [] \nfor feature in features_columns : \n    \n    expandeddf = expand_feature_column(df, feature)\n    features = features + list(expandeddf.columns)\n    df = pd.concat([df, expandeddf], axis=1)\n\ndf.head(2)\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T19:22:56.607601Z","iopub.execute_input":"2025-04-28T19:22:56.607935Z","iopub.status.idle":"2025-04-28T19:22:56.878389Z","shell.execute_reply.started":"2025-04-28T19:22:56.607911Z","shell.execute_reply":"2025-04-28T19:22:56.877547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Create the encoder\nle = LabelEncoder()\n\n# Fit and transform the column\ndf['primary_label_encoded'] = le.fit_transform(df['primary_label'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T19:23:46.884293Z","iopub.execute_input":"2025-04-28T19:23:46.884631Z","iopub.status.idle":"2025-04-28T19:23:46.890267Z","shell.execute_reply.started":"2025-04-28T19:23:46.884608Z","shell.execute_reply":"2025-04-28T19:23:46.889323Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"data splitting","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Suppose you have features X and labels y\nX_train, X_test, y_train, y_test = train_test_split(\n    df[features],          # your features\n    df['primary_label_encoded'],          # your labels\n    test_size=0.2,  # 20% for test set\n    random_state=42 # (optional) to make it reproducible\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T19:23:54.152748Z","iopub.execute_input":"2025-04-28T19:23:54.153076Z","iopub.status.idle":"2025-04-28T19:23:54.305833Z","shell.execute_reply.started":"2025-04-28T19:23:54.153053Z","shell.execute_reply":"2025-04-28T19:23:54.304821Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"some scalling","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\nscaler = MinMaxScaler()\n\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T19:24:58.162847Z","iopub.execute_input":"2025-04-28T19:24:58.163161Z","iopub.status.idle":"2025-04-28T19:24:58.185254Z","shell.execute_reply.started":"2025-04-28T19:24:58.163138Z","shell.execute_reply":"2025-04-28T19:24:58.184227Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"xgboost training with some default parameters ","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBClassifier \n\n# Create the XGBClassifier\nmodel = XGBClassifier(\n    n_estimators=10,   # number of trees\n    learning_rate=0.1,  # step size shrinkage\n    max_depth=6,        # maximum depth of a tree\n    random_state=42     # for reproducibility\n)\n\n# Fit the model on training data\nmodel.fit(X_train_scaled, y_train)\n\n# Predict on test data\ny_pred = model.predict(X_test_scaled)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T19:25:46.061203Z","iopub.execute_input":"2025-04-28T19:25:46.061575Z","iopub.status.idle":"2025-04-28T19:25:48.138746Z","shell.execute_reply.started":"2025-04-28T19:25:46.061549Z","shell.execute_reply":"2025-04-28T19:25:48.138059Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" Multi-Class Classification and Feature Contributions\n\n 100 samples,\n 31 classes,\n 257 features + bias\n \nPositive contributions for a class: This means that the feature is pushing the model towards that specific class. The larger the positive value, the more it favors that class.\n\nNegative contributions for a class: This means that the feature is pulling the model away from that specific class and towards other classes.\n\nso feature values that are close to 0  has no influce to the mdoel","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb\nbooster = model.get_booster()\n\n# Predict with feature contributions\nmodel_pred_detail = booster.predict(xgb.DMatrix(X_test_scaled), pred_contribs=True)\nprint(model_pred_detail.shape)\nmodel_pred_detail","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T20:06:22.986260Z","iopub.execute_input":"2025-04-28T20:06:22.986653Z","iopub.status.idle":"2025-04-28T20:06:23.027828Z","shell.execute_reply.started":"2025-04-28T20:06:22.986626Z","shell.execute_reply":"2025-04-28T20:06:23.026990Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"ideally we can ommit the features value that are in IQR range and work with nly extreme features ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\n# Calculate the mean contribution for each feature across all samples\nmean_contribution_per_feature = np.mean(model_pred_detail[:, :, :-1], axis=(0, 1))  # Exclude the last column (class probabilities)\n\n# Plot the mean contributions per feature using a box plot\nplt.figure(figsize=(12, 8))\nplt.boxplot(mean_contribution_per_feature.T, vert=False)  # Transpose to have features as boxes\nplt.xlabel(\"Mean Feature Contribution\")\nplt.ylabel(\"Feature Index\")\nplt.title(\"Box Plot of Mean Feature Contributions\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T19:57:35.442040Z","iopub.execute_input":"2025-04-28T19:57:35.442358Z","iopub.status.idle":"2025-04-28T19:57:35.585786Z","shell.execute_reply.started":"2025-04-28T19:57:35.442335Z","shell.execute_reply":"2025-04-28T19:57:35.584816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# If the mean_contribution_per_feature is 1D, you can reshape it into a 2D array\nmean_contribution_per_feature = np.mean(model_pred_detail[:, :, :-1], axis=(0, 1))\n\n# Reshaping to make it 2D (for heatmap compatibility, 1 feature per row)\nmean_contribution_per_feature_2d = mean_contribution_per_feature.reshape(1, -1)\n\n# Plot the heatmap\nplt.figure(figsize=(10, 2))  # Adjust the figure size as needed\nsns.heatmap(mean_contribution_per_feature_2d, cmap=\"YlGnBu\")\nplt.title(\"Mean Feature Contribution Heatmap\")\nplt.xlabel(\"Features\")\nplt.ylabel(\"Mean Contribution\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T19:56:10.770177Z","iopub.execute_input":"2025-04-28T19:56:10.770523Z","iopub.status.idle":"2025-04-28T19:56:11.167091Z","shell.execute_reply.started":"2025-04-28T19:56:10.770498Z","shell.execute_reply":"2025-04-28T19:56:11.165988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(np.mean(model_pred_detail, axis=0)[0] ) # axis=0 means the mean along rows (across samples)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T19:39:24.305229Z","iopub.execute_input":"2025-04-28T19:39:24.306006Z","iopub.status.idle":"2025-04-28T19:39:24.314294Z","shell.execute_reply.started":"2025-04-28T19:39:24.305978Z","shell.execute_reply":"2025-04-28T19:39:24.313320Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport librosa.display\n\ndef plot_mel_and_mfcc(df, index=0):\n    mel = df.loc[index, 'mel_spectrogram']\n    mfcc = df.loc[index, 'mfcc']\n    sr = df.loc[index, 'sample_rate']\n\n    plt.figure(figsize=(12, 6))\n\n    # Plot Mel Spectrogram\n    plt.subplot(1, 2, 1)\n    librosa.display.specshow(mel, sr=sr, x_axis='time', y_axis='mel', cmap='magma')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title('Mel Spectrogram')\n\n    # Plot MFCC\n    plt.subplot(1, 2, 2)\n    librosa.display.specshow(mfcc, sr=sr, x_axis='time', cmap='coolwarm')\n    plt.colorbar()\n    plt.title('MFCC')\n\n    plt.tight_layout()\n    plt.show()\nplot_mel_and_mfcc(df, index=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T08:38:26.963840Z","iopub.execute_input":"2025-04-22T08:38:26.964376Z","iopub.status.idle":"2025-04-22T08:38:28.478236Z","shell.execute_reply.started":"2025-04-22T08:38:26.964351Z","shell.execute_reply":"2025-04-22T08:38:28.477039Z"}},"outputs":[],"execution_count":null}]}