{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":8900,"databundleVersionId":862232,"sourceType":"competition"},{"sourceId":10053631,"sourceType":"datasetVersion","datasetId":6194737},{"sourceId":10054127,"sourceType":"datasetVersion","datasetId":6195136},{"sourceId":10055938,"sourceType":"datasetVersion","datasetId":6140397},{"sourceId":10160419,"sourceType":"datasetVersion","datasetId":6273900},{"sourceId":10176313,"sourceType":"datasetVersion","datasetId":6285534},{"sourceId":10195018,"sourceType":"datasetVersion","datasetId":6299433},{"sourceId":10195930,"sourceType":"datasetVersion","datasetId":6299925},{"sourceId":10200733,"sourceType":"datasetVersion","datasetId":6303495},{"sourceId":10202663,"sourceType":"datasetVersion","datasetId":6304968},{"sourceId":10203238,"sourceType":"datasetVersion","datasetId":6305393},{"sourceId":1378,"sourceType":"modelInstanceVersion","modelInstanceId":1163,"modelId":162}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Freesound Submission","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"# Initial Setup","metadata":{}},{"cell_type":"markdown","source":"- We will load previosuly tranformed test data based upon the data-preprocessing steps performed in the data preparation phase\n    * The transformed data for the test set is saved as (/kaggle/input/freesound-x-test-5-trim10/X_test_vggish_mffc_hubert_features_5_trim10.npy)\n    * The code to transform the data is commented out below, but could be used following a re-run of the data preparation notebook aand saving/loading the feature extraction pipleine","metadata":{}},{"cell_type":"markdown","source":"### Import Statements","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport joblib\nimport pickle\nimport librosa\nimport random\nimport wave\nimport torch\nimport torchaudio\n\nfrom scipy.signal import wiener\nfrom tensorflow import keras\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import load_model\nimport tensorflow as tf\nimport tensorflow_hub as hub\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom transformers import Wav2Vec2Processor, HubertModel","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:43:32.936762Z","iopub.execute_input":"2024-12-15T03:43:32.937371Z","iopub.status.idle":"2024-12-15T03:43:57.443010Z","shell.execute_reply.started":"2024-12-15T03:43:32.937311Z","shell.execute_reply":"2024-12-15T03:43:57.442124Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load Dataframe","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/freesound-audio-tagging/test_post_competition.csv')\n\nprint('Test Set Size:', test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-12-15T03:43:57.445271Z","iopub.execute_input":"2024-12-15T03:43:57.446514Z","iopub.status.idle":"2024-12-15T03:43:57.486387Z","shell.execute_reply.started":"2024-12-15T03:43:57.446459Z","shell.execute_reply":"2024-12-15T03:43:57.485228Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-15T03:43:57.487674Z","iopub.execute_input":"2024-12-15T03:43:57.488018Z","iopub.status.idle":"2024-12-15T03:43:57.510089Z","shell.execute_reply.started":"2024-12-15T03:43:57.487984Z","shell.execute_reply":"2024-12-15T03:43:57.508795Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load Saved X_test Features","metadata":{}},{"cell_type":"code","source":"# Load X_test file that has been saved after previous pre-processing\nX_test = np.load('/kaggle/input/freesound-x-test-5-trim10/X_test_vggish_mffc_hubert_features_5_trim10.npy')","metadata":{"execution":{"iopub.status.busy":"2024-12-15T03:43:57.512323Z","iopub.execute_input":"2024-12-15T03:43:57.513165Z","iopub.status.idle":"2024-12-15T03:43:57.889554Z","shell.execute_reply.started":"2024-12-15T03:43:57.513109Z","shell.execute_reply":"2024-12-15T03:43:57.888346Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:43:57.890967Z","iopub.execute_input":"2024-12-15T03:43:57.891324Z","iopub.status.idle":"2024-12-15T03:43:57.898836Z","shell.execute_reply.started":"2024-12-15T03:43:57.891288Z","shell.execute_reply":"2024-12-15T03:43:57.897703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing (As required)\n- Since we are already loading in previously transformed test data, we do not need to use the cells below","metadata":{}},{"cell_type":"markdown","source":"test_audio_dir = '/kaggle/input/freesound-audio-tagging/audio_test'\n\nvggish_model = hub.load('https://kaggle.com/models/google/vggish/frameworks/TensorFlow2/variations/vggish/versions/1')\nprocessor = Wav2Vec2Processor.from_pretrained(\"facebook/hubert-large-ls960-ft\")\nhubert_model = HubertModel.from_pretrained(\"facebook/hubert-large-ls960-ft\")\n\n# VGGish feature extraction\nclass VGGishFeatureExtractor(BaseEstimator, TransformerMixin):\n    def fit(self, X, y=None):\n        return self\n\n    def transform(self, X):\n        return np.array([self.extract_vggish_embeddings(file_path) for file_path in X])\n    \n    def extract_vggish_embeddings(self, file_path):\n        sample, _ = librosa.load(file_path, sr=16000)\n        sample, _ = librosa.effects.trim(sample, top_db=10)\n        if len(sample) < 5 * 16000:\n            sample = np.pad(sample, (0, 5 * 16000 - len(sample)))\n        else:\n            sample = sample[:5 * 16000]\n        if np.max(np.abs(sample)) > 0:\n            sample = sample / np.max(np.abs(sample))\n        embeddings = vggish_model(sample)\n        embeddings_flattened = embeddings.numpy().flatten()\n        return embeddings_flattened\n\n# HuBERT feature extraction\nclass HubertFeatureExtractor(BaseEstimator, TransformerMixin):\n    def fit(self, X, y=None):\n        return self\n\n    def transform(self, X):\n        return np.array([self.extract_features(path) for path in X])\n\n    def extract_features(self, file_path):\n        sample, sr = librosa.load(file_path, sr=16000)\n        sample, _ = librosa.effects.trim(sample, top_db=10)\n        \n        if len(sample) < 5 * 16000:\n            sample = np.pad(sample, (0, 5 * 16000 - len(sample)))\n        else:\n            sample = sample[:5 * 16000]\n\n        if np.max(np.abs(sample)) > 0:\n            sample = sample / np.max(np.abs(sample))\n\n        input_values = processor(sample, sampling_rate=16000, return_tensors=\"pt\").input_values\n        with torch.no_grad():\n            hidden_states = hubert_model(input_values).last_hidden_state\n        features = hidden_states.mean(dim=1).squeeze().numpy()\n\n        return features\n\n# MFCC feature extraction\nclass MFCCFeatureExtractor(BaseEstimator, TransformerMixin):\n    def __init__(self, n_mfcc=40, sr=16000):\n        self.n_mfcc = n_mfcc\n        self.sr = sr\n\n    def fit(self, X, y=None):\n        return self\n\n    def transform(self, X):\n        return np.array([self.extract_features(path) for path in X])\n\n    def extract_features(self, file_path):\n        y, sr = librosa.load(file_path, sr=self.sr)\n        y, _ = librosa.effects.trim(y, top_db=10)\n\n        if len(y) < 5 * sr:\n            y = np.pad(y, (0, 5 * sr - len(y)))\n        else:\n            y = y[:5 * sr]\n\n        if np.max(np.abs(y)) > 0:\n            y = y / np.max(np.abs(y))\n\n        mfccs = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=self.n_mfcc)\n        return mfccs.mean(axis=1)\n\n# LogMel feature extraction\nclass LogMelFeatureExtractor(BaseEstimator, TransformerMixin):\n    def __init__(self, n_mels=64, sr=16000):\n        self.n_mels = n_mels\n        self.sr = sr\n\n    def fit(self, X, y=None):\n        return self\n\n    def transform(self, X):\n        return np.array([self.extract_features(path) for path in X])\n\n    def extract_features(self, file_path):\n        y, sr = librosa.load(file_path, sr=self.sr)\n        y, _ = librosa.effects.trim(y, top_db=10)\n\n        if len(y) < 5 * sr:\n            y = np.pad(y, (0, 5 * sr - len(y)))\n        else:\n            y = y[:5 * sr]\n\n        if np.max(np.abs(y)) > 0:\n            y = y / np.max(np.abs(y))\n\n        mel_spectrogram = librosa.feature.melspectrogram(y=y, sr=sr, n_mels=self.n_mels)\n        log_mel_spectrogram = librosa.power_to_db(mel_spectrogram)\n        return log_mel_spectrogram.mean(axis=1)\n","metadata":{}},{"cell_type":"markdown","source":"features_pipeline = joblib.load('/kaggle/input/freesound-x-train-x-test-vhml-mv/feature_extraction_pipeline_mv_vhml.joblib')","metadata":{"execution":{"iopub.status.busy":"2024-11-30T01:18:51.655674Z","iopub.execute_input":"2024-11-30T01:18:51.656064Z","iopub.status.idle":"2024-11-30T01:18:51.663817Z","shell.execute_reply.started":"2024-11-30T01:18:51.656031Z","shell.execute_reply":"2024-11-30T01:18:51.662225Z"}}},{"cell_type":"markdown","source":"test_audio_paths = [os.path.join(test_audio_dir, fname) for fname in test['fname']] \nX_test = features_pipeline.transform(test_audio_paths)","metadata":{"execution":{"iopub.status.busy":"2024-11-30T01:18:55.103204Z","iopub.execute_input":"2024-11-30T01:18:55.103775Z","iopub.status.idle":"2024-11-30T07:00:19.600510Z","shell.execute_reply.started":"2024-11-30T01:18:55.103724Z","shell.execute_reply":"2024-11-30T07:00:19.597961Z"}}},{"cell_type":"markdown","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-30T07:00:41.251778Z","iopub.execute_input":"2024-11-30T07:00:41.253036Z","iopub.status.idle":"2024-11-30T07:00:41.261863Z","shell.execute_reply.started":"2024-11-30T07:00:41.252984Z","shell.execute_reply":"2024-11-30T07:00:41.260355Z"}}},{"cell_type":"markdown","source":"X_test = X_test.reshape(-1, 1768, 1)","metadata":{"execution":{"iopub.status.busy":"2024-11-30T07:00:43.536219Z","iopub.execute_input":"2024-11-30T07:00:43.537343Z","iopub.status.idle":"2024-11-30T07:00:43.543695Z","shell.execute_reply.started":"2024-11-30T07:00:43.537270Z","shell.execute_reply":"2024-11-30T07:00:43.542247Z"}}},{"cell_type":"markdown","source":"np.save('X_test_vggish_hubert_mffc_logmel_mv_features_5_trim10.npy', X_test)","metadata":{"execution":{"iopub.status.busy":"2024-11-30T07:00:44.999621Z","iopub.execute_input":"2024-11-30T07:00:45.000010Z","iopub.status.idle":"2024-11-30T07:00:45.106674Z","shell.execute_reply.started":"2024-11-30T07:00:44.999977Z","shell.execute_reply":"2024-11-30T07:00:45.105247Z"}}},{"cell_type":"markdown","source":"# Evaluation\n- We will use an ensemble approach to evaluate the final model. The best model from each fold during training will be used to create an average prediction for each sample","metadata":{}},{"cell_type":"markdown","source":"### Load Saved Models","metadata":{}},{"cell_type":"code","source":"model1 = tf.keras.models.load_model('/kaggle/input/freesound-v70/best_model_vggish_fold_1.keras')\nmodel2 = tf.keras.models.load_model('/kaggle/input/freesound-v70/best_model_vggish_fold_2.keras')\nmodel3 = tf.keras.models.load_model('/kaggle/input/freesound-v70/best_model_vggish_fold_3.keras')\nmodel4 = tf.keras.models.load_model('/kaggle/input/freesound-v70/best_model_vggish_fold_4.keras')\nmodel5 = tf.keras.models.load_model('/kaggle/input/freesound-v70/best_model_vggish_fold_5.keras')\nmodel6 = tf.keras.models.load_model('/kaggle/input/freesound-v70/best_model_vggish_fold_6.keras')\nmodel7 = tf.keras.models.load_model('/kaggle/input/freesound-v70/best_model_vggish_fold_7.keras')\nmodel8 = tf.keras.models.load_model('/kaggle/input/freesound-v70/best_model_vggish_fold_8.keras')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:57:39.113672Z","iopub.execute_input":"2024-12-15T03:57:39.114757Z","iopub.status.idle":"2024-12-15T03:58:42.735480Z","shell.execute_reply.started":"2024-12-15T03:57:39.114712Z","shell.execute_reply":"2024-12-15T03:58:42.734172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"models = [model1, model2, model3, model4, model5, model6]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:58:42.737591Z","iopub.execute_input":"2024-12-15T03:58:42.737960Z","iopub.status.idle":"2024-12-15T03:58:42.743449Z","shell.execute_reply.started":"2024-12-15T03:58:42.737927Z","shell.execute_reply":"2024-12-15T03:58:42.742095Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load Encoder","metadata":{}},{"cell_type":"code","source":"le = joblib.load('/kaggle/input/freesound-v70/label_encoder_v70.pkl')","metadata":{"execution":{"iopub.status.busy":"2024-12-15T03:58:42.745092Z","iopub.execute_input":"2024-12-15T03:58:42.745618Z","iopub.status.idle":"2024-12-15T03:58:42.766163Z","shell.execute_reply.started":"2024-12-15T03:58:42.745558Z","shell.execute_reply":"2024-12-15T03:58:42.764733Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Make Test Predictions\n- Below we define a function to perform an ensemble prediction using the best model from each training fold","metadata":{}},{"cell_type":"code","source":"# Define the ensemble prediction\ndef ensemble_predict(models, inputs):\n    predictions = [model.predict(inputs) for model in models]\n    avg_prediction = np.mean(predictions, axis=0)\n    return avg_prediction","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:58:42.768913Z","iopub.execute_input":"2024-12-15T03:58:42.769286Z","iopub.status.idle":"2024-12-15T03:58:42.775624Z","shell.execute_reply.started":"2024-12-15T03:58:42.769240Z","shell.execute_reply":"2024-12-15T03:58:42.774443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_predictions = ensemble_predict(models, X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:58:42.776955Z","iopub.execute_input":"2024-12-15T03:58:42.777332Z","iopub.status.idle":"2024-12-15T04:01:31.961676Z","shell.execute_reply.started":"2024-12-15T03:58:42.777284Z","shell.execute_reply":"2024-12-15T04:01:31.960518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_predictions_array = np.array(final_predictions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T04:01:31.963663Z","iopub.execute_input":"2024-12-15T04:01:31.964048Z","iopub.status.idle":"2024-12-15T04:01:31.969120Z","shell.execute_reply.started":"2024-12-15T04:01:31.964013Z","shell.execute_reply":"2024-12-15T04:01:31.967946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(final_predictions_array[0, :].round(2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T04:01:31.970421Z","iopub.execute_input":"2024-12-15T04:01:31.970783Z","iopub.status.idle":"2024-12-15T04:01:31.988967Z","shell.execute_reply.started":"2024-12-15T04:01:31.970751Z","shell.execute_reply":"2024-12-15T04:01:31.987500Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Create the Submission Dataframe","metadata":{}},{"cell_type":"code","source":"#Import the example submission to use in creating the final submission with our predictions\nsubmission = test[['fname', 'label']].copy()\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-15T04:01:31.990242Z","iopub.execute_input":"2024-12-15T04:01:31.990648Z","iopub.status.idle":"2024-12-15T04:01:32.020168Z","shell.execute_reply.started":"2024-12-15T04:01:31.990612Z","shell.execute_reply":"2024-12-15T04:01:32.018992Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Since the final submission requires Mean Average Precision at 3 (MAP@3), we need to create a loop that will take our prediction array and select and return the 3 highest percentage predictions for each sample in the label column","metadata":{}},{"cell_type":"code","source":"N_test = len(submission)\nfor i in range(N_test):\n    p = final_predictions[i, :]\n    idx = np.argsort(-p)[:3]\n    top3 = le.classes_[idx]\n    submission.at[i, 'label'] = ' '.join(top3)","metadata":{"execution":{"iopub.status.busy":"2024-12-15T04:01:32.021946Z","iopub.execute_input":"2024-12-15T04:01:32.022311Z","iopub.status.idle":"2024-12-15T04:01:32.272235Z","shell.execute_reply.started":"2024-12-15T04:01:32.022275Z","shell.execute_reply":"2024-12-15T04:01:32.271160Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Now we will look at our returned predictions for MAP@3","metadata":{}},{"cell_type":"code","source":"submission.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-12-15T04:01:32.275138Z","iopub.execute_input":"2024-12-15T04:01:32.275573Z","iopub.status.idle":"2024-12-15T04:01:32.287570Z","shell.execute_reply.started":"2024-12-15T04:01:32.275533Z","shell.execute_reply":"2024-12-15T04:01:32.286139Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save Sumbission Dataframe as csv","metadata":{}},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-12-15T04:01:32.289240Z","iopub.execute_input":"2024-12-15T04:01:32.289651Z","iopub.status.idle":"2024-12-15T04:01:32.325393Z","shell.execute_reply.started":"2024-12-15T04:01:32.289616Z","shell.execute_reply":"2024-12-15T04:01:32.324065Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}