{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":8900,"databundleVersionId":862232,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Source \n - https://huggingface.co/MIT/ast-finetuned-audioset-10-10-0.4593\n - https://huggingface.co/docs/transformers/main/en/model_doc/audio-spectrogram-transformer#transformers.ASTForAudioClassification.forward.example","metadata":{}},{"cell_type":"code","source":"from huggingface_hub import hf_hub_download\nimport IPython\n\nfilepath = hf_hub_download(repo_id=\"nielsr/audio-spectogram-transformer-checkpoint\",\n                           filename=\"sample_audio.flac\",\n                           repo_type=\"dataset\")\n\nIPython.display.Audio(filepath)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T04:06:27.873595Z","iopub.execute_input":"2023-11-12T04:06:27.874576Z","iopub.status.idle":"2023-11-12T04:06:29.356093Z","shell.execute_reply.started":"2023-11-12T04:06:27.874531Z","shell.execute_reply":"2023-11-12T04:06:29.354792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"-   prepare the audio using ASTFeatureExtractor, which turns it into a tensor of shape (batch_size, time_dimension, frequency_dimension).","metadata":{}},{"cell_type":"code","source":"from transformers import ASTFeatureExtractor\n\nfeature_extractor = ASTFeatureExtractor()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T04:06:29.357549Z","iopub.execute_input":"2023-11-12T04:06:29.358074Z","iopub.status.idle":"2023-11-12T04:06:36.491379Z","shell.execute_reply.started":"2023-11-12T04:06:29.358030Z","shell.execute_reply":"2023-11-12T04:06:36.490276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchaudio\n\nwaveform, sampling_rate = torchaudio.load(filepath)\nwaveform = waveform.squeeze().numpy()\n\nwaveform.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-12T04:06:36.494118Z","iopub.execute_input":"2023-11-12T04:06:36.494637Z","iopub.status.idle":"2023-11-12T04:06:36.536388Z","shell.execute_reply.started":"2023-11-12T04:06:36.494607Z","shell.execute_reply":"2023-11-12T04:06:36.535204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs = feature_extractor(waveform, sampling_rate=sampling_rate, padding=\"max_length\", return_tensors=\"pt\")\ninput_values = inputs.input_values\nprint(input_values.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T04:06:36.537553Z","iopub.execute_input":"2023-11-12T04:06:36.537893Z","iopub.status.idle":"2023-11-12T04:06:36.700804Z","shell.execute_reply.started":"2023-11-12T04:06:36.537865Z","shell.execute_reply":"2023-11-12T04:06:36.699458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoModelForAudioClassification\n\nmodel = AutoModelForAudioClassification.from_pretrained(\"MIT/ast-finetuned-audioset-10-10-0.4593\")","metadata":{"execution":{"iopub.status.busy":"2023-11-12T04:06:36.702434Z","iopub.execute_input":"2023-11-12T04:06:36.702985Z","iopub.status.idle":"2023-11-12T04:06:51.481548Z","shell.execute_reply.started":"2023-11-12T04:06:36.702944Z","shell.execute_reply":"2023-11-12T04:06:51.480365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Forward pass\n- forward the audio through the model! We perform an argmax on the model's logits to get the predicted class index. We use model.config.id2label to turn that back into text.","metadata":{}},{"cell_type":"code","source":"import torch\n\nwith torch.no_grad():\n  outputs = model(input_values)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T04:06:51.483388Z","iopub.execute_input":"2023-11-12T04:06:51.484194Z","iopub.status.idle":"2023-11-12T04:06:55.776888Z","shell.execute_reply.started":"2023-11-12T04:06:51.484151Z","shell.execute_reply":"2023-11-12T04:06:55.775668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_class_idx = outputs.logits.argmax(-1).item()\nprint(\"Predicted class:\", model.config.id2label[predicted_class_idx])","metadata":{"execution":{"iopub.status.busy":"2023-11-12T04:06:55.778505Z","iopub.execute_input":"2023-11-12T04:06:55.779430Z","iopub.status.idle":"2023-11-12T04:06:55.789511Z","shell.execute_reply.started":"2023-11-12T04:06:55.779357Z","shell.execute_reply":"2023-11-12T04:06:55.788510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install librosa","metadata":{"execution":{"iopub.status.busy":"2023-11-12T04:13:19.070802Z","iopub.execute_input":"2023-11-12T04:13:19.071235Z","iopub.status.idle":"2023-11-12T04:13:34.884330Z","shell.execute_reply.started":"2023-11-12T04:13:19.071205Z","shell.execute_reply":"2023-11-12T04:13:34.882915Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\n\ntry:\n    melspec = librosa.feature.melspec\nexcept AttributeError:\n    print(\"The `melspec` function is not available in this version of librosa.\")\nelse:\n    print(\"The `melspec` function is available.\")\n","metadata":{"execution":{"iopub.status.busy":"2023-11-12T04:13:58.329563Z","iopub.execute_input":"2023-11-12T04:13:58.330051Z","iopub.status.idle":"2023-11-12T04:13:58.337127Z","shell.execute_reply.started":"2023-11-12T04:13:58.330011Z","shell.execute_reply":"2023-11-12T04:13:58.336223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport matplotlib.pyplot as plt\n\n# Load the audio file\ny, sr = librosa.load('/kaggle/input/freesound-audio-tagging/audio_train/0033e230.wav')\n\n# Create the waveform plot\nplt.figure()\nplt.plot(y)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-12T04:21:05.901393Z","iopub.execute_input":"2023-11-12T04:21:05.901830Z","iopub.status.idle":"2023-11-12T04:21:06.141040Z","shell.execute_reply.started":"2023-11-12T04:21:05.901791Z","shell.execute_reply":"2023-11-12T04:21:06.139364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport matplotlib.pyplot as plt\n\n# Load the audio file\ny, sr = librosa.load('/kaggle/input/freesound-audio-tagging/audio_train/003b91e8.wav')\n\n# Create the spectrogram\nD = librosa.stft(y, n_fft=2048)\nmagnitude = np.abs(D)\nplt.figure()\nplt.imshow(magnitude)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-12T04:21:41.190619Z","iopub.execute_input":"2023-11-12T04:21:41.191085Z","iopub.status.idle":"2023-11-12T04:21:41.460721Z","shell.execute_reply.started":"2023-11-12T04:21:41.191054Z","shell.execute_reply":"2023-11-12T04:21:41.459210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport matplotlib.pyplot as plt\n\n# Load the audio file\ny, sr = librosa.load('/kaggle/input/freesound-audio-tagging/audio_train/002d256b.wav')\n\n# Create the chromagram\nC = librosa.feature.chroma_stft(y=y, sr=sr)\nplt.figure()\nplt.imshow(C)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-12T04:20:34.639977Z","iopub.execute_input":"2023-11-12T04:20:34.640361Z","iopub.status.idle":"2023-11-12T04:20:34.914909Z","shell.execute_reply.started":"2023-11-12T04:20:34.640333Z","shell.execute_reply":"2023-11-12T04:20:34.913587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}