{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook you will find how to classify audio by:\n- Transforming the audio data to the spectrogram\n- Use 'resnet18 bag of tricks model' without pretrained to classify the spectrogram image\n\n\nNotes:\n- torchaudio.load is way faster than librosa.load (I speed up 10 times of the loading by doing so). However the shape of the output is different to the librosa way, and it returns 2 channels. For now, I just take the 1st channel to put to the model, if you know what is going on and how to deal with multiple channels audio, please let me know in the comments","metadata":{}},{"cell_type":"code","source":"from fastai.vision.all import *\nimport torchaudio\nimport librosa","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-14T20:19:16.050423Z","iopub.execute_input":"2022-03-14T20:19:16.050778Z","iopub.status.idle":"2022-03-14T20:19:19.991157Z","shell.execute_reply.started":"2022-03-14T20:19:16.050696Z","shell.execute_reply":"2022-03-14T20:19:19.990320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/kaggle-pog-series-s01e02/train.csv')\ndf_test = pd.read_csv('../input/kaggle-pog-series-s01e02/test.csv')\nsubmission = pd.read_csv('../input/kaggle-pog-series-s01e02/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:19:19.992966Z","iopub.execute_input":"2022-03-14T20:19:19.993214Z","iopub.status.idle":"2022-03-14T20:19:20.054323Z","shell.execute_reply.started":"2022-03-14T20:19:19.993180Z","shell.execute_reply":"2022-03-14T20:19:20.053644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = Path('../input/kaggle-pog-series-s01e02/train/')\ntest_path = Path('../input/kaggle-pog-series-s01e02/test/')","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:19:20.055592Z","iopub.execute_input":"2022-03-14T20:19:20.056077Z","iopub.status.idle":"2022-03-14T20:19:20.059709Z","shell.execute_reply.started":"2022-03-14T20:19:20.056040Z","shell.execute_reply":"2022-03-14T20:19:20.059072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_files = get_files(train_path, extensions='.ogg')","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:19:20.062016Z","iopub.execute_input":"2022-03-14T20:19:20.062531Z","iopub.status.idle":"2022-03-14T20:19:38.202715Z","shell.execute_reply.started":"2022-03-14T20:19:20.062493Z","shell.execute_reply":"2022-03-14T20:19:38.202019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Excluded unusual music thanks to this thread: https://www.kaggle.com/c/kaggle-pog-series-s01e02/discussion/312842\ndef get_items(path): \n    excluded_files = [\"010449.ogg\" , \"005589.ogg\" , \"004921.ogg\", \"019511.ogg\" , \"013375.ogg\" , \"024247.ogg\", \"024156.ogg\"]\n    items = get_files(path, extensions='.ogg')\n    items = [item for item in items if item.name not in excluded_files]\n#     items.shuffle()\n    return L(items)","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:19:38.204080Z","iopub.execute_input":"2022-03-14T20:19:38.204322Z","iopub.status.idle":"2022-03-14T20:19:38.209476Z","shell.execute_reply.started":"2022-03-14T20:19:38.204289Z","shell.execute_reply":"2022-03-14T20:19:38.208641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_FFT = 2048\nHOP_LEN = 512","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:19:38.211016Z","iopub.execute_input":"2022-03-14T20:19:38.211522Z","iopub.status.idle":"2022-03-14T20:19:38.219047Z","shell.execute_reply.started":"2022-03-14T20:19:38.211485Z","shell.execute_reply":"2022-03-14T20:19:38.218337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_y(filename):\n    return df_train[df_train['filename']==filename.name]['genre'].values[0]","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:19:38.220691Z","iopub.execute_input":"2022-03-14T20:19:38.221193Z","iopub.status.idle":"2022-03-14T20:19:38.227949Z","shell.execute_reply.started":"2022-03-14T20:19:38.221157Z","shell.execute_reply":"2022-03-14T20:19:38.227282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_spectrogram(filename):\n    audio, sr = torchaudio.load(filename)\n    specgram = torchaudio.transforms.MelSpectrogram(sample_rate=sr, \n                                                    n_fft=N_FFT, \n                                                    win_length=N_FFT, \n                                                    hop_length=HOP_LEN*4,\n                                                    center=True,\n                                                    pad_mode=\"reflect\",\n                                                    power=2.0,\n                                                    norm='slaney',\n                                                    onesided=True,\n                                                    n_mels=128,\n                                                    mel_scale=\"htk\"\n                                                   )(audio)[0]\n    specgram = torchaudio.transforms.AmplitudeToDB()(specgram)\n    specgram = specgram - specgram.min()\n    specgram = specgram/specgram.max()*255\n    \n    \n    return specgram","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:19:38.229423Z","iopub.execute_input":"2022-03-14T20:19:38.230150Z","iopub.status.idle":"2022-03-14T20:19:38.238002Z","shell.execute_reply.started":"2022-03-14T20:19:38.230113Z","shell.execute_reply":"2022-03-14T20:19:38.237339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filename = train_files[0]\n# spec_default = create_spectrogram(filename)","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:19:38.239490Z","iopub.execute_input":"2022-03-14T20:19:38.239922Z","iopub.status.idle":"2022-03-14T20:19:38.249246Z","shell.execute_reply.started":"2022-03-14T20:19:38.239871Z","shell.execute_reply":"2022-03-14T20:19:38.248517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def create_spectrogram_filer(filename):\n#     specgram = create_spectrogram(filename)\n#     zero_pct = (specgram == 0).sum()/(specgram.shape[0]*specgram.shape[1])\n#     if zero_pct > 0.92:\n#         print(f'too much zero at: {filename}')\n#         return spec_default\n#     else:\n#         return specgram","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:19:38.252341Z","iopub.execute_input":"2022-03-14T20:19:38.252844Z","iopub.status.idle":"2022-03-14T20:19:38.259148Z","shell.execute_reply.started":"2022-03-14T20:19:38.252793Z","shell.execute_reply":"2022-03-14T20:19:38.258468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"db = DataBlock(\n    blocks=(ImageBlock, CategoryBlock),\n    get_items=get_items,\n    get_x=create_spectrogram,\n    get_y=get_y,\n    splitter=RandomSplitter(seed=42),\n    item_tfms=[Resize(256)])","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:19:38.260354Z","iopub.execute_input":"2022-03-14T20:19:38.261148Z","iopub.status.idle":"2022-03-14T20:19:38.270224Z","shell.execute_reply.started":"2022-03-14T20:19:38.261114Z","shell.execute_reply":"2022-03-14T20:19:38.269400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = db.dataloaders(train_path, bs=128)","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:19:38.271371Z","iopub.execute_input":"2022-03-14T20:19:38.272178Z","iopub.status.idle":"2022-03-14T20:20:39.411596Z","shell.execute_reply.started":"2022-03-14T20:19:38.272140Z","shell.execute_reply":"2022-03-14T20:20:39.410868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.show_batch()","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:20:39.413991Z","iopub.execute_input":"2022-03-14T20:20:39.414504Z","iopub.status.idle":"2022-03-14T20:20:53.197303Z","shell.execute_reply.started":"2022-03-14T20:20:39.414466Z","shell.execute_reply":"2022-03-14T20:20:53.195258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = cnn_learner(dls, xresnet18, metrics=accuracy, pretrained=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:20:53.198330Z","iopub.execute_input":"2022-03-14T20:20:53.198569Z","iopub.status.idle":"2022-03-14T20:20:53.462537Z","shell.execute_reply.started":"2022-03-14T20:20:53.198540Z","shell.execute_reply":"2022-03-14T20:20:53.461842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.to_fp16()","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:20:53.463695Z","iopub.execute_input":"2022-03-14T20:20:53.465753Z","iopub.status.idle":"2022-03-14T20:20:53.472905Z","shell.execute_reply.started":"2022-03-14T20:20:53.465721Z","shell.execute_reply":"2022-03-14T20:20:53.470804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## This is to make model works with 1-channel input\ndef alter_learner(learn, channels=1):\n    learn.model[0][0][0].in_channels=channels\n    learn.model[0][0][0].weight = torch.nn.parameter.Parameter(learn.model[0][0][0].weight[:,1,:,:].unsqueeze(1))","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:20:53.474442Z","iopub.execute_input":"2022-03-14T20:20:53.475420Z","iopub.status.idle":"2022-03-14T20:20:53.482534Z","shell.execute_reply.started":"2022-03-14T20:20:53.475389Z","shell.execute_reply":"2022-03-14T20:20:53.481689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nchannels = dls.one_batch()[0].shape[1]\nalter_learner(learn, nchannels)","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:20:53.483862Z","iopub.execute_input":"2022-03-14T20:20:53.484202Z","iopub.status.idle":"2022-03-14T20:21:07.326752Z","shell.execute_reply.started":"2022-03-14T20:20:53.484164Z","shell.execute_reply":"2022-03-14T20:21:07.326006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This is to get the good learning rate\nlearn.lr_find()","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:21:07.327907Z","iopub.execute_input":"2022-03-14T20:21:07.328959Z","iopub.status.idle":"2022-03-14T20:36:48.302238Z","shell.execute_reply.started":"2022-03-14T20:21:07.328926Z","shell.execute_reply":"2022-03-14T20:36:48.301544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fit_one_cycle(4, lr_max=1e-3, cbs=TerminateOnNaNCallback())","metadata":{"execution":{"iopub.status.busy":"2022-03-14T20:37:18.711851Z","iopub.execute_input":"2022-03-14T20:37:18.712568Z","iopub.status.idle":"2022-03-14T21:06:22.085237Z","shell.execute_reply.started":"2022-03-14T20:37:18.712532Z","shell.execute_reply":"2022-03-14T21:06:22.083719Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def genreid_from_genre(genre):\n    return int(genre2id[genre2id['genre'] == genre]['genre_id'].values[0])","metadata":{"execution":{"iopub.status.busy":"2022-03-14T21:06:26.948711Z","iopub.execute_input":"2022-03-14T21:06:26.949586Z","iopub.status.idle":"2022-03-14T21:06:26.953664Z","shell.execute_reply.started":"2022-03-14T21:06:26.949550Z","shell.execute_reply":"2022-03-14T21:06:26.952830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_items = get_items(test_path)\ntest_dl = dls.test_dl(test_items)\npreds = learn.get_preds(dl=test_dl)\npreds_idx = preds[0].argmax(axis=1)\ngenre2id = pd.read_csv('../input/kaggle-pog-series-s01e02/genres.csv')\nsongid_preds = {int(file_path.stem):genreid_from_genre(learn.dls.vocab[_id]) for file_path, _id in zip(test_items,preds_idx)}\nsubmission['genre_id'] = submission['song_id'].map(songid_preds)\nsubmission['genre_id'].fillna(0, inplace=True)\nsubmission.genre_id = submission.genre_id.astype(int)\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-14T21:06:27.798569Z","iopub.execute_input":"2022-03-14T21:06:27.799366Z","iopub.status.idle":"2022-03-14T21:13:25.953305Z","shell.execute_reply.started":"2022-03-14T21:06:27.799325Z","shell.execute_reply":"2022-03-14T21:13:25.952547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}