{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install speechbrain\n!pip install -U torchaudio","metadata":{"execution":{"iopub.status.busy":"2022-02-27T13:18:06.768370Z","iopub.execute_input":"2022-02-27T13:18:06.768707Z","iopub.status.idle":"2022-02-27T13:19:27.920335Z","shell.execute_reply.started":"2022-02-27T13:18:06.768640Z","shell.execute_reply":"2022-02-27T13:19:27.919469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport random\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\n\nimport torch\nimport torch.nn as nn\n\nimport torchaudio","metadata":{"execution":{"iopub.status.busy":"2022-02-27T13:19:27.925565Z","iopub.execute_input":"2022-02-27T13:19:27.926247Z","iopub.status.idle":"2022-02-27T13:19:28.774375Z","shell.execute_reply.started":"2022-02-27T13:19:27.926206Z","shell.execute_reply":"2022-02-27T13:19:28.773581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bundle = torchaudio.pipelines.WAV2VEC2_BASE\nwav2vec2 = bundle.get_model()\nwav2vec2.encoder.transformer.layers = wav2vec2.encoder.transformer.layers[:-4]","metadata":{"execution":{"iopub.status.busy":"2022-02-27T13:19:28.776050Z","iopub.execute_input":"2022-02-27T13:19:28.776298Z","iopub.status.idle":"2022-02-27T13:19:37.692481Z","shell.execute_reply.started":"2022-02-27T13:19:28.776260Z","shell.execute_reply":"2022-02-27T13:19:37.691742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_dir = 'hackaton_ds/train/'\ndevice = 'cuda' if torch.cuda.is_available() else 'cpu'\ndevice","metadata":{"execution":{"iopub.status.busy":"2022-02-27T13:19:37.696538Z","iopub.execute_input":"2022-02-27T13:19:37.698507Z","iopub.status.idle":"2022-02-27T13:19:37.754709Z","shell.execute_reply.started":"2022-02-27T13:19:37.698463Z","shell.execute_reply":"2022-02-27T13:19:37.753872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CommandClassifier(nn.Module):\n    def __init__(self, feature_extractor):\n        super(CommandClassifier, self).__init__()\n        self.feature_extractor = feature_extractor\n        self.linear = nn.Linear(768, 10)\n        \n    def forward(self, X):\n        features = self.get_embeddings(X)\n        logits = self.linear(features)\n        return logits\n    \n    def get_embeddings(self, X):\n        embeddings = self.feature_extractor(X)[0].mean(axis=1)\n        return nn.functional.normalize(embeddings)","metadata":{"execution":{"iopub.status.busy":"2022-02-27T13:19:37.759538Z","iopub.execute_input":"2022-02-27T13:19:37.761377Z","iopub.status.idle":"2022-02-27T13:19:37.769892Z","shell.execute_reply.started":"2022-02-27T13:19:37.761339Z","shell.execute_reply":"2022-02-27T13:19:37.769123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wav2vec2_model = CommandClassifier(wav2vec2)\nwav2vec2_model.load_state_dict(torch.load('/kaggle/input/command-recognition/wav2vec2 checkpoint/model.pth'))\nwav2vec2_model.to(device)\nwav2vec2_model.eval()","metadata":{"execution":{"iopub.status.busy":"2022-02-27T13:19:37.773896Z","iopub.execute_input":"2022-02-27T13:19:37.776052Z","iopub.status.idle":"2022-02-27T13:19:43.359099Z","shell.execute_reply.started":"2022-02-27T13:19:37.776015Z","shell.execute_reply":"2022-02-27T13:19:43.358425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from speechbrain.pretrained import EncoderClassifier\nspeechbrain_classifier = EncoderClassifier.from_hparams(source=\"/kaggle/input/command-recognition/CKPT+2022-02-27+03-56-53+00\", savedir=\"pretrained_models/google_speech_command_xvector\", run_opts={\"device\":\"cuda\"})","metadata":{"execution":{"iopub.status.busy":"2022-02-27T13:19:43.360497Z","iopub.execute_input":"2022-02-27T13:19:43.360989Z","iopub.status.idle":"2022-02-27T13:19:44.531108Z","shell.execute_reply.started":"2022-02-27T13:19:43.360951Z","shell.execute_reply":"2022-02-27T13:19:44.530334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dir = '/kaggle/input/classification-of-short-noisy-audio-speech/hackaton_ds/test/'","metadata":{"execution":{"iopub.status.busy":"2022-02-27T13:19:44.532505Z","iopub.execute_input":"2022-02-27T13:19:44.532768Z","iopub.status.idle":"2022-02-27T13:19:44.537168Z","shell.execute_reply.started":"2022-02-27T13:19:44.532731Z","shell.execute_reply":"2022-02-27T13:19:44.535978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = []\nfor i in tqdm(os.listdir(test_dir)):\n    \n    waveform, sample_rate = torchaudio.load(f'{test_dir}/{i}')\n    waveform = torchaudio.functional.resample(waveform, sample_rate, bundle.sample_rate)#[:, :10**5]\n    waveform = torch.nn.functional.pad(waveform, (16000-waveform.shape[1], 0))[0]\n    \n    with torch.no_grad():\n        predictions = wav2vec2_model(waveform.unsqueeze(0).to(device))[0]\n        predictions = torch.softmax(predictions, dim=0)\n    \n    out_prob, score, index, text_lab = speechbrain_classifier.classify_file(f'{test_dir}/{i}')\n    out_prob = torch.softmax(out_prob[0][:-2], dim=0)\n\n    prob = (out_prob + predictions)\n    label = speechbrain_classifier.hparams.label_encoder.ind2lab[prob.argmax().item()]\n    \n    pred.append({'id': i.replace('.wav', ''), 'category': label})\npred = pd.DataFrame(pred)","metadata":{"execution":{"iopub.status.busy":"2022-02-27T13:19:44.538587Z","iopub.execute_input":"2022-02-27T13:19:44.539186Z","iopub.status.idle":"2022-02-27T13:30:54.943800Z","shell.execute_reply.started":"2022-02-27T13:19:44.539148Z","shell.execute_reply":"2022-02-27T13:30:54.942546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred.category.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-02-27T13:30:54.945188Z","iopub.execute_input":"2022-02-27T13:30:54.945426Z","iopub.status.idle":"2022-02-27T13:30:54.964159Z","shell.execute_reply.started":"2022-02-27T13:30:54.945392Z","shell.execute_reply":"2022-02-27T13:30:54.963327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-27T13:30:54.965275Z","iopub.execute_input":"2022-02-27T13:30:54.965944Z","iopub.status.idle":"2022-02-27T13:30:55.035121Z","shell.execute_reply.started":"2022-02-27T13:30:54.965903Z","shell.execute_reply":"2022-02-27T13:30:55.034451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!find . -regex '^.*wav$' -delete","metadata":{"execution":{"iopub.status.busy":"2022-02-27T13:30:55.036078Z","iopub.execute_input":"2022-02-27T13:30:55.036303Z","iopub.status.idle":"2022-02-27T13:30:56.442977Z","shell.execute_reply.started":"2022-02-27T13:30:55.036271Z","shell.execute_reply":"2022-02-27T13:30:56.442049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}