{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":2660070,"sourceType":"datasetVersion","datasetId":686792},{"sourceId":5683788,"sourceType":"datasetVersion","datasetId":818525},{"sourceId":8335495,"sourceType":"datasetVersion","datasetId":4888994},{"sourceId":173386161,"sourceType":"kernelVersion"}],"dockerImageVersionId":30162,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import gc\nimport os\nimport sys\n# sys.path.append('../input/pytorch-image-models/pytorch-image-models-master')\nimport random\nimport time\nimport warnings\n\nimport librosa\nimport numpy as np\nimport pandas as pd\nimport soundfile as sf\n\n\nfrom contextlib import contextmanager\nfrom joblib import Parallel, delayed\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom glob import glob\n","metadata":{"execution":{"iopub.status.busy":"2024-04-27T06:32:24.428485Z","iopub.execute_input":"2024-04-27T06:32:24.428821Z","iopub.status.idle":"2024-04-27T06:32:26.952749Z","shell.execute_reply.started":"2024-04-27T06:32:24.428785Z","shell.execute_reply":"2024-04-27T06:32:26.951535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"negative_folder_path = \"/kaggle/input/negative-samples-bc2024\"\nnegative_audio_files = os.listdir(negative_folder_path)\nprint(len(negative_audio_files))\n\nD=[]\nnegative_paths=[]\nnegative_targets = ['noise'] * len(negative_audio_files)\nfor file in negative_audio_files:\n    path = os.path.join(negative_folder_path,file)\n    negative_paths.append(path)\n    signal,sr= librosa.load(path,sr=32000)\n    duration = len(signal)/sr\n#     print(file,duration)\n    D.append(duration)\nprint(\"Total duration of the negative samples:\",sum(D))","metadata":{"execution":{"iopub.status.busy":"2024-04-27T06:32:26.955583Z","iopub.execute_input":"2024-04-27T06:32:26.956766Z","iopub.status.idle":"2024-04-27T06:32:43.743657Z","shell.execute_reply.started":"2024-04-27T06:32:26.956700Z","shell.execute_reply":"2024-04-27T06:32:43.742095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SR = 32000\n # 60 # 90 # 60\nCAP_NUM_CLASSES = 10\nOUT_DIR = \"/kaggle/working/\"\nMIN_DUR = 0.5\nUSE_SEC=5\n\ndef save_npy(audio_path,target_file_number,folder_path):\n    total_samples = 0\n    main_num_iters = 50\n    for num_iter in range(main_num_iters):\n        if total_samples>=target_file_number:\n                break\n        y, sr = librosa.load(audio_path,sr=SR)\n        start_index = num_iter*USE_SEC*SR\n        stop_index  = (num_iter+1)*USE_SEC*SR\n        if start_index<len(y):\n            y = y[start_index:stop_index]\n            if len(y)>=  MIN_DUR*SR:\n                PP = audio_path.split(\"/\")\n                save_path = folder_path +\\\n                                PP[-1].split(\" \")[0] + \"_\" +\\\n                                        PP[-1].split(\" \")[1].split(\".\")[0] +\\\n                                \"_\" + str(num_iter)\n                print(save_path)\n                np.save(save_path, y)\n                total_samples+=1\n                if total_samples>=target_file_number:\n                    break\n                \n\ndef process_audio(audio_path,folder_path):\n    target_file_number=10\n    save_npy(audio_path,target_file_number,folder_path)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-27T06:32:43.745573Z","iopub.execute_input":"2024-04-27T06:32:43.745987Z","iopub.status.idle":"2024-04-27T06:32:43.760678Z","shell.execute_reply.started":"2024-04-27T06:32:43.745936Z","shell.execute_reply":"2024-04-27T06:32:43.759192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_folder_path = os.path.join(OUT_DIR,'noise/')\nprint(output_folder_path)\n_ = os.makedirs(output_folder_path, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-27T06:33:33.243408Z","iopub.execute_input":"2024-04-27T06:33:33.243804Z","iopub.status.idle":"2024-04-27T06:33:33.251810Z","shell.execute_reply.started":"2024-04-27T06:33:33.243755Z","shell.execute_reply":"2024-04-27T06:33:33.250354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# path = '/kaggle/input/negative-samples-bc2024/Recording 100.wav'\n# process_audio(path,output_folder_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-27T06:33:34.788030Z","iopub.execute_input":"2024-04-27T06:33:34.788396Z","iopub.status.idle":"2024-04-27T06:33:46.980379Z","shell.execute_reply.started":"2024-04-27T06:33:34.788361Z","shell.execute_reply":"2024-04-27T06:33:46.979414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"negative_audio_paths = glob('/kaggle/input/negative-samples-bc2024/*.wav')\nprint(len(negative_audio_paths))","metadata":{"execution":{"iopub.status.busy":"2024-04-27T06:32:55.858686Z","iopub.execute_input":"2024-04-27T06:32:55.858958Z","iopub.status.idle":"2024-04-27T06:32:55.866587Z","shell.execute_reply.started":"2024-04-27T06:32:55.858922Z","shell.execute_reply":"2024-04-27T06:32:55.865502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_WORKERS=4\n_ = Parallel(n_jobs=NUM_WORKERS)(delayed(process_audio)(audio_path,output_folder_path)\\\n                                 for audio_path in tqdm(negative_audio_paths))","metadata":{"execution":{"iopub.status.busy":"2024-04-27T06:19:35.663708Z","iopub.execute_input":"2024-04-27T06:19:35.664639Z","iopub.status.idle":"2024-04-27T06:25:33.211438Z","shell.execute_reply.started":"2024-04-27T06:19:35.664578Z","shell.execute_reply":"2024-04-27T06:25:33.210051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# npy_files = os.listdir(\"/kaggle/working/negative-samples-bc2024-npy/noise\")\n# print(npy_files[0])\n# print(len(npy_files))","metadata":{"execution":{"iopub.status.busy":"2024-04-27T06:26:45.064446Z","iopub.execute_input":"2024-04-27T06:26:45.065249Z","iopub.status.idle":"2024-04-27T06:26:45.071739Z","shell.execute_reply.started":"2024-04-27T06:26:45.065198Z","shell.execute_reply":"2024-04-27T06:26:45.070974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# output_dir = \"/kaggle/working/negative-samples-bc2024-npy/noise\"\n# for file in npy_files:\n#     path = os.path.join(output_dir,file)\n#     y= np.load(path,)\n#     print(file, len(y)/32000)","metadata":{"execution":{"iopub.status.busy":"2024-04-27T06:29:55.045211Z","iopub.execute_input":"2024-04-27T06:29:55.045528Z","iopub.status.idle":"2024-04-27T06:29:55.164494Z","shell.execute_reply.started":"2024-04-27T06:29:55.045495Z","shell.execute_reply":"2024-04-27T06:29:55.163489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}