{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd  #Import Pandas\nimport numpy as np  #Import NumPy\nimport matplotlib.pylab as plt  #Import matplotlib\nimport seaborn as sns  #Python data visualization library\nimport soundfile\n\nimport os\nimport shutil\nfrom pydub import AudioSegment\n\nimport random #2 make a random numer\nimport tensorflow as tf\nfrom tensorflow import keras\n\nfrom glob import glob #2 List the files in a directory\n\nimport librosa  #Python package for music and audio analysis\nimport librosa.display  #Python package for music and audio analysis\nimport IPython.display as ipd  #2 Display audio samples\n\nfrom pathlib import Path\nfrom IPython.display import display, Audio\n\nimport soundfile as sf","metadata":{"execution":{"iopub.status.busy":"2022-04-12T17:52:54.621455Z","iopub.execute_input":"2022-04-12T17:52:54.621805Z","iopub.status.idle":"2022-04-12T17:53:01.095925Z","shell.execute_reply.started":"2022-04-12T17:52:54.621679Z","shell.execute_reply":"2022-04-12T17:53:01.094989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_AUDIO_PATH = os.path.join('../input/music-classification-wav/TRAIN_V2/data_out_2')\n\n# Percentage of samples to use for validation\nVALID_SPLIT = 0.1\n\n# The sampling rate to use.\n# This is the one used in all of the audio samples.\n# We will resample all of the noise to this sampling rate.\n# This will also be the output size of the audio wave samples\n# (since all samples are of 1 second long)\nSAMPLING_RATE = 16000\n\n# Seed to use when shuffling the dataset\nSHUFFLE_SEED = 43\n\nSCALE = 0.5\n\nBATCH_SIZE = 128\nEPOCHS = 100\n#print(train_sng)  #Print the csv","metadata":{"execution":{"iopub.status.busy":"2022-04-12T17:53:01.097693Z","iopub.execute_input":"2022-04-12T17:53:01.097981Z","iopub.status.idle":"2022-04-12T17:53:01.103671Z","shell.execute_reply.started":"2022-04-12T17:53:01.097945Z","shell.execute_reply":"2022-04-12T17:53:01.102798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Here only is 2 look the numers of files in each folder\npath, dirs, files = next(os.walk(DATASET_AUDIO_PATH +'/0'))\nfile_count = len(files)\nprint(file_count)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T17:53:01.104988Z","iopub.execute_input":"2022-04-12T17:53:01.105450Z","iopub.status.idle":"2022-04-12T17:53:02.629158Z","shell.execute_reply.started":"2022-04-12T17:53:01.105392Z","shell.execute_reply":"2022-04-12T17:53:02.628434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def paths_and_labels_to_dataset(audio_paths, labels):\n    \"\"\"Constructs a dataset of audios and labels.\"\"\"\n    path_ds = tf.data.Dataset.from_tensor_slices(audio_paths)\n    audio_ds = path_ds.map(lambda x: path_to_audio(x))\n    label_ds = tf.data.Dataset.from_tensor_slices(labels)\n    return tf.data.Dataset.zip((audio_ds, label_ds))\n\n\ndef path_to_audio(path):\n    \"\"\"Reads and decodes an audio file.\"\"\"\n    audio = tf.io.read_file(path)\n    audio, _ = tf.audio.decode_wav(audio, 1, SAMPLING_RATE)\n    return audio\n\ndef audio_to_fft(audio):\n    # Since tf.signal.fft applies FFT on the innermost dimension,\n    # we need to squeeze the dimensions and then expand them again\n    # after FFT\n    audio = tf.squeeze(audio, axis=-1)\n    fft = tf.signal.fft(\n        tf.cast(tf.complex(real=audio, imag=tf.zeros_like(audio)), tf.complex64)\n    )\n    fft = tf.expand_dims(fft, axis=-1)\n\n    # Return the absolute value of the first half of the FFT\n    # which represents the positive frequencies\n    return tf.math.abs(fft[:, : (audio.shape[1] // 2), :])","metadata":{"execution":{"iopub.status.busy":"2022-04-12T17:53:02.631260Z","iopub.execute_input":"2022-04-12T17:53:02.631530Z","iopub.status.idle":"2022-04-12T17:53:02.641380Z","shell.execute_reply.started":"2022-04-12T17:53:02.631492Z","shell.execute_reply":"2022-04-12T17:53:02.640656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the list of audio file paths along with their corresponding labels\n\nclass_names = os.listdir(DATASET_AUDIO_PATH)\n#class_names.remove('16')\n#class_names.remove('17')\n\nprint(\"Our class names: {}\".format(class_names,))","metadata":{"execution":{"iopub.status.busy":"2022-04-12T17:53:02.643164Z","iopub.execute_input":"2022-04-12T17:53:02.643836Z","iopub.status.idle":"2022-04-12T17:53:02.657990Z","shell.execute_reply.started":"2022-04-12T17:53:02.643791Z","shell.execute_reply":"2022-04-12T17:53:02.657215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_paths = []\nlabels = []\nfor label, name in enumerate(class_names):\n    print(\"Processing speaker {}\".format(name,))\n    dir_path = Path(DATASET_AUDIO_PATH) / name\n    speaker_sample_paths = [\n        os.path.join(dir_path, filepath)\n        for filepath in os.listdir(dir_path)\n        if filepath.endswith(\".wav\")\n    ]\n    audio_paths += speaker_sample_paths\n    labels += [label] * len(speaker_sample_paths)\n\nprint(\n    \"Found {} files belonging to {} classes.\".format(len(audio_paths), len(class_names))\n)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T17:53:02.659273Z","iopub.execute_input":"2022-04-12T17:53:02.659706Z","iopub.status.idle":"2022-04-12T17:53:03.963274Z","shell.execute_reply.started":"2022-04-12T17:53:02.659670Z","shell.execute_reply":"2022-04-12T17:53:03.962430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shuffle\nrng = np.random.RandomState(SHUFFLE_SEED)\nrng.shuffle(audio_paths)\nrng = np.random.RandomState(SHUFFLE_SEED)\nrng.shuffle(labels)\n\n# Split into training and validation\nnum_val_samples = int(VALID_SPLIT * len(audio_paths))\nprint(\"Using {} files for training.\".format(len(audio_paths) - num_val_samples))\ntrain_audio_paths = audio_paths[:-num_val_samples]\ntrain_labels = labels[:-num_val_samples]\n\nprint(\"Using {} files for validation.\".format(num_val_samples))\nvalid_audio_paths = audio_paths[-num_val_samples:]\nvalid_labels = labels[-num_val_samples:]\n\n# Create 2 datasets, one for training and the other for validation\ntrain_ds = paths_and_labels_to_dataset(train_audio_paths, train_labels)\ntrain_ds = train_ds.shuffle(buffer_size=BATCH_SIZE * 8, seed=SHUFFLE_SEED).batch(\n    BATCH_SIZE\n)\n\nvalid_ds = paths_and_labels_to_dataset(valid_audio_paths, valid_labels)\nvalid_ds = valid_ds.shuffle(buffer_size=32 * 8, seed=SHUFFLE_SEED).batch(32)\n\n\n\n\n# Transform audio wave to the frequency domain using `audio_to_fft`\ntrain_ds = train_ds.map(\n    lambda x, y: (audio_to_fft(x), y), num_parallel_calls=tf.data.AUTOTUNE\n)\ntrain_ds = train_ds.prefetch(tf.data.AUTOTUNE)\n\nvalid_ds = valid_ds.map(\n    lambda x, y: (audio_to_fft(x), y), num_parallel_calls=tf.data.AUTOTUNE\n)\n\n#train_ds = train_ds.shuffle(buffer_size=700)\n\n#valid_ds = valid_ds.shuffle(buffer_size=300)\n\nvalid_ds = valid_ds.prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T17:53:03.965602Z","iopub.execute_input":"2022-04-12T17:53:03.966086Z","iopub.status.idle":"2022-04-12T17:53:06.706903Z","shell.execute_reply.started":"2022-04-12T17:53:03.966040Z","shell.execute_reply":"2022-04-12T17:53:06.706232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds","metadata":{"execution":{"iopub.status.busy":"2022-04-12T17:53:06.708072Z","iopub.execute_input":"2022-04-12T17:53:06.708305Z","iopub.status.idle":"2022-04-12T17:53:06.719284Z","shell.execute_reply.started":"2022-04-12T17:53:06.708271Z","shell.execute_reply":"2022-04-12T17:53:06.718650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def residual_block(x, filters, conv_num=3, activation=\"relu\"):\n    # Shortcut\n    s = keras.layers.Conv1D(filters, 1, padding=\"same\")(x)\n    for i in range(conv_num - 1):\n        x = keras.layers.Conv1D(filters, 3, padding=\"same\")(x)\n        x = keras.layers.Activation(activation)(x)\n    x = keras.layers.Conv1D(filters, 3, padding=\"same\")(x)\n    x = keras.layers.Add()([x, s])\n    x = keras.layers.Activation(activation)(x)\n    return keras.layers.MaxPool1D(pool_size=2, strides=2)(x)\n\n\ndef build_model(input_shape, num_classes):\n    inputs = keras.layers.Input(shape=input_shape, name=\"input\")\n\n    x = residual_block(inputs, 16, 2)\n    x = residual_block(x, 32, 2)\n    x = residual_block(x, 64, 3)\n    x = residual_block(x, 128, 3)\n\n    x = keras.layers.AveragePooling1D(pool_size=3, strides=3)(x)\n    x = keras.layers.Flatten()(x)\n    x = keras.layers.Dense(256, activation=\"relu\")(x)\n    #x = keras.layers.Dense(128, activation=\"relu\")(x)\n\n    outputs = keras.layers.Dense(num_classes, activation=\"softmax\", name=\"output\")(x)\n\n    return keras.models.Model(inputs=inputs, outputs=outputs)\n\n\nmodel = build_model((SAMPLING_RATE // 2, 1), len(class_names))\n\nmodel.summary()\n\n# Compile the model using Adam's default learning rate\nmodel.compile(\n    optimizer=\"Adam\", loss=\"sparse_categorical_crossentropy\", metrics=[\"accuracy\"]\n)\n\n# Add callbacks:\n# 'EarlyStopping' to stop training when the model is not enhancing anymore\n# 'ModelCheckPoint' to always keep the model that has the best val_accuracy\nmodel_save_filename = \"model.h5\"\n\nearlystopping_cb = keras.callbacks.EarlyStopping(patience=10, restore_best_weights=True)\nmdlcheckpoint_cb = keras.callbacks.ModelCheckpoint(\n    model_save_filename, monitor=\"val_accuracy\", save_best_only=True\n)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T17:53:06.720454Z","iopub.execute_input":"2022-04-12T17:53:06.720645Z","iopub.status.idle":"2022-04-12T17:53:08.187934Z","shell.execute_reply.started":"2022-04-12T17:53:06.720621Z","shell.execute_reply":"2022-04-12T17:53:08.187197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_ds,\n    epochs=EPOCHS,\n    validation_data=valid_ds,\n    callbacks=[earlystopping_cb, mdlcheckpoint_cb],\n)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T17:53:12.075104Z","iopub.execute_input":"2022-04-12T17:53:12.075451Z","iopub.status.idle":"2022-04-12T18:21:13.427488Z","shell.execute_reply.started":"2022-04-12T17:53:12.075410Z","shell.execute_reply":"2022-04-12T18:21:13.425447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(model.evaluate(valid_ds))","metadata":{"execution":{"iopub.status.busy":"2022-04-12T18:21:16.465076Z","iopub.execute_input":"2022-04-12T18:21:16.465361Z","iopub.status.idle":"2022-04-12T18:21:56.466493Z","shell.execute_reply.started":"2022-04-12T18:21:16.465316Z","shell.execute_reply":"2022-04-12T18:21:56.465280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAMPLES_TO_DISPLAY = 10\n\ntest_ds = paths_and_labels_to_dataset(valid_audio_paths, valid_labels)\ntest_ds = test_ds.shuffle(buffer_size=BATCH_SIZE * 8, seed=SHUFFLE_SEED).batch(\n    BATCH_SIZE\n)\n\n\nfor audios, labels in test_ds.take(1):\n    # Get the signal FFT\n    ffts = audio_to_fft(audios)\n    # Predict\n    y_pred = model.predict(ffts)\n    # Take random samples\n    rnd = np.random.randint(0, BATCH_SIZE, SAMPLES_TO_DISPLAY)\n    audios = audios.numpy()[rnd, :, :]\n    labels = labels.numpy()[rnd]\n    y_pred = np.argmax(y_pred, axis=-1)[rnd]\n\n    for index in range(SAMPLES_TO_DISPLAY):\n        # For every sample, print the true and predicted label\n        # as well as run the voice with the noise\n        print(\n            \"Speaker: {} - Predicted: {}\".format(\n                class_names[labels[index]],\n                class_names[y_pred[index]],\n            )\n        )\n        display(Audio(audios[index, :, :].squeeze(), rate=SAMPLING_RATE))","metadata":{"execution":{"iopub.status.busy":"2022-04-12T18:27:38.987815Z","iopub.execute_input":"2022-04-12T18:27:38.988586Z","iopub.status.idle":"2022-04-12T18:27:49.250620Z","shell.execute_reply.started":"2022-04-12T18:27:38.988532Z","shell.execute_reply":"2022-04-12T18:27:49.249872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}