{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd  #Import Pandas\nimport numpy as np  #Import NumPy\nimport matplotlib.pylab as plt  #Import matplotlib\nimport seaborn as sns  #Python data visualization library\nimport soundfile\n\nimport os\nimport shutil\nfrom pydub import AudioSegment\n\nimport random #2 make a random numer\nimport tensorflow as tf\nfrom tensorflow import keras\n\nfrom glob import glob #2 List the files in a directory\n\nimport librosa  #Python package for music and audio analysis\nimport librosa.display  #Python package for music and audio analysis\nimport IPython.display as ipd  #2 Display audio samples\n\nfrom pathlib import Path\nfrom IPython.display import display, Audio\n\nimport soundfile as sf","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-31T04:03:37.744746Z","iopub.execute_input":"2022-03-31T04:03:37.745052Z","iopub.status.idle":"2022-03-31T04:03:37.752938Z","shell.execute_reply.started":"2022-03-31T04:03:37.745021Z","shell.execute_reply":"2022-03-31T04:03:37.751560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data preparation**","metadata":{}},{"cell_type":"code","source":"train_sng = pd.read_csv('../input/kaggle-pog-series-s01e02/train.csv') #Read csv\ngenres = pd.read_csv('../input/kaggle-pog-series-s01e02/genres.csv') #Read csv\ndata_train ='../input/kaggle-pog-series-s01e02/train/' #Data of the songs 2 train\ndata_test ='../input/kaggle-pog-series-s01e02/test/' #Data of the songs 2 test\n#Delet the \"#\" in the next lines if you want see the DataFrame\nprint(genres) #Print the list of the genre\n#print('')\n#print(train_sng)  #Print the csv","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:03:40.745159Z","iopub.execute_input":"2022-03-31T04:03:40.745692Z","iopub.status.idle":"2022-03-31T04:03:40.809344Z","shell.execute_reply.started":"2022-03-31T04:03:40.745659Z","shell.execute_reply":"2022-03-31T04:03:40.808233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sngs_gen = []\n\nfor x in genres['genre_id']:\n    train_sng_temp = train_sng.loc[train_sng['genre_id'] == x] #Search in the column \"genre_id\" the rows with the same gender\n    #train_sng_temp = train_sng_temp[['filepath']]\n    sngs_gen.append(train_sng_temp) #Append those songs in anohter list\n\nprint(sngs_gen[0])","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:03:42.771608Z","iopub.execute_input":"2022-03-31T04:03:42.771991Z","iopub.status.idle":"2022-03-31T04:03:42.807391Z","shell.execute_reply.started":"2022-03-31T04:03:42.771951Z","shell.execute_reply":"2022-03-31T04:03:42.806370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_ROOT = os.path.join(\"../input/kaggle-pog-series-s01e02/train\")\nDATASET_AUDIO_PATH = os.path.join('./Data_Train/')\n\n# Percentage of samples to use for validation\nVALID_SPLIT = 0.1\n\n# The sampling rate to use.\n# This is the one used in all of the audio samples.\n# We will resample all of the noise to this sampling rate.\n# This will also be the output size of the audio wave samples\n# (since all samples are of 1 second long)\nSAMPLING_RATE = 22050\n\n# Seed to use when shuffling the dataset\nSHUFFLE_SEED = 43\n\nSCALE = 0.3\n\nBATCH_SIZE = 30\nEPOCHS = 100","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:24:38.841842Z","iopub.execute_input":"2022-03-31T04:24:38.842125Z","iopub.status.idle":"2022-03-31T04:24:38.850861Z","shell.execute_reply.started":"2022-03-31T04:24:38.842093Z","shell.execute_reply":"2022-03-31T04:24:38.849725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for x in range(genres[genres.columns[1]].count()-1): #Here have 1 less 2 delet the last gender that have a few examples\n    if os.path.exists(DATASET_AUDIO_PATH + '/' + str(x)) is False:\n        os.makedirs(DATASET_AUDIO_PATH + '/' + str(x))\n    print(x)    \n    for z in range(200):\n        try: \n            y, sr = librosa.load(\"../input/kaggle-pog-series-s01e02/train/\" + str(sngs_gen[x].iat[z,1])) #Read the file\n            samples_cut = y[0:360984]\n            \n            sf.write(DATASET_AUDIO_PATH + str(x) + '/'+ str(sngs_gen[x].iat[z,0]) + '.wav', samples_cut, sr, 'PCM_16')#Save the file in .wav in 16 bits           \n        except:\n            continue","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:03:50.953900Z","iopub.execute_input":"2022-03-31T04:03:50.954881Z","iopub.status.idle":"2022-03-31T04:06:36.407967Z","shell.execute_reply.started":"2022-03-31T04:03:50.954833Z","shell.execute_reply":"2022-03-31T04:06:36.406959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Here only is 2 look the numers of files in each folder\npath, dirs, files = next(os.walk(DATASET_AUDIO_PATH +'0'))\nfile_count = len(files)\n\nfor file in files:\n    print(file)\nprint(file_count)","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:08:00.576158Z","iopub.execute_input":"2022-03-31T04:08:00.576468Z","iopub.status.idle":"2022-03-31T04:08:00.585194Z","shell.execute_reply.started":"2022-03-31T04:08:00.576436Z","shell.execute_reply":"2022-03-31T04:08:00.584112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y, sr = librosa.load(DATASET_AUDIO_PATH + '0/' + '12181.wav')","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:08:20.594913Z","iopub.execute_input":"2022-03-31T04:08:20.595196Z","iopub.status.idle":"2022-03-31T04:08:20.602309Z","shell.execute_reply.started":"2022-03-31T04:08:20.595166Z","shell.execute_reply":"2022-03-31T04:08:20.601052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sr)","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:08:37.584619Z","iopub.execute_input":"2022-03-31T04:08:37.584984Z","iopub.status.idle":"2022-03-31T04:08:37.591898Z","shell.execute_reply.started":"2022-03-31T04:08:37.584952Z","shell.execute_reply":"2022-03-31T04:08:37.590256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ipd.Audio(y, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:08:44.879762Z","iopub.execute_input":"2022-03-31T04:08:44.880082Z","iopub.status.idle":"2022-03-31T04:08:44.905416Z","shell.execute_reply.started":"2022-03-31T04:08:44.880050Z","shell.execute_reply":"2022-03-31T04:08:44.904213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samples_cut = y[0:16000]\nipd.Audio(samples_cut, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:14:04.181674Z","iopub.execute_input":"2022-03-31T04:14:04.182004Z","iopub.status.idle":"2022-03-31T04:14:04.195188Z","shell.execute_reply.started":"2022-03-31T04:14:04.181973Z","shell.execute_reply":"2022-03-31T04:14:04.193832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sf.write('stereo_file.wav', samples_cut, 20000, 'PCM_16')","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:11:50.296702Z","iopub.execute_input":"2022-03-31T04:11:50.297029Z","iopub.status.idle":"2022-03-31T04:11:50.313606Z","shell.execute_reply.started":"2022-03-31T04:11:50.296996Z","shell.execute_reply":"2022-03-31T04:11:50.312422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Dataset generation**","metadata":{}},{"cell_type":"code","source":"def paths_and_labels_to_dataset(audio_paths, labels):\n    \"\"\"Constructs a dataset of audios and labels.\"\"\"\n    path_ds = tf.data.Dataset.from_tensor_slices(audio_paths)\n    audio_ds = path_ds.map(lambda x: path_to_audio(x))\n    label_ds = tf.data.Dataset.from_tensor_slices(labels)\n    return tf.data.Dataset.zip((audio_ds, label_ds))\n\n\ndef path_to_audio(path):\n    \"\"\"Reads and decodes an audio file.\"\"\"\n    audio = tf.io.read_file(path)\n    audio, _ = tf.audio.decode_wav(audio, 1, SAMPLING_RATE)\n    return audio\n\ndef audio_to_fft(audio):\n    # Since tf.signal.fft applies FFT on the innermost dimension,\n    # we need to squeeze the dimensions and then expand them again\n    # after FFT\n    audio = tf.squeeze(audio, axis=-1)\n    fft = tf.signal.fft(\n        tf.cast(tf.complex(real=audio, imag=tf.zeros_like(audio)), tf.complex64)\n    )\n    fft = tf.expand_dims(fft, axis=-1)\n\n    # Return the absolute value of the first half of the FFT\n    # which represents the positive frequencies\n    return tf.math.abs(fft[:, : (audio.shape[1] // 2), :])","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:24:43.629569Z","iopub.execute_input":"2022-03-31T04:24:43.629957Z","iopub.status.idle":"2022-03-31T04:24:43.640387Z","shell.execute_reply.started":"2022-03-31T04:24:43.629928Z","shell.execute_reply":"2022-03-31T04:24:43.639060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the list of audio file paths along with their corresponding labels\n\nclass_names = os.listdir(DATASET_AUDIO_PATH)\n\nprint(\"Our class names: {}\".format(class_names,))\n\n","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:24:44.230058Z","iopub.execute_input":"2022-03-31T04:24:44.230363Z","iopub.status.idle":"2022-03-31T04:24:44.237849Z","shell.execute_reply.started":"2022-03-31T04:24:44.230332Z","shell.execute_reply":"2022-03-31T04:24:44.236502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_paths = []\nlabels = []\nfor label, name in enumerate(class_names):\n    print(\"Processing speaker {}\".format(name,))\n    dir_path = Path(DATASET_AUDIO_PATH) / name\n    speaker_sample_paths = [\n        os.path.join(dir_path, filepath)\n        for filepath in os.listdir(dir_path)\n        if filepath.endswith(\".wav\")\n    ]\n    audio_paths += speaker_sample_paths\n    labels += [label] * len(speaker_sample_paths)\n\nprint(\n    \"Found {} files belonging to {} classes.\".format(len(audio_paths), len(class_names))\n)","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:24:45.923359Z","iopub.execute_input":"2022-03-31T04:24:45.923875Z","iopub.status.idle":"2022-03-31T04:24:45.942162Z","shell.execute_reply.started":"2022-03-31T04:24:45.923824Z","shell.execute_reply":"2022-03-31T04:24:45.941031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Model Definition**","metadata":{}},{"cell_type":"code","source":"# Shuffle\nrng = np.random.RandomState(SHUFFLE_SEED)\nrng.shuffle(audio_paths)\nrng = np.random.RandomState(SHUFFLE_SEED)\nrng.shuffle(labels)\n\n# Split into training and validation\nnum_val_samples = int(VALID_SPLIT * len(audio_paths))\nprint(\"Using {} files for training.\".format(len(audio_paths) - num_val_samples))\ntrain_audio_paths = audio_paths[:-num_val_samples]\ntrain_labels = labels[:-num_val_samples]\n\nprint(\"Using {} files for validation.\".format(num_val_samples))\nvalid_audio_paths = audio_paths[-num_val_samples:]\nvalid_labels = labels[-num_val_samples:]\n\n# Create 2 datasets, one for training and the other for validation\ntrain_ds = paths_and_labels_to_dataset(train_audio_paths, train_labels)\ntrain_ds = train_ds.shuffle(buffer_size=BATCH_SIZE * 8, seed=SHUFFLE_SEED).batch(\n    BATCH_SIZE\n)\n\nvalid_ds = paths_and_labels_to_dataset(valid_audio_paths, valid_labels)\nvalid_ds = valid_ds.shuffle(buffer_size=32 * 8, seed=SHUFFLE_SEED).batch(32)\n\n\n\n\n# Transform audio wave to the frequency domain using `audio_to_fft`\ntrain_ds = train_ds.map(\n    lambda x, y: (audio_to_fft(x), y), num_parallel_calls=tf.data.AUTOTUNE\n)\ntrain_ds = train_ds.prefetch(tf.data.AUTOTUNE)\n\nvalid_ds = valid_ds.map(\n    lambda x, y: (audio_to_fft(x), y), num_parallel_calls=tf.data.AUTOTUNE\n)\n\n#train_ds = train_ds.shuffle(buffer_size=700)\n\n#valid_ds = valid_ds.shuffle(buffer_size=300)\n\nvalid_ds = valid_ds.prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:24:48.189040Z","iopub.execute_input":"2022-03-31T04:24:48.189331Z","iopub.status.idle":"2022-03-31T04:24:48.373461Z","shell.execute_reply.started":"2022-03-31T04:24:48.189285Z","shell.execute_reply":"2022-03-31T04:24:48.372392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def residual_block(x, filters, conv_num=3, activation=\"relu\"):\n    # Shortcut\n    s = keras.layers.Conv1D(filters, 1, padding=\"same\")(x)\n    for i in range(conv_num - 1):\n        x = keras.layers.Conv1D(filters, 3, padding=\"same\")(x)\n        x = keras.layers.Activation(activation)(x)\n    x = keras.layers.Conv1D(filters, 3, padding=\"same\")(x)\n    x = keras.layers.Add()([x, s])\n    x = keras.layers.Activation(activation)(x)\n    return keras.layers.MaxPool1D(pool_size=2, strides=2)(x)\n\n\ndef build_model(input_shape, num_classes):\n    inputs = keras.layers.Input(shape=input_shape, name=\"input\")\n\n    x = residual_block(inputs, 16, 2)\n    x = residual_block(x, 32, 2)\n    x = residual_block(x, 64, 3)\n    x = residual_block(x, 128, 3)\n\n    x = keras.layers.AveragePooling1D(pool_size=3, strides=3)(x)\n    x = keras.layers.Flatten()(x)\n    x = keras.layers.Dense(256, activation=\"relu\")(x)\n    x = keras.layers.Dense(128, activation=\"relu\")(x)\n\n    outputs = keras.layers.Dense(num_classes, activation=\"softmax\", name=\"output\")(x)\n\n    return keras.models.Model(inputs=inputs, outputs=outputs)\n\n\nmodel = build_model((SAMPLING_RATE // 2, 1), len(class_names))\n\nmodel.summary()\n\n# Compile the model using Adam's default learning rate\nmodel.compile(\n    optimizer=\"Adam\", loss=\"sparse_categorical_crossentropy\", metrics=[\"accuracy\"]\n)\n\n# Add callbacks:\n# 'EarlyStopping' to stop training when the model is not enhancing anymore\n# 'ModelCheckPoint' to always keep the model that has the best val_accuracy\nmodel_save_filename = \"model.h5\"\n\nearlystopping_cb = keras.callbacks.EarlyStopping(patience=10, restore_best_weights=True)\nmdlcheckpoint_cb = keras.callbacks.ModelCheckpoint(\n    model_save_filename, monitor=\"val_accuracy\", save_best_only=True\n)","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:24:50.211338Z","iopub.execute_input":"2022-03-31T04:24:50.211874Z","iopub.status.idle":"2022-03-31T04:24:50.500665Z","shell.execute_reply.started":"2022-03-31T04:24:50.211839Z","shell.execute_reply":"2022-03-31T04:24:50.499608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Training**","metadata":{}},{"cell_type":"code","source":"history = model.fit(\n    train_ds,\n    epochs=EPOCHS,\n    validation_data=valid_ds,\n    callbacks=[earlystopping_cb, mdlcheckpoint_cb],\n)","metadata":{"execution":{"iopub.status.busy":"2022-03-31T04:24:54.470958Z","iopub.execute_input":"2022-03-31T04:24:54.471307Z","iopub.status.idle":"2022-03-31T04:26:08.150585Z","shell.execute_reply.started":"2022-03-31T04:24:54.471274Z","shell.execute_reply":"2022-03-31T04:26:08.149594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Evaluation**","metadata":{}},{"cell_type":"code","source":"print(model.evaluate(valid_ds))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}