{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":7634,"databundleVersionId":46676,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install py7zr -q","metadata":{"_uuid":"102bc939-f32a-4c27-b1bb-950e5a85fe55","_cell_guid":"f01aa6a5-31cd-494d-b022-12ecbbbddfb4","collapsed":false,"execution":{"iopub.status.busy":"2024-10-10T17:21:09.739548Z","iopub.execute_input":"2024-10-10T17:21:09.740309Z","iopub.status.idle":"2024-10-10T17:21:22.622578Z","shell.execute_reply.started":"2024-10-10T17:21:09.740267Z","shell.execute_reply":"2024-10-10T17:21:22.621469Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport py7zr\nimport random\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport seaborn as sns\n\nfrom IPython import display","metadata":{"_uuid":"7ad957cc-f905-4cfb-bb1f-3024f1124539","_cell_guid":"18f96018-1e27-4ce3-ab90-216219169fd1","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-10-10T18:19:04.390705Z","iopub.execute_input":"2024-10-10T18:19:04.391574Z","iopub.status.idle":"2024-10-10T18:19:04.406252Z","shell.execute_reply.started":"2024-10-10T18:19:04.391532Z","shell.execute_reply":"2024-10-10T18:19:04.405478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data and extraction directories\ndata_dir = \"/kaggle/input/tensorflow-speech-recognition-challenge/\"\nextract_dir = '/kaggle/working/extracted_data_train'\n\ndef extract_7z(filepath, dest_dir):\n    with py7zr.SevenZipFile(filepath, mode='r') as z:\n        z.extractall(path=dest_dir)\n\n# Extract the data if it hasn't been already\nif not os.path.exists(extract_dir):\n    filepath = os.path.join(data_dir, \"train.7z\")\n    print(f\"Extracting files from {filepath} to {extract_dir}...\")\n    extract_7z(filepath, extract_dir)\nelse:\n    print(f\"Data already extracted at {extract_dir}\")","metadata":{"_uuid":"be0ae6d2-3f05-4709-b603-493cde3fb179","_cell_guid":"0dbf8a76-94cc-4dda-aa39-7f81479784fc","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-10-10T17:30:01.217954Z","iopub.execute_input":"2024-10-10T17:30:01.218898Z","iopub.status.idle":"2024-10-10T17:32:19.349076Z","shell.execute_reply.started":"2024-10-10T17:30:01.218852Z","shell.execute_reply":"2024-10-10T17:32:19.348191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load command labels\ncommands = np.array([d for d in tf.io.gfile.listdir(os.path.join(extract_dir, \"train/audio\")) \n                    if d != '_background_noise_'])\nprint('Commands:', commands)\n\n# Load background noise files for data augmentation\nnoise_dir = os.path.join(extract_dir, \"train/audio/_background_noise_\")\nnoise_files = tf.io.gfile.glob(noise_dir + '/*.wav')\n\n# Function to decode audio files\ndef decode_audio(audio_binary):\n    audio, _ = tf.audio.decode_wav(contents=audio_binary)\n    return tf.squeeze(audio, axis=-1)\n\n# Function to extract label from file path\ndef get_label(file_path):\n    label = tf.strings.split(input=file_path, sep=os.path.sep)[-2]\n    return tf.cond(tf.reduce_any(tf.equal(commands, label)), lambda: label, lambda: tf.constant(\"unknown\", dtype=tf.string))\n\n# Function to convert waveform to spectrogram\ndef get_spectrogram(waveform):\n    input_len = 16000\n    waveform = waveform[:input_len]\n    # Pad with zeros if audio is shorter than 16000 samples\n    zero_padding = tf.zeros([input_len] - tf.shape(waveform), dtype=tf.float32)\n    waveform = tf.cast(waveform, dtype=tf.float32)\n    equal_length = tf.concat([waveform, zero_padding], 0)\n    \n    spectrogram = tf.signal.stft(equal_length, frame_length=255, frame_step=128)\n    spectrogram = tf.abs(spectrogram)\n    return spectrogram[..., tf.newaxis]\n\n# Function to add background noise for data augmentation\ndef add_background_noise(waveform, noise_files):\n    noise_file = random.choice(noise_files)\n    noise_audio_binary = tf.io.read_file(noise_file)\n    noise_waveform = decode_audio(noise_audio_binary)\n\n    # Adjust noise length to match the audio\n    waveform_len = tf.shape(waveform)[0]\n    noise_len = tf.shape(noise_waveform)[0]\n    if noise_len > waveform_len:\n        offset = tf.random.uniform(shape=[], minval=0, maxval=noise_len - waveform_len, dtype=tf.int32)\n        noise_waveform = noise_waveform[offset:offset + waveform_len]\n    else:\n        padding = tf.zeros([waveform_len - noise_len], dtype=tf.float32)\n        noise_waveform = tf.concat([noise_waveform, padding], axis=0)\n\n    noise_factor = tf.random.uniform(shape=[], minval=0.0, maxval=0.5)\n    augmented_waveform = waveform + noise_factor * noise_waveform\n    return tf.clip_by_value(augmented_waveform, -1.0, 1.0)\n\n# Function to load and preprocess audio data\ndef preprocess_dataset(files, augment=False):\n    files_ds = tf.data.Dataset.from_tensor_slices(files)\n    output_ds = files_ds.map(\n        lambda file_path: (tf.io.read_file(file_path), get_label(file_path)), \n        num_parallel_calls=tf.data.AUTOTUNE\n    )\n    output_ds = output_ds.map(\n        lambda audio_binary, label: (decode_audio(audio_binary), label),\n        num_parallel_calls=tf.data.AUTOTUNE\n    )\n    if augment:\n        output_ds = output_ds.map(\n            lambda waveform, label: (add_background_noise(waveform, noise_files), label), \n            num_parallel_calls=tf.data.AUTOTUNE\n        )\n    output_ds = output_ds.map(\n        lambda waveform, label: (get_spectrogram(waveform), tf.argmax(label == commands)),\n        num_parallel_calls=tf.data.AUTOTUNE\n    )\n    return output_ds","metadata":{"_uuid":"9c8ff4ca-377a-431f-864d-1ff776e542ca","_cell_guid":"1fdf503a-2277-4aad-8a01-c379ac11e190","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-10-10T17:32:19.350844Z","iopub.execute_input":"2024-10-10T17:32:19.351172Z","iopub.status.idle":"2024-10-10T17:32:19.370516Z","shell.execute_reply.started":"2024-10-10T17:32:19.351139Z","shell.execute_reply":"2024-10-10T17:32:19.369547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath_data = os.path.join(extract_dir, \"train/audio\")\n\n# Get all filenames, excluding noise files\nfilenames = tf.io.gfile.glob([os.path.join(filepath_data, d,  '*') for d in commands])\nfilenames = tf.random.shuffle(filenames)\n\ntotal_samples = len(filenames)\ntrain_size = int(0.8 * total_samples)  # 80% for training\nval_size = int(0.1 * total_samples)  # 10% for validation\ntest_size = total_samples - train_size - val_size  # Remaining 10% for testing\n\n# Split dataset into train, validation, and test sets\ntrain_files = filenames[:train_size]\nval_files = filenames[train_size:train_size + val_size]\ntest_files = filenames[train_size + val_size:]\n\nprint('Training set size:', len(train_files))\nprint('Validation set size:', len(val_files))\nprint('Test set size:', len(test_files))","metadata":{"_uuid":"17c6ac8b-9a8e-42a1-976a-23c780dd4ae8","_cell_guid":"e3c478ad-b19b-43f4-8712-77524314ccea","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-10-10T17:32:19.371834Z","iopub.execute_input":"2024-10-10T17:32:19.372250Z","iopub.status.idle":"2024-10-10T17:32:19.705094Z","shell.execute_reply.started":"2024-10-10T17:32:19.372205Z","shell.execute_reply":"2024-10-10T17:32:19.704100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = preprocess_dataset(train_files, augment=True)\nval_ds = preprocess_dataset(val_files)\ntest_ds = preprocess_dataset(test_files)\n\n# Batch the datasets for training\nbatch_size = 32\ntrain_ds = train_ds.batch(batch_size).cache().prefetch(tf.data.AUTOTUNE)\nval_ds = val_ds.batch(batch_size).cache().prefetch(tf.data.AUTOTUNE)","metadata":{"_uuid":"e1a2a808-d029-457e-be85-8a21ae8ed86e","_cell_guid":"b647bfa6-db6b-452b-8926-5e2c0b6ef14b","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-10-10T17:32:19.707457Z","iopub.execute_input":"2024-10-10T17:32:19.707855Z","iopub.status.idle":"2024-10-10T17:32:20.573771Z","shell.execute_reply.started":"2024-10-10T17:32:19.707812Z","shell.execute_reply":"2024-10-10T17:32:20.572965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get input shape from the data\nfor spectrogram, _ in train_ds.take(1):\n    input_shape = spectrogram.shape[1:]\n\nprint('Input shape:', input_shape)\nnum_labels = len(commands)\n\n# Normalize input\nnorm_layer = layers.Normalization()\nnorm_layer.adapt(data=train_ds.map(lambda spec, label: spec))\n\ntimesteps = 16\n\n# Build the model\nmodel = models.Sequential([\n    layers.Input(shape=input_shape),\n    layers.Resizing(32, 32),\n    norm_layer,\n    layers.Conv2D(32, 3, activation='relu'),\n    layers.Conv2D(64, 3, activation='relu'),\n    layers.Conv2D(128, 3, activation='relu'),\n    layers.MaxPooling2D(),\n    layers.Dropout(0.25),\n    layers.Flatten(),\n    layers.Reshape((-1, timesteps, 21632 // timesteps)), \n    tf.keras.layers.Lambda(lambda x: tf.squeeze(x, axis=1)),\n    layers.GRU(64),\n    layers.Dense(128, activation='relu'),\n    layers.Dropout(0.5),\n    layers.Dense(num_labels)\n])\n\nmodel.summary()\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(),\n    loss=tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n    metrics=['accuracy']\n)\n\nEPOCHS = 20\nhistory = model.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=EPOCHS,\n    callbacks=tf.keras.callbacks.EarlyStopping(verbose=1, patience=2),\n)","metadata":{"_uuid":"665d980c-5a40-48a2-b279-cf22f207f1a2","_cell_guid":"7a6deeb8-42b5-4f58-98eb-c1263ff368da","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-10-10T17:46:21.897704Z","iopub.execute_input":"2024-10-10T17:46:21.897999Z","iopub.status.idle":"2024-10-10T17:49:05.687680Z","shell.execute_reply.started":"2024-10-10T17:46:21.897968Z","shell.execute_reply":"2024-10-10T17:49:05.686759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metrics = history.history\nplt.plot(history.epoch, metrics['loss'], metrics['val_loss'])\nplt.legend(['loss', 'val_loss'])\nplt.show()\n\ntest_audio = []\ntest_labels = []\n\nfor audio, label in test_ds:\n    test_audio.append(audio.numpy())\n    test_labels.append(label.numpy())\n\ntest_audio = np.array(test_audio)\ntest_labels = np.array(test_labels)\n\ny_pred = np.argmax(model.predict(test_audio), axis=1)\ny_true = test_labels\n\ntest_acc = sum(y_pred == y_true) / len(y_true)\nprint(f'Test set accuracy: {test_acc:.0%}')\n\nconfusion_mtx = tf.math.confusion_matrix(y_true, y_pred)\nplt.figure(figsize=(10, 8))\nsns.heatmap(confusion_mtx, xticklabels=commands, yticklabels=commands, annot=True, fmt='g')\nplt.xlabel('Prediction')\nplt.ylabel('Label')\nplt.show()","metadata":{"_uuid":"38e6f71e-c345-4a3c-b7f5-cf6b0ddbd730","_cell_guid":"fcac6c72-0c1a-4ef0-a105-24238c20a5b5","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-10-10T17:36:00.299704Z","iopub.execute_input":"2024-10-10T17:36:00.300535Z","iopub.status.idle":"2024-10-10T17:36:09.261784Z","shell.execute_reply.started":"2024-10-10T17:36:00.300490Z","shell.execute_reply":"2024-10-10T17:36:09.260821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data and extraction directories\ndata_dir = \"/kaggle/input/tensorflow-speech-recognition-challenge/\"\nextract_dir = '/kaggle/working/extracted_data_test/'\n\n# Extract the data if it hasn't been already\nif not os.path.exists(extract_dir):\n    filepath = os.path.join(data_dir, \"test.7z\")\n    print(f\"Extracting files from {filepath} to {extract_dir}...\")\n    extract_7z(filepath, extract_dir)\nelse:\n    print(f\"Data already extracted at {extract_dir}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-10T18:21:09.319932Z","iopub.execute_input":"2024-10-10T18:21:09.320794Z","iopub.status.idle":"2024-10-10T18:21:09.326828Z","shell.execute_reply.started":"2024-10-10T18:21:09.320752Z","shell.execute_reply":"2024-10-10T18:21:09.325902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -r /kaggle/working/sample_submission.csv","metadata":{"execution":{"iopub.status.busy":"2024-10-10T19:37:59.715085Z","iopub.execute_input":"2024-10-10T19:37:59.715879Z","iopub.status.idle":"2024-10-10T19:38:00.770125Z","shell.execute_reply.started":"2024-10-10T19:37:59.715839Z","shell.execute_reply":"2024-10-10T19:38:00.768762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath_data = \"/kaggle/working/extracted_data_test/test/audio/\"\nfilenames = os.listdir(filepath_data)\nfilenames_path = [filepath_data + i for i in filenames]","metadata":{"execution":{"iopub.status.busy":"2024-10-10T19:35:18.542972Z","iopub.execute_input":"2024-10-10T19:35:18.543365Z","iopub.status.idle":"2024-10-10T19:35:18.594715Z","shell.execute_reply.started":"2024-10-10T19:35:18.543327Z","shell.execute_reply":"2024-10-10T19:35:18.593736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = preprocess_dataset(filenames_path)\ntest_ds = test_ds.batch(1)\n\nlabels = []\n\nfor spectrogram in tqdm(test_ds):\n    spectrogram = spectrogram[0]\n    prediction = model(spectrogram)\n    \n    probabilities = tf.nn.softmax(prediction[0])\n    predicted_index = tf.argmax(probabilities).numpy()\n    predicted_command = commands[predicted_index]\n    \n    labels.append(predicted_command)\n\n#     print(f\"Probabilities: {probabilities.numpy()}\")\n#     print(f\"Predicted Command: {predicted_command}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-10T18:21:14.871891Z","iopub.execute_input":"2024-10-10T18:21:14.872297Z","iopub.status.idle":"2024-10-10T19:05:00.233260Z","shell.execute_reply.started":"2024-10-10T18:21:14.872258Z","shell.execute_reply":"2024-10-10T19:05:00.232244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\"fname\":filenames,\"label\":labels})\nsubmission.to_csv(\"submission.csv\", index=False)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2024-10-10T19:35:53.990210Z","iopub.execute_input":"2024-10-10T19:35:53.990698Z","iopub.status.idle":"2024-10-10T19:35:54.224520Z","shell.execute_reply.started":"2024-10-10T19:35:53.990657Z","shell.execute_reply":"2024-10-10T19:35:54.223397Z"},"trusted":true},"execution_count":null,"outputs":[]}]}