{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":8900,"databundleVersionId":862232,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Freesound Classification","metadata":{}},{"cell_type":"markdown","source":"# Import Packages","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nimport os\nimport time\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport IPython\nimport IPython.display as ipd\nimport librosa\nimport librosa.display\nimport pickle\nimport joblib\nimport random\nimport cv2\n\nfrom scipy.signal import wiener\nfrom sklearn.model_selection import train_test_split, KFold, ShuffleSplit\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import KFold\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.python.keras.utils.data_utils import Sequence\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\nfrom tensorflow.keras.applications import MobileNetV2\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.optimizers import AdamW\nfrom tensorflow.image import grayscale_to_rgb\nfrom tensorflow.keras.applications import Xception\nfrom sklearn.utils.class_weight import compute_class_weight","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-17T00:18:42.116114Z","iopub.execute_input":"2024-11-17T00:18:42.116501Z","iopub.status.idle":"2024-11-17T00:18:46.100028Z","shell.execute_reply.started":"2024-11-17T00:18:42.116462Z","shell.execute_reply":"2024-11-17T00:18:46.098943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the Training Data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/freesound-audio-tagging/train_post_competition.csv\", dtype=str)\nprint(train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:18:46.102039Z","iopub.execute_input":"2024-11-17T00:18:46.102794Z","iopub.status.idle":"2024-11-17T00:18:46.128897Z","shell.execute_reply.started":"2024-11-17T00:18:46.102743Z","shell.execute_reply":"2024-11-17T00:18:46.127948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:18:46.129959Z","iopub.execute_input":"2024-11-17T00:18:46.130262Z","iopub.status.idle":"2024-11-17T00:18:46.145208Z","shell.execute_reply.started":"2024-11-17T00:18:46.130230Z","shell.execute_reply":"2024-11-17T00:18:46.144215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Label Distribution","metadata":{}},{"cell_type":"code","source":"train_labels = (train.label.value_counts() / len(train)).to_frame().sort_index().T\n\ntrain_labels","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:18:46.147185Z","iopub.execute_input":"2024-11-17T00:18:46.147492Z","iopub.status.idle":"2024-11-17T00:18:46.172343Z","shell.execute_reply.started":"2024-11-17T00:18:46.147460Z","shell.execute_reply":"2024-11-17T00:18:46.171270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'{train_labels.shape[1]} possible classes.')","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:18:46.173483Z","iopub.execute_input":"2024-11-17T00:18:46.173818Z","iopub.status.idle":"2024-11-17T00:18:46.181881Z","shell.execute_reply.started":"2024-11-17T00:18:46.173783Z","shell.execute_reply":"2024-11-17T00:18:46.180876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"verification_counts = train['manually_verified'].value_counts()\nprint(verification_counts)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:18:46.183403Z","iopub.execute_input":"2024-11-17T00:18:46.183798Z","iopub.status.idle":"2024-11-17T00:18:46.193725Z","shell.execute_reply.started":"2024-11-17T00:18:46.183764Z","shell.execute_reply":"2024-11-17T00:18:46.192819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 5))\nplt.bar(['Not Verified', 'Manually Verified'], verification_counts, color=['blue', 'red'])\nplt.xlabel('Verification Status')\nplt.ylabel('Count')\nplt.title('Number of Manually Verified vs Not Verified Samples')\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:18:46.194803Z","iopub.execute_input":"2024-11-17T00:18:46.195077Z","iopub.status.idle":"2024-11-17T00:18:46.454992Z","shell.execute_reply.started":"2024-11-17T00:18:46.195047Z","shell.execute_reply":"2024-11-17T00:18:46.453961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load a random wav file and check the length, label, and sample rate, \nrandom_row = train.sample(n=1)\nrandom_filename = random_row['fname'].values[0]\nlabel = random_row['label'].values[0]\n\n#Set the audio path and sample rate to load with librosa\naudio_dir = '/kaggle/input/freesound-audio-tagging/audio_train'\naudio_path = f\"{audio_dir}/{random_filename}\"\nsample, sr = librosa.load(audio_path, sr=None)\n\nprint(f\"Playing audio: {random_filename}\")\nipd.display(ipd.Audio(sample, rate=sr))\n\nprint(f'Length: {len(sample)/sr:.2f}s')\nprint(f'Label: {label}')\nprint(f'Sample Rate: {sr}')\n\n# Plot Different Spectrograms\nfig, ax = plt.subplots(4, 1, figsize=(16, 10))\n\n# Temporal - audio over time\nlibrosa.display.waveshow(sample, sr=sr, ax=ax[0])\nax[0].set(title='Temporal Signal', xlabel='Time (s)', ylabel='Amplitude')\n\n# STFT - Short Term Fourier Transform: Computes the amplitude of the frequencies for different bands over time.\nstft_result = librosa.stft(sample)\nstft_db = librosa.amplitude_to_db(abs(stft_result))\nlibrosa.display.specshow(stft_db, sr=sr, x_axis='time', y_axis='log', ax=ax[1])\nax[1].set(title='STFT Spectrogram', xlabel='Time (s)', ylabel='Frequency (Hz)')\n\n# MFCC Mel Frequency Cepstral Coefficients: Describe the instantaneous spectral envelope shape of the speech signal.\nmfccs = librosa.feature.mfcc(y=sample, sr=sr, n_mfcc=13)\nlibrosa.display.specshow(mfccs, sr=sr, x_axis='time', ax=ax[2])\nax[2].set(title='MFCC', xlabel='Time (s)', ylabel='MFCC Coefficients')\n\n#Log-Mel spectrogram: Similar to STFT but represented in the Mel scale (log transformation of the frequency scale)\nmel_spec = librosa.feature.melspectrogram(y=sample, sr=sr)\nlog_mel_spec = librosa.power_to_db(mel_spec)\nlibrosa.display.specshow(log_mel_spec, sr=sr, x_axis='time', y_axis='mel', ax=ax[3])\nax[3].set(title='Log-Mel Spectrogram', xlabel='Time (s)', ylabel='Mel Frequency (Hz)')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:18:46.456212Z","iopub.execute_input":"2024-11-17T00:18:46.456522Z","iopub.status.idle":"2024-11-17T00:18:50.030086Z","shell.execute_reply.started":"2024-11-17T00:18:46.456489Z","shell.execute_reply":"2024-11-17T00:18:50.029187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plot the distribution of labels\nlabel_counts = train['label'].value_counts()\n\nplt.figure(figsize=(12, 6))\nlabel_counts.plot(kind='bar')\nplt.xlabel('Labels')\nplt.ylabel('Frequency')\nplt.title('Distribution of Training Labels')\nplt.xticks(rotation=45, ha='right') \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:18:50.031413Z","iopub.execute_input":"2024-11-17T00:18:50.032086Z","iopub.status.idle":"2024-11-17T00:18:50.556063Z","shell.execute_reply.started":"2024-11-17T00:18:50.032043Z","shell.execute_reply":"2024-11-17T00:18:50.555079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a list to explore the length of the files\ntrain_len_data = []\n\nfor file in train['fname']:\n    audio_path = os.path.join(audio_dir, file)\n    sample, sr = librosa.load(audio_path, sr=None)\n    fname = file\n    length = len(sample) / sr\n    train_len_data.append({'fname': fname, 'length': length})\n\ntrain_len = pd.DataFrame(train_len_data)\n\ntrain_len.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:18:50.559487Z","iopub.execute_input":"2024-11-17T00:18:50.559850Z","iopub.status.idle":"2024-11-17T00:19:18.727669Z","shell.execute_reply.started":"2024-11-17T00:18:50.559802Z","shell.execute_reply":"2024-11-17T00:19:18.726666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the lengths and calculate simple statistics\nfig, ax = plt.subplots(figsize=(10,4))\ntrain_len['length'].sort_values(ascending=False).plot(kind='hist', bins=100, ax=ax)\nax.set_xlim([0,31])\nax.set_ylabel('Count')\nax.set_xlabel('Duration (s)')\nax.set_title('Distribution of signal durations');\n\nprint(f'Smallest duration: {train_len[\"length\"].min():.2f}s')\nprint(f'Largest duration: {train_len[\"length\"].max():.2f}s')\nprint(f'Mean duration: {train_len[\"length\"].mean():.2f}s')\nprint(f'Median duration: {train_len[\"length\"].median():.2f}s')","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:19:18.728938Z","iopub.execute_input":"2024-11-17T00:19:18.729247Z","iopub.status.idle":"2024-11-17T00:19:19.115092Z","shell.execute_reply.started":"2024-11-17T00:19:18.729214Z","shell.execute_reply":"2024-11-17T00:19:19.114077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['length'] = train_len['length'] ","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:19:19.116385Z","iopub.execute_input":"2024-11-17T00:19:19.116715Z","iopub.status.idle":"2024-11-17T00:19:19.122135Z","shell.execute_reply.started":"2024-11-17T00:19:19.116669Z","shell.execute_reply":"2024-11-17T00:19:19.121300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:19:19.123177Z","iopub.execute_input":"2024-11-17T00:19:19.123478Z","iopub.status.idle":"2024-11-17T00:19:19.140470Z","shell.execute_reply.started":"2024-11-17T00:19:19.123447Z","shell.execute_reply":"2024-11-17T00:19:19.139548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"grouped = train.groupby('label')['length'].agg(['min', 'max', 'median']).reset_index()\n\n# Sort each label by min, max, and median\ngrouped_sorted_by_min = grouped.sort_values(by='min', ascending=True)\ngrouped_sorted_by_max = grouped.sort_values(by='max', ascending=True)\ngrouped_sorted_by_median = grouped.sort_values(by='median', ascending=True)\n\nprint(\"Sorted by min length:\")\nprint(grouped_sorted_by_min)\n\nprint(\"\\nSorted by max length:\")\nprint(grouped_sorted_by_max)\n\nprint(\"\\nSorted by median length:\")\nprint(grouped_sorted_by_median)","metadata":{}},{"cell_type":"markdown","source":"train.shape","metadata":{}},{"cell_type":"markdown","source":"grouped_sorted_by_median.head()","metadata":{}},{"cell_type":"markdown","source":"grouped_sorted_by_median['min_threshold'] = grouped_sorted_by_median['median'] * 0.20\ngrouped_sorted_by_median['max_threshold'] = grouped_sorted_by_median['median'] / 0.20","metadata":{}},{"cell_type":"markdown","source":"grouped_sorted_by_median.head(41)","metadata":{}},{"cell_type":"markdown","source":"train = pd.merge(train, grouped_sorted_by_median[['label', 'min_threshold', 'max_threshold']], on='label', how='left')","metadata":{}},{"cell_type":"markdown","source":"train.head()","metadata":{}},{"cell_type":"markdown","source":"train.shape","metadata":{}},{"cell_type":"markdown","source":"train = train[(train['length'] >= train['min_threshold']) & (train['length'] <= train['max_threshold'])]","metadata":{}},{"cell_type":"markdown","source":"train.shape","metadata":{}},{"cell_type":"markdown","source":"train.head()","metadata":{}},{"cell_type":"markdown","source":"train_labels2 = (train.label.value_counts() / len(train)).to_frame().sort_index().T\n\ntrain_labels2","metadata":{}},{"cell_type":"markdown","source":"#Plot the distribution of labels\nlabel_counts = train['label'].value_counts()\n\nplt.figure(figsize=(12, 6))\nlabel_counts.plot(kind='bar')\nplt.xlabel('Labels')\nplt.ylabel('Frequency')\nplt.title('Distribution of Training Labels')\nplt.xticks(rotation=45, ha='right') \nplt.tight_layout()\nplt.show()","metadata":{}},{"cell_type":"code","source":"# Create a list to explore the length of the files\ntrain_len_data = []\n\nfor file in train['fname']:\n    audio_path = os.path.join(audio_dir, file)\n    sample, sr = librosa.load(audio_path, sr=None)\n    fname = file\n    length = len(sample) / sr\n    train_len_data.append({'fname': fname, 'length': length})\n\ntrain_len = pd.DataFrame(train_len_data)\n\ntrain_len.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:19:19.141606Z","iopub.execute_input":"2024-11-17T00:19:19.141936Z","iopub.status.idle":"2024-11-17T00:19:45.365397Z","shell.execute_reply.started":"2024-11-17T00:19:19.141904Z","shell.execute_reply":"2024-11-17T00:19:45.364437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the lengths and calculate simple statistics\nfig, ax = plt.subplots(figsize=(10,4))\ntrain_len['length'].sort_values(ascending=False).plot(kind='hist', bins=100, ax=ax)\nax.set_xlim([0,31])\nax.set_ylabel('Count')\nax.set_xlabel('Duration (s)')\nax.set_title('Distribution of signal durations');\n\nprint(f'Smallest duration: {train_len[\"length\"].min():.2f}s')\nprint(f'Largest duration: {train_len[\"length\"].max():.2f}s')\nprint(f'Mean duration: {train_len[\"length\"].mean():.2f}s')\nprint(f'Median duration: {train_len[\"length\"].median():.2f}s')","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:19:45.366783Z","iopub.execute_input":"2024-11-17T00:19:45.367170Z","iopub.status.idle":"2024-11-17T00:19:46.001689Z","shell.execute_reply.started":"2024-11-17T00:19:45.367127Z","shell.execute_reply":"2024-11-17T00:19:46.000747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Explaination of log mel sampling\n- sr = 44100: Sampling rate of 44.1 kHz.\n- nmels = 128: Number of Mel bands to generate.\n- nfft = 1024: Length of the FFT window.\n- hop_length = 512: Number of samples between successive frames.\n- fmin = 20: Minimum frequency to consider (20 Hz).\n- fmax = sr // 2: Maximum frequency to consider, set to half the sampling rate (Nyquist frequency).","metadata":{}},{"cell_type":"markdown","source":"## Transform the Audio Files Into Log-Mel Specgrograms","metadata":{}},{"cell_type":"markdown","source":"sr = 44100\nnmels = 128\nnfft = 1024\nhop_length = 512\nfmin = 20\nfmax = sr // 2\n\nlogmel_features = [] \nepsilon = 1e-6\ntotal_width = 0\ntotal_height = 0\nnum_spectrograms = 0\n\nfor file in train['fname']:\n    audio_path = os.path.join(audio_dir, file)\n    sample, sr = librosa.load(audio_path, sr=sr)\n\n#   Optional transformations\n#    if random.random() < 0.5:\n#        sample = librosa.effects.time_stretch(sample, rate=random.uniform(0.8, 1.2))\n\n#    if random.random() < 0.5:\n#        sample = librosa.effects.pitch_shift(y=sample, sr=sr, n_steps=random.randint(-2, 2))\n\n    #Apply filter to remove background noise\n#    filtered_sample = wiener(sample + epsilon)\n\n#    sample_trimmed, _ = librosa.effects.trim(sample)\n\n    mel_spec = librosa.feature.melspectrogram(\n        y=sample, \n        sr=sr, \n        n_fft=nfft, \n        hop_length=hop_length, \n        n_mels=nmels, \n        fmin=fmin, \n        fmax=fmax if fmax else sr // 2\n    )\n\n    log_mel_spec = librosa.power_to_db(mel_spec)\n    \n    total_height += log_mel_spec.shape[0]\n    total_width += log_mel_spec.shape[1]\n    num_spectrograms += 1\n\n\n    log_mel_resized = np.resize(log_mel_spec, (128, 128))\n\n    # Map to RGB\n    #log_mel_resized_uint8 = (log_mel_resized * 255.0 / np.max(log_mel_resized)).astype(np.uint8)\n    #colored_spectrogram = cv2.applyColorMap(log_mel_resized_uint8, cv2.COLORMAP_VIRIDIS)\n\n    logmel_features.append({'fname': file, 'log_mel': log_mel_resized})\n\navg_height = total_height / num_spectrograms\navg_width = total_width / num_spectrograms\nprint(f\"Average spectrogram size before resizing: Height = {avg_height:.2f}, Width = {avg_width:.2f}\")    \n    \nlogmel_df = pd.DataFrame(logmel_features)\n\nlogmel_df.head(5)","metadata":{"jupyter":{"source_hidden":true}}},{"cell_type":"markdown","source":"logmel_df.shape","metadata":{}},{"cell_type":"markdown","source":"# Plot the Log-Mel Spectrograms\nfor i in range(5):\n    log_mel_spec = logmel_df.iloc[i]['log_mel']\n    fname = logmel_df.iloc[i]['fname']\n    \n    # Plotting the log Mel spectrogram\n    plt.figure(figsize=(10, 4))\n    librosa.display.specshow(log_mel_spec, sr=sr, hop_length=hop_length, x_axis='time', y_axis='mel', cmap='magma')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title(f'Log Mel Spectrogram - {fname}')\n    plt.xlabel('Time (s)')\n    plt.ylabel('Frequency (Hz)')\n    plt.tight_layout()\n    plt.show()","metadata":{"jupyter":{"source_hidden":true}}},{"cell_type":"markdown","source":"# Create a Pipeline to Transform Train/Test Data Into Log-Mel Spectrograms","metadata":{}},{"cell_type":"markdown","source":"class AudioPreprocessor(BaseEstimator, TransformerMixin):\n    def __init__(self, sr=22050, nmels=128, nfft=1024, hop_length=512, fmin=20, fmax=None, target_dim=(128, 128)):\n        self.sr = sr\n        self.nmels = nmels\n        self.nfft = nfft\n        self.hop_length = hop_length\n        self.fmin = fmin\n        self.fmax = fmax if fmax else sr // 2\n        self.target_dim = target_dim\n\n    def fit(self, X, y=None):\n        return self\n\n    def transform(self, X):\n        log_mel_features = []\n        epsilon = 1e-6\n        \n        def add_noise(audio, noise_level=0.005):\n            noise = np.random.randn(len(audio)) * noise_level\n            return audio + noise\n        \n        for file in X:\n            sample, _ = librosa.load(file, sr=self.sr)\n\n            if random.random() < 0.5:\n                sample = add_noise(sample, noise_level=0.005)\n            \n#            if random.random() < 0.5:\n#                sample = librosa.effects.time_stretch(sample, rate=random.uniform(0.8, 1.2))\n\n            if random.random() < 0.5:\n                sample = librosa.effects.pitch_shift(y=sample, sr=self.sr, n_steps=random.randint(-2, 2))\n\n#            filtered_sample = wiener(sample + epsilon)\n\n            sample_trimmed, _ = librosa.effects.trim(sample)\n\n            mel_spec = librosa.feature.melspectrogram(\n                y=sample_trimmed, \n                sr=self.sr, \n                n_fft=self.nfft, \n                hop_length=self.hop_length, \n                n_mels=self.nmels, \n                fmin=self.fmin, \n                fmax=self.fmax\n            )\n\n            log_mel_spec = librosa.power_to_db(mel_spec)\n            \n#            log_mel_spec = (log_mel_spec - log_mel_spec.min()) / (log_mel_spec.max() - log_mel_spec.min())\n\n            log_mel_resized = np.resize(log_mel_spec, self.target_dim)\n    \n            # Map to RGB\n            #log_mel_resized_uint8 = (log_mel_resized * 255.0 / np.max(log_mel_resized)).astype(np.uint8)\n            #colored_spectrogram = cv2.applyColorMap(log_mel_resized_uint8, cv2.COLORMAP_VIRIDIS)\n\n            log_mel_features.append(log_mel_resized)\n\n        return np.array(log_mel_features)\n\naudio_pipeline = Pipeline([\n    ('audio_preprocessor', AudioPreprocessor(target_dim=(128, 128)))\n])\n\naudio_paths = [os.path.join(audio_dir, fname) for fname in train['fname']]\nlogmel_features_array = audio_pipeline.fit_transform(audio_paths)\n\ntrain['log_mel'] = list(logmel_features_array)","metadata":{"execution":{"iopub.status.busy":"2024-11-15T21:58:10.940973Z","iopub.execute_input":"2024-11-15T21:58:10.942240Z","iopub.status.idle":"2024-11-15T22:13:20.658899Z","shell.execute_reply.started":"2024-11-15T21:58:10.942186Z","shell.execute_reply":"2024-11-15T22:13:20.657558Z"},"jupyter":{"source_hidden":true}}},{"cell_type":"code","source":"train_files, test_files = train_test_split(train, test_size=0.2, random_state=42)\n\ntrain_audio_paths = [os.path.join(audio_dir, fname) for fname in train_files['fname']]\ntest_audio_paths = [os.path.join(audio_dir, fname) for fname in test_files['fname']]\n\nclass AudioPreprocessor(BaseEstimator, TransformerMixin):\n    def __init__(self, sr=44100, nmels=128, nfft=1024, hop_length=512, fmin=20, fmax=None, target_dim=(128, 256), augment=False):\n        self.sr = sr\n        self.nmels = nmels\n        self.nfft = nfft\n        self.hop_length = hop_length\n        self.fmin = fmin\n        self.fmax = fmax if fmax else sr // 2\n        self.target_dim = target_dim\n        self.augment = augment\n\n    def fit(self, X, y=None):\n        return self\n\n    def transform(self, X):\n        log_mel_features = []\n        \n        def add_noise(audio, noise_level=0.005):\n            noise = np.random.randn(len(audio)) * noise_level\n            return audio + noise\n\n        for file in X:\n            sample, _ = librosa.load(file, sr=self.sr)\n\n            if self.augment:\n                if random.random() < 0.5:\n                    sample = add_noise(sample, noise_level=0.005)\n                \n                if random.random() < 0.5:\n                    sample = librosa.effects.pitch_shift(y=sample, sr=self.sr, n_steps=random.randint(-2, 2))\n\n            sample_trimmed, _ = librosa.effects.trim(sample)\n\n            mel_spec = librosa.feature.melspectrogram(\n                y=sample_trimmed, \n                sr=self.sr, \n                n_fft=self.nfft, \n                hop_length=self.hop_length, \n                n_mels=self.nmels, \n                fmin=self.fmin, \n                fmax=self.fmax\n            )\n\n            log_mel_spec = librosa.power_to_db(mel_spec)\n            \n            log_mel_resized = np.resize(log_mel_spec, self.target_dim)\n\n            log_mel_features.append(log_mel_resized)\n\n        return np.array(log_mel_features)\n\ntrain_audio_pipeline = Pipeline([\n    ('audio_preprocessor', AudioPreprocessor(target_dim=(128, 256), augment=True))\n])\n\ntrain_logmel_features_array = train_audio_pipeline.fit_transform(train_audio_paths)\n\ntest_audio_pipeline = Pipeline([\n    ('audio_preprocessor', AudioPreprocessor(target_dim=(128, 256), augment=False))\n])\n\ntest_logmel_features_array = test_audio_pipeline.fit_transform(test_audio_paths)\n\ntrain_df = train_files.copy()\ntrain_df['log_mel'] = list(train_logmel_features_array)\n\n# Create a new test DataFrame with log-mel features\ntest_df = test_files.copy()\ntest_df['log_mel'] = list(test_logmel_features_array)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:19:46.003058Z","iopub.execute_input":"2024-11-17T00:19:46.003452Z","iopub.status.idle":"2024-11-17T00:49:14.386334Z","shell.execute_reply.started":"2024-11-17T00:19:46.003407Z","shell.execute_reply":"2024-11-17T00:49:14.385405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_indices = random.sample(range(len(train_df)), 5)\n\nfor i, idx in enumerate(random_indices):\n    log_mel_spec = train_df['log_mel'].iloc[idx] \n\n    plt.figure(figsize=(10, 4))\n    librosa.display.specshow(log_mel_spec, sr=44100, hop_length=512, x_axis='time', y_axis='mel')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title(f\"Sample {idx}\")\n    plt.axis('off')\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:49:14.387673Z","iopub.execute_input":"2024-11-17T00:49:14.387984Z","iopub.status.idle":"2024-11-17T00:49:16.023818Z","shell.execute_reply.started":"2024-11-17T00:49:14.387950Z","shell.execute_reply":"2024-11-17T00:49:16.022845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:49:16.024986Z","iopub.execute_input":"2024-11-17T00:49:16.025267Z","iopub.status.idle":"2024-11-17T00:49:16.646688Z","shell.execute_reply.started":"2024-11-17T00:49:16.025237Z","shell.execute_reply":"2024-11-17T00:49:16.645675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:49:16.647990Z","iopub.execute_input":"2024-11-17T00:49:16.648311Z","iopub.status.idle":"2024-11-17T00:49:16.654501Z","shell.execute_reply.started":"2024-11-17T00:49:16.648278Z","shell.execute_reply":"2024-11-17T00:49:16.653534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create a Label Encoder","metadata":{}},{"cell_type":"code","source":"label_encoder = LabelEncoder()\ntrain_df['label_encoded'] = label_encoder.fit_transform(train_df['label'])\ntest_df['label_encoded'] = label_encoder.transform(test_df['label'])","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:49:16.655839Z","iopub.execute_input":"2024-11-17T00:49:16.656773Z","iopub.status.idle":"2024-11-17T00:49:16.667851Z","shell.execute_reply.started":"2024-11-17T00:49:16.656727Z","shell.execute_reply":"2024-11-17T00:49:16.666706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train_df['label_encoded']\ny_test = test_df['label_encoded']\nprint(y_train)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:49:16.669253Z","iopub.execute_input":"2024-11-17T00:49:16.669638Z","iopub.status.idle":"2024-11-17T00:49:16.679761Z","shell.execute_reply.started":"2024-11-17T00:49:16.669596Z","shell.execute_reply":"2024-11-17T00:49:16.678687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label_encoder = LabelEncoder()\n#train['label_encoded'] = label_encoder.fit_transform(train['label'])\n\n#Encode target values\n#y_train = train.label_encoded\n#print(y_train)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-11-17T00:49:16.681049Z","iopub.execute_input":"2024-11-17T00:49:16.681388Z","iopub.status.idle":"2024-11-17T00:49:16.689609Z","shell.execute_reply.started":"2024-11-17T00:49:16.681356Z","shell.execute_reply":"2024-11-17T00:49:16.688843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create X_train Using the Log-Mel Spectrograms","metadata":{}},{"cell_type":"code","source":"X_train = np.expand_dims(np.array(train_df['log_mel'].tolist()), axis=-1)\nX_test = np.expand_dims(np.array(test_df['log_mel'].tolist()), axis=-1)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:49:16.691054Z","iopub.execute_input":"2024-11-17T00:49:16.691309Z","iopub.status.idle":"2024-11-17T00:49:17.356562Z","shell.execute_reply.started":"2024-11-17T00:49:16.691281Z","shell.execute_reply":"2024-11-17T00:49:17.355741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_train = np.expand_dims(logmel_features_array, axis=-1)\n#X_train_split, X_valid_split, y_train_split, y_valid_split = train_test_split(\n#    X_train,\n#    y_train,\n#    test_size=0.2,\n#    random_state=42,\n#    stratify=y_train\n#)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-11-17T00:49:17.357830Z","iopub.execute_input":"2024-11-17T00:49:17.358142Z","iopub.status.idle":"2024-11-17T00:49:17.362734Z","shell.execute_reply.started":"2024-11-17T00:49:17.358110Z","shell.execute_reply":"2024-11-17T00:49:17.361654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use image datagenerator to create train/val data and augmentation\ntrain_datagen = ImageDataGenerator(\n    rescale=1.0 / 255,\n    rotation_range=20,\n    width_shift_range=0.15,\n    height_shift_range=0.15,\n    shear_range=0.15,\n    zoom_range=0.15,\n    horizontal_flip=True\n)\nvalid_datagen = ImageDataGenerator(rescale=1.0 / 255)\n\n# Creating data generators in CNN with K-folds (reserved for use with other CNN experimentation)\n#train_generator = train_datagen.flow(\n#    X_train_split,\n#    y_train_split,\n#    batch_size=32,\n#    shuffle=True\n#)\n\n#validation_generator = valid_datagen.flow(\n#    X_valid_split,\n#    y_valid_split,\n#    batch_size=32,\n#    shuffle=False\n#)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:49:17.363856Z","iopub.execute_input":"2024-11-17T00:49:17.364141Z","iopub.status.idle":"2024-11-17T00:49:17.373513Z","shell.execute_reply.started":"2024-11-17T00:49:17.364111Z","shell.execute_reply":"2024-11-17T00:49:17.372592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create CNN","metadata":{}},{"cell_type":"code","source":"#np.random.seed(1)\n#tf.random.set_seed(1)\n#cnn = Sequential([\n#    Input(shape=(128, 128, 1)),\n#    Conv2D(32, (3, 3), padding='same'),\n#    Conv2D(32, (3, 3), padding='same'),\n#    BatchNormalization(),\n#    Activation('relu'),\n#    MaxPooling2D(2,2),\n#    Dropout(0.1),\n\n#    Conv2D(64, (3, 3), padding='same'),\n#    Conv2D(64, (3, 3), padding='same'),\n#    BatchNormalization(),\n#    Activation('relu'),\n#    MaxPooling2D(2,2),\n#    Dropout(0.1),\n\n#   Conv2D(128, (3, 3), padding='same'),\n#    Conv2D(128, (3, 3), padding='same'),\n#    BatchNormalization(),\n#    Activation('relu'),\n#    MaxPooling2D(2,2),\n#    Dropout(0.1),\n    \n#    Conv2D(264, (3, 3), padding='same'),\n#    Conv2D(264, (3, 3), padding='same'),\n#    BatchNormalization(),\n#    Activation('relu'),\n#    MaxPooling2D(2,2),\n#    Dropout(0.1),\n    \n#    Flatten(),\n    \n#    Dense(128, activation='relu'),\n#    Dropout(0.5),\n#    BatchNormalization(),\n#    Dense(len(label_encoder.classes_), activation='softmax')\n#])\n\n#cnn.summary()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-11-17T00:49:17.374759Z","iopub.execute_input":"2024-11-17T00:49:17.375051Z","iopub.status.idle":"2024-11-17T00:49:17.384592Z","shell.execute_reply.started":"2024-11-17T00:49:17.375020Z","shell.execute_reply":"2024-11-17T00:49:17.383758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#opt = tf.keras.optimizers.Adam(0.001)\n#cnn.compile(loss='sparse_categorical_crossentropy', optimizer=opt, metrics=['accuracy'])","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-11-17T00:49:17.385912Z","iopub.execute_input":"2024-11-17T00:49:17.386403Z","iopub.status.idle":"2024-11-17T00:49:17.397533Z","shell.execute_reply.started":"2024-11-17T00:49:17.386360Z","shell.execute_reply":"2024-11-17T00:49:17.396638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# detect and init the TPU\n#tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n\n# instantiate a distribution strategy\n#tf.tpu.experimental.initialize_tpu_system(tpu)\n#tpu_strategy = tf.distribute.TPUStrategy(tpu)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:49:17.403505Z","iopub.execute_input":"2024-11-17T00:49:17.403849Z","iopub.status.idle":"2024-11-17T00:49:17.408093Z","shell.execute_reply.started":"2024-11-17T00:49:17.403817Z","shell.execute_reply":"2024-11-17T00:49:17.407091Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_train_split = X_train_split.values if isinstance(X_train_split, pd.DataFrame) else X_train_split\n#y_train_split = y_train_split.values if isinstance(y_train_split, pd.Series) else y_train_split\n\nfold_losses = [] \ntraining_histories = []\n\n# Setup crossvalidation\nn_splits = 5\nkf = KFold(n_splits=n_splits, shuffle=True, random_state=1)\n\n#with tpu_strategy.scope():\nfold = 1\nfor train_index, val_index in kf.split(X_train, y_train):\n    print(f\"Training Fold {fold}/{n_splits}\")\n\n    X_train_fold, y_train_fold = X_train[train_index], y_train.iloc[train_index]\n    X_val_fold, y_val_fold = X_train[val_index], y_train.iloc[val_index]\n    \n    #X_train, y_train = X_train_split[train_index], y_train_split[train_index]\n    #X_val, y_val = X_train_split[val_index], y_train_split[val_index]\n\n    #X_train = np.squeeze(X_train, axis=-1) if X_train.ndim == 5 else X_train\n    #X_val = np.squeeze(X_val, axis=-1) if X_val.ndim == 5 else X_val\n\n    #Fold Generators\n    \n    train_fold_generator = train_datagen.flow(X_train_fold, y_train_fold, batch_size=64, shuffle=True)\n    val_fold_generator = valid_datagen.flow(X_val_fold, y_val_fold, batch_size=64, shuffle=False)\n\n    #Set input layer based upon Log-Mel Spectrogram size\n    input_layer = Input(shape=(128, 256, 1))\n\n    # Convert to RGB for compatability with Trabsfer Model\n    rgb_layer = tf.keras.layers.Concatenate(axis=-1)([input_layer, input_layer, input_layer])\n\n    #Setup ransfer learning\n    base_model = Xception(input_tensor=rgb_layer, include_top=False, weights='imagenet', input_shape=(128, 256, 3))\n    base_model.trainable = True\n    for layer in base_model.layers[:60]:\n        layer.trainable = False\n\n    # Build model\n    x = base_model.output\n#    x = Conv2D(64, (1, 1), padding='same', activation='relu', kernel_regularizer=l2(1e-4))(x)\n#    x = Dropout(0.2)(x)\n    x = GlobalAveragePooling2D()(x)\n    x = Dropout(0.3)(x)\n    x = Dense(128, kernel_regularizer=l2(1e-3))(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = Dropout(0.3)(x)\n    output = Dense(41, activation='softmax')(x)\n\n    cnn = Model(inputs=input_layer, outputs=output)\n\n    # Create checkpoints, early stopping, and lr reduction\n    checkpoint = ModelCheckpoint(f'best_model_fold_{fold}.keras', monitor='val_loss', save_best_only=True, verbose=1)\n    early_stopping = EarlyStopping(monitor='val_loss', patience=5, verbose=1)\n    reduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=3, min_lr=1e-6, verbose=1)\n\n    class_weights = compute_class_weight('balanced', classes=np.unique(y_train), y=y_train)\n    class_weight_dict = dict(enumerate(class_weights))\n\n    # Compile the Model\n    optimizer = AdamW(learning_rate=0.0001, weight_decay=1e-5)\n    cnn.compile(optimizer=optimizer, loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n    #Start Training\n    history = cnn.fit(\n        train_fold_generator,\n        validation_data=val_fold_generator,\n        epochs=50,  \n        verbose=1,\n        callbacks=[checkpoint, early_stopping, reduce_lr],\n        class_weight=class_weight_dict\n    )\n    training_histories.append(history)\n    final_val_loss = min(history.history['val_loss'])\n    fold_losses.append((fold, final_val_loss))\n    fold += 1\n\ncnn.summary()","metadata":{"execution":{"iopub.status.busy":"2024-11-17T00:49:17.409474Z","iopub.execute_input":"2024-11-17T00:49:17.409856Z","iopub.status.idle":"2024-11-17T01:57:23.383188Z","shell.execute_reply.started":"2024-11-17T00:49:17.409815Z","shell.execute_reply":"2024-11-17T01:57:23.382178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras_tuner as kt\n\ndef build_model(hp):\n    input_layer = Input(shape=(128, 256, 1))\n\n    rgb_layer = tf.keras.layers.Concatenate(axis=-1)([input_layer, input_layer, input_layer])\n\n    base_model = Xception(input_tensor=rgb_layer, include_top=False, weights='imagenet', input_shape=(128, 256, 3))\n    base_model.trainable = True\n    for layer in base_model.layers[:hp.Int('trainable_layers', min_value=40, max_value=70, step=10)]:\n        layer.trainable = False\n\n    x = base_model.output\n\n    dropout_rate = hp.Float('dropout_rate', min_value=0.2, max_value=0.7, step=0.1)\n    dense_units = hp.Int('dense_units', min_value=16, max_value=128, step=16)\n    l2_reg = hp.Float('l2_reg', min_value=1e-4, max_value=1e-2, sampling='log')\n\n    x = GlobalAveragePooling2D()(x)\n    x = Dropout(dropout_rate)(x)\n    x = Dense(dense_units, kernel_regularizer=tf.keras.regularizers.l2(l2_reg))(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = Dropout(dropout_rate)(x)\n    output = Dense(41, activation='softmax')(x)\n\n    model = Model(inputs=input_layer, outputs=output)\n\n    learning_rate = hp.Float('learning_rate', min_value=1e-5, max_value=1e-3, sampling='log')\n    optimizer = tf.keras.optimizers.Adam(learning_rate=learning_rate)\n    model.compile(optimizer=optimizer, loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n    return model\n\ntuner = kt.Hyperband(\n    build_model,\n    objective='val_accuracy',\n    max_epochs=30,\n    factor=3,\n    directory='kt_dir',\n    project_name='audio_classification'\n)\n\nn_splits = 5\nkf = KFold(n_splits=n_splits, shuffle=True, random_state=1)\n\nfold = 1\nfold_losses = []\ntraining_histories = []\n\nfor train_index, val_index in kf.split(X_train, y_train):\n    print(f\"Training Fold {fold}/{n_splits}\")\n\n    X_train_fold, y_train_fold = X_train[train_index], y_train.iloc[train_index]\n    X_val_fold, y_val_fold = X_train[val_index], y_train.iloc[val_index]\n\n    train_fold_generator = train_datagen.flow(X_train_fold, y_train_fold, batch_size=64, shuffle=True)\n    val_fold_generator = valid_datagen.flow(X_val_fold, y_val_fold, batch_size=64, shuffle=False)\n\n    tuner.search(\n        train_fold_generator,\n        validation_data=val_fold_generator,\n        epochs=30,\n        callbacks=[EarlyStopping(monitor='val_loss', patience=5)]\n    )\n\n    best_hps = tuner.get_best_hyperparameters(num_trials=1)[0]\n\n    model = tuner.hypermodel.build(best_hps)\n\n    checkpoint = ModelCheckpoint(f'best_model_fold_{fold}.keras', monitor='val_loss', save_best_only=True, verbose=1)\n    early_stopping = EarlyStopping(monitor='val_loss', patience=5, verbose=1)\n    reduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=3, min_lr=1e-6, verbose=1)\n\n    class_weights = compute_class_weight('balanced', classes=np.unique(y_train), y=y_train)\n    class_weight_dict = dict(enumerate(class_weights))\n\n    history = model.fit(\n        train_fold_generator,\n        validation_data=val_fold_generator,\n        epochs=50,\n        verbose=1,\n        callbacks=[checkpoint, early_stopping, reduce_lr],\n        class_weight=class_weight_dict\n    )\n\n    training_histories.append(history)\n    final_val_loss = min(history.history['val_loss'])\n    fold_losses.append((fold, final_val_loss))\n    fold += 1\n\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-17T02:15:34.282629Z","iopub.execute_input":"2024-11-17T02:15:34.283582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluate the Results","metadata":{}},{"cell_type":"code","source":"best_fold, best_val_loss = min(fold_losses, key=lambda x: x[1])\nprint(f'The best fold is Fold {best_fold} with validation loss {best_val_loss}')\n\nbest_model = tf.keras.models.load_model(f'best_model_fold_{best_fold}.keras')\n\nbest_model.save('final_best_model.keras')","metadata":{"execution":{"iopub.status.busy":"2024-11-17T01:57:23.384695Z","iopub.execute_input":"2024-11-17T01:57:23.385456Z","iopub.status.idle":"2024-11-17T01:57:33.417148Z","shell.execute_reply.started":"2024-11-17T01:57:23.385408Z","shell.execute_reply":"2024-11-17T01:57:33.416214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_training_history(history, fold_number):\n    training_loss = history.history['loss']\n    validation_loss = history.history['val_loss']\n    training_accuracy = history.history['accuracy']\n    validation_accuracy = history.history['val_accuracy']\n\n    plt.figure(figsize=(14, 5))\n    plt.subplot(1, 2, 1)\n    plt.plot(training_loss, label='Training Loss')\n    plt.plot(validation_loss, label='Validation Loss')\n    plt.title(f'Fold {fold_number} - Loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend()\n\n    plt.subplot(1, 2, 2)\n    plt.plot(training_accuracy, label='Training Accuracy')\n    plt.plot(validation_accuracy, label='Validation Accuracy')\n    plt.title(f'Fold {fold_number} - Accuracy')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.legend()\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-17T01:57:33.418297Z","iopub.execute_input":"2024-11-17T01:57:33.418612Z","iopub.status.idle":"2024-11-17T01:57:33.426888Z","shell.execute_reply.started":"2024-11-17T01:57:33.418577Z","shell.execute_reply":"2024-11-17T01:57:33.425861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, history in enumerate(training_histories):\n    plot_training_history(history, i + 1)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T01:57:33.427974Z","iopub.execute_input":"2024-11-17T01:57:33.428232Z","iopub.status.idle":"2024-11-17T01:57:35.746151Z","shell.execute_reply.started":"2024-11-17T01:57:33.428203Z","shell.execute_reply":"2024-11-17T01:57:35.745214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#%%time \n#checkpoint = ModelCheckpoint(f'best_model_fold_{fold}.keras', monitor='val_loss', save_best_only=True, verbose=1)\n#early_stopping = EarlyStopping(monitor='val_loss', patience=5, verbose=1)\n\n#lr_scheduler = ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=3, verbose=1)\n\n\n#h1 = cnn.fit(\n#    train_fold_generator,\n#    validation_data=val_fold_generator,\n#    epochs=100,\n#    callbacks=[checkpoint, early_stopping]\n#)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-11-17T01:57:35.747527Z","iopub.execute_input":"2024-11-17T01:57:35.747915Z","iopub.status.idle":"2024-11-17T01:57:35.752751Z","shell.execute_reply.started":"2024-11-17T01:57:35.747872Z","shell.execute_reply":"2024-11-17T01:57:35.751761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = history.history\nprint(history.keys())","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-11-17T01:57:35.754317Z","iopub.execute_input":"2024-11-17T01:57:35.754644Z","iopub.status.idle":"2024-11-17T01:57:35.763054Z","shell.execute_reply.started":"2024-11-17T01:57:35.754612Z","shell.execute_reply":"2024-11-17T01:57:35.762129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epoch_range = range(1, len(history['loss'])+1)\n\nplt.figure(figsize=[14,4])\nplt.subplot(1,2,1)\nplt.plot(epoch_range, history['loss'], label='Training')\nplt.plot(epoch_range, history['val_loss'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\nplt.legend()\nplt.subplot(1,2,2)\nplt.plot(epoch_range, history['accuracy'], label='Training')\nplt.plot(epoch_range, history['val_accuracy'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Accuracy'); plt.title('Accuracy')\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-11-17T01:57:35.764178Z","iopub.execute_input":"2024-11-17T01:57:35.764560Z","iopub.status.idle":"2024-11-17T01:57:36.373841Z","shell.execute_reply.started":"2024-11-17T01:57:35.764528Z","shell.execute_reply":"2024-11-17T01:57:36.372917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#cnn.optimizer.learning_rate = 0.0001","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-11-17T01:57:36.375084Z","iopub.execute_input":"2024-11-17T01:57:36.375394Z","iopub.status.idle":"2024-11-17T01:57:36.379548Z","shell.execute_reply.started":"2024-11-17T01:57:36.375360Z","shell.execute_reply":"2024-11-17T01:57:36.378542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#%%time \n\n#h2 = cnn.fit(\n#    train_fold_generator,\n#    validation_data=val_fold_generator,\n#    epochs=100,\n#    callbacks=[checkpoint, early_stopping]\n#)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-11-17T01:57:36.380994Z","iopub.execute_input":"2024-11-17T01:57:36.381662Z","iopub.status.idle":"2024-11-17T01:57:36.387491Z","shell.execute_reply.started":"2024-11-17T01:57:36.381620Z","shell.execute_reply":"2024-11-17T01:57:36.386675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for k in history.keys():\n#    history[k] += h2.history[k]\n\n#    epoch_range = range(1, len(history['loss'])+1)\n\n#plt.figure(figsize=[14,4])\n#plt.subplot(1,2,1)\n#plt.plot(epoch_range, history['loss'], label='Training')\n#plt.plot(epoch_range, history['val_loss'], label='Validation')\n#plt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\n#plt.legend()\n#plt.subplot(1,2,2)\n#plt.plot(epoch_range, history['accuracy'], label='Training')\n#plt.plot(epoch_range, history['val_accuracy'], label='Validation')\n#plt.xlabel('Epoch'); plt.ylabel('Accuracy'); plt.title('Accuracy')\n#plt.legend()\n#plt.tight_layout()\n#plt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-11-17T01:57:36.388798Z","iopub.execute_input":"2024-11-17T01:57:36.389881Z","iopub.status.idle":"2024-11-17T01:57:36.399589Z","shell.execute_reply.started":"2024-11-17T01:57:36.389815Z","shell.execute_reply":"2024-11-17T01:57:36.398714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_datagen = ImageDataGenerator(rescale=1.0 / 255)\ntest_generator = test_datagen.flow(X_test, y_test, batch_size=64, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T01:57:36.400775Z","iopub.execute_input":"2024-11-17T01:57:36.401999Z","iopub.status.idle":"2024-11-17T01:57:36.408780Z","shell.execute_reply.started":"2024-11-17T01:57:36.401926Z","shell.execute_reply":"2024-11-17T01:57:36.407968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model = tf.keras.models.load_model('final_best_model.keras')\nfinal_loss, final_accuracy = final_model.evaluate(test_generator)\nprint(f\"Final Validation Loss: {final_loss}, Final Validation Accuracy: {final_accuracy}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-17T01:57:36.409789Z","iopub.execute_input":"2024-11-17T01:57:36.410099Z","iopub.status.idle":"2024-11-17T01:57:53.858117Z","shell.execute_reply.started":"2024-11-17T01:57:36.410060Z","shell.execute_reply":"2024-11-17T01:57:53.857191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_predictions = cnn.predict(test_generator)\n\nprint(val_predictions.shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T01:57:53.859288Z","iopub.execute_input":"2024-11-17T01:57:53.859602Z","iopub.status.idle":"2024-11-17T01:58:00.493032Z","shell.execute_reply.started":"2024-11-17T01:57:53.859569Z","shell.execute_reply":"2024-11-17T01:58:00.492076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(val_predictions[0, :].round(2))","metadata":{"execution":{"iopub.status.busy":"2024-11-17T01:58:00.494267Z","iopub.execute_input":"2024-11-17T01:58:00.494580Z","iopub.status.idle":"2024-11-17T01:58:00.500922Z","shell.execute_reply.started":"2024-11-17T01:58:00.494540Z","shell.execute_reply":"2024-11-17T01:58:00.499672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top1 = train.copy()\n\ntop1.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-17T01:58:00.502537Z","iopub.execute_input":"2024-11-17T01:58:00.503075Z","iopub.status.idle":"2024-11-17T01:58:00.524996Z","shell.execute_reply.started":"2024-11-17T01:58:00.503030Z","shell.execute_reply":"2024-11-17T01:58:00.523768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top1 = pd.DataFrame(columns=['label'], index=range(len(X_val)))\n\nlabels = label_encoder.classes_\n# Loop through each validation prediction and get the top 1 prediction\nN = len(X_val)\nfor i in range(N):\n    p = val_predictions[i, :]\n    idx = np.argmax(p)\n    top1.at[i, 'label'] = labels[idx]\n\nprint(top1.head())","metadata":{"execution":{"iopub.status.busy":"2024-11-17T01:58:00.526341Z","iopub.execute_input":"2024-11-17T01:58:00.526714Z","iopub.status.idle":"2024-11-17T01:58:00.823157Z","shell.execute_reply.started":"2024-11-17T01:58:00.526659Z","shell.execute_reply":"2024-11-17T01:58:00.821757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top1.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T01:58:00.824475Z","iopub.status.idle":"2024-11-17T01:58:00.824848Z","shell.execute_reply.started":"2024-11-17T01:58:00.824659Z","shell.execute_reply":"2024-11-17T01:58:00.824676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save Models","metadata":{}},{"cell_type":"code","source":"best_model.save('freesound_model_v08.h5')\n#cnn.save('freesound_model_v02.h5')\n#pickle.dump(history, open(f'freesound_history_v02.pkl', 'wb'))","metadata":{"execution":{"iopub.status.busy":"2024-11-17T01:58:00.825975Z","iopub.status.idle":"2024-11-17T01:58:00.826475Z","shell.execute_reply.started":"2024-11-17T01:58:00.826209Z","shell.execute_reply":"2024-11-17T01:58:00.826235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"joblib.dump(test_audio_pipeline, 'audio_preprocessing_pipeline_v8.pkl')\njoblib.dump(label_encoder, 'label_encoder_v8.pkl')","metadata":{"execution":{"iopub.status.busy":"2024-11-17T01:58:00.827747Z","iopub.status.idle":"2024-11-17T01:58:00.828262Z","shell.execute_reply.started":"2024-11-17T01:58:00.827995Z","shell.execute_reply":"2024-11-17T01:58:00.828026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}