{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":8900,"databundleVersionId":862232,"sourceType":"competition"},{"sourceId":9936501,"sourceType":"datasetVersion","datasetId":6108541},{"sourceId":9957997,"sourceType":"datasetVersion","datasetId":6109117},{"sourceId":9979171,"sourceType":"datasetVersion","datasetId":6140362},{"sourceId":9995454,"sourceType":"datasetVersion","datasetId":6152006},{"sourceId":9995463,"sourceType":"datasetVersion","datasetId":6152012},{"sourceId":10025680,"sourceType":"datasetVersion","datasetId":6173973},{"sourceId":10043496,"sourceType":"datasetVersion","datasetId":6186901},{"sourceId":10052765,"sourceType":"datasetVersion","datasetId":6194156},{"sourceId":10053631,"sourceType":"datasetVersion","datasetId":6194737},{"sourceId":10054127,"sourceType":"datasetVersion","datasetId":6195136},{"sourceId":10061714,"sourceType":"datasetVersion","datasetId":6200615},{"sourceId":10071165,"sourceType":"datasetVersion","datasetId":6207520},{"sourceId":10071452,"sourceType":"datasetVersion","datasetId":6207731},{"sourceId":10089688,"sourceType":"datasetVersion","datasetId":6221425},{"sourceId":10111129,"sourceType":"datasetVersion","datasetId":6237841},{"sourceId":10114909,"sourceType":"datasetVersion","datasetId":6240643},{"sourceId":10132904,"sourceType":"datasetVersion","datasetId":6253761},{"sourceId":10165066,"sourceType":"datasetVersion","datasetId":6268267},{"sourceId":10195022,"sourceType":"datasetVersion","datasetId":6299437}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Training with Convoluational Neural Network\n- During the training we will be using a Convolutional Neural Network combined with dense layers to use the extracted features from the data preparation steps. Given the time sequence of the extracted featrues, a convolutional neural network is a good choice for this problem set","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"## Import Packages","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport IPython\nimport IPython.display as ipd\nimport librosa\nimport librosa.display\nimport pickle\nimport joblib\nimport random\n\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\nfrom sklearn.metrics import classification_report, accuracy_score, f1_score, precision_score, recall_score\n\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.optimizers import AdamW\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras.layers import Conv1D, Dense, Dropout, Flatten, Input, Reshape, BatchNormalization, MaxPooling1D, Activation\nimport keras_tuner\nfrom keras_tuner.tuners import RandomSearch, BayesianOptimization\nfrom imblearn.over_sampling import SMOTE\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.losses import SparseCategoricalCrossentropy\nimport tensorflow as tf","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Import train/test Dataframes\n- Import the transformed training and test data from the data preparation steps","metadata":{}},{"cell_type":"code","source":"X_train = np.load('/kaggle/input/freesound-x-train-test-vmh-no-aug-final/X_train_vmh_mv_final.npy')\nX_test = np.load('/kaggle/input/freesound-x-train-test-vmh-no-aug-final/X_test_vmh_mv_final.npy')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load Original Train/Test Dataframe for y_train/test Creation\n- Import the original train/test data in a dataframe in order to create the y_labels","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_pickle('/kaggle/input/freesound-x-train-test-vmh-no-aug-final/train_mv_final.pkl')\ntest_df = pd.read_pickle('/kaggle/input/freesound-x-train-test-vmh-no-aug-final/test_mv_final.pkl')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create Label Encoder","metadata":{}},{"cell_type":"code","source":"# Encode the y labels\nlabel_encoder = LabelEncoder()\ntrain_df['label_encoded'] = label_encoder.fit_transform(train_df['label'])\ntest_df['label_encoded'] = label_encoder.transform(test_df['label'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create label mapping to see how each label is assigned - this will also be used in the evaluation section\nlabel_mapping_df = pd.DataFrame({\n    'Label': label_encoder.classes_,\n    'Encoded_Value': label_encoder.transform(label_encoder.classes_)\n})\n\nprint(label_mapping_df)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train = train_df['label_encoded']\ny_test = test_df['label_encoded']\nprint(y_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### SMOTE - Synthetic Minority Over-sampling Technique\n- We will use this technique to create sysnthetic examples from the under represented classes to create a balanced dataset for training","metadata":{}},{"cell_type":"code","source":"# Apply SMOTE to account for class imbalance in the training data\nsmote = SMOTE(random_state=42)\nX_train_smote, y_train_smote = smote.fit_resample(X_train, y_train) \n\nprint(f\"Original X_train shape: {X_train.shape}\") \nprint(f\"New X_train shape after SMOTE: {X_train_smote.shape}\")\nprint(f\"Original y_train distribution:\\n{pd.Series(y_train).value_counts()}\")\nprint()\nprint(f\"New y_train distribution after SMOTE:\\n{pd.Series(y_train_smote).value_counts()}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_smote.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reshape data for input into model\nX_train_smote = X_train_smote.reshape(-1, 1704, 1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_smote.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create Model\n- For the training we will use a comvoluational neural netwrok combined with dense layers. The Conv1D filters will extract the spacial features from our input features features\n- The dense layers will be used to process high-level features\n- RandomSearch is used to explore the hyperparameter space, and 8-fold cross validation is used to ensure a robust evaluation on different subsets of the data for the model\n\n* After experimenting with several different model architectures to include, transfer learning, XGBoost, Random Forest, and ensemble models, this approach has achieved the best results","metadata":{}},{"cell_type":"code","source":"# Setup 8-fold cross validation\nn_splits = 8\nkf = KFold(n_splits=n_splits, shuffle=True, random_state=1)\n\n# Define the CNN\ndef build_model(hp):\n    input_layer = Input(shape=(1704, 1))\n\n    l2_reg_1 = hp.Choice('l2_reg_1', values=[1e-7, 1e-6, 1e-5, 1e-4])  # Regularization for first layer\n    l2_reg_2 = hp.Choice('l2_reg_2', values=[1e-7, 1e-6, 1e-5, 1e-4])  # Regularization for second layer\n    dense_l2_reg = hp.Choice('dense_l2_reg', values=[1e-6, 1e-5, 1e-4])\n    \n    filters_1 = hp.Int('filters_1', min_value=16, max_value=128, step=16)\n    kernel_size_1 = hp.Choice('kernel_size_1', values=[3, 5])\n    x = Conv1D(filters_1, kernel_size=kernel_size_1, strides=1, padding='same', activation='relu', kernel_regularizer=tf.keras.regularizers.l2(l2_reg_1))(input_layer)\n    x = BatchNormalization()(x)\n    x = MaxPooling1D(pool_size=2)(x)\n\n    filters_2 = hp.Int('filters_2', min_value=32, max_value=256, step=32)\n    kernel_size_2 = hp.Choice('kernel_size_2', values=[3, 5])\n    x = Conv1D(filters_2, kernel_size=kernel_size_2, strides=1, padding='same', activation='relu', kernel_regularizer=tf.keras.regularizers.l2(l2_reg_2))(x)\n    x = BatchNormalization()(x)\n    x = MaxPooling1D(pool_size=2)(x)\n\n    x = Flatten()(x)\n    \n    units_1 = hp.Int('units_1', min_value=128, max_value=512, step=128)\n    x = Dense(units_1, kernel_regularizer=tf.keras.regularizers.l2(dense_l2_reg))(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    dropout_1 = hp.Float('dropout_1', min_value=0.3, max_value=0.7, step=0.1)\n    x = Dropout(dropout_1)(x)\n\n    units_2 = hp.Int('units_2', min_value=32, max_value=256, step=32)\n    x = Dense(units_2, kernel_regularizer=tf.keras.regularizers.l2(dense_l2_reg))(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    dropout_2 = hp.Float('dropout_2', min_value=0.3, max_value=0.7, step=0.1)\n    x = Dropout(dropout_2)(x)\n\n    output = Dense(41, activation='softmax')(x)\n\n    model = Model(inputs=input_layer, outputs=output)\n\n    learning_rate = hp.Choice('learning_rate', values=[1e-2, 1e-3, 1e-4, 1e-5, 1e-6])\n    optimizer = Adam(learning_rate=learning_rate, clipnorm=1.0)\n    model.compile(optimizer=optimizer, loss=tf.keras.losses.SparseCategoricalCrossentropy(), metrics=['accuracy'])\n    \n    return model\n\n# Use RandomSearch for hyperparameter tuning\ntuner = RandomSearch(\n    build_model,\n    objective='val_loss',\n    max_trials=30, \n    executions_per_trial=1,\n    directory='hyperparam_tuning',\n    project_name='vggish_tuning'\n)\n\n# Setup early stopping \nearly_stopping = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True, verbose=1)\n\nclass_weights = compute_class_weight('balanced', classes=np.unique(y_train_smote), y=y_train_smote)\nclass_weight_dict = dict(enumerate(class_weights))\n\n\ntuner.search(\n    X_train_smote, y_train_smote,\n    validation_data=(X_test, y_test),\n    epochs=30,\n    batch_size=64,\n    callbacks=[early_stopping],\n    class_weight=class_weight_dict,\n    verbose=1\n)\n\n\nbest_hps = tuner.get_best_hyperparameters(num_trials=1)[0]\nprint(f\"Best hyperparameters: {best_hps.values}\")\n\nbest_model = tuner.hypermodel.build(best_hps)\n\n\nfold_losses = []\ntraining_histories = []\nfold = 1\n\n# Setup training after hyperparameter tuning\nfor train_index, val_index in kf.split(X_train_smote, y_train_smote):\n    print(f\"Training Fold {fold}/{n_splits}\")\n    \n    X_train_fold, y_train_fold = X_train_smote[train_index], y_train_smote.iloc[train_index].values\n    X_val_fold, y_val_fold = X_test, y_test \n\n    class_weights = compute_class_weight('balanced', classes=np.unique(y_train_fold), y=y_train_fold)\n    class_weight_dict = dict(enumerate(class_weights))\n\n    best_model = tuner.hypermodel.build(best_hps)\n\n    early_stopping = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True, verbose=1)\n    checkpoint = ModelCheckpoint(f'best_model_vggish_fold_{fold}.keras', monitor='val_loss', save_best_only=True, verbose=1)\n    reduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=3, min_lr=1e-6, verbose=1)\n\n    \n    history = best_model.fit(\n        X_train_fold, y_train_fold,\n        validation_data=(X_val_fold, y_val_fold),\n        epochs=50,\n        batch_size=64,\n        verbose=1,\n        callbacks=[checkpoint, early_stopping, reduce_lr],\n        class_weight=class_weight_dict\n    )\n    \n    training_histories.append(history)\n    final_val_loss = min(history.history['val_loss'])\n    fold_losses.append((fold, final_val_loss))\n    fold += 1\n\nprint(\"Cross-validation fold losses:\")\nfor fold_num, loss in fold_losses:\n    print(f\"Fold {fold_num}: {loss:.4f}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Evaluate the Results","metadata":{}},{"cell_type":"code","source":"train_acc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\ntrain_loss = history.history['loss']\nval_loss = history.history['val_loss']\n\n# Plot accuracy\nplt.figure(figsize=(10,4))\nplt.subplot(1,2,1)\nplt.plot(train_acc, label='Train Accuracy')\nplt.plot(val_acc, label='Val Accuracy')\nplt.title('Accuracy over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\n\n# Plot loss\nplt.subplot(1,2,2)\nplt.plot(train_loss, label='Train Loss')\nplt.plot(val_loss, label='Val Loss')\nplt.title('Loss over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Predict on X_test and transform y_pred to select the highest value for each sample","metadata":{}},{"cell_type":"code","source":"y_prob = best_model.predict(X_test)\ny_pred = np.argmax(y_prob, axis=1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Evaluate at sample 42","metadata":{}},{"cell_type":"code","source":"print(y_prob[42, :].round(2))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(y_pred[42])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Manually verify classification\n- Compare the prediction for sample 42 to the actual label","metadata":{}},{"cell_type":"code","source":"first_test_index = test_df.index[42]\nfirst_test_file = test_df.loc[first_test_index, 'fname']\n\naudio_dir = '/kaggle/input/freesound-audio-tagging/audio_train' \naudio_path = os.path.join(audio_dir, first_test_file)\nsample, sr = librosa.load(audio_path, sr=None)\n\nprint(f\"Playing audio: {first_test_file}\")\nprint(f'Predicted label: {label_encoder.inverse_transform([y_pred[42]])[0]}')\nprint(f'Actual label:    {label_encoder.inverse_transform([y_test.iloc[42]])[0]}')\nipd.display(ipd.Audio(sample, rate=sr))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Confusion Matrix","metadata":{}},{"cell_type":"code","source":"cm = confusion_matrix(test_df['label_encoded'], y_pred)\n\nclass_names = label_encoder.classes_\n\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=class_names)\nfig, ax = plt.subplots(figsize=(10,10))\ndisp.plot(ax=ax, cmap=plt.cm.Blues, xticks_rotation='vertical')\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Classificaton Report","metadata":{}},{"cell_type":"code","source":"print(\"Classification Report:\")\nprint(classification_report(y_test, y_pred, target_names=label_encoder.classes_))\n\nacc = accuracy_score(y_test, y_pred)\nprecision = precision_score(y_test, y_pred, average='weighted')  # or 'macro'/'micro'\nrecall = recall_score(y_test, y_pred, average='weighted')\nf1 = f1_score(y_test, y_pred, average='weighted')\n\nprint(f\"Accuracy: {acc:.4f}\")\nprint(f\"Precision (weighted): {precision:.4f}\")\nprint(f\"Recall (weighted): {recall:.4f}\")\nprint(f\"F1-score (weighted): {f1:.4f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save Encoders\n- Save the encoder to use in the Submission process","metadata":{}},{"cell_type":"code","source":"joblib.dump(label_encoder, 'label_encoder_v70.pkl')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Save Best Model\n- We will save the best model for reference, but we will be using the best model from each fold to perform an ensemble prediction during the submission phase\n- The best folds are saved manually from the kaggle working output following running the notebook","metadata":{}},{"cell_type":"code","source":"best_model.save('final_best_model_70.keras')","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}