{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pip install python_speech_features","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pip install librosa","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pip install soundfile","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport librosa\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nfrom python_speech_features import mfcc\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix\nfrom keras.layers import Conv2D, MaxPool2D, Flatten, Dense, Dropout\nfrom keras.models import Sequential","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/freesound-audio-tagging/train.csv')\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = df[df['label'].isin(['Cello','Saxophone','Acoustic_guitar','Double_bass', 'Clarinet'])]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path = '../input/freesound-audio-tagging/audio_train/'\naudio_data = list()\nfor i in tqdm(range(df.shape[0])):\n    audio_data.append(librosa.load(path+df['fname'].iloc[i]))\naudio_data = np.array(audio_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['audio_waves'] = audio_data[:,0]\ndf['samplerate'] = audio_data[:,1]\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bit_lengths = list()\nfor i in range(df.shape[0]):\n    bit_lengths.append(len(df['audio_waves'].iloc[i]))\nbit_lengths = np.array(bit_lengths)\ndf['bit_lengths'] = bit_lengths\ndf['seconds_lengths'] = df['bit_lengths']/df['samplerate']\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = df[df['seconds_lengths'] >= 2.0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"min_bits = np.min(df['bit_lengths'])\nprint(min_bits)\nmin_seconds = np.min(df['seconds_lengths'])\nprint(min_seconds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with open('audio_df.pickle', 'wb') as f:\n    pickle.dump(df, f)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with open('audio_df.pickle', 'rb') as f:\n    df = pickle.load(f)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"num_samples = 6000\ngenerated_audio_waves = list()\ngenerated_audio_labels = list()\nfor i in tqdm(range(num_samples)):\n    try:\n        chosen_file = np.random.choice(df['fname'].values)\n        chosen_initial = np.random.choice(np.arange(0,df[df['fname']==chosen_file]['bit_lengths'].values-min_bits))\n        generated_audio_waves.append(df[df['fname']==chosen_file]['audio_waves'].values[0][chosen_initial:chosen_initial+min_bits])\n        \n        generated_audio_labels.append(df[df['fname']==chosen_file]['label'].values)\n    except ValueError:\n        continue\ngenerated_audio_waves = np.array(generated_audio_waves)\ngenerated_audio_labels = np.array(generated_audio_labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"mfcc_features = list()\nfor i in tqdm(range(len(generated_audio_waves))):\n    mfcc_features.append(mfcc(generated_audio_waves[i]))\nmfcc_features = np.array(mfcc_features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(generated_audio_waves.shape)\nprint(mfcc_features.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(12,2))\nplt.plot(generated_audio_waves[30])\nplt.title(generated_audio_labels[30])\nplt.show()\nplt.figure(figsize=(12, 2))\nplt.imshow(mfcc_features[30].T, cmap='hot')\nplt.title(generated_audio_labels[30])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"label_encoder = LabelEncoder()\nlabel_encoded = label_encoder.fit_transform(generated_audio_labels)\nprint(label_encoded)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"label_encoded = label_encoded[:, np.newaxis]\nlabel_encoded","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"one_hot_encoder = OneHotEncoder(sparse=False)\none_hot_encoded = one_hot_encoder.fit_transform(label_encoded)\none_hot_encoded","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Model Train CNN"},{"metadata":{"trusted":true},"cell_type":"code","source":"X = mfcc_features\ny = one_hot_encoded\nX = (X-X.min())/(X.max()-X.min())\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"input_shape = (X_train.shape[1], X_train.shape[2], 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train = X_train.reshape(X_train.shape[0], X_train.shape[1], X_train.shape[2], 1)\nprint(X_train.shape)\nX_test = X_test.reshape(X_test.shape[0], X_test.shape[1], X_test.shape[2], 1)\nprint(X_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\nmodel.add(Conv2D(16, (3, 3), activation='relu', strides=(1, 1), \n    padding='same', input_shape=input_shape))\nmodel.add(Conv2D(32, (3, 3), activation='relu', strides=(1, 1), \n    padding='same'))\nmodel.add(MaxPool2D((2, 2)))\nmodel.add(Dropout(0.5))\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(5, activation='softmax'))\n\nmodel.compile(loss='categorical_crossentropy', \n     optimizer='adam',\n     metrics=['acc'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit(X_train, y_train, epochs=30, validation_data=(X_test, y_test))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(8,8))\nplt.title('Loss Value')\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.legend(['loss', 'val_loss'])\nprint('loss:', history.history['loss'][-1])\nprint('val_loss:', history.history['val_loss'][-1])\nplt.show()\nplt.figure(figsize=(8,8))\nplt.title('Accuracy')\nplt.plot(history.history['acc'])\nplt.plot(history.history['val_acc'])\nplt.legend(['acc', 'val_acc'])\nprint('acc:', history.history['acc'][-1])\nprint('val_acc:', history.history['val_acc'][-1])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions = model.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"array([3.0511815e-06, 2.2099694e-05, 9.9997330e-01, 1.0746862e-12,1.5381156e-06], dtype=float32)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions = np.argmax(predictions, axis=1)\ny_test = one_hot_encoder.inverse_transform(y_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cm = confusion_matrix(y_test, predictions)\nplt.figure(figsize=(8,8))\nsns.heatmap(cm, annot=True, xticklabels=label_encoder.classes_, yticklabels=label_encoder.classes_, fmt='d', cmap=plt.cm.Blues, cbar=False)\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}