{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv_file=pd.read_csv('../input/train.csv')\ntrain_csv_file","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(os.listdir(\"../input/audio_train/audio_train\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"filename_label={\"../input/audio_train/audio_train/\"+k:v for k,v in zip(train_csv_file.fname.values, train_csv_file.label.values)}","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"filename_label will store the file names as the keys and  the lable as the vlaue"},{"metadata":{"trusted":true},"cell_type":"code","source":"filename_label['../input/audio_train/audio_train/fff81f55.wav']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import librosa","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\ndef audio_norm(data):\n\n    max_data = np.max(data)\n    min_data = np.min(data)\n    data = (data-min_data)/(max_data-min_data)\n    return data\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"input_length = 16000*2\ndef load_audio_file(file_path, input_length=input_length):\n    data = librosa.core.load(file_path, sr=16000)[0] #, sr=16000\n    if len(data)>input_length:\n        \n        \n        max_offset = len(data)-input_length\n        \n        offset = np.random.randint(max_offset)\n        \n        data = data[offset:(input_length+offset)]\n        \n        \n    else:\n        \n        if input_length > len(data):\n            max_offset = input_length - len(data)\n\n            offset = np.random.randint(max_offset)\n        else:\n            offset = 0\n        \n        \n        data = np.pad(data, (offset, input_length - len(data) - offset), \"constant\")\n        \n        \n    data = audio_norm(data)\n    return data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import glob\ntrain_files = glob.glob(\"../input/audio_train/audio_train/*.wav\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_files","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_base = load_audio_file(train_files[9])\nfig = plt.figure(figsize=(14, 8))\nplt.title('Raw wave : %s ' % (filename_label[train_files[0]]))\nplt.ylabel('Amplitude')\nplt.plot(np.linspace(0, 1, input_length), data_base)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"list_labels = sorted(list(set(train_csv_file.label.values)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"list_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"label_to_int = {k:v for v,k in enumerate(list_labels)}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"label_to_int","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"int_to_label = {v:k for k,v in label_to_int.items()}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"int_to_label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"file_to_int = {k:label_to_int[v] for k,v in filename_label.items()}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"file_to_int","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n_class=41","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Conv1D,MaxPool1D,Dropout,GlobalMaxPool1D,Dense","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model=Sequential()\nmodel.add(Conv1D(16,kernel_size=9,activation='relu',input_shape=(input_length,1)))\nmodel.add(Conv1D(16,kernel_size=9,activation='relu'))\nmodel.add(MaxPool1D(pool_size=16))\nmodel.add(Dropout(rate=0.1))\nmodel.add(Conv1D(32,kernel_size=3,activation='relu'))\nmodel.add(Conv1D(32,kernel_size=3,activation='relu'))\nmodel.add(MaxPool1D(pool_size=4))\nmodel.add(Dropout(rate=0.1))\nmodel.add(Conv1D(32,kernel_size=3,activation='relu'))\nmodel.add(Conv1D(32,kernel_size=3,activation='relu'))\nmodel.add(MaxPool1D(pool_size=4))\nmodel.add(Dropout(rate=0.1))\nmodel.add(Conv1D(64,kernel_size=3,activation='relu'))\nmodel.add(Conv1D(64,kernel_size=3,activation='relu'))\nmodel.add(GlobalMaxPool1D())\nmodel.add(Dropout(rate=0.2))\nmodel.add(Dense(128,activation='relu'))\nmodel.add(Dense(256,activation='relu'))\nmodel.add(Dense(n_class,activation='softmax'))\n          \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(optimizer='adam',loss='sparse_categorical_crossentropy', metrics=['acc'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":""},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size=32","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from random import shuffle\nfrom sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def chunker(seq, size):\n    return (seq[pos:pos + size] for pos in range(0, len(seq), size))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def train_generator(list_files, batch_size=batch_size):\n    while True:\n        shuffle(list_files)\n        for batch_files in chunker(list_files, size=batch_size):\n            batch_data = [load_audio_file(fpath) for fpath in batch_files]\n            batch_data = np.array(batch_data)[:,:,np.newaxis]\n            batch_labels = [file_to_int[fpath] for fpath in batch_files]\n            batch_labels = np.array(batch_labels)\n            \n            yield batch_data, batch_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_files, val_files = train_test_split(train_files, test_size=0.1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(tr_files)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.fit_generator(train_generator(tr_files), steps_per_epoch=len(tr_files)//batch_size, epochs=2,\n                    validation_data=train_generator(val_files), validation_steps=len(val_files)//batch_size,use_multiprocessing=True, workers=8, max_queue_size=20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save_weights(\"audio_tagging.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}