{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\nfrom utils import *\nimport array \n\nfrom pydub import AudioSegment\nimport numba \nimport regex as re\nfrom glob import glob\nimport numpy as np\nimport pandas as pd\n\nimport tensorflow as tf\n\nfrom keras.models import Model, Sequential\nfrom keras.layers import Input, Conv2D, Flatten, MaxPooling2D, Activation, BatchNormalization, GlobalAveragePooling2D, GlobalMaxPool2D, concatenate, Dense, Dropout\nfrom keras.optimizers import Adam\nfrom tensorflow.python.keras.utils import to_categorical\nfrom keras_tqdm import TQDMNotebookCallback\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\n\n%matplotlib inline\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"all_labels = [x[0].split('/')[-1] for x in os.walk(\"../input/train/audio/\")]\n \nexclusions = [\"\",\"_background_noise_\"]\nPOSSIBLE_LABELS = [item for item in all_labels if item not in exclusions]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n# POSSIBLE_LABELS = 'yes no up down left right on off stop go silence unknown'.split()\nid2name = {i: name for i, name in enumerate(POSSIBLE_LABELS)}\nname2id = {name: i for i, name in id2name.items()}\nlen(id2name)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"all_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_data(data_dir):\n    np.random.seed = 1\n    \n    \"\"\" Return 2 lists of tuples:\n    [(class_id, user_id, path), ...] for train\n    [(class_id, user_id, path), ...] for validation\n    \"\"\"\n    # Just a simple regexp for paths with three groups:\n    # prefix, label, user_id\n#     pattern = re.compile(\"(.+\\/)?(\\w+)\\/([^_]+)_.+wav\")\n    pattern  =  re.compile(\"(.+[\\/\\\\\\\\])?(\\w+)[\\/\\\\\\\\]([^_]+)_.+wav\")\n    all_files = glob(os.path.join(data_dir, '../input/train/audio/*/*wav'))\n\n    with open(os.path.join(data_dir, 'train/validation_list.txt'), 'r') as fin:\n        validation_files = fin.readlines()\n        \n    valset = set()\n    for entry in validation_files:\n        r = re.match(pattern, entry)\n        if r:\n            valset.add(r.group(3))\n    \n    possible = set(POSSIBLE_LABELS)\n    \n    train, val, silent, unknown = [], [],[],[]\n    \n    for entry in all_files:\n        r = re.match(pattern, entry)\n        if r:\n            label, uid = r.group(2), r.group(3)\n            \n            if label == '_background_noise_': #we've already split up noise files into 1 seg chunks under 'silence' folder\n                continue\n                \n#             if label not in possible:\n#                 label = 'unknown'\n\n            label_id = name2id[label]\n            sample = (label, label_id, uid, entry)\n        \n            \n            if label == \"unknown\":\n                unknown.append(sample)\n            elif label == \"silence\":\n                silent.append(sample)\n                \n            elif uid in valset:    \n                val.append(sample)\n            else:\n                train.append(sample)\n\n    print('There are {} train and {} val samples'.format(len(train), len(val)))\n    \n    columns_list = ['label', 'label_id', 'user_id', 'wav_file']\n    \n\n    train_df = pd.DataFrame(train, columns = columns_list)\n    valid_df = pd.DataFrame(val, columns = columns_list)\n    silent_df = pd.DataFrame(silent, columns = columns_list)\n    unknown_df = pd.DataFrame(unknown, columns = columns_list)\n    \n    return train_df, valid_df, unknown_df, silent_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df, valid_df, unknown_df, silent_df = load_data('../input/')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['label'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"int(valid_df.shape[0]*0.1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"unknown_df.shape,silent_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#augment validation set with silence and unknown files, made with step=250 when generating silence files\nextra_data_size = int(valid_df.shape[0]*0.1)\n\nunknown_val = unknown_df.sample(extra_data_size,random_state=2)\nunknown_df = unknown_df[~unknown_df.index.isin(unknown_val.index.values)]\n\nsilent_val = silent_df.sample(extra_data_size,random_state=2)\nsilent_df = silent_df[~silent_df.index.isin(silent_val.index.values)]\n\n\nvalid_df = pd.concat([valid_df,silent_val,unknown_val],axis=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"silence_files_AS = [AudioSegment.from_wav(x) for x in silent_df.wav_file.values]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import random\nrandom.choice(silence_files_AS)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}