{"nbformat": 4, "metadata": {"language_info": {"mimetype": "text/x-python", "name": "python", "version": "3.6.3", "file_extension": ".py", "nbconvert_exporter": "python", "codemirror_mode": {"name": "ipython", "version": 3}, "pygments_lexer": "ipython3"}, "kernelspec": {"name": "python3", "display_name": "Python 3", "language": "python"}}, "nbformat_minor": 1, "cells": [{"execution_count": null, "cell_type": "code", "metadata": {"_cell_guid": "d95f486f-0873-4ced-b14f-e4033a1f8a4b", "_uuid": "155aa61569972fe6b2fe0433e604428ef16a72ac"}, "outputs": [], "source": ["# This Python 3 environment comes with many helpful analytics libraries installed\n", "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n", "# For example, here's several helpful packages to load in \n", "\n", "import os\n", "from pathlib import Path\n", "import IPython.display as ipd\n", "\n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "\n", "# Input data files are available in the \"../input/\" directory.\n", "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n", "\n", "from subprocess import check_output\n", "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n", "\n", "# Any results you write to the current directory are saved as output."]}, {"execution_count": null, "cell_type": "code", "metadata": {}, "outputs": [], "source": ["print(check_output([\"ls\", \"../input/train\"]).decode(\"utf8\"))\n", "\n", "folders = os.listdir(\"../input/train/audio\")\n", "print(folders)"]}, {"cell_type": "markdown", "metadata": {}, "source": ["## Some starter script to help with the data processing"]}, {"execution_count": null, "cell_type": "code", "metadata": {}, "outputs": [], "source": ["train_audio_path = '../input/train/audio'\n", "\n", "train_labels = os.listdir(train_audio_path)\n", "train_labels.remove('_background_noise_')\n", "print(f'Number of labels: {len(train_labels)}')\n", "\n", "labels_to_keep = ['yes', 'no', 'up', 'down', 'left',\n", "                  'right', 'on', 'off', 'stop', 'go', 'silence']\n", "\n", "train_file_labels = dict()\n", "for label in train_labels:\n", "    files = os.listdir(train_audio_path + '/' + label)\n", "    for f in files:\n", "        train_file_labels[label + '/' + f] = label\n", "\n", "train = pd.DataFrame.from_dict(train_file_labels, orient='index')\n", "train = train.reset_index(drop=False)\n", "train = train.rename(columns={'index': 'file', 0: 'folder'})\n", "train = train[['folder', 'file']]\n", "train = train.sort_values('file')\n", "train = train.reset_index(drop=True)\n", "print(train.shape)\n", "\n", "def remove_label_from_file(label, fname):\n", "    return fname[len(label)+1:]\n", "\n", "train['file'] = train.apply(lambda x: remove_label_from_file(*x), axis=1)\n", "train['label'] = train['folder'].apply(lambda x: x if x in labels_to_keep else 'unknown')"]}, {"execution_count": null, "cell_type": "code", "metadata": {}, "outputs": [], "source": ["train.sample(5)"]}, {"cell_type": "markdown", "metadata": {}, "source": ["## You can listen to the audio directly from the Kernel"]}, {"execution_count": null, "cell_type": "code", "metadata": {}, "outputs": [], "source": ["ipd.Audio(str(train_audio_path) + '/house/61e50f62_nohash_1.wav')"]}, {"cell_type": "markdown", "metadata": {}, "source": ["## Benchmark"]}, {"execution_count": null, "cell_type": "code", "metadata": {"collapsed": true}, "outputs": [], "source": ["sample_submission = pd.read_csv('../input/sample_submission.csv', index_col='fname')\n", "sample_submission['label'] = 'silence'\n", "sample_submission.to_csv('silence_is_golden.csv')"]}, {"execution_count": null, "cell_type": "code", "metadata": {}, "outputs": [], "source": ["sample_submission.head()"]}, {"execution_count": null, "cell_type": "code", "metadata": {"collapsed": true}, "outputs": [], "source": []}]}