{"cells":[{"metadata":{"_uuid":"4e89fe6456d8a286096c4f4e6424f4a7e85ad255","_cell_guid":"931fdfe5-4c85-4545-8074-a9effd51004b","trusted":false},"cell_type":"code","source":"\n\nimport os\nfrom os.path import isdir, join\nfrom pathlib import Path\nimport pandas as pd\n\n# Math\nimport numpy as np\nfrom scipy.fftpack import fft\nfrom scipy import signal\nfrom scipy.io import wavfile\nimport librosa\n\nfrom sklearn.decomposition import PCA\n\n# Visualization\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport IPython.display as ipd\nimport librosa.display\n\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nimport plotly.tools as tls\nimport pandas as pd\n\n%matplotlib inline\n\n","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"ba1b14483fd449db1a7636eb72a79527dccc71d3","_cell_guid":"ac11c73a-a6a1-4bfb-9c10-462247389532","trusted":false},"cell_type":"code","source":"train_audio_path = \"../input/train/audio/\"\nfilename = '../input/train/audio/cat/004ae714_nohash_0.wav'\nsample_rate,samples = wavfile.read(filename)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"cdc38f949d8bbf031a33bf16445702bcae637fdd","_cell_guid":"23bc02b1-2ddc-4882-90a0-34483144744b","trusted":false},"cell_type":"code","source":"# defining a function that calculates the spectrograms\n\ndef specgram(audio, sample_rate, ):\n    window_size=20 \n    step_size = 10\n    esp = 1e-10\n    nperseg = int(round(window_size*sample_rate/1e3))\n    noverlap = int(round(step_size*sample_rate/1e3))\n    \n    freqs, times, spec = signal.spectrogram(audio,\n                                    fs=sample_rate,\n                                    window='hann',\n                                    nperseg=nperseg,\n                                    noverlap=noverlap,\n                                    detrend=False)\n    return freqs, times, np.log(spec.T.astype(np.float32) + 1e-10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3a62183f2ddb9d22d167a8a98b8fc75a1c16ac29","_cell_guid":"4836f7e8-0dc2-4928-bea9-3d5bfcc8ef94","trusted":false},"cell_type":"code","source":"freqs, times, spectrogram = specgram(samples, sample_rate)\nfig = plt.figure(figsize=(14, 8))\nax1 = fig.add_subplot(211)\nax1.set_title('Raw wave of ' + filename)\nax1.set_ylabel('Amplitude')\nax1.plot(np.linspace(0, sample_rate/len(samples), sample_rate), samples)\n\nax2 = fig.add_subplot(212)\nax2.imshow(spectrogram.T, aspect='auto', origin='lower', \n           extent=[times.min(), times.max(), freqs.min(), freqs.max()])\nax2.set_yticks(freqs[::16])\nax2.set_xticks(times[::16])\nax2.set_title('Spectrogram of ' + filename)\nax2.set_ylabel('Freqs in Hz')\nax2.set_xlabel('Seconds')","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"abd14e44d0be56140308aaddc676e8ea4194eefb","_cell_guid":"024f0f8b-75ee-400d-b203-42cdb789fa7b","trusted":false},"cell_type":"code","source":"mean = np.mean(spectrogram, axis=0)\nstd = np.std(spectrogram, axis=0)\nspectrogram = (spectrogram - mean) / std","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6cfa95a51f70563b946fc5c7f66726e331b96828","_cell_guid":"cf02069e-19df-4394-84e7-70fa539e1277","trusted":false},"cell_type":"code","source":"spectrogram","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e9d600d7ccc262e561ca0e0b0de8dd604856bce8","_cell_guid":"4c037aca-0130-4140-be11-97d0bcaaaf46","trusted":false},"cell_type":"code","source":"dirs = [f for f in os.listdir(train_audio_path) if isdir(join(train_audio_path, f))]\ndirs.sort()\nprint('Number of labels: ' + str(len(dirs)))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"75122633cc6a66a88362f1732d00bc2477c67c40","_cell_guid":"01a32c5f-76b2-4478-a0e6-b650f585ed4b","trusted":false},"cell_type":"code","source":"# Calculate\nnumber_of_recordings = []\nfor direct in dirs:\n    waves = [f for f in os.listdir(join(train_audio_path, direct)) if f.endswith('.wav')]\n    number_of_recordings.append(len(waves))\n\n# Plot\ndata = [go.Histogram(x=dirs, y=number_of_recordings)]\ntrace = go.Bar(\n    x=dirs,\n    y=number_of_recordings,\n    marker=dict(color = number_of_recordings, colorscale='Viridius', showscale=True\n    ),\n)\nlayout = go.Layout(\n    title='Number of recordings in given label',\n    xaxis = dict(title='Words'),\n    yaxis = dict(title='Number of recordings')\n)\npy.iplot(go.Figure(data=[trace], layout=layout))","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"5af9f9113242388c03d9fd93b852d8df98862d90","_cell_guid":"aa305715-0480-429f-b995-612ad37ccaad","trusted":false},"cell_type":"code","source":"def violinplot_frequency(dirs, freq_ind):\n    \"\"\" Plot violinplots for given words (waves in dirs) and frequency freq_ind\n    from all frequencies freqs.\"\"\"\n\n    spec_all = []  # Contain spectrograms\n    ind = 0\n    for direct in dirs:\n        spec_all.append([])\n\n        waves = [f for f in os.listdir(join(train_audio_path, direct)) if\n                 f.endswith('.wav')]\n        for wav in waves[:100]:\n            sample_rate, samples = wavfile.read(\n                train_audio_path + direct + '/' + wav)\n            freqs, times, spec = specgram(samples, sample_rate)\n            spec_all[ind].extend(spec[:, freq_ind])\n        ind += 1\n\n    # Different lengths = different num of frames. Make number equal\n    minimum = min([len(spec) for spec in spec_all])\n    spec_all = np.array([spec[:minimum] for spec in spec_all])\n\n    plt.figure(figsize=(13,7))\n    plt.title('Frequency ' + str(freqs[freq_ind]) + ' Hz')\n    plt.ylabel('Amount of frequency in a word')\n    plt.xlabel('Words')\n    sns.violinplot(data=pd.DataFrame(spec_all.T, columns=dirs))\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f871c69d39b709e5af6e4ac61d3c5330cec63223","_cell_guid":"53d97ac8-6321-4c30-98b8-1978db23fa3a","trusted":false},"cell_type":"code","source":"violinplot_frequency(dirs, 120)\n","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"dce0836a5352b9e6ac79f7466860c0cd760db0b1","_cell_guid":"ecc157fd-aedb-4d3e-92db-d2bccf3c1fef","trusted":false},"cell_type":"code","source":"annon = 0\nfor i in os.listdir(str(train_audio_path)+\"go/\"):\n    sample_rate,samples = wavfile.read(str(train_audio_path)+\"go/\"+str(i))\n    if sample_rate != samples.shape[0]:\n        annon+=1\n\nprint(annon)\nprint(len(os.listdir(str(train_audio_path)+\"go/\")))\nprint(len(samples))\n    ","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"f8f9b87badad8d1d9e6c2db068084c187cb1355b","_cell_guid":"579f6cb0-aebc-437c-a87b-8f9073028d07","trusted":false},"cell_type":"code","source":"# defining a fast furiour transformation\ndef custom_fft(y, fs):\n    T = 1.0 / fs\n    N = y.shape[0]\n    yf = fft(y)\n    xf = np.linspace(0.0, 1.0/(2.0*T), N//2)\n    vals = 2.0/N * np.abs(yf[0:N//2])  # FFT is simmetrical, so we take just the first half\n    # FFT is also complex, to we take just the real part (abs)\n    return xf, vals","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"9afc9bf46aa11ffc5d2632637779b163d374c1f0","_cell_guid":"891d4203-399f-488e-b832-355b26203847","trusted":false},"cell_type":"code","source":"# fft_all = []\n# names = []\n# for direct in dirs:\n#     waves = [f for f in os.listdir(join(train_audio_path, direct)) if f.endswith('.wav')]\n#     for wav in waves:\n#         sample_rate, samples = wavfile.read(train_audio_path + direct + '/' + wav)\n#         if samples.shape[0] != sample_rate:\n#             samples = np.append(samples, np.zeros(abs(sample_rate - samples.shape[0] )))\n#         x, val = custom_fft(samples, sample_rate)\n#         print(x, val)\n#         fft_all.append(val)\n#         names.append(direct + '/' + wav)\n\n# fft_all = np.array(fft_all)\n\n# # Normalization\n# fft_all = (fft_all - np.mean(fft_all)) / np.std(fft_all)\n\n# # Dim reduction\n# pca = PCA(n_components=3)\n# fft_all = pca.fit_transform(fft_all)\n\n# def interactive_3d_plot(data, names):\n#     scatt = go.Scatter3d(x=data[:, 0], y=data[:, 1], z=data[:, 2], mode='markers', text=names)\n#     data = go.Data([scatt])\n#     layout = go.Layout(title=\"Anomaly detection\")\n#     figure = go.Figure(data=data, layout=layout)\n#     py.iplot(figure)\n    \n# interactive_3d_plot(fft_all, names)\n            ","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"a0124eb1086e9039f02a6e33b344dbcc0c15f3fe","_cell_guid":"e9d02b4e-d307-4590-903a-91d61058ad55","trusted":false},"cell_type":"code","source":"# print(fft_all.shape)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"e9a71e8da61c855a16c0c6dfb1ed14da2cd166c1","_cell_guid":"490198fa-2a9b-4fdd-80f9-da177e7e40e2","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"137a5c05c2aa293c99c5b3521238648a4b6256f5","_cell_guid":"f332ee27-09a8-48ab-b731-0042b1f358df","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"927625c972f769a213224d9febdccf11aa52f9de","_cell_guid":"479c43e0-bcd6-4433-82d4-1ce8e1f71472","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"12a1f4ee95c0ad2da64c995da56681c5591a524f","_cell_guid":"f79cbbcf-5e29-4008-8ec3-080032975162","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"8548b6b898212774bb743c60655b1d0bf5179666","_cell_guid":"466dcc29-dd8f-4b40-a368-5d68d412b51b","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"4a19900abedaf67228a8a7054ec7a7255788f0da","_cell_guid":"4fdde448-255b-435d-9e8d-2f2df7fb1c68","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"fbbfdc49c51f36353814353fef671cb8f4f2714f","_cell_guid":"6fef505f-e836-427b-974c-bffee90235bb","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"e633385bc2b69d0227fa506f19c99842d4c0b4ca","_cell_guid":"8c74d395-19af-4d37-8811-55ac80762978","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"89535a7587c12b660cab0a5938662a31a0c1212d","_cell_guid":"0a44ea93-3010-4df2-a28e-bb41cbb4b4bf","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"4dbb84da14dad3dc4204ae1cc00920b5609665c5","_cell_guid":"ed438c5e-94ba-4d06-a21f-2c51c91898af","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}