{"cells": [{"metadata": {"_uuid": "163f64b0834bef96ab093583d4bb32b4f760fb10", "_cell_guid": "0295f7c2-3924-4852-95ec-ab8cbef3ac1c"}, "source": ["# Analysis of Variational Autoencoder variables on Tensorflow Speech Recognition challenge\n", "\n", "The idea behind the VAE was to do 'one-class classification', i.e. to \n", "engineer features which could potentially be useful to distinguish classes to be\n", "predicted ('known classes') from others ('unknown classes' in the test set).\n", "\n", "The VAE code used to generate the input dataset is here: https://www.kaggle.com/holzner/variational-autoencoder-for-speech-dataset\n"], "cell_type": "markdown"}, {"metadata": {"_uuid": "2264e2d313f68422db86339ceabd87b5b6eaa760", "collapsed": true, "_cell_guid": "819dc0dd-2f44-4f01-bfb8-f3e24a40883a"}, "source": ["%matplotlib inline\n", "\n", "import matplotlib.pyplot as plt\n", "import numpy as np\n", "import pandas as pd"], "execution_count": null, "outputs": [], "cell_type": "code"}, {"metadata": {"_uuid": "6388cae19e2e93ba3c2b79ed826419f42e3460de", "_cell_guid": "9cdc62b0-9ba6-4a55-870a-0dfd0b59bfae"}, "source": ["load the dataset and add some columns"], "cell_type": "markdown"}, {"metadata": {"_uuid": "43c7c81470bcd109e8e210a0ae9d6d846b9bd826", "collapsed": true, "_cell_guid": "2b073ca9-9218-46ae-8251-ff1f9418b49b"}, "source": ["df = pd.read_csv(\"../input/tensorflow-speech-recognition-vae-latent-variables/autoencoder-results.csv\")\n", "\n", "# add some columns\n", "\n", "# train or test\n", "df['sample'] = df['fname'].str.split('/').str[0]\n", "\n", "# labels for train data samples\n", "df['label'] = None\n", "df.loc[df['sample'] == 'train', 'label'] = df['fname'].str.split('/').str[2]\n", "\n", "# whether this was part of the classes this was trained\n", "# on (which is equivalent whether this was one of the\n", "# classes to be predicted other than 'unknown' and 'silence')\n", "labels_to_predict = 'yes no up down left right on off stop go silence unknown'.split()\n", "\n", "df['label_to_predict'] = df['label'].isin(labels_to_predict)"], "execution_count": null, "outputs": [], "cell_type": "code"}, {"metadata": {"scrolled": true, "_uuid": "c16fb558473735efe09e8c029bb6d4712189f47e", "collapsed": true, "_cell_guid": "00007871-603d-4afa-9b96-b4f71ad66297"}, "source": ["df.head()"], "execution_count": null, "outputs": [], "cell_type": "code"}, {"metadata": {"_uuid": "bd12fb63eb00a79681ad07a2d6b160135ca86ce7", "_cell_guid": "63179307-6c47-41c6-8aac-9231a4429c73"}, "source": ["all column names"], "cell_type": "markdown"}, {"metadata": {"_uuid": "cb2b0f0b0fa42e8b86b542a1ce69833fa3fb07c2", "collapsed": true, "_cell_guid": "feb594b3-7859-43a2-9980-db1275c17415"}, "source": ["df.columns.tolist()"], "execution_count": null, "outputs": [], "cell_type": "code"}, {"metadata": {"_uuid": "23cc79f97b3271493d315d6a314ccc7948e95489", "_cell_guid": "690fb346-6865-405f-95db-3f11b0109f1c"}, "source": ["get number of latent variables from column names"], "cell_type": "markdown"}, {"metadata": {"_uuid": "518ee7ee4e1b62d801417f42d52b8784c142c3ae", "collapsed": true, "_cell_guid": "b127ebaf-e8f9-4670-b840-3104ca852799"}, "source": ["mu_names    = sorted([ col for col in df.columns if col.startswith('mu')])\n", "sigma_names = sorted([ col for col in df.columns if col.startswith('sigma')])"], "execution_count": null, "outputs": [], "cell_type": "code"}, {"metadata": {"_uuid": "bb10b564bdfd87a1c17fcd5d07ea8351965c3ad9", "_cell_guid": "b5639827-8486-4c39-b2ca-43a98e27b608"}, "source": ["helper function to plot latent variables\n"], "cell_type": "markdown"}, {"metadata": {"_uuid": "9b78410d66da430c6a87355256e7a242fd3886c6", "collapsed": true, "_cell_guid": "de5c50ed-7e26-456f-b627-6921d97a3517"}, "source": ["def plot_latent_variables(varnames, groups, normed = False, log = True):\n", "    \n", "    plt.figure(figsize = (15,15))\n", "\n", "    # find range for common binning of histograms\n", "    bins = np.linspace(df[varnames].min().min(), \n", "                     df[varnames].max().max(),\n", "                 101)\n", "\n", "    for index, colname in enumerate(varnames):\n", "        plt.subplot(4,3, index + 1)\n", "\n", "        for group in groups:\n", "            \n", "            rows = group['selector'](df)\n", "        \n", "            plt.hist(df[rows][colname], \n", "                     bins = bins, \n", "                     alpha = 0.3, \n", "                     log = log, \n", "                     color = group.get('color', None),\n", "                     histtype = 'stepfilled',\n", "                     label    = group.get('label', None),\n", "                     normed   = normed\n", "                    )\n", "\n", "\n", "        plt.title(colname)\n", "        plt.legend()\n", "        plt.grid()"], "execution_count": null, "outputs": [], "cell_type": "code"}, {"metadata": {"_uuid": "ee74c1df51921056c13e20ed75a10fc6ee048bb2", "collapsed": true, "_cell_guid": "0c316fa8-de7f-4b23-898b-0415c4436bd0"}, "source": ["# filters for plotting train vs. test sample distributions\n", "train_test_groups = [\n", "    dict(selector = lambda df: df['sample'] == 'test', label = 'test', color = 'red'),\n", "    dict(selector = lambda df: df['sample'] == 'train', label = 'train', color = 'blue'),\n", "                      ]"], "execution_count": null, "outputs": [], "cell_type": "code"}, {"metadata": {"_uuid": "e12b7c158ec0f4323b11cac2fba588abb3c903da", "collapsed": true, "_cell_guid": "2b3c906f-f608-4e08-95a5-c6422fb8311c"}, "source": ["# filters for plotting train label to predict ('core') vs. train other label\n", "core_other_label_groups = [\n", "    dict(selector = lambda df: (df['sample'] == 'train') & (df['label_to_predict']), label = 'train core', color = 'red'),\n", "    dict(selector = lambda df: (df['sample'] == 'train') & (~ df['label_to_predict']), label = 'train other', color = 'blue'),\n", "                      ]"], "execution_count": null, "outputs": [], "cell_type": "code"}, {"metadata": {"scrolled": false, "_uuid": "f3642a7d88f55ec2476b2c239f6fa151e93b090d", "_cell_guid": "a540bbdc-b694-41aa-97d9-5f8eeadfd8d7"}, "source": [], "cell_type": "markdown"}, {"metadata": {"_uuid": "b324105422e5cafcb8c542814d9fd87c7e95c764", "_cell_guid": "bd3b4597-3b75-42bc-9e6e-e51214beeb52"}, "source": ["## labels to be predicted vs. others in train sample\n", "\n", "for these it looks like that the autoencoder is not able to distinguish between them\n", "(at least not from the individual variables alone)\n"], "cell_type": "markdown"}, {"metadata": {"_uuid": "8b5b2a7ff677c320fc49ee8f20313a79a71de557", "_cell_guid": "f65e7689-f436-45c4-a3b4-63fe903e4bf8"}, "source": ["### latent $\\mu$ distributions"], "cell_type": "markdown"}, {"metadata": {"_uuid": "68585fe2ed51d130cd2e11523a7cc70f1bf9f007", "collapsed": true, "_cell_guid": "7827445b-772c-4a18-a548-acfe73643b3e"}, "source": ["plot_latent_variables(mu_names, \n", "                      core_other_label_groups,\n", "                      True)"], "execution_count": null, "outputs": [], "cell_type": "code"}, {"metadata": {"_uuid": "3a65d208be14c7b6322a6f279406420752086d9b", "_cell_guid": "fac9bb9d-07ec-4e26-96f1-dd3cd5006224"}, "source": ["### latent $\\sigma$ distributions"], "cell_type": "markdown"}, {"metadata": {"scrolled": false, "_uuid": "7998e6a3fdfdc8cd56fec3e2e252b7318658713a", "collapsed": true, "_cell_guid": "cf0932df-99e7-4c67-8e92-8c8ab4e15035"}, "source": ["plot_latent_variables(sigma_names, \n", "                      core_other_label_groups,\n", "                      True)"], "execution_count": null, "outputs": [], "cell_type": "code"}, {"metadata": {"_uuid": "bdade7c9dfbb1ef0eb3c6bb864bb6002f2e1839d", "_cell_guid": "6e7baf26-c572-4bd9-8998-c84df80405ea"}, "source": ["## Train vs. test sample\n", "\n", "For these it looks like there are some differences in shape between train and test samples"], "cell_type": "markdown"}, {"metadata": {"_uuid": "4c1e6fa407c5a628f8c1612c495da2889d9cda79", "_cell_guid": "5e93a39b-592f-4868-8169-b4044e2a517f"}, "source": ["### latent $\\mu$ distributions"], "cell_type": "markdown"}, {"metadata": {"_uuid": "80c7cc9059eeebde1a673ff1f3ce72ad016cdb86", "collapsed": true, "_cell_guid": "4e314aa2-805b-42c6-9db0-bc360793867d"}, "source": ["plot_latent_variables(mu_names, \n", "                      train_test_groups,\n", "                      True)"], "execution_count": null, "outputs": [], "cell_type": "code"}, {"metadata": {"_uuid": "136a45b8d5a62df360e82b9c9cb6609dffef5ed7", "_cell_guid": "fc0a500e-89fb-4f25-87e7-8239d9ebafca"}, "source": ["### latent $\\sigma$ distributions"], "cell_type": "markdown"}, {"metadata": {"_uuid": "8761f88c2a508a237bec58a32d554c48860f4544", "collapsed": true, "_cell_guid": "7694be58-a85d-43fd-9ae9-d142455d2740"}, "source": ["plot_latent_variables(sigma_names, \n", "                      train_test_groups,\n", "                      True)"], "execution_count": null, "outputs": [], "cell_type": "code"}, {"metadata": {"_uuid": "19449537adc539366cb4e8a7290be8ad03c9c38a", "collapsed": true, "_cell_guid": "9a990005-1111-4f95-92c0-6aeb18186e47"}, "source": [], "execution_count": null, "outputs": [], "cell_type": "code"}], "metadata": {"kernelspec": {"name": "python3", "display_name": "Python 3", "language": "python"}, "language_info": {"mimetype": "text/x-python", "nbconvert_exporter": "python", "file_extension": ".py", "name": "python", "pygments_lexer": "ipython3", "version": "3.6.4", "codemirror_mode": {"name": "ipython", "version": 3}}}, "nbformat_minor": 1, "nbformat": 4}