{"cells":[{"metadata":{"_uuid":"37fb09139395a1355af8ddd953b7bedc45e80613"},"cell_type":"markdown","source":"# Tengwar digit recognizer"},{"metadata":{"_uuid":"1bc376ae7ae51e4c2043b54ad218b744e5ad3d00"},"cell_type":"markdown","source":"This is an image classification project based on the handwritten unique Elvish/Tengwar digit dataset (0-11)."},{"metadata":{"_uuid":"523058d9a9e36a8352f9f57c9849c1add467128d"},"cell_type":"markdown","source":"## Data exploration"},{"metadata":{"trusted":true,"_uuid":"9089a22184bf69ba83f68583786d83bc63f4c058"},"cell_type":"code","source":"%matplotlib inline\n\n# For simple vectorized calculations\nimport numpy as np\n\n# Mainly data handling and representation\nimport pandas as pd\n\n# Models\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Conv2D, Flatten, MaxPooling2D, BatchNormalization, Activation, Dropout\nfrom keras.losses import categorical_crossentropy\nfrom keras.optimizers import Adam\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras import Model\n\n# For checking GPU backend\nfrom keras import backend\nfrom tensorflow.python.client import device_lib\n\n# Data preparation\nfrom sklearn.preprocessing import MinMaxScaler, PowerTransformer, OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.decomposition import PCA\n\n# Plotting and display\nfrom IPython.display import display\nfrom matplotlib import pyplot as plt\n\n# Image manipulation\nfrom PIL import Image\n\nnp.random.seed(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ad32e837c4cafb12536be6b140cf61c6313ef1d8"},"cell_type":"code","source":"# Path of the file to read.\ntrain_file_path = \"../input/tengwar-digits-dataset/tengwar_data.csv\"\n\n# Read the file\ndigit_data_orig = pd.read_csv(train_file_path)\n\n# The shape of the data\ndigit_data_orig.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"acef16495a932f9a86a03d9e8a0b16c04357eb41"},"cell_type":"code","source":"# Separate the label from the data\ny_orig = digit_data_orig.iloc[:, 0].values.reshape(-1,1)\n\nm = y_orig.shape[0]\nprint(\"The number of images: m = {}\".format(m))\n\n# There are 308 images so 308 labels\nprint(\"The shape of y: {}\".format(y_orig.shape))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fd6ffabe11219757db3d17fe5d78b1741a5b0498"},"cell_type":"code","source":"# One-hot encode the categorical values\ndef one_hot_encode_categories(y):\n    \n    encoder = OneHotEncoder(handle_unknown='ignore', sparse=False)\n\n    y_one_hot = pd.DataFrame(encoder.fit_transform(y), columns=encoder.get_feature_names())\n        \n    return y_one_hot, encoder","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"b7e6e7097e1fed163771ff32dea58b848a5834a1"},"cell_type":"code","source":"# One hot encode y\ny_one_hot, encoder = one_hot_encode_categories(y_orig)\n\n# Strip the category names\ny_one_hot.columns = pd.DataFrame(y_one_hot.columns)[0].apply(lambda x: x[3:-2])\n\n# The number of categories\nn_y = y_one_hot.shape[1]\nprint(\"The number of categories: n_y = {}\".format(n_y))\n\n# A few examples\nprint(np.random.choice(y_orig.reshape(-1), 10))\ny_one_hot.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"20b61c79160c5cfb1e27a25808f6224432917521"},"cell_type":"code","source":"# Let's see how many examples are ther from each category\ndisplay(pd.DataFrame(y_one_hot.sum(axis=0)).transpose())\n\n# Plot the values\nplt.bar(y_one_hot.columns, y_one_hot.sum(axis=0))\n\n# There is approximately the same number of examples there are from each category","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"daaa4126353c7057a76c41ed9fb6c200a8d88e2c"},"cell_type":"code","source":"# Separate the image data\nX_orig = digit_data_orig.iloc[:, 1:].values.reshape(-1,1)\n\nprint(\"The shape of X without reshaping: {}\".format(X_orig.shape))\n\n# There are 42000 images and 64x64 pixel each image which is 32928000 total","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"23cf7ce40c196c4ac83c7d33b6f8feb8a15fe09a"},"cell_type":"code","source":"# In case of square image we can calculate the edge size\nedge_size = 64\n\n# Let's reshape the images and view some\nX_reshaped = X_orig.reshape(-1, edge_size, edge_size)\n\ndef plot_sample_images(X, y, images_to_show=10, random=True):\n\n    fig = plt.figure(1)\n\n    images_to_show = min(X.shape[0], images_to_show)\n\n    # Set the canvas based on the numer of images\n    fig.set_size_inches(18.5, max(3, images_to_show * 0.3))\n\n    # Generate random integers (non repeating)\n    if random == True:\n        idx = np.random.choice(range(X.shape[0]), images_to_show, replace=False)\n    else:\n        idx = np.arange(images_to_show)\n        \n    # Print the images with labels\n    for i in range(images_to_show):\n        plt.subplot(images_to_show/10 + 1, 10, i+1)\n        plt.title(str(y[idx[i]]))\n        plt.imshow(X[idx[i], :, :], cmap='Greys')\n        \n\n# Choose how many images you would like to see\nimages_to_show = 30\n\nplot_sample_images(X_reshaped, y_orig, images_to_show=images_to_show)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a8cb6aa57dd5a8768c5196d08fdd191cf2bc45a1"},"cell_type":"code","source":"# The number of X features are 64*64 = 4096\nn_x = edge_size*edge_size\n\nprint(\"The number of X features are: n_x = {}\".format(n_x))","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"957c43e542861cc33ff73a3be96025868d993e18"},"cell_type":"code","source":"# Scale the image pixel values from 0-255 to 0-1 range so the neural net can to converge faster\nX_scaled = X_reshaped / 255\n\nprint(\"Original scale: {} - {}\".format(X_reshaped.min(), X_reshaped.max()))\nprint(\"New scale: {} - {}\".format(X_scaled.min(), X_scaled.max()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"647c3b7d6830794c07e30e5f0ca390374cff06b3"},"cell_type":"code","source":"X = X_scaled.reshape(-1, edge_size, edge_size, 1)\ny = y_one_hot","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"50436292b2612025308bc1487609529c026e3cf3"},"cell_type":"markdown","source":"## Simple model definition"},{"metadata":{"trusted":true,"_uuid":"a7d6b7ba43fbe0095f3e277d6cbe66a5d3391fab"},"cell_type":"code","source":"# We can train a model using directly the images, let's first do that","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f6535a795041c36585d94eadeb60b11e161ae92c"},"cell_type":"code","source":"def model_definition():\n    # Define a simple model in Keras\n    model = Sequential()\n\n    # Add layers to the model\n\n    # Add convolutional layer\n    model.add(Conv2D(10, kernel_size=(5,5), input_shape=(edge_size, edge_size, 1)))\n\n    # Add ReLu activation function\n    model.add(Activation('relu'))\n\n    # Add dropout layer for generalization\n    model.add(Dropout(rate = 0.1))\n\n    # Add maxpool layer\n    model.add(MaxPooling2D(pool_size=(2, 2), strides=None, padding='valid', data_format=None))\n\n    # Add batch normalization to help learning and avoid vanishing or exploding gradient\n    model.add(BatchNormalization())\n\n    # Add convolutional layer\n    model.add(Conv2D(10, kernel_size=(3,3)))\n\n    # Add ReLu activation function\n    model.add(Activation('relu'))\n\n    # Add dropout layer for generalization\n    model.add(Dropout(rate = 0.1))\n\n    # Maxpool layer\n    model.add(MaxPooling2D(pool_size=(2, 2), strides=None, padding='valid', data_format=None))\n\n    # Add batch normalization to help learning and avoid vanishing or exploding gradient\n    model.add(BatchNormalization())\n\n    # Add flatten layer to get 1d data for dense layer\n    model.add(Flatten())\n\n    # Dense layer\n    model.add(Dense(10))#, input_dim=650))\n    \n    # Add ReLu activation function\n    model.add(Activation('relu'))\n\n    # Dense layer\n    model.add(Dense(n_y))\n    \n    # Add sigmoid activation function to get values beteween 0-1\n    model.add(Activation('softmax'))\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a4bb7f9d5d62f571edf432098451f5ea5171fd00"},"cell_type":"code","source":"model = model_definition()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c5fcf577509141abbbc7ec91ee78c75e9865189d"},"cell_type":"code","source":"# Define the hyperparameters\n\nbatch_size = 64\nepochs = 300","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"166b44d92e2b905a5fa187459d3b4e4038b48c49"},"cell_type":"code","source":"# Define the loss function, this is a categorical cross entropy\nloss = categorical_crossentropy","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"966541f4a0eefe63e1b659a566f8309780e23690"},"cell_type":"code","source":"# Define the optimizer\noptimizer = Adam(lr=0.0002)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"738f282d22fc528b577896c3e5b161a1f4591140"},"cell_type":"code","source":"# Compile the model\nmodel.compile(loss=loss, optimizer=optimizer, metrics=[\"categorical_accuracy\"])","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"50f840e6cf4e8e43d917ee75bebbdb448e35e845"},"cell_type":"code","source":"# Let's see the model configuration\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":true,"_uuid":"636a088fa43e74c89465aee39498b6eb57cc82d0"},"cell_type":"code","source":"# Split the data into train and validation parts\ntrain_X, val_X, train_y, val_y = train_test_split(X, y.values, random_state=1)\n\nprint(\"The train image shape: {}\".format(train_X.shape))\nprint(\"The train label shape: {}\".format(train_y.shape))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8cd8193532efba4815c951eb4bd63409700790dd"},"cell_type":"markdown","source":"### Image augmentation"},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"9a2d5b3a8188ccae6d5f3f579638090a8f1fe282"},"cell_type":"code","source":"# Define the augmentation properties\ngenerator = ImageDataGenerator(#featurewise_center=True,\n                               #samplewise_center=True,\n                               #featurewise_std_normalization=True,\n                               #samplewise_std_normalization=True,\n                               #zca_whitening=False,\n                               #zca_epsilon=1e-06,\n                               rotation_range=10,\n                               width_shift_range=0.1,\n                               height_shift_range=0.1,\n                               #brightness_range=None,\n                               shear_range=0.1,\n                               zoom_range=0.1,\n                               cval=0.0,)\n\n# Fit the augmentation to the images\ngenerator.fit(X)\n\nX_augmented, y_augmented = generator.flow(train_X, train_y, batch_size=batch_size).next()\n\n# Plot some augmented images\nplot_sample_images(X_augmented[:10,:,:,0], encoder.inverse_transform(y_augmented)[:10,0], 10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fca0938c4ab0d45511e7871a0ffdd63a9da56a0d"},"cell_type":"code","source":"def kfold(X, y_one_hot, n_splits=5, seed=0):\n    \"\"\"\n    This function randomly splits the input and output data equally probable by label to train and validation set.\n    X: nd array of shape: (m, n_x, n_x, colors)\n    y: pandas DataFrame with shape: (m, n_y) and column names as label names\n    n_splits: the number of splits that kfold splits the data into\n    \n    m: number of samples\n    n_h: height in pixels\n    n_w: width in pixels\n    colors: nr of color layers, colors >= 1\n    n_y: number of label types, n_y >= 1\n    \"\"\"\n    \n    # Initialize the already chosen for validation because we do not want to choose a validation sample twice\n    already_chosen_for_validation = []\n    \n    for i in range(n_splits):\n        \n        # The number of samples\n        m = X.shape[0]\n\n        # Define the validation and train size\n        validation_size = int(m / n_splits)\n        train_size = m - validation_size\n\n        # The number of labels\n        n_y = y_one_hot.shape[1]\n\n        # Number of example by label\n        n_label = int(train_size / n_y)\n\n        # Initialize train set list\n        X_train_by_label = []\n        y_train_by_label = []\n\n        # Initialize validation set list\n        X_validation_by_label = []\n        y_validation_by_label = []\n\n        # choose equally randomly from each label type about equally\n        for label in y_one_hot.columns:\n            # Get the indexes from the y_one_hot where the \"label\" is the label variable\n            indexes_by_label = y_one_hot[y_one_hot[label] == 1].index.values.tolist()\n\n            # The number of samples in this type of label\n            m_label = len(indexes_by_label)\n            \n            # Remove the already once chosen validation data\n            available_indexes_by_label = list(set(indexes_by_label).difference(set(already_chosen_for_validation)))\n            \n            if i < n_splits - 1:\n                # Choose from the indexes randomly\n                val_indexes_by_label = np.random.choice(available_indexes_by_label, int(m_label / n_splits), replace=False)\n            else:\n                val_indexes_by_label = available_indexes_by_label\n            \n            # New set with elements in indexes_by_label but not in train_indexes_by_label is equals the indexes of the training set by label\n            train_indexes_by_label = list(set(indexes_by_label).difference(set(val_indexes_by_label)))\n\n            # Append the selected categories of the train data by label\n            X_train_by_label.append(X[train_indexes_by_label, :, :, :])\n            y_train_by_label.append(y_one_hot.iloc[train_indexes_by_label, :])\n\n            # Append the selected categories of the validation data by label\n            X_validation_by_label.append(X[val_indexes_by_label, :, :, :])\n            y_validation_by_label.append(y_one_hot.iloc[val_indexes_by_label, :])\n            \n            # Extend the already_chosen_for_validation list with the newly chosen validation\n            already_chosen_for_validation.extend(val_indexes_by_label)\n\n\n        # Create final train data\n        X_train = np.concatenate(X_train_by_label)\n        y_train = np.concatenate(y_train_by_label)\n\n        # Create final validation data\n        X_validation = np.concatenate(X_validation_by_label)\n        y_validation = np.concatenate(y_validation_by_label)\n    \n        yield X_train, X_validation, y_train, y_validation","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c6d5604d6e3f9ada9f3b96162c61a707451e8e54"},"cell_type":"code","source":"# Train with cross validation\nhistories = []\nscores = []\nmodels = []\n\n# Cross validation train\nfor X_train, X_val, y_train, y_val in kfold(X, y, n_splits=5):\n    # Define model\n    model = model_definition()\n    \n    # Compile model\n    model.compile(loss=loss, optimizer=optimizer, metrics=[\"categorical_accuracy\"])\n    \n    # Fit the model\n    histories.append(model.fit_generator(generator.flow(X_train, y_train, batch_size=batch_size),\n                                         steps_per_epoch=len(X_train) / batch_size,\n                                         validation_data=[X_val, y_val],\n                                         epochs=epochs,\n                                         verbose=0))\n    \n    # Append models\n    models.append(model)\n    \n    # Calculate the score of the model\n    score = model.evaluate(X_val, y_val, verbose=0)\n    print(\"{}: {}\".format(model.metrics_names, score))\n    scores.append(score)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ed203728ffc8cdd97363dea4f9ca100926f1985f"},"cell_type":"code","source":"for i in range(len(model.metrics_names)):\n    print(\"Average {}: {}\".format(model.metrics_names[i], np.mean(np.array(scores)[:, i])))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"661312c0f479637b944f3a67813058e3c161823a"},"cell_type":"code","source":"def plot_history(history):# Plot the loss and accuracy\n    # Format the train history\n    history_df = pd.DataFrame(history.history, columns=history.history.keys())\n\n    \n    # Plot the accuracy\n    fig = plt.figure()\n    fig.set_size_inches(18.5, 10)\n    ax = plt.subplot(211)\n    ax.plot(history_df[\"categorical_accuracy\"], label=\"categorical_accuracy\")\n    ax.plot(history_df[\"val_categorical_accuracy\"], label=\"val_categorical_accuracy\")\n    ax.legend()\n    plt.title('Score during training.')\n    plt.xlabel('Training step')\n    plt.ylabel('Accuracy')\n    plt.grid(b=True, which='major', axis='both')\n    \n    # Plot the loss\n    ax = plt.subplot(212)\n    ax.plot(history_df[\"loss\"], label=\"loss\")\n    ax.plot(history_df[\"val_loss\"], label=\"val_loss\")\n    ax.legend()\n    plt.title('Loss during training.')\n    plt.xlabel('Training step')\n    plt.ylabel('Loss')\n    plt.grid(b=True, which='major', axis='both')\n    \n    plt.show()\n    \n    #display(history_df)","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"4382684db9ce1bb1344321d97fcd15d81072c38c"},"cell_type":"code","source":"for history in histories:\n    plot_history(history)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"12f80f813062fe6c7ccea199fb43b5da4808df34"},"cell_type":"code","source":"final_model = model_definition()\n\n# Compile the model\nfinal_model.compile(loss=loss, optimizer=optimizer, metrics=[\"categorical_accuracy\"])\n\n# Train the model on the full train data\nfinal_history = final_model.fit_generator(generator.flow(X, y, batch_size=batch_size),\n                                          steps_per_epoch=len(X) / batch_size,\n                                          epochs=epochs,\n                                          verbose=0)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"962a22728f700cee7cc8b7a2f8a9a43883a36ba1"},"cell_type":"markdown","source":"### Show the learned visualization by PCA"},{"metadata":{"scrolled":true,"trusted":false,"_uuid":"4c724e42b5dfaf9ec78bb1d1e3078be1bb410a15"},"cell_type":"code","source":"# Get the output of the last activation function\nlayer_name = 'dense_2'\n\n# Define an intermediate model\nintermediate_layer_model = Model(inputs=model.input,\n                                 outputs=model.get_layer(layer_name).output)\n\n# Calculate the values of the intermediate model\nintermediate_output = intermediate_layer_model.predict(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"37ebc7b9ae67da4a98da668ea3a5d78f5cca6552"},"cell_type":"code","source":"PCA_transformer = PCA()\n        \n# Fit and transform\nactivation_7_PCA = PCA_transformer.fit_transform(intermediate_output)","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"a0e194e362737433c95f6f132484f3cf314b764f"},"cell_type":"code","source":"%matplotlib inline\nnumber_of_points = 308\n\nx1 = activation_7_PCA[:number_of_points, 0]\nx2 = activation_7_PCA[:number_of_points, 1]\n\nfig = plt.figure()\nfig.set_size_inches(18.5, 10)\n\nplt.scatter(x=x1,\n            y=x2,\n            c=(y_orig[:number_of_points])[:, 0],\n            cmap=\"tab10\")\n\nax = plt.subplot(111)\n\nfor i in range(number_of_points):\n    if not i % 10:\n        ax.annotate(str(y_orig[i]), (x1[i], x2[i]))\n\nplt.title('Last layer visualization using PCA 2D')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"548501b3bd85a024600948ba9b4e7a5514ff5449"},"cell_type":"code","source":"from mpl_toolkits.mplot3d import Axes3D\n\nnumber_of_points = 308\n\nx1 = activation_7_PCA[:number_of_points, 0]\nx2 = activation_7_PCA[:number_of_points, 1]\nx3 = activation_7_PCA[:number_of_points, 2]\n\nfig = plt.figure()\nfig.set_size_inches(10, 10)\n\n\n\nax = plt.subplot(111, projection='3d')\n\nax.scatter(xs=x1,\n            ys=x2,\n            zs=x3,\n            c=(y_orig[:number_of_points])[:,0],\n            cmap=\"tab10\",)\n\nfor i in range(number_of_points):\n    if not i % 10:\n        ax.text(x1[i], x2[i], x3[i], str(y_orig[i]))\n\nplt.title('Last layer visualization using PCA 3D')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7f7f634e698f1f29e25a9663d845e0a8b89f17dc"},"cell_type":"markdown","source":"### Error analysis"},{"metadata":{"trusted":false,"_uuid":"cf82e9290abf28b485fb2d71bb897200a00c61a3"},"cell_type":"code","source":"augmented_data_batch = val_X.shape[0]\n#X_aug, y_aug_real_unscaled = generator.flow(val_X, val_y, batch_size=augmented_data_batch).next()\nX_aug = val_X\n# Print examples when the model made bad decisions\ny_aug_preds_unscaled = model.predict(X_aug)\n\n# Inverse transform the predictions to the original scale\ny_aug_preds = encoder.inverse_transform(y_aug_preds_unscaled)\ny_aug_real = encoder.inverse_transform(val_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"4fbf170d0dd4cead3cd7f5d79f0aaba7e0b92ed5"},"cell_type":"code","source":"y_aug_all = pd.DataFrame([y_aug_preds[:,0], y_aug_real[:,0]]).transpose()\ny_aug_all.columns = [\"y predicted\", \"y real\"]\nprint(y_aug_all.head(10))\nplot_sample_images(X_aug[:10, :, :, 0], y_aug_real, 10, random=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"a4d63c297a1fe90968f6f6d7594ba8f780d3127a"},"cell_type":"code","source":"pred_errors = y_aug_all[y_aug_all['y predicted'] != y_aug_all['y real']]\nprint(pred_errors)\ntotal_errors = pred_errors.shape[0]\nprint(total_errors)\n\nprint(\"The total number of errors from {} augmented image: {}\".format(augmented_data_batch, total_errors))\nprint(\"Which is {0:.3f}%\".format(total_errors/augmented_data_batch * 100))","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":false,"_uuid":"18d9d1d77882a8b0b6f83c6a630a62760533990e"},"cell_type":"code","source":"errors_by_category = []\nerror_count_by_category = []\n\nfor i in range(n_y):\n    \n    errors_by_category.append(pred_errors[pred_errors[\"y real\"] == i])\n\n    error_count_by_category.append(errors_by_category[i].shape[0])\n\nerror_count_by_category_df = pd.DataFrame(error_count_by_category).transpose()\n\nprint(\"Number of errors by category: \")\ndisplay(error_count_by_category_df)\n\nfig = plt.figure()\nax = plt.subplot(111)\nplt.bar(x=error_count_by_category_df.columns, height=error_count_by_category_df.values[0]/total_errors * 100)\n\nplt.title('Percentage of error by category')\nplt.xlabel('Categories')\nplt.ylabel('Percentage of error (%)')\n#plt.xticks(range(n_y))\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"f43b33e7d7c5cf22c93454a015a728b390988bdf"},"cell_type":"code","source":"for errors in errors_by_category:\n    plot_sample_images(X_aug[errors.index, :, :, 0], errors[\"y predicted\"].values, 10, random=False)\n    plt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}