{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\nfrom scipy.misc import imread, imsave\nimport cv2\nimport matplotlib.pyplot as plt\nfrom scipy.misc import imread, imsave\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"## Read Data"},{"metadata":{"trusted":true,"_uuid":"49712666b07d4438bf6851c038d59429209ae6ae"},"cell_type":"code","source":"path_data = '../input/train.csv'\ndata = pd.read_csv(path_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"55c43e8a802f5846e870b07b892f45f8191a63dc"},"cell_type":"code","source":"data.head(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"48354bf20acfcde607a2c54f0ad0b31617744d28"},"cell_type":"markdown","source":"#### Show Example"},{"metadata":{"trusted":true,"_uuid":"14417bf2f95d154c94b950f336717b21d736d675"},"cell_type":"code","source":"# Take example\ndata_example = data.loc[0].values\n\nimg_dir = data_example[0]\nlabel = data_example[1]\n\n# Read image\npath_join = os.path.join('../input/train', img_dir)\nimage = imread(path_join)\n\n# Plot image\nimgplot = plt.imshow(image)\nplt.title(label)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"90dfecbc44139563b2e9ce74686434e7ed81b246"},"cell_type":"markdown","source":"We can observe that there are some images with text. We have built a function to clean automatically the text, it works at 85/90 %, this is the first approach. I'm going to improve it. Below, an example:"},{"metadata":{"trusted":true,"_uuid":"906cb8186650bd4b1fc6bbd47c5959cee9d25529"},"cell_type":"code","source":"path_join = os.path.join('../input/train', '2b96cac5a.jpg')\nimage = imread(path_join)\n\nbacktorgb = cv2.cvtColor(image,cv2.COLOR_GRAY2RGB)\nplt.imshow(backtorgb)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f92525eafb95f73568293180afff21d88c7afea6"},"cell_type":"markdown","source":"## Function to Clean Text Automatically\n\nTake the images and convert it to gray scale and we measure the number of white pixels (between 245 - 255) in the bottom of the image. Then we convert the image to blur scale and calculate the lines, if the slope of the lines is between -0.06 and 0.06 we return the values. Finally, we delete that part of the images."},{"metadata":{"trusted":true,"_uuid":"ff982e0fb9a9e765deb8388a497bb25570397159"},"cell_type":"code","source":"def clean_image(image):\n\n    gray = cv2.cvtColor(image,cv2.COLOR_BGR2GRAY)\n\n    kernel_size = 5\n    blur_gray = cv2.GaussianBlur(gray,(kernel_size, kernel_size),0)\n\n    low_threshold = 20\n    high_threshold = 50\n    edges = cv2.Canny(blur_gray, low_threshold, high_threshold)\n\n    rho = 3  # distance resolution in pixels of the Hough grid\n    theta = np.pi / 180  # angular resolution in radians of the Hough grid\n    threshold = 15  # minimum number of votes (intersections in Hough grid cell)\n    min_line_length = 50  # minimum number of pixels making up a line\n    max_line_gap = 20  # maximum gap in pixels between connectable line segments\n    line_image = np.copy(image) * 0  # creating a blank to draw lines on\n\n    # Run Hough on edge detected image\n    # Output \"lines\" is an array containing endpoints of detected line segments\n    lines = cv2.HoughLinesP(edges, rho, theta, threshold, np.array([]),\n                        min_line_length, max_line_gap)\n    y_1 = []\n    for line in lines:\n        for x1,y1,x2,y2 in line:\n            slope  = (y2 - y1) / (x2 - x1)\n            if slope < 0.06 and slope > -0.06:\n                region = gray.shape[0] - (gray.shape[0] * 0.30)\n                if y1 > region:\n                    y_1.append(y1)    \n            else:\n                continue\n    return y_1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c230da779f02fdbe042e63f585cebba3b1ea7ca7"},"cell_type":"markdown","source":"These are the lines to apply this function for every image, take time to preprocess all images and return 2 arrays (`images_preprocess` and `labels_preprocess`). "},{"metadata":{"trusted":true,"_uuid":"5c3270745197b76f7e1c93e90f0fed6533a7c452"},"cell_type":"code","source":"\ndata_images = data['Image'].values\ndata_labels = data['Id'].values\n\nimages_preprocess = []\nlabels_preprocess = []\n\n##############################################################\n# NOTE: DELETE data_images[:10] to preprocess all images. \n##############################################################\nprint('DELETE -> [:10] // in data_images[:10] to preprocess all images')\n\ni = 0\nfor items in data_images[:10]:\n    path_join = os.path.join('../input/train', items)\n    image = cv2.imread(path_join)\n    image_original = image\n\n    if len(image.shape) < 3:\n        image = cv2.cvtColor(image, cv2.COLOR_GRAY2RGB)\n\n    # Crop the image and check if has white pixels\n    bottom_percent = 0.25\n    bottom = image.shape[0] - int(np.ceil(image.shape[0] * bottom_percent))\n    img = image[bottom:image.shape[0], :]\n\n    n_white_pix = np.sum(img >= 250)\n\n    if n_white_pix >= 90000:\n        y1 = clean_image(image)\n\n        # Crop image\n        if y1:\n            min_y1 = min(y1)\n            image_original = image_original[0:min_y1, 0:image_original.shape[1]]\n            images_preprocess.append(image_original)\n            labels_preprocess.append(data_labels[i])\n    else:\n        images_preprocess.append(image_original)\n        labels_preprocess.append(data_labels[i])\n\n    i += 1\n    \nimages_preprocess = np.array(images_preprocess)\nlabels_preprocess = np.array(labels_preprocess)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9776b4ae978a5220fe8b6830bd02e15d714a00ec"},"cell_type":"code","source":"print(images_preprocess.shape)\nprint(labels_preprocess.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9a2b3aa477059e687b54f4d50b19a88ea17b85e9"},"cell_type":"markdown","source":"### Example with image"},{"metadata":{"trusted":true,"_uuid":"8a4f499c1dd0f13725d04c518b995b14c0e60acb"},"cell_type":"code","source":"path_join = os.path.join('../input/train', '2b96cac5a.jpg')\n# image = imread(path_join)\nimage = cv2.imread(path_join)\nimage_original = image\nplt.imshow(image)\nplt.title('BEFORE PREPROCESS')\nplt.show()\n\nif len(image.shape) < 3:\n    image = cv2.cvtColor(image, cv2.COLOR_GRAY2RGB)\n\n# Crop the image and check if has white pixels\nbottom_percent = 0.25\nbottom = image.shape[0] - int(np.ceil(image.shape[0] * bottom_percent))\nimg = image[bottom:image.shape[0], :]\n\nn_white_pix = np.sum(img >= 250)\n\nif n_white_pix >= 90000:\n    y1 = clean_image(image)\n\n    # Crop image\n    if y1:\n        min_y1 = min(y1)\n        image_original = image_original[0:min_y1, 0:image_original.shape[1]]\nelse:\n    pass\n\nplt.imshow(image_original)\nplt.title('AFTER PREPROCESS')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d81b3341013c3886be087ff034270bd01970b129"},"cell_type":"markdown","source":"Works in the most cases but in some images doesn't work correctly, I'm going to try to improve it. I hope this is helpful for you!!"},{"metadata":{"trusted":true,"_uuid":"8fce53ab6a3c08cb504835c20aff9b44bf21aa96"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}