{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### Extracting small digits from UMNIST dataset (latest) using image processing\n\n\nThe core idea is - \"wherever there are digits, there is curvature\". To extract digits, extract curves. Most simplest way to detect curves is by detecting circles. \n\n> - Apply similar techniques to extract for bigger digits (relatively easier)\n> - Discrete pixel values makes it easier to apply traditional image processing techniques. There are more than one way to do this. \n> - This is the most simplest example. More can be found in books like https://szeliski.org/Book/\n> - Applying these ideas may get us on top of leaderboard - but I am more interested in building *elegant and well-defined pipeline to train CNN with these kind of high-resolution images*. For that, I am using semi-supervised learning and some SOTA techniques - hopefully, they may get me to top 10% of leaderboard atleast -  will share my analyses soon.","metadata":{}},{"cell_type":"code","source":"import shutil\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nnp.random.seed(42)\n\nimages = sorted(list(Path(\"../input/ultra-mnist/train\").glob(\"*.jpeg\")))\nimages = np.random.choice(images, 5)\nfor p in images:\n    shutil.copy(str(p), \"./\")","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:42:16.025274Z","iopub.execute_input":"2022-03-16T11:42:16.025567Z","iopub.status.idle":"2022-03-16T11:42:16.503685Z","shell.execute_reply.started":"2022-03-16T11:42:16.025527Z","shell.execute_reply":"2022-03-16T11:42:16.502812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\nnp.random.seed(42)\n\n\nfor i in range(5):\n    images = Path(\"./\").glob(\"*.jpeg\")\n    images = sorted(list(images))\n    image = cv2.imread(str(images[i]))\n    image = image.astype(np.uint8)\n\n    ### very old template code that i've been using for sometime\n    ### more: https://docs.opencv.org/4.x/da/d53/tutorial_py_houghcircles.html\n    img = image.copy()\n    gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    img_blur = cv2.medianBlur(gray, 5)\n    dp, min_dist = 1, img.shape[0]/64, \n    circles = cv2.HoughCircles(img_blur, cv2.HOUGH_GRADIENT, dp, min_dist, \n                               param1=200, param2=10, minRadius=5, maxRadius=30)\n    if circles is not None:\n        circles = np.uint16(np.around(circles))\n        for i in circles[0, :]:\n            # cv2.circle(img, (i[0], i[1]), i[2], (0, 255, 0), 10)\n            # cv2.circle(img, (i[0], i[1]), 2, (0, 0, 255), 6)\n            cv2.circle(img,(i[0],i[1]),i[2]*10,(0,255,0, 150), 15)\n\n\n    fig = plt.figure(figsize=(20,20))\n    plt.imshow(img)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:42:31.408519Z","iopub.execute_input":"2022-03-16T11:42:31.408947Z","iopub.status.idle":"2022-03-16T11:42:43.861570Z","shell.execute_reply.started":"2022-03-16T11:42:31.408916Z","shell.execute_reply":"2022-03-16T11:42:43.860652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}