{"cells":[{"cell_type":"markdown","metadata":{"_cell_guid":"a7d0a331-42b8-fe3e-a798-17761ed27916"},"source":"# This part is taken from @vfdev Data Exploration kernel"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"4e5ee4f2-526f-4432-2153-74c46eddce2e"},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"45838b83-dd0e-74bd-34b1-aa4de5a19eb6"},"outputs":[],"source":"import os\nfrom glob import glob\nTRAIN_DATA = \"../input/train\"\ntype_1_files = glob(os.path.join(TRAIN_DATA, \"Type_1\", \"*.jpg\"))\ntype_1_ids = np.array([s[len(os.path.join(TRAIN_DATA, \"Type_1\"))+1:-4] for s in type_1_files])\ntype_2_files = glob(os.path.join(TRAIN_DATA, \"Type_2\", \"*.jpg\"))\ntype_2_ids = np.array([s[len(os.path.join(TRAIN_DATA, \"Type_2\"))+1:-4] for s in type_2_files])\ntype_3_files = glob(os.path.join(TRAIN_DATA, \"Type_3\", \"*.jpg\"))\ntype_3_ids = np.array([s[len(os.path.join(TRAIN_DATA, \"Type_3\"))+1:-4] for s in type_3_files])\n\nprint(len(type_1_files), len(type_2_files), len(type_3_files))\nprint(\"Type 1\", type_1_ids[:10])\nprint(\"Type 2\", type_2_ids[:10])\nprint(\"Type 3\", type_3_ids[:10])"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"78590ba3-a73d-a8fd-ddf7-875842125a3e"},"outputs":[],"source":"TEST_DATA = \"../input/test\"\ntest_files = glob(os.path.join(TEST_DATA, \"*.jpg\"))\ntest_ids = np.array([s[len(TEST_DATA)+1:-4] for s in test_files])\nprint(len(test_ids))\nprint(test_ids[:10])"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"e71d80c8-f599-0abe-9ead-211e79fa0f8c"},"outputs":[],"source":"ADDITIONAL_DATA = \"../input/additional\"\nadditional_type_1_files = glob(os.path.join(ADDITIONAL_DATA, \"Type_1\", \"*.jpg\"))\nadditional_type_1_ids = np.array([s[len(os.path.join(ADDITIONAL_DATA, \"Type_1\"))+1:-4] for s in additional_type_1_files])\nadditional_type_2_files = glob(os.path.join(ADDITIONAL_DATA, \"Type_2\", \"*.jpg\"))\nadditional_type_2_ids = np.array([s[len(os.path.join(ADDITIONAL_DATA, \"Type_2\"))+1:-4] for s in additional_type_2_files])\nadditional_type_3_files = glob(os.path.join(ADDITIONAL_DATA, \"Type_3\", \"*.jpg\"))\nadditional_type_3_ids = np.array([s[len(os.path.join(ADDITIONAL_DATA, \"Type_3\"))+1:-4] for s in additional_type_3_files])\n\nprint(len(additional_type_1_files), len(additional_type_2_files), len(additional_type_2_files))\nprint(\"Type 1\", additional_type_1_ids[:10])\nprint(\"Type 2\", additional_type_2_ids[:10])\nprint(\"Type 3\", additional_type_3_ids[:10])"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"d02c46f8-681f-f674-4f9d-aabcf13bcb1a"},"outputs":[],"source":"def get_filename(image_id, image_type):\n    \"\"\"\n    Method to get image file path from its id and type   \n    \"\"\"\n    if image_type == \"Type_1\" or \\\n        image_type == \"Type_2\" or \\\n        image_type == \"Type_3\":\n        data_path = os.path.join(TRAIN_DATA, image_type)\n    elif image_type == \"Test\":\n        data_path = TEST_DATA\n    elif image_type == \"AType_1\" or \\\n          image_type == \"AType_2\" or \\\n          image_type == \"AType_3\":\n        data_path = os.path.join(ADDITIONAL_DATA, image_type[1:])\n    else:\n        raise Exception(\"Image type '%s' is not recognized\" % image_type)\n\n    ext = 'jpg'\n    return os.path.join(data_path, \"{}.{}\".format(image_id, ext))\n\n\ndef get_image_data(image_id, image_type):\n    \"\"\"\n    Method to get image data as np.array specifying image id and type\n    \"\"\"\n    fname = get_filename(image_id, image_type)\n    img = cv2.imread(fname)\n    assert img is not None, \"Failed to read image : %s, %s\" % (image_id, image_type)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    return img"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"526e5457-56c4-dc70-4117-e303812f2395"},"outputs":[],"source":"import matplotlib.pylab as plt\nimport math"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"55c2ea7e-4de9-0d41-fd77-905f1635fbd0"},"outputs":[],"source":"img_1 = get_image_data('1023', 'Type_1')\nimg_2 = get_image_data('531', 'Type_1')\nimg_3 = get_image_data('596', 'Type_1')\nimg_4 = get_image_data('1061', 'Type_1')\nimg_5 = get_image_data('1365', 'Type_2')"},{"cell_type":"markdown","metadata":{"_cell_guid":"c88cdb88-f7e1-d29b-757b-3a46c8894d9e"},"source":"# This part is based on @chattob Cervix Segmentation kernel"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f9576114-6223-d3ef-b19b-6143eec57c7a"},"outputs":[],"source":"def Ra_space(img, Ra_ratio, a_threshold):\n    imgLab = cv2.cvtColor(img, cv2.COLOR_RGB2LAB);\n    w = img.shape[0]\n    h = img.shape[1]\n    Ra = np.zeros((w*h, 2))\n    for i in range(w):\n        for j in range(h):\n            R = math.sqrt((w/2-i)*(w/2-i) + (h/2-j)*(h/2-j))\n            Ra[i*h+j, 0] = R\n            Ra[i*h+j, 1] = min(imgLab[i][j][1], a_threshold)\n            \n    Ra[:,0] /= max(Ra[:,0])\n    Ra[:,0] *= Ra_ratio\n    Ra[:,1] /= max(Ra[:,1])\n\n    return Ra"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"419dbab3-3831-57c6-a28d-eba74d1e4fee"},"outputs":[],"source":"plt.imshow(img_1)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"c6b8ed96-bfc6-8327-7bd9-495692684de9"},"outputs":[],"source":"img = img_1 #get_image_data(image_id, 'Type_%i' % (k+1))\n\n#img = cropCircle(img)\nw = img.shape[0]\nh = img.shape[1]\n                        \nimgLab = cv2.cvtColor(img, cv2.COLOR_RGB2LAB);\n        \n# Saturating the a-channel at 150 helps avoiding wrong segmentation\n# in the case of close-up cervix pictures where the bloody os is falsly segemented as the cervix.\nRa = Ra_space(img, 1.0, 300) \na_channel = np.reshape(Ra[:,1], (w,h))\nplt.subplot(121)\nplt.imshow(a_channel) \n"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"d27b2b72-5eb2-ba04-eba2-bd7fa8f800ea"},"outputs":[],"source":"from sklearn import mixture\nfrom sklearn.utils import shuffle\n\ng = mixture.GaussianMixture(n_components = 2, covariance_type = 'diag', random_state = 0, init_params = 'kmeans')\nimage_array_sample = shuffle(Ra, random_state=0)[:1000]\ng.fit(image_array_sample)\nlabels = g.predict(Ra)\nlabels += 1 # Add 1 to avoid labeling as 0 since regionprops ignores the 0-label.\n\nprint(labels[:10])"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"ffcf2d15-583f-382a-6254-5c753000a389"},"outputs":[],"source":"plt.scatter(Ra[:,0], Ra[:,1])\nplt.show()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"0e20367b-0758-cc64-026f-67cb4f6bfd1c"},"outputs":[],"source":"len(labels)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"5d374f72-b5fd-453b-50a0-7775301430f7"},"outputs":[],"source":"labels -= 1\nplt.scatter(Ra[:,0], Ra[:,1], c=labels)\nplt.show()"}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.0"}},"nbformat":4,"nbformat_minor":0}