{"cells":[{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"979efbe9-c386-5d7c-73ca-7533a25009a9"},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2\nimport os\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelBinarizer\nimport keras\nfrom keras.models import Sequential, load_model\nfrom keras.layers import Dense, Dropout, Activation, Flatten, Conv2D, MaxPooling2D, Lambda, Cropping2D\nfrom keras.utils import np_utils\n\n#VISUALIZERS\nfrom skimage.viewer import ImageViewer\nfrom skimage import feature, color\n\nfrom collections import Counter\n\n%matplotlib inline\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"2d5bafc2-f4dd-52d6-69f8-b03598705cab"},"outputs":[],"source":"tgt_classes = [\"adult_males\", \"subadult_males\", \"adult_females\", \"juveniles\", \"pups\", \"notasealion\"]\n\nfile_names = filter( lambda f: not f.startswith('.'), os.listdir('../input/Train/'))\nfile_names = sorted(file_names, key=lambda \n                    item: (int(item.partition('.')[0]) if item[0].isdigit() else float('inf'), item))\n\nfile_names = file_names[0:3]"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"7c598a59-6d11-2451-8527-d74d98e1f9f4"},"outputs":[],"source":"coords_df = pd.DataFrame(index=file_names, columns=tgt_classes)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"2c067dd6-de46-730b-e383-4d30900d2339"},"outputs":[],"source":"for filename in file_names:\n    print(filename)\n    # read the Train and Train Dotted images\n    trn_dot_img = cv2.imread(\"../input/TrainDotted/\" + filename)\n    trn_img = cv2.imread(\"../input/Train/\" + filename)\n    \n    cut = np.copy(trn_img)\n    \n    # absolute difference between Train and Train Dotted\n    diff_img = cv2.absdiff(trn_dot_img,trn_img)\n    \n    # convert to grayscale to be accepted by skimage.feature.blob_log\n    diff_img = cv2.cvtColor(diff_img, cv2.COLOR_BGR2GRAY)\n    \n    trn_dot_mask = cv2.cvtColor(trn_dot_img, cv2.COLOR_BGR2GRAY)\n    trn_dot_mask[trn_dot_mask < 20] = 0\n    trn_dot_mask[trn_dot_mask > 0] = 255\n\n    trn_mask = cv2.cvtColor(trn_img, cv2.COLOR_BGR2GRAY)\n    trn_mask[trn_mask < 20] = 0\n    trn_mask[trn_mask > 0] = 255\n    \n    diff_img = cv2.bitwise_or(diff_img, diff_img, mask=trn_dot_mask)\n    diff_img = cv2.bitwise_or(diff_img, diff_img, mask=trn_mask) \n\n    # detect blobs using laplaccian of guassian\n    blobs = feature.blob_log(diff_img, min_sigma=3, max_sigma=4, num_sigma=1, threshold=0.02)\n    \n    adult_males = []\n    subadult_males = []\n    pups = []\n    juveniles = []\n    adult_females = []\n    notasealion = []\n    \n    for blob in blobs:\n        # get the coordinates for each blob\n        y, x, s = blob\n        # get the color of the pixel from Train Dotted in the center of the blob\n        g,b,r = trn_dot_img[int(y)][int(x)][:]\n        \n        # decision tree to pick the class of the blob by looking at the color in Train Dotted\n        if r > 200 and g < 50 and b < 50: # RED\n            adult_males.append((int(x),int(y)))        \n        elif r > 200 and g > 200 and b < 50: # MAGENTA\n            subadult_males.append((int(x),int(y)))         \n        elif r < 100 and g < 100 and 150 < b < 200: # GREEN\n            pups.append((int(x),int(y)))\n        elif r < 100 and  100 < g and b < 100: # BLUE\n            juveniles.append((int(x),int(y))) \n        elif r < 150 and g < 50 and b < 100:  # BROWN\n            adult_females.append((int(x),int(y)))\n        else:\n            notasealion.append((int(x),int(y)))\n            \n    coords_df[\"adult_males\"][filename] = adult_males\n    coords_df[\"subadult_males\"][filename] = subadult_males\n    coords_df[\"adult_females\"][filename] = adult_females\n    coords_df[\"juveniles\"][filename] = juveniles\n    coords_df[\"pups\"][filename] = pups\n    coords_df[\"notasealion\"][filename] = notasealion"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"2e6f57c1-c906-b488-c1d7-3837f22f71af"},"outputs":[],"source":"x = []\ny = []\n\nfor filename in file_names:    \n    image = cv2.imread(\"../input/Train/\" + filename)\n    for lion_class in tgt_classes:\n        for coordinates in coords_df[lion_class][filename]:\n            thumb = image[coordinates[1]-16:coordinates[1]+16,coordinates[0]-16:coordinates[0]+16,:]\n            if np.shape(thumb) == (32, 32, 3):\n                x.append(thumb)\n                y.append(lion_class)\nx = np.array(x)\ny = np.array(y)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"3dd57b63-abab-cd72-6abd-a01dec887721"},"outputs":[],"source":"reference = pd.read_csv('../input/Train/train.csv')\nreference.ix[0:0]"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"348c4a6e-67bc-f02f-d0d2-4d0434e9d6d0"},"outputs":[],"source":"img = cv2.imread(\"../input/Train/0.jpg\")\n\nx_test = []\n\nfor i in range(0,np.shape(img)[0],32):\n    for j in range(0,np.shape(img)[1],32):                \n        thumb = img[i:i+32,j:j+23,:]        \n        if np.shape(thumb) == (32,23,3):\n            x_test.append(thumb)\n\nx_test = np.array(x_test)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"eb68b3d7-eb5d-55a3-1e9a-ee5c4f7efbee"},"outputs":[],"source":"cut = np.copy(\"../input/Train/0.jpg\")\nnp.shape(cut)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"96be037a-2014-d8b9-d904-2852e01b2b45"},"outputs":[],"source":"for i in range(0,np.shape(cut)[0],224):\n    for j in range(0,np.shape(cut)[1],224):                \n        thumb = cut[i:i+32,j:j+32,:]\n        if np.amin(cv2.cvtColor(thumb, cv2.COLOR_BGR2GRAY)) != 0:\n            if np.shape(thumb) == (32,32,3):\n                x.append(thumb)\n                y.append(\"negative\")              \n\ntgt_classes.append(\"negative\")\nx = np.array(x)\ny = np.array(y)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"232b6073-6f22-9810-1fe2-7058807c312f"},"outputs":[],"source":"for lion_class in tgt_classes:\n    f, ax = plt.subplots(1,10,figsize=(12,1.5))\n    f.suptitle(lion_class)\n    axes = ax.flatten()\n    j = 0\n    for a in axes:\n        a.set_xticks([])\n        a.set_yticks([])\n        for i in range(j,len(x)):\n            if y[i] == lion_class:\n                j = i+1\n                a.imshow(cv2.cvtColor(x[i], cv2.COLOR_BGR2RGB))\n                break"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"1aef00b8-1d0d-d2be-7e0b-cc96f6342409"},"outputs":[],"source":""}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.0"}},"nbformat":4,"nbformat_minor":0}