{"cells":[{"cell_type":"markdown","metadata":{"_cell_guid":"03514d69-b592-29b5-6165-1b3b582a3208"},"source":""},{"cell_type":"markdown","metadata":{"_cell_guid":"493bdc83-5458-77d0-a569-f91fcc5dec39"},"source":"## Data exploration ##\n - Visualisation of all data"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f25844a9-3eec-6e46-ad7f-b28764550748"},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n#Any results you write to the current directory are saved as output."},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"e5f8dee6-6825-efd5-6aad-0cc5b4433079"},"outputs":[],"source":"from glob import glob\ninputDir = \"../input/\"\ntraincsv = pd.read_csv(inputDir + 'Train/train.csv')\ntrain = glob(inputDir + 'Train/*.jpg')\ntraindot = glob(inputDir + 'TrainDotted/*.jpg')\nsubm = pd.read_csv(inputDir + 'sample_submission.csv')\nprint(len(traincsv),len(train), len(traindot), len(subm))\n"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"48b7459c-c8bd-a18a-73cb-ef0227b3fd9d"},"outputs":[],"source":"print(traincsv.head())"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"e7f59069-b186-d514-6d03-df5a2ea32d6e"},"outputs":[],"source":"import cv2\nimport matplotlib.pylab as plt\n\nsize = 512\ndef read_imgs(img):\n    print('{}'.format(img))\n    im = cv2.imread(img, cv2.IMREAD_COLOR)\n    img = cv2.cvtColor(im, cv2.COLOR_BGR2RGB)\n    return img\ntrain = glob(inputDir + 'Train/*.jpg')\nprint('Reading Train images... ')\nfor trn in (train):\n    img = read_imgs(trn)\n    plt.figure()\n    plt.imshow(img)\n    plt.axis('off')\n"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"d7b6477a-b0b9-6e23-c0d3-b179ee4a66df"},"outputs":[],"source":"trainDot = glob(inputDir + 'TrainDotted/*.jpg')\nprint('Reading TrainDotted images... ')\nfor trn in (trainDot):\n    img = read_imgs(trn)\n    plt.figure()\n    plt.imshow(img)\n    plt.axis('off')"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"537b2230-a8b7-d201-72f2-8a8fd244df36"},"outputs":[],"source":"subm = pd.read_csv(inputDir + 'sample_submission.csv')\nprint(subm.head())"}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.0"}},"nbformat":4,"nbformat_minor":0}