{"cells":[{"cell_type":"markdown","metadata":{"_cell_guid":"0cba9a75-eea9-10eb-dbce-b0bef8915546"},"source":"Thought: \" My first  competition and notebook... Stoked!\" \n\nCorresponding facial reaction: smile :)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"82a5f3d0-bc44-a218-5f52-b0040b06d6f5"},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom matplotlib import pyplot as plt\n%matplotlib inline\nimport cv2\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n\n# Any results you write to the current directory are saved as output."},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"6adce42b-9631-b25d-b598-e7f9d08e6368"},"outputs":[],"source":"#I stole the below code from vfdev \"Data exploration\" notebook. Thx vfdev.\n\nimport os\nfrom glob import glob\nTRAIN_DATA = \"../input/train\"\ntype_1_files = glob(os.path.join(TRAIN_DATA, \"Type_1\", \"*.jpg\"))\ntype_1_ids = np.array([s[len(os.path.join(TRAIN_DATA, \"Type_1\"))+1:-4] for s in type_1_files])\ntype_2_files = glob(os.path.join(TRAIN_DATA, \"Type_2\", \"*.jpg\"))\ntype_2_ids = np.array([s[len(os.path.join(TRAIN_DATA, \"Type_2\"))+1:-4] for s in type_2_files])\ntype_3_files = glob(os.path.join(TRAIN_DATA, \"Type_3\", \"*.jpg\"))\ntype_3_ids = np.array([s[len(os.path.join(TRAIN_DATA, \"Type_3\"))+1:-4] for s in type_3_files])\ndef get_imagesizes(type_x):\n    j=0\n    sizes=  {}\n    for i in range (1, len(type_x)):\n                img=cv2.imread(type_x[i])\n                if img.size in sizes:\n                    sizes[img.size] += 1\n                else:\n                    sizes[img.size] = 1\n    print(sizes)\n \n\n            \nget_imagesizes(type_1_files) \nget_imagesizes(type_2_files)\nget_imagesizes(type_3_files)\nimg=cv2.imread(type_1_files[20])\nprint(img.shape, img.size)\nprint(len(type_1_files), len(type_2_files), len(type_3_files))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"9de44fba-e6b2-0f49-651a-71e97bb91471"},"outputs":[],"source":"def get_filename(image_id, image_type):\n    \"\"\"\n    Method to get image file path from its id and type   \n    \"\"\"\n    if image_type == \"Type_1\" or \\\n        image_type == \"Type_2\" or \\\n        image_type == \"Type_3\":\n        data_path = os.path.join(TRAIN_DATA, image_type)\n    elif image_type == \"Test\":\n        data_path = TEST_DATA\n    elif image_type == \"AType_1\" or \\\n          image_type == \"AType_2\" or \\\n          image_type == \"AType_3\":\n        data_path = os.path.join(ADDITIONAL_DATA, image_type[1:])\n    else:\n        raise Exception(\"Image type '%s' is not recognized\" % image_type)\n\n    ext = 'jpg'\n    return os.path.join(data_path, \"{}.{}\".format(image_id, ext))\n\n\ndef get_image_data(image_id, image_type):\n    \"\"\"\n    Method to get image data as np.array specifying image id and type\n    \"\"\"\n    fname = get_filename(image_id, image_type)\n    img = cv2.imread(fname)\n    assert img is not None, \"Failed to read image : %s, %s\" % (image_id, image_type)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    return img\n\nplt.imshow(get_image_data(7,\"Type_1\"),cmap=\"hot\")\nplt.show"},{"cell_type":"markdown","metadata":{"_cell_guid":"28336f5e-866b-0d3e-92e6-ea6511db1360"},"source":"Lets try to create a neural network, based on this awesome book: http://neuralnetworksanddeeplearning.com/\n\n - 1st problem: The images differ in size: {image.size : number of images}\n\n         *type 1:  {38340864: 145, 23970816: 86, 28755648: 2, 38937600: 14, 921600: 2}\n\n         *type 2: {23970816: 332, 38340864: 420, 38937600: 15, 28755648: 9, 38241792: 1, 921600: 3}\n\n         *type 3: {23970816: 310, 38340864: 127, 28755648: 6, 38937600: 5, 921600: 1}\n"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"5c2321b2-0428-a9f7-5ae7-465ac1db18a4"},"outputs":[],"source":"#Main class to construct and train Networks\nclass Network(object):\n    \n    def __init__(self, layers, mini_batch_size):\n        \"\"\" Takes a list of layers, describing the network architecture, and a value for the 'mini_batch_size' to be used during trainig by stochastic gradient descent\"\"\"\n        \n        self.layers = layers\n        self.mini_batch_size = mini_batch_size\n        self.params = [param for layer in self.layers for param in layer.params]\n            #bundles up the parameters for each layer into a single list\n        self.x = T.matrix(\"x\")\n        self.y = T.ivector(\"y\")\n        init_layer = self.layers[0]\n        init_layer.set_inpt(self.x, self.x, self.mini_batch_size)\n        for j in xrange(1,len(self.layers)):\n            prev_layer, layer = self.layers[j-1], self.layers[j]\n            layer.set_inp(prev_layer.output, prev_layer.output_dropout, self.mini_batch_size)\n        self.output = self.layers[-1].output\n        self.output_dropout=self.layers[-1].output_dropout\n      "},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"0205f9a4-05b1-6da2-33b2-3b76741c6a7d"},"outputs":[],"source":""}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.0"}},"nbformat":4,"nbformat_minor":0}