{"cells":[{"cell_type":"markdown","metadata":{"_cell_guid":"ddea7d13-c6ab-02b8-d2ab-32dc5b7d3cff"},"source":"**Feature Extraction with pretrained deep model**"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"115cf240-805c-6eae-7c56-bcf11cda6ef9"},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"5c81d9e0-1cb3-bbba-c682-00035088b2bc"},"outputs":[],"source":"from keras import backend as K\nK.set_image_dim_ordering('tf')"},{"cell_type":"markdown","metadata":{"_cell_guid":"796d7536-5716-22c9-9f77-52c479591674"},"source":"The Data"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"b4f62768-6fbb-587a-033a-d4fe757934a3"},"outputs":[],"source":"data_path = \"../input/\""},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"11903d88-5688-0e3b-34da-f1b2ffa6965e"},"outputs":[],"source":"#1. List of training images\n\nimport os, shutil\nimport pandas as pd\ndataset_path = data_path + \"train/\"\ntrain_images = pd.DataFrame(columns=[\"Class\", \"Image\", \"Imagepath\"])\nfor (folder, subs, files) in os.walk(dataset_path):\n    for filename in files:\n        label = folder.split(\"/\")[-1]\n        imagepath = os.path.join(folder, filename)\n        train_images = train_images.append({\"Class\":label, \n                                                \"Image\":filename, \n                                                \"Imagepath\":imagepath}, \n                                               ignore_index=True)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"a33b6ace-fc6f-16af-8527-a821f1ff9552"},"outputs":[],"source":"train_images.head()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"71e955b8-b2f0-e47d-fc47-154b0f16da3e"},"outputs":[],"source":"train_images = train_images[train_images[\"Image\"] != \".DS_Store\"]"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"6cac41f8-0269-2ded-9d97-bf650bb4e51e"},"outputs":[],"source":"#Class distribution\n\n%matplotlib inline\ntrain_images[\"Class\"].value_counts().plot(kind='bar')"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"eb360b17-1774-15f6-0221-0e8b10b9a305"},"outputs":[],"source":"#2. List of test images\n\ndataset_path = data_path + \"test/\"\ntest_images = pd.DataFrame(columns=[\"Image\", \"Imagepath\"])\nfor (folder, subs, files) in os.walk(dataset_path):\n    for filename in files:\n        imagepath = os.path.join(folder, filename)\n        test_images = test_images.append({\"Image\":filename, \n                                                \"Imagepath\":imagepath}, \n                                               ignore_index=True)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f1a391d0-a685-3e53-0385-55d1e74fdea0"},"outputs":[],"source":"test_images.head()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"a1b13a28-353d-21e8-e045-608de755f1ce"},"outputs":[],"source":"test_images = test_images[test_images[\"Image\"] != \".DS_Store\"]"},{"cell_type":"markdown","metadata":{"_cell_guid":"336b2c66-ed78-49bb-da2f-ff333b02afcb"},"source":"Feature Extraction"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"3cc11ebb-435f-160c-a0e6-56846228a05e"},"outputs":[],"source":"#Function to load an image as a numpy matrix\n\nimport numpy as np\nfrom keras.preprocessing.image import img_to_array, load_img\ndef get_image_as_X(path_to_image_file, target_size, dim_ordering='tf'):\n    img = load_img(path_to_image_file, grayscale=False, target_size=target_size)\n    x = img_to_array(img, dim_ordering=dim_ordering)\n    x = np.expand_dims(x, axis=0)\n    return x"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"2749d965-9d45-4a17-b8ce-91af96398f4e"},"outputs":[],"source":"#Function to extract features from an image with a pretrained model\n\ndef extract_features(image_list, image_size, model, preprocessing_function):\n    features = []\n    failed = []\n    for image in image_list:\n        try:\n            x = get_image_as_X(image, image_size)\n            x = preprocessing_function(x)\n            p = model.predict(x)\n            features.extend(p)\n        except Exception as e:\n            failed.append(image)\n            print(\"Fail with image:\", image)\n            print(e)\n            continue\n    \n    return features, failed"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"b6d68284-f08e-795b-b8b8-7aa29268339e"},"outputs":[],"source":"#Loading the model: ResNet50\n\nimg_size=(224, 224)\nfrom keras.applications import resnet50\nmodel = resnet50.ResNet50(weights='imagenet', include_top=False)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"2bbe3abf-07a6-2a80-412c-b4ac9f18bd0b"},"outputs":[],"source":"#Image lists\ntrain_image_list = train_images[\"Imagepath\"].as_matrix()\ntest_image_list = test_images[\"Imagepath\"].as_matrix()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"76eb6db3-e083-4128-9329-6b84423f8f14"},"outputs":[],"source":""}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.0"}},"nbformat":4,"nbformat_minor":0}