{"metadata": {"kernelspec": {"name": "python3", "language": "python", "display_name": "Python 3"}, "_is_fork": false, "language_info": {"name": "python", "pygments_lexer": "ipython3", "nbconvert_exporter": "python", "mimetype": "text/x-python", "codemirror_mode": {"name": "ipython", "version": 3}, "version": "3.6.1", "file_extension": ".py"}, "_change_revision": 0}, "nbformat_minor": 0, "cells": [{"metadata": {"_uuid": "3d35aaae5e489f10a70d1f021918b252054c2c77", "_cell_guid": "392043f3-2ff2-3a71-ade8-c71e5c5975b5"}, "execution_count": null, "source": "This notebook goes through a simple process of finding all the images, generating a few basic features, building a classifier and then applying to classifier on the images", "outputs": [], "cell_type": "markdown"}, {"metadata": {"trusted": false, "_uuid": "7ab89d1cd06d59e643ea8edaf1da995ed896b94b", "_cell_guid": "3f13ee40-96b3-440b-7258-abf6b4d42bf8"}, "execution_count": null, "source": "import matplotlib.pylab as plt\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom glob import glob\nimport os\nCAT_COLUMN = 'type_cat'", "outputs": [], "cell_type": "code"}, {"metadata": {"trusted": false, "_uuid": "ccc93c93ac1ba38193a3260e066f040b50a0a3fd", "_cell_guid": "e956ee3d-5969-c579-8d49-1515b8b7a1be"}, "execution_count": null, "source": "# read in the file paths\ntrain_df = pd.DataFrame([{'path': c_path, \n                           'image_name': os.path.basename(c_path),\n                          CAT_COLUMN: os.path.basename(os.path.dirname(c_path))}\n              for c_path in glob('../input/train/*/*')])\nprint('Total Training Data',train_df.shape[0])\nprint('Sample Summary\\n', pd.value_counts(train_df['type_cat']))\ntrain_df.sample(3)", "outputs": [], "cell_type": "code"}, {"metadata": {"trusted": false, "_uuid": "0c6b6562099e6c4f6261abfa18f76e347748e157", "_cell_guid": "011aaf96-6de3-6b22-2478-98bd1fe59393"}, "execution_count": null, "source": "test_df = pd.DataFrame([dict(path = c_path, \n                           image_name = os.path.basename(c_path)) \n              for c_path in glob('../input/test/*')])\nprint('Total Testing',test_df.shape[0])\ntest_df.sample(3)", "outputs": [], "cell_type": "code"}, {"metadata": {"_uuid": "c6f3a409e07fc7900b5eda1aa62978a944c3e29b", "_cell_guid": "70b95158-6948-e13b-629e-703c9129e05c"}, "execution_count": null, "source": "# Feature Generation\nHere we make a very simple feature (file-size)", "outputs": [], "cell_type": "markdown"}, {"metadata": {"trusted": false, "_uuid": "34a7e9793a1cce5dddee687c3acf250f6ed6a56a", "_cell_guid": "98b4c939-26e7-971c-6b83-0937011ce01c"}, "execution_count": null, "source": "from skimage.io import imread\ndef safe_image_read(in_path):\n    try:\n        return imread(in_path)\n    except:\n        return np.zeros((1,1)) \ndef generate_feature_vector(in_df):\n    current_df = in_df.copy()\n    current_df['file_size'] = in_df['path'].map(lambda x: os.stat(x).st_size)\n    current_df['creation_time'] = in_df['path'].map(lambda x: os.stat(x).st_ctime)\n    current_df['pixel_count'] = in_df['path'].map(lambda x: np.prod(safe_image_read(x).shape))\n    current_df['bits_per_pixel'] = current_df['file_size']/current_df['pixel_count']\n    keep_cols = ['image_name', \n                 'file_size', \n                 'creation_time',\n                 'pixel_count',\n                 'bits_per_pixel',\n                 CAT_COLUMN]\n    return current_df[[ccol for ccol in current_df.columns if ccol in keep_cols]]", "outputs": [], "cell_type": "code"}, {"metadata": {"trusted": false, "_uuid": "de00c7b3cd091c06ba3d84f90b67cee3d9b59f4a", "_cell_guid": "e4a886b2-85c9-b267-ba7e-ad6fbe4ea99d"}, "execution_count": null, "source": "%%time\n# generate the features for the training set\nftrain_df = generate_feature_vector(train_df)\n# generate the features for the test set\nftest_df = generate_feature_vector(test_df)", "outputs": [], "cell_type": "code"}, {"metadata": {"trusted": false, "_uuid": "e3f62ddfc34135aefc316860157d17e9c57df6ef", "_cell_guid": "41dc5d73-c6ac-d2c1-ba23-67bf5e572f72"}, "execution_count": null, "source": "ftrain_df.sample(3)", "outputs": [], "cell_type": "code"}, {"metadata": {"trusted": false, "_uuid": "3905cb8f522f28f0e46cd74aa76eaeca8ff38d1f", "_cell_guid": "aa76899f-9477-019f-23c3-68234ac05bc9"}, "execution_count": null, "source": "ftrain_df.head()", "outputs": [], "cell_type": "code"}, {"metadata": {"_uuid": "7ff4b836e698dd461e0d1ca4a7af5aae097a7656", "_cell_guid": "4120fcab-f987-0afb-45ad-4c38415d4a30"}, "execution_count": null, "source": "# Train a simple classifier\nWe use the TPOT package to handle the cross validation and hyperparameters for us", "outputs": [], "cell_type": "markdown"}, {"metadata": {"trusted": false, "_uuid": "02871e6fc8272e4f3f3c259c11666ab57d3af864", "_cell_guid": "91d1214d-00b8-fa5a-0d99-8bd2fbcb86b5"}, "execution_count": null, "source": "from tpot import TPOTClassifier\nauto_classifier = TPOTClassifier(generations=2, population_size=8, verbosity=2)", "outputs": [], "cell_type": "code"}, {"metadata": {"trusted": false, "_uuid": "1172bd69f3deeee294f79e41be438ede9c89a932", "_cell_guid": "460a35e0-74a0-636b-3a8f-3644152166cd"}, "execution_count": null, "source": "y_train = ftrain_df[CAT_COLUMN]\nx_train = ftrain_df[[ccol for ccol in ftrain_df.columns if ccol not in [CAT_COLUMN, 'image_name']]]\nauto_classifier.fit(x_train, y_train)", "outputs": [], "cell_type": "code"}, {"metadata": {"trusted": false, "_uuid": "8008779e35c34a25c48eb3a1ea86342acf2cd29e", "_cell_guid": "3256b5b8-8acb-3868-c8e7-88fd13c8c0b2"}, "execution_count": null, "source": "x_test = ftest_df[[ccol for ccol in ftrain_df.columns if ccol not in [CAT_COLUMN, 'image_name']]]\n# we need access to the pipeline to get the probabilities\ntest_prob = auto_classifier._fitted_pipeline.predict_proba(x_test)\nguess_df = test_df[['image_name']]\nfor i, class_name in enumerate(auto_classifier._fitted_pipeline.classes_):\n    guess_df[class_name] = test_prob[:,i]\nguess_df.sample(3)", "outputs": [], "cell_type": "code"}, {"metadata": {"trusted": false, "_uuid": "ec6bb8400afd99eeedc284bac90603355ec1e9b3", "_cell_guid": "32edea05-2e70-77c5-fac0-7bb3f4c5590f"}, "execution_count": null, "source": "guess_df.to_csv('guess_03_24th.csv', index = False)", "outputs": [], "cell_type": "code"}, {"metadata": {"trusted": false, "_uuid": "4f59dd389d0d182dcfd7a78f11aa91213abbfc06", "_cell_guid": "64c88f03-11b0-3977-545d-2111d226822b"}, "execution_count": null, "source": "", "outputs": [], "cell_type": "code"}], "nbformat": 4}