{"cells":[{"cell_type":"markdown","metadata":{"_cell_guid":"392043f3-2ff2-3a71-ade8-c71e5c5975b5"},"source":"This notebook goes through a simple process of finding all the images, generating a few basic features, building a classifier and then applying to classifier on the images"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"3f13ee40-96b3-440b-7258-abf6b4d42bf8"},"outputs":[],"source":"import matplotlib.pylab as plt\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom glob import glob\nimport os\nCAT_COLUMN = 'class'"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"e956ee3d-5969-c579-8d49-1515b8b7a1be"},"outputs":[],"source":"# read in the file paths\ntrain_df = pd.DataFrame([{'path': c_path, \n                           'image_name': os.path.basename(c_path),\n                          CAT_COLUMN: os.path.basename(os.path.dirname(c_path))}\n              for c_path in glob('../input/train/*/*')])\nprint('Total Training Data',train_df.shape[0])\nprint('Sample Summary\\n', pd.value_counts(train_df['class']))\ntrain_df.sample(3)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"011aaf96-6de3-6b22-2478-98bd1fe59393"},"outputs":[],"source":"test_df = pd.DataFrame([dict(path = c_path, \n                           image_name = os.path.basename(c_path)) \n              for c_path in glob('../input/test/*')])\nprint('Total Testing',test_df.shape[0])\ntest_df.sample(3)"},{"cell_type":"markdown","metadata":{"_cell_guid":"70b95158-6948-e13b-629e-703c9129e05c"},"source":"# Feature Generation\nHere we make a very simple feature (file-size)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"98b4c939-26e7-971c-6b83-0937011ce01c"},"outputs":[],"source":"from skimage.io import imread\ndef safe_image_read(in_path):\n    try:\n        return imread(in_path)\n    except:\n        return np.zeros((1,1)) \ndef generate_feature_vector(in_df):\n    current_df = in_df.copy() # [['image_name']]\n    current_df['file_size'] = in_df['path'].map(lambda x: os.stat(x).st_size)\n    current_df['creation_time'] = in_df['path'].map(lambda x: os.stat(x).st_ctime)\n    current_df['pixel_count'] = in_df['path'].map(lambda x: np.prod(safe_image_read(x).shape))\n    current_df['bits_per_pixel'] = current_df['file_size']/current_df['pixel_count']\n    keep_cols = ['image_name', 'file_size', CAT_COLUMN]\n    return current_df[[ccol for ccol in current_df.columns if ccol in keep_cols]]"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"e4a886b2-85c9-b267-ba7e-ad6fbe4ea99d"},"outputs":[],"source":"%%time\n# generate the features for the training set\nftrain_df = generate_feature_vector(train_df)\n# generate the features for the test set\nftest_df = generate_feature_vector(test_df)\n\nftrain_df.sample(3)"},{"cell_type":"markdown","metadata":{"_cell_guid":"4120fcab-f987-0afb-45ad-4c38415d4a30"},"source":"# Train a simple classifier\nWe use the TPOT package to handle the cross validation and hyperparameters for us"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"91d1214d-00b8-fa5a-0d99-8bd2fbcb86b5"},"outputs":[],"source":"from tpot import TPOTClassifier\nauto_classifier = TPOTClassifier(generations=3, population_size=8, verbosity=2)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"460a35e0-74a0-636b-3a8f-3644152166cd"},"outputs":[],"source":"y_train = ftrain_df[CAT_COLUMN]\nx_train = ftrain_df[[ccol for ccol in ftrain_df.columns if ccol not in [CAT_COLUMN, 'image_name']]]\nauto_classifier.fit(x_train, y_train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"0ef6dcba-8621-9575-5ad8-cfab3b98ca22"},"outputs":[],"source":"auto_classifier.export('exported_pipeline_26thRun.py')"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"3256b5b8-8acb-3868-c8e7-88fd13c8c0b2"},"outputs":[],"source":"x_test = ftest_df[[ccol for ccol in ftrain_df.columns if ccol not in [CAT_COLUMN, 'image_name']]]\n# we need access to the pipeline to get the probabilities\ntest_prob = auto_classifier._fitted_pipeline.predict_proba(x_test)\nguess_df = test_df[['image_name']]\nfor i, class_name in enumerate(auto_classifier._fitted_pipeline.classes_):\n    guess_df[class_name] = test_prob[:,i]\nguess_df.sample(3)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"32edea05-2e70-77c5-fac0-7bb3f4c5590f"},"outputs":[],"source":"guess_df.to_csv('guess26th_04_9th.csv', index = False)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"64c88f03-11b0-3977-545d-2111d226822b"},"outputs":[],"source":""}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.0"}},"nbformat":4,"nbformat_minor":0}