{"nbformat_minor": 1, "cells": [{"cell_type": "code", "metadata": {"_uuid": "010af0a7ee1e100ec2afbf62503ec4eb110c5554", "_cell_guid": "26afe6fa-2950-4402-96a4-c2d859de2cea", "collapsed": true}, "execution_count": null, "source": ["import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "import os\n", "from skimage.io import imread # read image\n", "from PIL import Image \n", "# imread fails on some of the tiffs so we use PIL\n", "pil_imread = lambda c_file: np.array(Image.open(c_file)) \n", "from skimage.exposure import equalize_adapthist\n", "from glob import glob\n", "\n", "%matplotlib inline\n", "import matplotlib.pyplot as plt"], "outputs": []}, {"cell_type": "markdown", "metadata": {}, "source": ["# Load previous results"]}, {"cell_type": "code", "metadata": {}, "execution_count": null, "source": ["import h5py\n", "t_h5 = os.path.join('..','input', 'fourier-analysis-for-spatial-resolution-estimates', 'training_subset.h5')\n", "with h5py.File(t_h5, 'r') as fd:\n", "    for i in fd.keys():\n", "        print(i, fd[i].shape)\n", "    full_train_df = pd.DataFrame({c_lab: [x for x in fd[c_lab]] for c_lab in fd.keys()})\n", "full_train_df['category'] = full_train_df['category'].map(lambda x: x.decode())\n", "full_train_df['psd'] = full_train_df['psd'].map(lambda x: np.log10(np.mean(x, 1))[30:])\n", "full_train_df.sample(3)"], "outputs": []}, {"cell_type": "code", "metadata": {}, "execution_count": null, "source": ["from sklearn.model_selection import train_test_split\n", "from sklearn.preprocessing import LabelEncoder\n", "cat_enc = LabelEncoder()\n", "cat_enc.fit(full_train_df['category'])\n", "train_df, test_df = train_test_split(full_train_df, \n", "                                     test_size = 0.4,\n", "                                    random_state = 2018,\n", "                                    stratify = full_train_df['category'])\n", "print('Train', train_df.shape[0], \n", "      'Test', test_df.shape[0])"], "outputs": []}, {"cell_type": "code", "metadata": {}, "execution_count": null, "source": ["from sklearn.ensemble import ExtraTreesClassifier\n", "\n", "rfc = ExtraTreesClassifier(n_estimators = 25)\n", "rfc.fit(np.stack(train_df['psd'], 0), \n", "        train_df['category'])"], "outputs": []}, {"cell_type": "code", "metadata": {}, "execution_count": null, "source": ["from sklearn.metrics import classification_report, confusion_matrix\n", "out_pred = rfc.predict(np.stack(test_df['psd'], 0))\n", "print(classification_report(test_df['category'], \n", "                            out_pred))\n", "plt.matshow(confusion_matrix(test_df['category'], out_pred))"], "outputs": []}, {"cell_type": "markdown", "metadata": {}, "source": ["# Try some AutoML \n", "Some preprocessing steps might improve the results so we see what TPOT suggests"]}, {"cell_type": "code", "metadata": {}, "execution_count": null, "source": ["from tpot import TPOTClassifier\n", "tpt = TPOTClassifier(generations = 3, population_size = 10, max_eval_time_mins = 1, verbosity=1)\n", "tpt.fit(np.stack(train_df['psd'], 0), \n", "        train_df['category'])"], "outputs": []}, {"cell_type": "code", "metadata": {}, "execution_count": null, "source": ["from sklearn.metrics import classification_report, confusion_matrix\n", "out_pred = tpt.predict(np.stack(test_df['psd'], 0))\n", "print(classification_report(test_df['category'], \n", "                            out_pred))\n", "plt.matshow(confusion_matrix(test_df['category'], out_pred))"], "outputs": []}, {"cell_type": "code", "metadata": {}, "execution_count": null, "source": ["list_train = glob(os.path.join('..', 'input', 'sp-society-camera-model-identification', 'train', '*', '*.jpg'))\n", "print('Train Files found', len(list_train), 'first file:', list_train[0])\n", "list_test = glob(os.path.join('..', 'input', 'sp-society-camera-model-identification', '*', '*.tif'))\n", "print('Test Files found', len(list_test), 'first file:', list_test[0])"], "outputs": []}, {"cell_type": "code", "metadata": {"collapsed": true}, "execution_count": null, "source": [], "outputs": []}], "metadata": {"kernelspec": {"display_name": "Python 3", "language": "python", "name": "python3"}, "language_info": {"file_extension": ".py", "codemirror_mode": {"version": 3, "name": "ipython"}, "mimetype": "text/x-python", "nbconvert_exporter": "python", "version": "3.6.4", "name": "python", "pygments_lexer": "ipython3"}}, "nbformat": 4}