{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "import numpy as np\nimport pandas as pd\nfrom time import time\nfrom sklearn.decomposition import RandomizedPCA\nfrom sklearn.grid_search import GridSearchCV\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\nimport matplotlib.pyplot as plt\nimport sklearn\nimport glob, os\n%matplotlib inline\nimport skimage\nfrom skimage.feature import greycomatrix, greycoprops,corner_harris\nfrom skimage.filters import sobel,gaussian\nfrom skimage.color import rgb2gray\nfrom skimage.transform import resize \nfrom sklearn.cross_validation import train_test_split\nfrom sklearn.cross_validation import StratifiedShuffleSplit\n\n"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "drivers = pd.read_csv('../input/driver_imgs_list.csv')\ntrain_files = [f for f in glob.glob(\"../input/train/*/*.jpg\")]\ntest_files = [\"../input/test/\" + f for f in os.listdir(\"../input/test/\")]\nprint(train_files[:10])\nprint(test_files[:10])"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# read training images\nimg_all = []\nimg_all_avg = []\nimg_all_gs = []\ndata_all = []\n\nimg_all_rz = []\nimg_all_avg_rz = []\nimg_all_gs_rz = []\ndata_all_rz = []\ntarget_all = []\nsubject_all = []\nfor i in range(0,len(train_files)):\n    if (i%5000==0):\n        print(str(i) + ' images read')\n    path = train_files[i]\n    #print path\n    im_read = plt.imread(path)\n    im_read_avg = im_read[:,:,0]+im_read[:,:,1]+im_read[:,:,2]\n    img_gray = rgb2gray(im_read)\n    dims = np.shape(img_gray)\n    img_data= np.reshape(img_gray, (dims[0] * dims[1], 1))\n    \n    img_gray_rz = gaussian(resize(img_gray,(84,112)),sigma=1)\n    im_read_rz = resize(im_read,(84,112))\n    im_read_avg_rz = im_read_rz[:,:,0]+im_read_rz[:,:,1]+im_read_rz[:,:,2]\n    dims = np.shape(img_gray_rz)\n    img_data_rz= np.reshape(img_gray_rz, (dims[0] * dims[1], 1))\n    data_all_rz.append(img_data_rz)\n    target_all.append(drivers.loc[i]['classname'])\n    subject_all.append(drivers.loc[i]['subject'])\n"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "## Converting data to NP-array\ndata_all_model = np.asarray(data_all_rz)\ntarget_all = np.asarray(target_all)\nsubject_all =np.asarray(subject_all)\ndata_all_model = data_all_model[:,:,0]\n"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "n_components = 200\nprint(\"Extracting the top %d PCs from %d images\"\n      % (n_components, data_all_model.shape[0]))\nt0 = time()\npca = RandomizedPCA(n_components=n_components, whiten=True).fit(data_all_model)\nprint(\"done in %0.3fs\" % (time() - t0))"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "path = train_files[0]\nprint path\nimg = plt.imread(path)\nimg_rz = resize(img,(84,112))\nimg_rz_gs = rgb2gray(img_rz)\nplt.figure(figsize=(15,20))\nplt.subplot(2,2,1)\nplt.imshow(img)\nplt.subplot(2,2,2)\nplt.imshow(img_rz_gs,cmap='gray')\n\nprint(np.shape(img_rz_gs))\nimg_rz_gs_v = np.reshape(img_rz_gs,(1,9408))\nimg_PC = pca.transform(img_rz_gs_v)\n\nimg_PC2 = pca.inverse_transform(img_PC)\n\nplt.figure(figsize=(15,20))\nplt.subplot(2,2,1)\nplt.imshow(resize(img_PC2,(84,112)),cmap='gray')\nplt.subplot(2,2,2)\nplt.imshow(img_PC_rz,cmap='gray')\n\nnp.shape(img_PC_rz)\nprint img_PC2\nprint img_rz_gs_v"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": ""
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}