{
 "cells": [
  {
   "cell_type": "markdown",
   "metadata": {
    "_cell_guid": "085d9848-e618-b058-fa96-16302b73dc4f"
   },
   "source": []
  },
  {
   "cell_type": "code",
   "execution_count": 1,
   "metadata": {
    "_cell_guid": "76c6c1c2-ee11-8dcc-76d5-dd0dd98c4d9c"
   },
   "outputs": [],
   "source": [
    "import pandas as pd\n",
    "import numpy as np"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 2,
   "metadata": {
    "_cell_guid": "f006da23-48bd-5539-b720-b4e1ea3719f0"
   },
   "outputs": [],
   "source": [
    "from glob import glob\n",
    "\n",
    "def make_base_df():\n",
    "    base_path = '../input/train'\n",
    "    image_paths = []\n",
    "    for type_base_path in sorted(glob(base_path +'/*')):\n",
    "        image_paths = image_paths + glob(type_base_path + '/*')\n",
    "    df = pd.DataFrame({'path':image_paths})\n",
    "    df['type'] = df.path.map(lambda x: x.split('/')[-2])\n",
    "    df['filetype'] = df.path.map(lambda x: x.split('.')[-1])\n",
    "    df['num_id'] = df.path.map(lambda x:x.split('/')[-1].split('.')[0])\n",
    "    return df"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 3,
   "metadata": {
    "_cell_guid": "12b2c343-7eb6-580c-d246-9ce2996e7f8c"
   },
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "[[[ 15  13  12]\n",
      "  [ 12  11   7]\n",
      "  [ 12  13   9]\n",
      "  ..., \n",
      "  [ 41  29  27]\n",
      "  [ 40  29  25]\n",
      "  [ 40  30  23]]\n",
      "\n",
      " [[ 14  12  11]\n",
      "  [ 13  12   8]\n",
      "  [ 14  15  11]\n",
      "  ..., \n",
      "  [ 42  30  28]\n",
      "  [ 43  32  28]\n",
      "  [ 44  34  27]]\n",
      "\n",
      " [[ 11  12  10]\n",
      "  [ 12  13   9]\n",
      "  [ 14  15  11]\n",
      "  ..., \n",
      "  [ 42  30  28]\n",
      "  [ 43  32  28]\n",
      "  [ 44  34  27]]\n",
      "\n",
      " ..., \n",
      " [[162 115 101]\n",
      "  [163 116 102]\n",
      "  [163 115 103]\n",
      "  ..., \n",
      "  [ 45  26  21]\n",
      "  [ 42  24  17]\n",
      "  [ 43  25  18]]\n",
      "\n",
      " [[162 114 102]\n",
      "  [162 114 102]\n",
      "  [159 114 101]\n",
      "  ..., \n",
      "  [ 43  24  19]\n",
      "  [ 42  24  17]\n",
      "  [ 41  24  15]]\n",
      "\n",
      " [[164 116 104]\n",
      "  [162 114 102]\n",
      "  [159 114 101]\n",
      "  ..., \n",
      "  [ 44  25  20]\n",
      "  [ 42  24  17]\n",
      "  [ 41  24  15]]]\n"
     ]
    }
   ],
   "source": [
    "import cv2\n",
    "\n",
    "df = make_base_df()\n",
    "path = df.path[1]\n",
    "img = cv2.imread(path)\n",
    "print(img)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 4,
   "metadata": {
    "_cell_guid": "d3e62342-ddb2-1e29-f821-b2d61aa75173"
   },
   "outputs": [],
   "source": [
    "import cv2\n",
    "\n",
    "def get_grayscale_img(path, rescale_dim):\n",
    "    img = cv2.imread(path)\n",
    "    rescaled = cv2.resize(img, (rescale_dim, rescale_dim), cv2.INTER_LINEAR)\n",
    "    gray = cv2.cvtColor(rescaled, cv2.COLOR_RGB2GRAY).astype('float')\n",
    "    return gray\n",
    "\n",
    "def save_img(img, num_id, desc):\n",
    "    path = desc+num_id+\".jpg\"\n",
    "    cv2.imwrite(path,img)\n",
    "    return path\n",
    "    \n",
    "def save_grayscale(row, desc):\n",
    "    path = row.path\n",
    "    num_id = row.num_id\n",
    "    gray = get_grayscale_img(path, 100)\n",
    "    return save_img(gray, num_id, desc)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 5,
   "metadata": {
    "_cell_guid": "1c769433-d47b-0157-c1d7-f61999d07e39"
   },
   "outputs": [
    {
     "name": "stderr",
     "output_type": "stream",
     "text": [
      "/opt/conda/lib/python3.6/site-packages/pandas/core/indexing.py:297: SettingWithCopyWarning: \n",
      "A value is trying to be set on a copy of a slice from a DataFrame.\n",
      "Try using .loc[row_indexer,col_indexer] = value instead\n",
      "\n",
      "See the caveats in the documentation: http://pandas.pydata.org/pandas-docs/stable/indexing.html#indexing-view-versus-copy\n",
      "  self.obj[key] = _infer_fill_value(value)\n",
      "/opt/conda/lib/python3.6/site-packages/pandas/core/indexing.py:477: SettingWithCopyWarning: \n",
      "A value is trying to be set on a copy of a slice from a DataFrame.\n",
      "Try using .loc[row_indexer,col_indexer] = value instead\n",
      "\n",
      "See the caveats in the documentation: http://pandas.pydata.org/pandas-docs/stable/indexing.html#indexing-view-versus-copy\n",
      "  self.obj[item] = s\n",
      "/opt/conda/lib/python3.6/site-packages/pandas/core/indexing.py:141: SettingWithCopyWarning: \n",
      "A value is trying to be set on a copy of a slice from a DataFrame\n",
      "\n",
      "See the caveats in the documentation: http://pandas.pydata.org/pandas-docs/stable/indexing.html#indexing-view-versus-copy\n",
      "  self._setitem_with_indexer(indexer, value)\n",
      "/opt/conda/lib/python3.6/site-packages/ipykernel/__main__.py:5: SettingWithCopyWarning: \n",
      "A value is trying to be set on a copy of a slice from a DataFrame\n",
      "\n",
      "See the caveats in the documentation: http://pandas.pydata.org/pandas-docs/stable/indexing.html#indexing-view-versus-copy\n"
     ]
    },
    {
     "data": {
      "text/html": [
       "<div>\n",
       "<table border=\"1\" class=\"dataframe\">\n",
       "  <thead>\n",
       "    <tr style=\"text-align: right;\">\n",
       "      <th></th>\n",
       "      <th>path</th>\n",
       "      <th>type</th>\n",
       "      <th>filetype</th>\n",
       "      <th>num_id</th>\n",
       "      <th>gray_path</th>\n",
       "    </tr>\n",
       "  </thead>\n",
       "  <tbody>\n",
       "    <tr>\n",
       "      <th>0</th>\n",
       "      <td>../input/train/Type_1/0.jpg</td>\n",
       "      <td>Type_1</td>\n",
       "      <td>jpg</td>\n",
       "      <td>0</td>\n",
       "      <td>gray_0.jpg</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>1</th>\n",
       "      <td>../input/train/Type_1/10.jpg</td>\n",
       "      <td>Type_1</td>\n",
       "      <td>jpg</td>\n",
       "      <td>10</td>\n",
       "      <td>gray_10.jpg</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>2</th>\n",
       "      <td>../input/train/Type_1/1013.jpg</td>\n",
       "      <td>Type_1</td>\n",
       "      <td>jpg</td>\n",
       "      <td>1013</td>\n",
       "      <td>gray_1013.jpg</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>3</th>\n",
       "      <td>../input/train/Type_1/1014.jpg</td>\n",
       "      <td>Type_1</td>\n",
       "      <td>jpg</td>\n",
       "      <td>1014</td>\n",
       "      <td>gray_1014.jpg</td>\n",
       "    </tr>\n",
       "    <tr>\n",
       "      <th>4</th>\n",
       "      <td>../input/train/Type_1/1019.jpg</td>\n",
       "      <td>Type_1</td>\n",
       "      <td>jpg</td>\n",
       "      <td>1019</td>\n",
       "      <td>gray_1019.jpg</td>\n",
       "    </tr>\n",
       "  </tbody>\n",
       "</table>\n",
       "</div>"
      ],
      "text/plain": [
       "                             path    type filetype num_id      gray_path\n",
       "0     ../input/train/Type_1/0.jpg  Type_1      jpg      0     gray_0.jpg\n",
       "1    ../input/train/Type_1/10.jpg  Type_1      jpg     10    gray_10.jpg\n",
       "2  ../input/train/Type_1/1013.jpg  Type_1      jpg   1013  gray_1013.jpg\n",
       "3  ../input/train/Type_1/1014.jpg  Type_1      jpg   1014  gray_1014.jpg\n",
       "4  ../input/train/Type_1/1019.jpg  Type_1      jpg   1019  gray_1019.jpg"
      ]
     },
     "execution_count": 5,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    "df = make_base_df()\n",
    "small_df = df[:100]\n",
    "desc = 'gray_'\n",
    "for index, row in small_df.iterrows():\n",
    "    small_df.loc[index,'gray_path'] = save_grayscale(row, desc)\n",
    "    \n",
    "small_df.head()"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 6,
   "metadata": {
    "_cell_guid": "91ab5ce0-8693-2abc-986e-dc262e00afaa"
   },
   "outputs": [
    {
     "ename": "SyntaxError",
     "evalue": "invalid syntax (<ipython-input-6-43385c974fb5>, line 5)",
     "output_type": "error",
     "traceback": [
      "\u001b[0;36m  File \u001b[0;32m\"<ipython-input-6-43385c974fb5>\"\u001b[0;36m, line \u001b[0;32m5\u001b[0m\n\u001b[0;31m    for\u001b[0m\n\u001b[0m        ^\u001b[0m\n\u001b[0;31mSyntaxError\u001b[0m\u001b[0;31m:\u001b[0m invalid syntax\n"
     ]
    }
   ],
   "source": [
    "import matplotlib.pyplot as plt\n",
    "\n",
    "df = make_base_df()\n",
    "small_df = df[:100]\n",
    "for \n",
    "path = df.path[1]\n",
    "grey = get_grayscale_img(path, 100)\n",
    "cv2.imwrite(\"grey.jpg\", grey)\n",
    "plt.imshow(plt.imread(\"grey.jpg\"))"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 7,
   "metadata": {
    "_cell_guid": "2f77dd83-b374-06bf-49d3-780f3e1cea62"
   },
   "outputs": [
    {
     "ename": "NameError",
     "evalue": "name 'plt' is not defined",
     "output_type": "error",
     "traceback": [
      "\u001b[0;31m---------------------------------------------------------------------------\u001b[0m",
      "\u001b[0;31mNameError\u001b[0m                                 Traceback (most recent call last)",
      "\u001b[0;32m<ipython-input-7-0079abc176f7>\u001b[0m in \u001b[0;36m<module>\u001b[0;34m()\u001b[0m\n\u001b[0;32m----> 1\u001b[0;31m \u001b[0mplt\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mimshow\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mplt\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mimread\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mpath\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m",
      "\u001b[0;31mNameError\u001b[0m: name 'plt' is not defined"
     ]
    }
   ],
   "source": [
    "plt.imshow(plt.imread(path))"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 8,
   "metadata": {
    "_cell_guid": "189e709e-8714-ea7e-08b1-8b5f34484995"
   },
   "outputs": [
    {
     "ename": "NameError",
     "evalue": "name 'plt' is not defined",
     "output_type": "error",
     "traceback": [
      "\u001b[0;31m---------------------------------------------------------------------------\u001b[0m",
      "\u001b[0;31mNameError\u001b[0m                                 Traceback (most recent call last)",
      "\u001b[0;32m<ipython-input-8-e0d828a3e6da>\u001b[0m in \u001b[0;36m<module>\u001b[0;34m()\u001b[0m\n\u001b[0;32m----> 1\u001b[0;31m \u001b[0mplt\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mimshow\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mplt\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mimread\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m\"grey.jpg\"\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m",
      "\u001b[0;31mNameError\u001b[0m: name 'plt' is not defined"
     ]
    }
   ],
   "source": [
    "plt.imshow(plt.imread(\"grey.jpg\"))"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 9,
   "metadata": {
    "_cell_guid": "d871da50-0ab0-389b-e229-4718f363a55f"
   },
   "outputs": [],
   "source": [
    "from sklearn.ensemble import RandomForestClassifier as RFC\n",
    "\n",
    "forest = RFC(n_jobs=2,n_estimators=50)\n",
    "\n"
   ]
  }
 ],
 "metadata": {
  "_change_revision": 518,
  "_is_fork": false,
  "kernelspec": {
   "display_name": "Python 3",
   "language": "python",
   "name": "python3"
  },
  "language_info": {
   "codemirror_mode": {
    "name": "ipython",
    "version": 3
   },
   "file_extension": ".py",
   "mimetype": "text/x-python",
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.6.0"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 0
}
