{"cells":[{"metadata":{"_uuid":"17afbb3c35ea294eb092c790d633084a73a0e461"},"cell_type":"markdown","source":"### if  you want to understand competition and protein related terminology please refer below kernel ###\n\nhttps://www.kaggle.com/nikitpatel/eda-with-human-protein-information"},{"metadata":{"_uuid":"66564197608a4ecb0221a3022f37581fdfc4d084"},"cell_type":"markdown","source":"### Import Required Libraries ###"},{"metadata":{"trusted":true,"_uuid":"7402b7f9af8118a4d1fbd46731a091d74fe58be2"},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nprint(os.listdir(\"../input\"))\nimport matplotlib.pyplot as plt\nimport cv2\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.model_selection import train_test_split\nimport gc\nfrom keras.preprocessing.image import ImageDataGenerator\n\n#================================\n# import the necessary packages\nfrom keras.models import Sequential\nfrom keras.layers.normalization import BatchNormalization\nfrom keras.layers.convolutional import Conv2D\nfrom keras.layers.convolutional import MaxPooling2D\nfrom keras.layers.core import Activation\nfrom keras.layers.core import Flatten\nfrom keras.layers.core import Dropout\nfrom keras.layers.core import Dense\nfrom keras import backend as K\n\n#================================\n\nimport matplotlib\n#matplotlib.use(\"Agg\")\n \n# import the necessary packages\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.optimizers import Adam\nfrom keras.preprocessing.image import img_to_array\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.model_selection import train_test_split\n#from pyimagesearch.smallervggnet import SmallerVGGNet\nimport matplotlib.pyplot as plt\n#from imutils import paths\nimport numpy as np\n#import argparse\nimport random\nimport pickle\nimport cv2\nimport os","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c065d60c6b53b9397a4ab6ca813e7efbfc4726a0"},"cell_type":"code","source":"filepath0=\"../input/train/\"\nfilepath1=\"../input/test/\"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ef4b0ca48c014af2614d09580ca5a7c83e35c068"},"cell_type":"markdown","source":"## make final data with only green filter images ##\n\n* per sample four color images available but for this whole kernal we are going to use greeen color samples and it is mapped with train.csv \n* after concat we get finalize df"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_df=pd.read_csv(\"../input/train.csv\")\ntrain_image=os.listdir(\"../input/train/\")\ngreenimage= [n for n in train_image if \"green\" in n]\ngdf=pd.DataFrame({\"imagename\":greenimage})\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1dfe56edc9c9fc3faf896f3e3006019a0408edab"},"cell_type":"code","source":"gdf.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9e2b7c24db27bc9c8fc40f6c752a06afabf4f2e4"},"cell_type":"code","source":"dff=pd.concat([train_df,gdf],axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"da85a263aca225091f15025251eabdedc5235319"},"cell_type":"code","source":"dff.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f538ce3876d1cf12fbfc47accf6d553a6c59aeea"},"cell_type":"code","source":"df=dff[0:1000]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bf65645430c60e2c17642cc6e69cc94f191cbceb"},"cell_type":"code","source":"df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a94da04d4844a92d644fce9bc22880cbb4b805a3"},"cell_type":"code","source":"img_height=512\nimg_width=512","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c5d3ae49188e83c4fca75522383ad8760fee6f6b"},"cell_type":"markdown","source":"### Read images and stored in image list ###\n\n* images read using openCV library"},{"metadata":{"trusted":true,"_uuid":"6148a0ba559464202809d3e221a4f79db2e7a864"},"cell_type":"code","source":"image=[]\n#labels = []\nfor i in df['imagename']:\n        images = cv2.imread(filepath0+i,0) \n        images = cv2.resize(images, (img_width, img_height))\n        image.append(images)\n       ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d8e0451c2dc8451c91422058b434d73af88d19c2"},"cell_type":"code","source":"len(image)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2fc5533d9d535b5eee3833791205cc57e13c85c6"},"cell_type":"code","source":"plt.figure(figsize=(12, 12))\nfor i in range(0, 9):\n    plt.subplot(330 + 1 + i)\n    plt.imshow(image[i])\n    plt.gca().get_xaxis().set_ticks([])\n    plt.gca().get_yaxis().set_ticks([])\n   \nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ff96fdaff7c8b801ab7510b542fa41da43fd45d6"},"cell_type":"markdown","source":"### Put Labels in labels list ###"},{"metadata":{"trusted":true,"_uuid":"e10205d10524b1db4abda1513fdd6f59ab76bf47"},"cell_type":"code","source":"labels = []\nfor i in df['Target']:\n    li = list(i.split(\" \")) \n    labels.append(li)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fbd26a39cc1bdb740200a50847009a24ea02bda8"},"cell_type":"code","source":"len(labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"71b5811136ebbbb6f443b57df7ad9ba9c41d1074"},"cell_type":"code","source":"labels[0:5]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cb4b6a8cc58c3cce567da5a399b41b9fb5dd0972"},"cell_type":"markdown","source":"### Convert image and labels list to numpy array ###"},{"metadata":{"trusted":true,"_uuid":"9bdbf08dc1e7f3734a78f4ed0f97a7a1bcec497a"},"cell_type":"code","source":"image = np.array(image)\nlabels = np.array(labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"33815a65bb8935613199d24db7e3037e502e4486"},"cell_type":"code","source":"# binarize the labels using scikit-learn's special multi-label\n# binarizer implementation\nprint(\"[INFO] class labels:\")\nmlb = MultiLabelBinarizer()\nlabels = mlb.fit_transform(labels)\n \n# loop over each of the possible class labels and show them\nfor (i, label) in enumerate(mlb.classes_):\n    print(\"{}. {}\".format(i + 1, label))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8c2ce6f0b480a2aefad381d0cdbc16bff7048220"},"cell_type":"code","source":"gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ae1ce7056f8357ab5df25804b82cbc462719e420"},"cell_type":"markdown","source":"### Data splited for train and validation ###\n\n* reshape images"},{"metadata":{"trusted":true,"_uuid":"38816d57312b9c1639790d67d14dc0b91db25d8d"},"cell_type":"code","source":"(trainX, testX, trainY, testY) = train_test_split(image,labels, test_size=0.2, random_state=42)\n\ntrainX = trainX.reshape(trainX.shape[0], img_width, img_height,1) \ntestX = testX.reshape(testX.shape[0], img_width, img_height,1) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b1f8b0ce423bc4e662215077ec1d4ed70c5ac9f1"},"cell_type":"code","source":"trainY.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"32bea8d116eb2e53dab995e5a43caeb1c99719e6"},"cell_type":"code","source":"aug = ImageDataGenerator()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9e5c1a5d320de71f1a084aa2bdd492d9687b0bee"},"cell_type":"code","source":"EPOCHS = 20\nINIT_LR = 1e-3\nBS = 32\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"22143ee9ea48b68730f44977f1d9b4aeb62cc139"},"cell_type":"markdown","source":"### Make keras deep learning model for image classification ###"},{"metadata":{"trusted":true,"_uuid":"c26c5ea49472a004f3aeca8b529ec0403abe9ed9"},"cell_type":"code","source":"height=512\nwidth=512\ndepth=1\nchanDim = -1\nclasses=28, \nfinalAct=\"sigmoid\"\n\n\ninputShape = (height, width, depth)\n\nmodel = Sequential()\n# CONV => RELU => POOL\nmodel.add(Conv2D(32, (3, 3), padding=\"same\",\ninput_shape=inputShape))\nmodel.add(Activation(\"relu\"))\nmodel.add(BatchNormalization(axis=chanDim))\nmodel.add(MaxPooling2D(pool_size=(3, 3)))\nmodel.add(Dropout(0.25))\n# (CONV => RELU) * 2 => POOL\nmodel.add(Conv2D(64, (3, 3), padding=\"same\"))\nmodel.add(Activation(\"relu\"))\nmodel.add(BatchNormalization(axis=chanDim))\nmodel.add(Conv2D(64, (3, 3), padding=\"same\"))\nmodel.add(Activation(\"relu\"))\nmodel.add(BatchNormalization(axis=chanDim))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n# (CONV => RELU) * 2 => POOL\nmodel.add(Conv2D(128, (3, 3), padding=\"same\"))\nmodel.add(Activation(\"relu\"))\nmodel.add(BatchNormalization(axis=chanDim))\nmodel.add(Conv2D(128, (3, 3), padding=\"same\"))\nmodel.add(Activation(\"relu\"))\nmodel.add(BatchNormalization(axis=chanDim))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\n# first (and only) set of FC => RELU layers\nmodel.add(Flatten())\nmodel.add(Dense(1024))\nmodel.add(Activation(\"relu\"))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.5))\n\n# use a *softmax* activation for single-label classification\n# and *sigmoid* activation for multi-label classification\nmodel.add(Dense(27))\nmodel.add(Activation(finalAct))\n \nopt = Adam(lr=INIT_LR, decay=INIT_LR / EPOCHS)\n\nmodel.compile(loss=\"binary_crossentropy\", optimizer=opt,metrics=[\"accuracy\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f692eb27132b316a08261cfea96705fd6d567dc8"},"cell_type":"code","source":"H = model.fit_generator(aug.flow(trainX, trainY, batch_size=1),validation_data=(testX, testY),steps_per_epoch=len(trainX) // BS,epochs=EPOCHS, verbose=1)\n\n#H=model.fit(trainX, trainY, batch_size=BS,validation_data=(testX, testY),steps_per_epoch=len(trainX) // BS,epochs=EPOCHS)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fdc5074155aa208f456c03c80c280577b4ba2085"},"cell_type":"markdown","source":"### train vs validation acuraccy and loss ###**"},{"metadata":{"trusted":true,"_uuid":"bea32dae2ddfdebef6e2f9b7c21f3a24517834e0"},"cell_type":"code","source":"plt.plot(H.history['acc'])\nplt.plot(H.history['val_acc'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'Validation'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"67bdbc7bf987a3918d2b6ffc3274825c3e33a999"},"cell_type":"code","source":"\nplt.plot(H.history['loss'])\nplt.plot(H.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'validaiton'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1f8a92c444421bb2bb9fe3dfe9fff6f093fe72bf"},"cell_type":"markdown","source":"### Read test data greeen filter images ###"},{"metadata":{"trusted":true,"_uuid":"4598e74069b5ddc5efde83a89320de0b927996a4"},"cell_type":"code","source":"sub_df=pd.read_csv(\"../input/sample_submission.csv\")\ntest_image=os.listdir(\"../input/test/\")\ntestgreenimage= [n for n in test_image if \"green\" in n]\n#testgdf=pd.DataFrame({\"imagename\":testgreenimage})\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6d79a6cdcc4b5307bc87be94d659390f618a7a93"},"cell_type":"code","source":"sub_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"586f7364151b592c94f45383d7894ebfc676a51a"},"cell_type":"code","source":"len(testgreenimage)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e907e736046b809cc9329e5c5d6c0b6a4c503fd3"},"cell_type":"code","source":"X_test=[]\nY_test=[]\n\nfor i in testgreenimage[20:21]:\n    image = cv2.imread(filepath1+i,0) \n    images = cv2.resize(image, (img_width, img_height))\n    X_test.append(images)\n    Y_test.append(images)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d8a58b46acdfd4e3693478049ef47d1d5d8e30ad"},"cell_type":"markdown","source":"### Convert test data images to numpy array and then reshaped ###"},{"metadata":{"trusted":true,"_uuid":"7d8984fff365f3edc5d18044209db496e094f595"},"cell_type":"code","source":"X_test=np.array(X_test)\nX_test = X_test.reshape(X_test.shape[0], img_width, img_height,1) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1499ee261290f3a561d6c7249b142a311ab59af2"},"cell_type":"code","source":"X_test.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"782f95079e98b2e36b8a3f5f70beb56211c56ced"},"cell_type":"markdown","source":"### Predict selected one test image result and stored into proba ###"},{"metadata":{"trusted":true,"_uuid":"cdb7bc6f3f6eb909963d1d8af8114386d631c5af"},"cell_type":"code","source":"proba = model.predict(X_test)[0]\nproba.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"80898bec33fdd23913496a2ff5794ac8a7252d98"},"cell_type":"code","source":"idxs = np.argsort(proba)[::-1][:2]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"31724230228770fd9313f15336228b3585c898af"},"cell_type":"markdown","source":"### Selected image fall in 25th class there where result is 99% nearer to 100%"},{"metadata":{"trusted":true,"_uuid":"124b4862ba52b6888aa8dfc2f14fb2468aa6076f"},"cell_type":"code","source":"# loop over the indexes of the high confidence class labels\nfor (i, j) in enumerate(idxs):\n\n    label = \"{}: {:}%\".format(mlb.classes_[j], proba[j] * 100)\n\n# show the probabilities for each of the individual labels\nfor (label, p) in zip(mlb.classes_, proba):\n    print(\"{}: {:}%\".format(label, int(p * 100)))\n\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"53b2332688d07dedd763810c49a1fdb7f558716b"},"cell_type":"markdown","source":"### this notebook applied on green color channel we can apply it to more three channel using same approach ###"},{"metadata":{"trusted":true,"_uuid":"87d3ec12e7b2d5adfe5cd54fdd3ff91e5580ccb5"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}