{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2\n%env KERAS_BACKEND = tensorflow\n%matplotlib inline \nimport matplotlib.pylab as plt \nimport tensorflow as tf\nimport numpy as np \nfrom keras.preprocessing.image import ImageDataGenerator\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-23T05:39:26.643781Z","iopub.execute_input":"2021-06-23T05:39:26.644323Z","iopub.status.idle":"2021-06-23T05:39:30.649254Z","shell.execute_reply.started":"2021-06-23T05:39:26.644272Z","shell.execute_reply":"2021-06-23T05:39:30.648233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/plant-pathology-2021-fgvc8/train.csv')\ntrain['labels'] = train['labels'].apply(lambda string: string.split(' '))","metadata":{"execution":{"iopub.status.busy":"2021-06-23T05:39:30.651730Z","iopub.execute_input":"2021-06-23T05:39:30.652212Z","iopub.status.idle":"2021-06-23T05:39:30.710321Z","shell.execute_reply.started":"2021-06-23T05:39:30.652178Z","shell.execute_reply":"2021-06-23T05:39:30.709102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen = ImageDataGenerator(\n    rescale = 1./255,\n    preprocessing_function = None,\n    validation_split= 0.1\n)\n\ntrain_data = datagen.flow_from_dataframe(\n    train,\n    directory = '../input/resized-plant2021/img_sz_256',\n    x_col = 'image',\n    y_col = 'labels',\n    subset=\"training\",\n    color_mode=\"rgb\",\n    #target_size = (224,224),\n    class_mode=\"categorical\",\n    batch_size=16,\n    shuffle=False,\n    seed=40,\n)\nvalid_data = datagen.flow_from_dataframe(\n    train,\n    directory = '../input/resized-plant2021/img_sz_256',\n    x_col = 'image',\n    y_col = 'labels',\n    subset=\"validation\",\n    color_mode=\"rgb\",\n    #target_size = (224,224),\n    class_mode=\"categorical\",\n    batch_size=16,\n    shuffle=False,\n    seed=40,\n)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T05:39:30.714082Z","iopub.execute_input":"2021-06-23T05:39:30.714403Z","iopub.status.idle":"2021-06-23T05:40:08.794194Z","shell.execute_reply.started":"2021-06-23T05:39:30.714361Z","shell.execute_reply":"2021-06-23T05:40:08.792955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.applications.resnet50 import ResNet50\nfrom keras.applications.inception_v3 import InceptionV3\nfrom keras.optimizers import SGD\nfrom keras.optimizers import Adam\nbase_model = InceptionV3(\n    include_top=False, weights='../input/inceptionv3/inception_v3_weights_tf_dim_ordering_tf_kernels_notop.h5'\n    )\n'''base_model = ResNet50(\n    include_top=False, # 是否包含最後的全連接層 (fully-connected layer)\n    weights='../input/tf-keras-resnet/resnet50_notop.h5', # None: 權重隨機初始化、'imagenet': 載入預訓練權重\n    )'''\nfrom tensorflow.keras.layers import Dense,MaxPooling2D,GlobalAveragePooling2D,BatchNormalization,Activation\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\n#fully connected layer\nx = Dense(32, activation='relu')(x)\nx = Dense(16, activation='relu')(x)\n# finally, the softmax for the classifier \npredictions = Dense(6, activation='sigmoid')(x)\nmodel = tf.keras.Model(inputs=base_model.input ,outputs = predictions)\nmodel.compile(optimizer= 'SGD',\n              loss='binary_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-06-23T05:40:08.795898Z","iopub.execute_input":"2021-06-23T05:40:08.796601Z","iopub.status.idle":"2021-06-23T05:40:16.848994Z","shell.execute_reply.started":"2021-06-23T05:40:08.796553Z","shell.execute_reply":"2021-06-23T05:40:16.847652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(\n            train_data,  # data from generator\n             #steps_per_epoch=1,    # should be number of batches per epoch\n            epochs=20,\n            validation_data=valid_data,\n            steps_per_epoch=train_data.samples//20,\n            validation_steps=valid_data.samples//20,\n            #validation_steps = 1,\n            verbose=True)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T05:40:16.852070Z","iopub.execute_input":"2021-06-23T05:40:16.852381Z","iopub.status.idle":"2021-06-23T06:13:06.303958Z","shell.execute_reply.started":"2021-06-23T05:40:16.852349Z","shell.execute_reply":"2021-06-23T06:13:06.302962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs=range(len(history.history['accuracy']))\nplt.figure()\nplt.plot(epochs,history.history['accuracy'],'b',label='Training acc')\nplt.plot(epochs,history.history['val_accuracy'],'r',label='Validation acc')\nplt.title('Traing and Validation accuracy')\nplt.legend()\n\nplt.figure()\nplt.plot(epochs,history.history['loss'],'b',label='Training loss')\nplt.plot(epochs,history.history['val_loss'],'r',label='Validation val_loss')\nplt.title('Traing and Validation loss')\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T06:13:06.308179Z","iopub.execute_input":"2021-06-23T06:13:06.308545Z","iopub.status.idle":"2021-06-23T06:13:06.756665Z","shell.execute_reply.started":"2021-06-23T06:13:06.308512Z","shell.execute_reply":"2021-06-23T06:13:06.755621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nimport PIL\n\ntest = pd.read_csv('../input/plant-pathology-2021-fgvc8/sample_submission.csv')\n#將圖片站存至./\nfor img_name in tqdm(test['image']):\n    path = '../input/plant-pathology-2021-fgvc8/test_images/'+str(img_name)\n    with PIL.Image.open(path) as img:\n        img = img.resize((256,256))\n        img.save(f'./{img_name}')","metadata":{"execution":{"iopub.status.busy":"2021-06-23T06:13:06.760327Z","iopub.execute_input":"2021-06-23T06:13:06.760662Z","iopub.status.idle":"2021-06-23T06:13:07.790882Z","shell.execute_reply.started":"2021-06-23T06:13:06.760630Z","shell.execute_reply":"2021-06-23T06:13:07.788586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#讀取./圖片至test_data\ntest_data = datagen.flow_from_dataframe(\n    test,\n    directory = './',\n    x_col=\"image\",\n    y_col= None,\n    color_mode=\"rgb\",\n    #target_size = (224,224),\n    classes=None,\n    class_mode=None,\n    batch_size=16,\n    shuffle=False,\n    seed=40,\n)\n#進行預測\nbest_threshold = 0.4#\npreds = model.predict(test_data)\nprint(preds)\npreds = preds.tolist()\n\nindices = []\nfor pred in preds:\n    temp = []\n    for category in pred:\n        if category>=best_threshold:\n            temp.append(pred.index(category))\n    if temp!=[]:\n        indices.append(temp)\n    else:\n        temp.append(np.argmax(pred))\n        indices.append(temp)\n    \nprint(indices)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T06:13:07.792721Z","iopub.execute_input":"2021-06-23T06:13:07.793205Z","iopub.status.idle":"2021-06-23T06:13:09.590667Z","shell.execute_reply.started":"2021-06-23T06:13:07.793129Z","shell.execute_reply":"2021-06-23T06:13:09.589718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = (train_data.class_indices)\nlabels = dict((v,k) for k,v in labels.items())\nprint(labels)\n\ntestlabels = []\nfor image in indices:\n    temp = []\n    for i in image:\n        temp.append(str(labels[i]))\n    testlabels.append(' '.join(temp))\n\nprint(testlabels)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T06:13:09.594274Z","iopub.execute_input":"2021-06-23T06:13:09.594598Z","iopub.status.idle":"2021-06-23T06:13:09.609001Z","shell.execute_reply.started":"2021-06-23T06:13:09.594566Z","shell.execute_reply":"2021-06-23T06:13:09.607673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\ndelfiles = tf.io.gfile.glob('./*.jpg')\n \nfor file in delfiles:\n    os.remove(file)\n    \nsub = pd.read_csv('../input/plant-pathology-2021-fgvc8/sample_submission.csv')\nsub['labels'] = testlabels\nsub.to_csv('submission.csv', index=False)\nsub","metadata":{"execution":{"iopub.status.busy":"2021-06-23T06:13:09.612134Z","iopub.execute_input":"2021-06-23T06:13:09.612472Z","iopub.status.idle":"2021-06-23T06:13:09.850642Z","shell.execute_reply.started":"2021-06-23T06:13:09.612441Z","shell.execute_reply":"2021-06-23T06:13:09.849273Z"},"trusted":true},"execution_count":null,"outputs":[]}]}