{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 概要\nTensorflow, Kerasによる画像分類の方法について説明します。<br> \nこのnotebookでは学習済みモデルを使用して予測結果の提出を行います。<br> \n\n学習 : https://www.kaggle.com/takuyatone/cassava-keras-tf-baseline-training/notebook\n\n### 1. 準備 \n- ファイル構成\n- ライブラリのインポート\n- 設定\n- データの読み込み\n\n### 2. 推論\n- 分類モデルの定義\n- 推論用データセットの作成\n- 評価用データに対しての推論\n- 提出物の作成\n- 評価スコアの改善に向けて","metadata":{}},{"cell_type":"markdown","source":"# 1. 準備 ","metadata":{}},{"cell_type":"markdown","source":"## ファイル構成","metadata":{}},{"cell_type":"code","source":"!ls /kaggle/input/cassava-leaf-disease-classification","metadata":{"execution":{"iopub.status.busy":"2021-07-14T02:24:23.219233Z","iopub.execute_input":"2021-07-14T02:24:23.219616Z","iopub.status.idle":"2021-07-14T02:24:23.875484Z","shell.execute_reply.started":"2021-07-14T02:24:23.219526Z","shell.execute_reply":"2021-07-14T02:24:23.874636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ライブラリのインポート","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import accuracy_score\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.models import Sequential, Model,load_model\nfrom tensorflow.keras.applications.vgg16 import VGG16,preprocess_input\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.keras.layers import Conv2D, MaxPool2D, GlobalAveragePooling2D, Flatten, Dense, Dropout\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2021-07-14T02:24:26.925950Z","iopub.execute_input":"2021-07-14T02:24:26.926297Z","iopub.status.idle":"2021-07-14T02:24:32.654427Z","shell.execute_reply.started":"2021-07-14T02:24:26.926268Z","shell.execute_reply":"2021-07-14T02:24:32.653620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# seed固定\ndef seed_everything(seed=1234):\n    #random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n    \nseed_everything(seed=42)","metadata":{"execution":{"iopub.status.busy":"2021-07-14T02:24:32.656098Z","iopub.execute_input":"2021-07-14T02:24:32.656419Z","iopub.status.idle":"2021-07-14T02:24:32.664970Z","shell.execute_reply.started":"2021-07-14T02:24:32.656382Z","shell.execute_reply":"2021-07-14T02:24:32.664169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GPUの確認\nfrom tensorflow.python.client import device_lib\nprint(device_lib.list_local_devices())","metadata":{"execution":{"iopub.status.busy":"2021-07-14T02:24:32.668449Z","iopub.execute_input":"2021-07-14T02:24:32.668702Z","iopub.status.idle":"2021-07-14T02:24:34.756293Z","shell.execute_reply.started":"2021-07-14T02:24:32.668677Z","shell.execute_reply":"2021-07-14T02:24:34.754977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 設定","metadata":{}},{"cell_type":"code","source":"class CFG:\n    debug=True\n    size=64\n    epochs=8\n    batch_size=64\n    val_batch_size=128\n    seed=42\n    target_size=5\n    target_col='label'\n    n_fold=5\n    trn_fold=[0, 1, 2, 3, 4]","metadata":{"execution":{"iopub.status.busy":"2021-07-14T02:24:34.757992Z","iopub.execute_input":"2021-07-14T02:24:34.758388Z","iopub.status.idle":"2021-07-14T02:24:34.771184Z","shell.execute_reply.started":"2021-07-14T02:24:34.758332Z","shell.execute_reply":"2021-07-14T02:24:34.770462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## データの読み込み","metadata":{}},{"cell_type":"code","source":"# ====================================================\n# Directory settings\n# ====================================================\nif os.path.exists('/kaggle/input'):\n    # kaggle環境\n    DATA_DIR = '/kaggle/input/cassava-leaf-disease-classification/'\nelse:\n    # ローカル環境\n    DATA_DIR = '../../data/raw/'\n    \nOUTPUT_DIR = './'\nif not os.path.exists(OUTPUT_DIR):\n    os.makedirs(OUTPUT_DIR)","metadata":{"execution":{"iopub.status.busy":"2021-07-14T02:24:39.250577Z","iopub.execute_input":"2021-07-14T02:24:39.250917Z","iopub.status.idle":"2021-07-14T02:24:39.257808Z","shell.execute_reply.started":"2021-07-14T02:24:39.250889Z","shell.execute_reply":"2021-07-14T02:24:39.255765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(DATA_DIR + 'train.csv')\ntest = pd.read_csv(DATA_DIR + 'sample_submission.csv')\nlabel_map = pd.read_json(DATA_DIR + 'label_num_to_disease_map.json', \n                         orient='index')\ndisplay(train.head())\ndisplay(test.head())\ndisplay(label_map)","metadata":{"execution":{"iopub.status.busy":"2021-07-14T02:24:39.543384Z","iopub.execute_input":"2021-07-14T02:24:39.543701Z","iopub.status.idle":"2021-07-14T02:24:39.768024Z","shell.execute_reply.started":"2021-07-14T02:24:39.543673Z","shell.execute_reply":"2021-07-14T02:24:39.767360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. 推論","metadata":{}},{"cell_type":"markdown","source":"## 分類モデルの定義","metadata":{}},{"cell_type":"code","source":"def vgg16_model(num_classes=None):\n\n    base_model = VGG16(weights=None, include_top=False, input_shape=(CFG.size, CFG.size, 3), pooling='avg')\n    output = Dense(num_classes, activation='softmax')(base_model.output)\n    model = Model(base_model.input, output)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2021-07-14T02:25:33.088750Z","iopub.execute_input":"2021-07-14T02:25:33.089104Z","iopub.status.idle":"2021-07-14T02:25:33.096570Z","shell.execute_reply.started":"2021-07-14T02:25:33.089066Z","shell.execute_reply":"2021-07-14T02:25:33.095846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 推論用データセットの作成","metadata":{}},{"cell_type":"code","source":"test['label'] = test['label'].astype(str)\n\nmodel = vgg16_model(num_classes=CFG.target_size)\nweights_path = [f'../input/cassava-tf-vgg16/fold-{fold}.h5' for fold in CFG.trn_fold]\ntest_datagen = ImageDataGenerator(rescale=1./255)\ntest_generator = test_datagen.flow_from_dataframe(dataframe = test,\n                                        directory = DATA_DIR + \"test_images\",\n                                        x_col = 'image_id',\n                                        y_col = 'label',\n                                        target_size = (CFG.size, CFG.size),\n                                        color_mode = \"rgb\",\n                                        class_mode = \"categorical\",\n                                        batch_size = CFG.val_batch_size,\n                                        shuffle = False)","metadata":{"execution":{"iopub.status.busy":"2021-07-14T02:26:12.248754Z","iopub.execute_input":"2021-07-14T02:26:12.249089Z","iopub.status.idle":"2021-07-14T02:26:12.861462Z","shell.execute_reply.started":"2021-07-14T02:26:12.249057Z","shell.execute_reply":"2021-07-14T02:26:12.860492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights_path","metadata":{"execution":{"iopub.status.busy":"2021-07-14T02:26:12.868565Z","iopub.execute_input":"2021-07-14T02:26:12.868813Z","iopub.status.idle":"2021-07-14T02:26:12.874350Z","shell.execute_reply.started":"2021-07-14T02:26:12.868786Z","shell.execute_reply":"2021-07-14T02:26:12.873506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 評価用データに対しての推論","metadata":{}},{"cell_type":"code","source":"def inference(model, weights_path, test_generator):\n    \n    preds = []\n    for weight in weights_path:\n        print('Loading best model...')\n        model.load_weights(weight)\n        print('Predicting Test...')\n        y_preds = model.predict(test_generator, verbose=1)\n        preds.append(y_preds)\n    probs = np.mean(preds, axis=0)\n    return probs","metadata":{"execution":{"iopub.status.busy":"2021-07-14T02:26:30.851829Z","iopub.execute_input":"2021-07-14T02:26:30.852202Z","iopub.status.idle":"2021-07-14T02:26:30.859226Z","shell.execute_reply.started":"2021-07-14T02:26:30.852170Z","shell.execute_reply":"2021-07-14T02:26:30.858321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = inference(model, weights_path, test_generator)\npredictions","metadata":{"execution":{"iopub.status.busy":"2021-07-14T02:26:31.785774Z","iopub.execute_input":"2021-07-14T02:26:31.786105Z","iopub.status.idle":"2021-07-14T02:26:39.943491Z","shell.execute_reply.started":"2021-07-14T02:26:31.786075Z","shell.execute_reply":"2021-07-14T02:26:39.942472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 提出物の作成","metadata":{}},{"cell_type":"code","source":"test['label'] = predictions.argmax(1)\ntest[['image_id', 'label']].to_csv(OUTPUT_DIR+'submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-14T02:28:04.386615Z","iopub.execute_input":"2021-07-14T02:28:04.386962Z","iopub.status.idle":"2021-07-14T02:28:04.397247Z","shell.execute_reply.started":"2021-07-14T02:28:04.386930Z","shell.execute_reply":"2021-07-14T02:28:04.396433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[['image_id', 'label']]","metadata":{"execution":{"iopub.status.busy":"2021-07-14T02:28:17.254621Z","iopub.execute_input":"2021-07-14T02:28:17.254949Z","iopub.status.idle":"2021-07-14T02:28:17.268204Z","shell.execute_reply.started":"2021-07-14T02:28:17.254918Z","shell.execute_reply":"2021-07-14T02:28:17.267364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 評価スコアの改善に向けて","metadata":{}},{"cell_type":"markdown","source":"・より高性能なモデルの採用<br>\n・学習時のデータオーギュメンテーション(データ水増し)の変更<br>\n・損失関数・最適化手法の変更<br>\n・推論時のデータオーギュメンテーション(Test Time Augmentation)<br>\n・複数モデルのアンサンブル<br>\n・データ特有の課題への対応<br>","metadata":{}}]}