{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ライブラリのインポート","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\nos.environ['TF_CPP_MIN_LOG_LEVEL']='2'\n\nimport glob\n\nimport tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Sequential\nimport tensorflow.keras.layers as layers\n\nAUTOTUNE = tf.data.experimental.AUTOTUNE\ntf.get_logger().setLevel(\"ERROR\")\n\nfrom sklearn.model_selection import train_test_split\n\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:36:19.342328Z","iopub.execute_input":"2022-07-15T18:36:19.343257Z","iopub.status.idle":"2022-07-15T18:36:25.304569Z","shell.execute_reply.started":"2022-07-15T18:36:19.343121Z","shell.execute_reply":"2022-07-15T18:36:25.303484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GPUの確認","metadata":{}},{"cell_type":"code","source":"device_name = tf.test.gpu_device_name()\nprint(tf.test.gpu_device_name())","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:36:25.306526Z","iopub.execute_input":"2022-07-15T18:36:25.307203Z","iopub.status.idle":"2022-07-15T18:36:27.790551Z","shell.execute_reply.started":"2022-07-15T18:36:25.307166Z","shell.execute_reply":"2022-07-15T18:36:27.787654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ファイルの解凍","metadata":{}},{"cell_type":"code","source":"import zipfile\nwith zipfile.ZipFile('/kaggle/input/dogs-vs-cats-redux-kernels-edition/train.zip') as zf:\n    zf.extractall()\nwith zipfile.ZipFile('/kaggle/input/dogs-vs-cats-redux-kernels-edition/test.zip') as zf:\n    zf.extractall()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:36:27.792275Z","iopub.execute_input":"2022-07-15T18:36:27.792652Z","iopub.status.idle":"2022-07-15T18:36:45.743554Z","shell.execute_reply.started":"2022-07-15T18:36:27.792617Z","shell.execute_reply":"2022-07-15T18:36:45.742499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 学習データ読み込み","metadata":{}},{"cell_type":"code","source":"import cv2\n\ntraining_data_X = []\ntraining_data_Y = []\nIMG_SIZE = 224\nDIR_PATH = '/kaggle/working/train'\nfor img in os.listdir(DIR_PATH):\n    if 'dog.' == img[:4]:\n        category = 1\n    else:\n        category = 0\n        \n    training_data_X.append(DIR_PATH + \"/\" + img)\n    training_data_Y.append(category)\n\nprint(len(training_data_X))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:47:48.336048Z","iopub.execute_input":"2022-07-15T18:47:48.336403Z","iopub.status.idle":"2022-07-15T18:47:48.375791Z","shell.execute_reply.started":"2022-07-15T18:47:48.336372Z","shell.execute_reply":"2022-07-15T18:47:48.374741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 学習データとバリデーションデータに分割","metadata":{}},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(training_data_X, training_data_Y, test_size = 0.3, random_state=50)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:47:51.109566Z","iopub.execute_input":"2022-07-15T18:47:51.109945Z","iopub.status.idle":"2022-07-15T18:47:51.129608Z","shell.execute_reply.started":"2022-07-15T18:47:51.109916Z","shell.execute_reply":"2022-07-15T18:47:51.128428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def image_load(path, label):\n    image = tf.image.decode_jpeg(tf.io.read_file(path), channels=3)\n    image = tf.image.resize(image, [IMG_SIZE, IMG_SIZE])\n    return image, tf.one_hot(label, 2)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:47:55.775644Z","iopub.execute_input":"2022-07-15T18:47:55.776115Z","iopub.status.idle":"2022-07-15T18:47:55.783067Z","shell.execute_reply.started":"2022-07-15T18:47:55.776074Z","shell.execute_reply":"2022-07-15T18:47:55.782093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_train = tf.data.Dataset.from_tensor_slices((x_train, y_train))\nds_val = tf.data.Dataset.from_tensor_slices((x_val, y_val))\n\nds_train = ds_train.map(image_load)\nds_val = ds_val.map(image_load)\n\nprint('train dataset:',len(ds_train),'validation dataset:',len(ds_val))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:47:58.335790Z","iopub.execute_input":"2022-07-15T18:47:58.336159Z","iopub.status.idle":"2022-07-15T18:47:58.523283Z","shell.execute_reply.started":"2022-07-15T18:47:58.336127Z","shell.execute_reply":"2022-07-15T18:47:58.522215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 64\n\nds_batch_train = ds_train.batch(batch_size=batch_size, drop_remainder=True)\nds_batch_train = ds_batch_train.prefetch(tf.data.AUTOTUNE)\nds_batch_val = ds_val.batch(batch_size=batch_size, drop_remainder=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:48:01.106100Z","iopub.execute_input":"2022-07-15T18:48:01.106792Z","iopub.status.idle":"2022-07-15T18:48:01.114327Z","shell.execute_reply.started":"2022-07-15T18:48:01.106754Z","shell.execute_reply":"2022-07-15T18:48:01.113020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_augmentation = Sequential(\n    [\n        layers.RandomRotation(factor=0.15),\n        layers.RandomTranslation(height_factor=0.1, width_factor=0.1),\n        layers.RandomFlip(),\n        layers.RandomContrast(factor=0.1),\n    ],\n    name=\"img_augmentation\",\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:48:03.147762Z","iopub.execute_input":"2022-07-15T18:48:03.149532Z","iopub.status.idle":"2022-07-15T18:48:03.166119Z","shell.execute_reply.started":"2022-07-15T18:48:03.149483Z","shell.execute_reply":"2022-07-15T18:48:03.165012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# モデルの作成","metadata":{}},{"cell_type":"code","source":"def build_model(num_classes):\n    inputs = layers.Input(shape=(IMG_SIZE, IMG_SIZE, 3))\n    x = img_augmentation(inputs)\n    model = EfficientNetB0(include_top=False, input_tensor=x, weights=\"imagenet\")\n\n    # Freeze the pretrained weights\n    model.trainable = False\n\n    # Rebuild top\n    x = layers.GlobalAveragePooling2D(name=\"avg_pool\")(model.output)\n    x = layers.BatchNormalization()(x)\n\n    top_dropout_rate = 0.2\n    x = layers.Dropout(top_dropout_rate, name=\"top_dropout\")(x)\n    outputs = layers.Dense(num_classes, activation=\"softmax\", name=\"pred\")(x)\n\n    # Compile\n    model = tf.keras.Model(inputs, outputs, name=\"EfficientNet\")\n    optimizer = tf.keras.optimizers.Adam(learning_rate=1e-2)\n    model.compile(\n        optimizer=optimizer, loss=\"categorical_crossentropy\", metrics=[\"accuracy\"]\n    )\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:48:07.353833Z","iopub.execute_input":"2022-07-15T18:48:07.354526Z","iopub.status.idle":"2022-07-15T18:48:07.362776Z","shell.execute_reply.started":"2022-07-15T18:48:07.354488Z","shell.execute_reply":"2022-07-15T18:48:07.361510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 学習","metadata":{}},{"cell_type":"code","source":"model = build_model(num_classes=2)\n\nepochs = 10  \nhistory = model.fit(ds_batch_train, epochs=epochs, validation_data=ds_batch_val, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:48:09.735632Z","iopub.execute_input":"2022-07-15T18:48:09.735995Z","iopub.status.idle":"2022-07-15T18:50:00.388870Z","shell.execute_reply.started":"2022-07-15T18:48:09.735945Z","shell.execute_reply":"2022-07-15T18:50:00.385231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_dict = history.history\n\nloss_values = history_dict['loss']\nval_loss_values = history_dict['val_loss']\n\nepochs = range(1, len(loss_values)+1)\n\nline1 = plt.plot(epochs, val_loss_values, label ='Validation/Test Loss')\nline2 = plt.plot(epochs, loss_values, label='Training Loss')\nplt.setp(line1, linewidth = 2.0, marker='+', markersize=10.0)\nplt.setp(line2, linewidth=2.0, marker='4', markersize=10.0)\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:43:48.419898Z","iopub.execute_input":"2022-07-15T18:43:48.420518Z","iopub.status.idle":"2022-07-15T18:43:48.655335Z","shell.execute_reply.started":"2022-07-15T18:43:48.420479Z","shell.execute_reply":"2022-07-15T18:43:48.654362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_dict = history.history\n\nplt.plot(history_dict[\"accuracy\"])\nplt.plot(history_dict[\"val_accuracy\"])\nplt.title(\"model accuracy\")\nplt.ylabel(\"accuracy\")\nplt.xlabel(\"epoch\")\nplt.legend([\"train\", \"validation\"], loc=\"upper left\")\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:51:43.324339Z","iopub.execute_input":"2022-07-15T18:51:43.324695Z","iopub.status.idle":"2022-07-15T18:51:43.526782Z","shell.execute_reply.started":"2022-07-15T18:51:43.324655Z","shell.execute_reply":"2022-07-15T18:51:43.525834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# テストデータの読み込み","metadata":{}},{"cell_type":"code","source":"testing_data = []\ntesting_id = []\n\nDIR_PATH = '/kaggle/working/test'\nfor img in os.listdir(DIR_PATH):\n    testing_data.append(DIR_PATH + '/' + img)\n    testing_id.append(int(img.split('.')[0]))\n\ndef test_image_load(path, id):\n    image = tf.image.decode_jpeg(tf.io.read_file(path), channels=3)\n    image = tf.image.resize(image, [IMG_SIZE, IMG_SIZE])\n    return image, id\n\nprint(len(testing_data))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:43:48.873369Z","iopub.execute_input":"2022-07-15T18:43:48.874005Z","iopub.status.idle":"2022-07-15T18:43:48.901320Z","shell.execute_reply.started":"2022-07-15T18:43:48.873939Z","shell.execute_reply":"2022-07-15T18:43:48.900236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 予測","metadata":{}},{"cell_type":"code","source":"ds_test = tf.data.Dataset.from_tensor_slices((testing_data, testing_id))\n\nds_test = ds_test.map(test_image_load)\nds_batch_test = ds_test.batch(batch_size=100, drop_remainder=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:43:48.902820Z","iopub.execute_input":"2022-07-15T18:43:48.903448Z","iopub.status.idle":"2022-07-15T18:43:49.044554Z","shell.execute_reply.started":"2022-07-15T18:43:48.903407Z","shell.execute_reply":"2022-07-15T18:43:49.043544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nsubmission = {'id':[],'label':[]}\ndog_prediction = lambda x:x[1]\n\nfor batch in tqdm(ds_batch_test):\n    results = model.predict(batch[0])\n    id = batch[1].numpy()\n    \n    submission['id'].extend(id)\n    submission['label'].extend(map(dog_prediction,results))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:43:49.046159Z","iopub.execute_input":"2022-07-15T18:43:49.046540Z","iopub.status.idle":"2022-07-15T18:44:30.014685Z","shell.execute_reply.started":"2022-07-15T18:43:49.046502Z","shell.execute_reply":"2022-07-15T18:44:30.013447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 結果を保存","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\nsubmission_df=pd.DataFrame(submission)\nsubmission_df.to_csv('my_submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T18:44:30.019284Z","iopub.execute_input":"2022-07-15T18:44:30.019910Z","iopub.status.idle":"2022-07-15T18:44:30.080978Z","shell.execute_reply.started":"2022-07-15T18:44:30.019868Z","shell.execute_reply":"2022-07-15T18:44:30.079845Z"},"trusted":true},"execution_count":null,"outputs":[]}]}