{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-15T04:22:06.184299Z","iopub.execute_input":"2022-07-15T04:22:06.185002Z","iopub.status.idle":"2022-07-15T04:22:06.228329Z","shell.execute_reply.started":"2022-07-15T04:22:06.184896Z","shell.execute_reply":"2022-07-15T04:22:06.226737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ライブラリのインポート","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport sys\nimport tensorflow as tf\nimport os\nimport sys\nimport pandas as pd\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPooling2D\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n\n%matplotlib inline\nimport matplotlib.image as img\nimport matplotlib.pyplot as plt\n\nfrom sklearn.metrics import confusion_matrix\nimport plotly.graph_objects as go\nimport itertools\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2022-07-14T04:08:37.809035Z","iopub.execute_input":"2022-07-14T04:08:37.809451Z","iopub.status.idle":"2022-07-14T04:08:37.82662Z","shell.execute_reply.started":"2022-07-14T04:08:37.809417Z","shell.execute_reply":"2022-07-14T04:08:37.825419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# シード値の設定\nseed = 0\nnp.random.seed(seed)\ntf.random.set_seed(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T04:08:44.913472Z","iopub.execute_input":"2022-07-14T04:08:44.914624Z","iopub.status.idle":"2022-07-14T04:08:44.927401Z","shell.execute_reply.started":"2022-07-14T04:08:44.914582Z","shell.execute_reply":"2022-07-14T04:08:44.926172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ファイルの解凍","metadata":{}},{"cell_type":"code","source":"import zipfile\n\nzip_files = ['test', 'train']\n\nfor zip_file in zip_files:\n    with zipfile.ZipFile(\"../input/dogs-vs-cats-redux-kernels-edition/{}.zip\".format(zip_file),\"r\") as z:\n        z.extractall(\".\")\n        print(\"{} unzipped\".format(zip_file))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T04:08:46.770852Z","iopub.execute_input":"2022-07-14T04:08:46.77185Z","iopub.status.idle":"2022-07-14T04:09:08.362824Z","shell.execute_reply.started":"2022-07-14T04:08:46.771807Z","shell.execute_reply":"2022-07-14T04:09:08.361628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## パスの設定","metadata":{}},{"cell_type":"code","source":"IMAGE_FOLDER_PATH = \"../working/train\"\nFILE_NAMES = os.listdir(IMAGE_FOLDER_PATH)\n#データセット内の画像サイズを統一するため使用\nWIDTH = 150\nHEIGHT = 150\n\nlabels = []\nfor i in os.listdir(IMAGE_FOLDER_PATH):\n    labels+=[i]","metadata":{"execution":{"iopub.status.busy":"2022-07-14T04:09:17.03191Z","iopub.execute_input":"2022-07-14T04:09:17.032356Z","iopub.status.idle":"2022-07-14T04:09:17.08606Z","shell.execute_reply.started":"2022-07-14T04:09:17.032318Z","shell.execute_reply":"2022-07-14T04:09:17.085207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## データフレームの作成","metadata":{}},{"cell_type":"code","source":"targets = list()\nfull_paths = list()\ntrain_cats_dir = list()\ntrain_dogs_dir = list()\n\nfor file_name in FILE_NAMES:\n    target = file_name.split(\".\")[0] # target name\n    full_path = os.path.join(IMAGE_FOLDER_PATH, file_name)\n    \n    if(target == \"dog\"):\n        train_dogs_dir.append(full_path)\n    if(target == \"cat\"):\n        train_cats_dir.append(full_path)\n    \n    full_paths.append(full_path)\n    targets.append(target)\n\ndataset = pd.DataFrame() # make dataframe\ndataset['image_path'] = full_paths # file path\ndataset['target'] = targets # file's target","metadata":{"execution":{"iopub.status.busy":"2022-07-14T04:09:21.806186Z","iopub.execute_input":"2022-07-14T04:09:21.807415Z","iopub.status.idle":"2022-07-14T04:09:21.912607Z","shell.execute_reply.started":"2022-07-14T04:09:21.807374Z","shell.execute_reply":"2022-07-14T04:09:21.911427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## データの分割","metadata":{}},{"cell_type":"code","source":"dataset_train, dataset_test = train_test_split(dataset, test_size=0.5, random_state=seed)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T04:09:27.205907Z","iopub.execute_input":"2022-07-14T04:09:27.206483Z","iopub.status.idle":"2022-07-14T04:09:27.229341Z","shell.execute_reply.started":"2022-07-14T04:09:27.206443Z","shell.execute_reply":"2022-07-14T04:09:27.228112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 訓練データの生成","metadata":{}},{"cell_type":"code","source":"#データ拡張\ntrain_datagen=ImageDataGenerator(\nrotation_range=15,\nrescale=1./255,\nzoom_range=0.2,\nchannel_shift_range=100,\nhorizontal_flip=True,\nwidth_shift_range=0.1,\nheight_shift_range=0.1)\n\n\ntrain_datagenerator=train_datagen.flow_from_dataframe(dataframe=dataset_train,\n                                                     x_col=\"image_path\",\n                                                     y_col=\"target\",\n                                                     target_size=(WIDTH, HEIGHT),\n                                                     class_mode=\"binary\",\n                                                     batch_size=100)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T04:09:31.615191Z","iopub.execute_input":"2022-07-14T04:09:31.615858Z","iopub.status.idle":"2022-07-14T04:09:31.873348Z","shell.execute_reply.started":"2022-07-14T04:09:31.615824Z","shell.execute_reply":"2022-07-14T04:09:31.87232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## テストデータの生成","metadata":{}},{"cell_type":"code","source":"test_datagen = ImageDataGenerator(rescale=1./255)\ntest_datagenerator=test_datagen.flow_from_dataframe(dataframe=dataset_test,\n                                                   x_col=\"image_path\",\n                                                   y_col=\"target\",\n                                                   target_size=(WIDTH, HEIGHT),\n                                                   class_mode=\"binary\",\n                                                   batch_size=100)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T04:09:37.095298Z","iopub.execute_input":"2022-07-14T04:09:37.096026Z","iopub.status.idle":"2022-07-14T04:09:37.176313Z","shell.execute_reply.started":"2022-07-14T04:09:37.095975Z","shell.execute_reply":"2022-07-14T04:09:37.174982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CNNモデルの作成","metadata":{}},{"cell_type":"code","source":"model = Sequential() # implement model layer \nmodel.add(Conv2D(32, kernel_size=(3,3), input_shape=(WIDTH, HEIGHT, 3), activation='relu'))\nmodel.add(Conv2D(64, kernel_size=(3,3), activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size=2))\nmodel.add(Dropout(0.25))\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(1, activation='sigmoid'))\n\nmodel.compile(loss=\"binary_crossentropy\", optimizer='adam', metrics=['accuracy'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T04:09:44.23133Z","iopub.execute_input":"2022-07-14T04:09:44.231711Z","iopub.status.idle":"2022-07-14T04:09:44.479987Z","shell.execute_reply.started":"2022-07-14T04:09:44.231681Z","shell.execute_reply":"2022-07-14T04:09:44.479168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 学習","metadata":{}},{"cell_type":"code","source":"History=model.fit(train_datagenerator,\n                       epochs=30,\n                       validation_data=test_datagenerator,\n                       validation_steps=dataset_test.shape[0]/150,\n                       steps_per_epoch=dataset_train.shape[0]/150)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T04:09:50.435484Z","iopub.execute_input":"2022-07-14T04:09:50.436513Z","iopub.status.idle":"2022-07-14T09:05:59.055611Z","shell.execute_reply.started":"2022-07-14T04:09:50.43647Z","shell.execute_reply":"2022-07-14T09:05:59.051959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 学習の推移グラフ作成","metadata":{}},{"cell_type":"code","source":"acc = History.history['accuracy']\nval_acc = History.history['val_accuracy']\nloss = History.history['loss']\nval_loss = History.history['val_loss']\n\nepochs = range(len(acc))\n\nplt.plot(epochs, acc, 'bo', label='Training accuracy')\nplt.plot(epochs, val_acc, 'b', label='Validation accuracy')\nplt.title('Training and validation accuracy')\nplt.legend()\n\nplt.figure()\n\nplt.plot(epochs, loss, 'go', label='Training Loss')\nplt.plot(epochs, val_loss, 'g', label='Validation Loss')\nplt.title('Training and validation loss')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:23:11.060991Z","iopub.execute_input":"2022-07-14T09:23:11.061549Z","iopub.status.idle":"2022-07-14T09:23:11.586852Z","shell.execute_reply.started":"2022-07-14T09:23:11.061512Z","shell.execute_reply":"2022-07-14T09:23:11.584805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 精度検証","metadata":{}},{"cell_type":"code","source":"test_loss, test_acc = model.evaluate(test_datagenerator, steps=len(test_datagenerator), verbose=1)\nprint('Loss: %.3f' % (test_loss * 100.0))\nprint('Accuracy: %.3f' % (test_acc * 100.0)) ","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:23:21.324555Z","iopub.execute_input":"2022-07-14T09:23:21.324974Z","iopub.status.idle":"2022-07-14T09:24:06.054868Z","shell.execute_reply.started":"2022-07-14T09:23:21.324939Z","shell.execute_reply":"2022-07-14T09:24:06.053606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport itertools\n\npredictions = model.predict(x=test_datagenerator, steps= len(test_datagenerator), verbose=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:24:39.881785Z","iopub.execute_input":"2022-07-14T09:24:39.882256Z","iopub.status.idle":"2022-07-14T09:25:25.267975Z","shell.execute_reply.started":"2022-07-14T09:24:39.882202Z","shell.execute_reply":"2022-07-14T09:25:25.265971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CSVデータ作成","metadata":{}},{"cell_type":"code","source":"def gen_image_label(directory):\n    ''' A generator that yields (label, id, jpg_filename) tuple.'''\n    for root, dirs, files in os.walk(directory):\n        for f in files:\n            _, ext = os.path.splitext(f)\n            if ext != '.jpg':\n                continue\n            basename = os.path.basename(f)\n            splits = basename.split('.')\n            if len(splits) == 3:\n                label, id_, ext = splits\n            else:\n                label = None\n                id_, ext = splits\n            fullname = os.path.join(root, f)\n            yield label, int(id_), fullname","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:25:34.537589Z","iopub.execute_input":"2022-07-14T09:25:34.538082Z","iopub.status.idle":"2022-07-14T09:25:34.550102Z","shell.execute_reply.started":"2022-07-14T09:25:34.538042Z","shell.execute_reply":"2022-07-14T09:25:34.548516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# テストデータのデータフレームを作成\ntest_data_dir = \"../working/test/\"\nlst = list(gen_image_label(test_data_dir))\ntest_df = pd.DataFrame(lst, columns=['label', 'id', 'filename'])\ntest_df = test_df.sort_values(by=['label', 'id'])\ntest_df['label_code'] = test_df.label.map({'cat':0, 'dog':1})\n\ntest_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:25:38.37576Z","iopub.execute_input":"2022-07-14T09:25:38.37615Z","iopub.status.idle":"2022-07-14T09:25:38.567512Z","shell.execute_reply.started":"2022-07-14T09:25:38.376118Z","shell.execute_reply":"2022-07-14T09:25:38.566312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# csvの作成\nresults = pd.DataFrame({'id': pd.Series(test_df.id.values[:predictions.shape[0]]),\n                        'label': pd.Series(predictions.T[0])})\nresults.to_csv('submission.csv', index=False)\nresults.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:25:43.215701Z","iopub.execute_input":"2022-07-14T09:25:43.21629Z","iopub.status.idle":"2022-07-14T09:25:43.254254Z","shell.execute_reply.started":"2022-07-14T09:25:43.216245Z","shell.execute_reply":"2022-07-14T09:25:43.253146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}