{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-15T08:37:23.736444Z","iopub.execute_input":"2022-07-15T08:37:23.736970Z","iopub.status.idle":"2022-07-15T08:37:23.786652Z","shell.execute_reply.started":"2022-07-15T08:37:23.736864Z","shell.execute_reply":"2022-07-15T08:37:23.785661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# データ読み込み","metadata":{}},{"cell_type":"code","source":"import glob\nimport shutil\n\n# /kaggle/input のcsvファイルを kaggle/working にコピー\nshutil.copyfile('/kaggle/input/dogs-vs-cats-redux-kernels-edition/sample_submission.csv',\n                '/kaggle/working/sample_submission.csv')\n\n\n# /kaggle/input のzipファイルを /kaggle/working に解凍\nimport zipfile\nwith zipfile.ZipFile('/kaggle/input/dogs-vs-cats-redux-kernels-edition/train.zip') as existing_zip:\n    existing_zip.extractall()\nwith zipfile.ZipFile('/kaggle/input/dogs-vs-cats-redux-kernels-edition/test.zip') as existing_zip:\n    existing_zip.extractall()\n    \n# /kaggle/working に入っているファイルを確認\nfiles = glob.glob('/kaggle/working/*')\nfor file in files:\n    print(file)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:37:23.791338Z","iopub.execute_input":"2022-07-15T08:37:23.791714Z","iopub.status.idle":"2022-07-15T08:37:40.790236Z","shell.execute_reply.started":"2022-07-15T08:37:23.791679Z","shell.execute_reply":"2022-07-15T08:37:40.789224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 画像ファイル確認","metadata":{}},{"cell_type":"code","source":"# /kaggle/working/trainの中身を確認\nfiles = sorted(glob.glob('/kaggle/working/train/*.jpg')) # 訓練用画像フォルダ\nprint('学習フォルダ内の画像数', len(files))\n\n# /kaggle/working/testの中身を確認\nfiles = sorted(glob.glob('/kaggle/working/test/*.jpg')) # テスト画像フォルダ\nprint('評価フォルダ内の画像数', len(files))\n\n# 学習させる犬と猫の画像数を確認\nn_dog = 0 # 犬の画像数\nn_cat = 0 # 猫の画像数\nfiles = glob.glob('/kaggle/working/train/*') # 学習用フォルダ\nfor file in files:\n    if 'dog.' in file:\n        n_dog += 1\n    elif 'cat.' in file:\n        n_cat += 1\n        \nprint('学習させる画像数...')\nprint('犬の画像数：', n_dog)\nprint('猫の画像数：', n_cat)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:37:40.792264Z","iopub.execute_input":"2022-07-15T08:37:40.792879Z","iopub.status.idle":"2022-07-15T08:37:41.020515Z","shell.execute_reply.started":"2022-07-15T08:37:40.792840Z","shell.execute_reply":"2022-07-15T08:37:41.019354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 犬と猫の画像を別々のフォルダに保存","metadata":{}},{"cell_type":"code","source":"# 犬・猫の画像を，それぞれ保存するフォルダのパスを定義\ndog_dir = '/kaggle/working/train/dog'\ncat_dir = '/kaggle/working/train/cat'\n\n# フォルダの作成\nif not os.path.exists(dog_dir):\n    os.mkdir(dog_dir)\nif not os.path.exists(cat_dir):\n    os.mkdir(cat_dir)\n\n# 犬・猫の画像を，それぞれのフォルダに移動する\nfiles = glob.glob('/kaggle/working/train/*.jpg') # 訓練用フォルダ内の全ファイルパスを取得\nfor file in files:\n    file_name = os.path.basename(file)\n    if 'dog' in file:\n        shutil.move(file,'/kaggle/working/train/dog/' + file_name)\n    else:\n        shutil.move(file,'/kaggle/working/train/cat/' + file_name)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:37:41.022248Z","iopub.execute_input":"2022-07-15T08:37:41.023208Z","iopub.status.idle":"2022-07-15T08:37:41.739003Z","shell.execute_reply.started":"2022-07-15T08:37:41.023181Z","shell.execute_reply":"2022-07-15T08:37:41.738012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 学習中に，学習精度を評価するために，学習データの一部を訓練に使わず，予測に用いる．\n# この予測のことをバリデーションと言い，データをバリデーションデータと言う．\n\nfrom sklearn.model_selection import train_test_split\n\n#バリデーションフォルダのパスを定義\nval = '/kaggle/working/val'\nval_dog = '/kaggle/working/val/dog'\nval_cat = '/kaggle/working/val/cat'\n\n# フォルダの作成\nif not os.path.exists(val):\n    os.mkdir(val)\nif not os.path.exists(val_dog):\n    os.mkdir(val_dog)\nif not os.path.exists(val_cat):\n    os.mkdir(val_cat)\n\n# 学習画像の一部をバリデーションフォルダに移動する(犬)\ndog_files = glob.glob('/kaggle/working/train/dog/*.jpg')\ndog_train, dog_val = train_test_split(dog_files, test_size=0.2, random_state=42)\n# 画像の一部をバリデーションフォルダに移動する\nfor file in dog_val:\n    file_name = os.path.basename(file)\n    shutil.move(file,'/kaggle/working/val/dog/' + file_name)\n    \n# 学習画像の一部をバリデーションフォルダに移動する(猫)\ncat_files = glob.glob('/kaggle/working/train/cat/*.jpg')\ncat_train, cat_val = train_test_split(cat_files, test_size=0.2, random_state=42)\n# 画像の一部をバリデーションフォルダに移動する\nfor file in cat_val:\n    file_name = os.path.basename(file)\n    shutil.move(file,'/kaggle/working/val/cat/' + file_name)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:37:41.741713Z","iopub.execute_input":"2022-07-15T08:37:41.742053Z","iopub.status.idle":"2022-07-15T08:37:42.358739Z","shell.execute_reply.started":"2022-07-15T08:37:41.742018Z","shell.execute_reply":"2022-07-15T08:37:42.357713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# データ拡張","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 学習モデルの定義","metadata":{}},{"cell_type":"code","source":"import tensorflow.keras.layers as layers\nfrom tensorflow.keras import layers, models, optimizers\n\nmodel = models.Sequential()\nmodel.add(layers.Conv2D(32,(3,3),activation='relu',input_shape=(150,150,3)))\nmodel.add(layers.MaxPooling2D((2,2)))\n \nmodel.add(layers.Conv2D(64,(3,3),activation='relu'))\nmodel.add(layers.MaxPooling2D((2,2)))\n \nmodel.add(layers.Conv2D(128,(3,3),activation='relu'))\nmodel.add(layers.MaxPooling2D((2,2)))\n \nmodel.add(layers.Conv2D(128,(3,3),activation='relu'))\nmodel.add(layers.MaxPooling2D((2,2)))\n\n#model.add(layers.Dropout(0.5))\n\nmodel.add(layers.Flatten())\n\nmodel.add(layers.Dense(512,activation='relu'))\nmodel.add(layers.Dense(1,activation='sigmoid'))\n\nmodel.compile(loss='binary_crossentropy',\n             optimizer=optimizers.Adam(learning_rate=1e-4),\n             metrics=['acc'])\n \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:37:42.360308Z","iopub.execute_input":"2022-07-15T08:37:42.360751Z","iopub.status.idle":"2022-07-15T08:37:51.706397Z","shell.execute_reply.started":"2022-07-15T08:37:42.360712Z","shell.execute_reply":"2022-07-15T08:37:51.705437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 学習データの前処理","metadata":{}},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\n# データの正規化 \ntrain_datagen = ImageDataGenerator(rescale=1./255)\nvalidation_datagen = ImageDataGenerator(rescale=1./255)\n# 学習データ\ntrain_dir = ('/kaggle/working/train') \n# 評価データ\nvalidation_dir = ('/kaggle/working/val') \n\ntrain_generator = train_datagen.flow_from_directory(\n    train_dir,\n    target_size=(150,150),\n    batch_size=32,\n    class_mode='binary'\n)\n\nvalidation_generator = validation_datagen.flow_from_directory(\n    validation_dir,\n    target_size=(150,150),\n    batch_size=32,\n    class_mode='binary'\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:37:51.708061Z","iopub.execute_input":"2022-07-15T08:37:51.708740Z","iopub.status.idle":"2022-07-15T08:37:52.488420Z","shell.execute_reply.started":"2022-07-15T08:37:51.708701Z","shell.execute_reply":"2022-07-15T08:37:52.487367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 学習","metadata":{}},{"cell_type":"code","source":"# 学習\nhistory = model.fit(train_generator,\n                    steps_per_epoch=20000/32,\n                    epochs=20,\n                    validation_data=validation_generator,\n                    validation_steps=5000/32)\n# 学習済みモデルの保存\nmodel.save(\"/kaggle/working/dog_cat.h5\")","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:37:52.489825Z","iopub.execute_input":"2022-07-15T08:37:52.490508Z","iopub.status.idle":"2022-07-15T09:01:31.939535Z","shell.execute_reply.started":"2022-07-15T08:37:52.490446Z","shell.execute_reply":"2022-07-15T09:01:31.938540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 学習の様子の可視化","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nmetrics = ['loss', 'acc']  # 使用する評価関数を指定\n \nplt.figure(figsize=(10, 5))  # グラフを表示するスペースを用意\n \nfor i in range(len(metrics)):\n \n    metric = metrics[i]\n \n    plt.subplot(1, 2, i+1)  # figureを1×2のスペースに分け、i+1番目のスペースを使う\n    plt.title(metric)  # グラフのタイトルを表示\n    \n    plt_train = history.history[metric]  # historyから訓練データの評価を取り出す\n    plt_test = history.history['val_' + metric]  # historyからテストデータの評価を取り出す\n    \n    plt.plot(plt_train, label='training')  # 訓練データの評価をグラフにプロット\n    plt.plot(plt_test, label='test')  # テストデータの評価をグラフにプロット\n    plt.legend()  # ラベルの表示\n    \nplt.show()  # グラフの表示","metadata":{"execution":{"iopub.status.busy":"2022-07-15T09:01:31.940984Z","iopub.execute_input":"2022-07-15T09:01:31.941386Z","iopub.status.idle":"2022-07-15T09:01:32.243459Z","shell.execute_reply.started":"2022-07-15T09:01:31.941301Z","shell.execute_reply":"2022-07-15T09:01:32.242540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 予測","metadata":{}},{"cell_type":"code","source":"import re\nimport csv\nfrom keras.preprocessing import image\nfrom tqdm import tqdm\n\n# テスト画像のパスを取得\np = re.compile(r'\\d+')\ntest_list = glob.glob('/kaggle/working/test/*.jpg') # 全画像ファイルのパスを読み込み\ntest_list = sorted(test_list,key=lambda s: int(p.search(s).group())) # ソート\n\n# 学習済みモデルの読み込み\nmodel = models.load_model('/kaggle/working/dog_cat.h5')\n\nimcount = 0# 画像の数をカウント\n\n# csvファイルを開く\nwith open('/kaggle/working/sample_submission.csv', 'w') as f:\n    writer= csv.writer(f, lineterminator='\\n')\n    writer.writerow(['id', 'label'])\n    \n# テスト画像ごとに予測し，予測結果をcsvファイルに書き込んでいく\n    for test in tqdm(test_list): # 1画像ずつ繰り返す\n        \n        # 画像の読み込み\n        img = image.load_img(test, target_size=(150, 150))# 画像を読み込み，150×150にリサイズ\n        \n        # 画像を変形\n        x = image.img_to_array(img) # 画像をnumpy型(行列型)に変更\n        x = np.expand_dims(x, axis=0) # 1次元増やす\n        x = x / 255.0 # 正規化(全値を0～1の数値に変換)←モデルに入力する際はこの形\n        \n        # 予測\n        result_predict = model.predict(x)\n        \n        # 予測結果をcsvファイルに書き込み\n        imcount += 1\n        writer.writerow([imcount, result_predict[0][0]])","metadata":{"execution":{"iopub.status.busy":"2022-07-15T09:01:32.244981Z","iopub.execute_input":"2022-07-15T09:01:32.245569Z","iopub.status.idle":"2022-07-15T09:10:38.070648Z","shell.execute_reply.started":"2022-07-15T09:01:32.245531Z","shell.execute_reply":"2022-07-15T09:10:38.069571Z"},"trusted":true},"execution_count":null,"outputs":[]}]}