{"nbformat_minor":4,"nbformat":4,"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T01:10:32.062819Z","iopub.execute_input":"2022-07-28T01:10:32.064028Z","iopub.status.idle":"2022-07-28T01:10:32.106758Z","shell.execute_reply.started":"2022-07-28T01:10:32.063901Z","shell.execute_reply":"2022-07-28T01:10:32.105677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport sys\nimport tensorflow as tf\nimport os\nimport sys\nimport pandas as pd\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPooling2D\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n\n%matplotlib inline\nimport matplotlib.image as img\nimport matplotlib.pyplot as plt\n\nfrom sklearn.metrics import confusion_matrix\nimport plotly.graph_objects as go\nimport itertools\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2022-07-28T01:10:38.313046Z","iopub.execute_input":"2022-07-28T01:10:38.313460Z","iopub.status.idle":"2022-07-28T01:10:48.069279Z","shell.execute_reply.started":"2022-07-28T01:10:38.313427Z","shell.execute_reply":"2022-07-28T01:10:48.068133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# シード値の設定\nseed = 0\nnp.random.seed(seed)\ntf.random.set_seed(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T01:11:09.624638Z","iopub.execute_input":"2022-07-28T01:11:09.625964Z","iopub.status.idle":"2022-07-28T01:11:09.631675Z","shell.execute_reply.started":"2022-07-28T01:11:09.625909Z","shell.execute_reply":"2022-07-28T01:11:09.630503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import zipfile\n\nzip_files = ['test', 'train']\n\nfor zip_file in zip_files:\n    with zipfile.ZipFile(\"../input/dogs-vs-cats-redux-kernels-edition/{}.zip\".format(zip_file),\"r\") as z:\n        z.extractall(\".\")\n        print(\"{} unzipped\".format(zip_file))","metadata":{"execution":{"iopub.status.busy":"2022-07-28T01:11:19.747749Z","iopub.execute_input":"2022-07-28T01:11:19.748140Z","iopub.status.idle":"2022-07-28T01:11:42.613602Z","shell.execute_reply.started":"2022-07-28T01:11:19.748108Z","shell.execute_reply":"2022-07-28T01:11:42.611963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_FOLDER_PATH = \"../working/train\"\nFILE_NAMES = os.listdir(IMAGE_FOLDER_PATH)\n#データセット内の画像サイズを統一するため使用\nWIDTH = 150\nHEIGHT = 150\n\nlabels = []\nfor i in os.listdir(IMAGE_FOLDER_PATH):\n    labels+=[i]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T01:17:08.839574Z","iopub.execute_input":"2022-07-28T01:17:08.840756Z","iopub.status.idle":"2022-07-28T01:17:08.891071Z","shell.execute_reply.started":"2022-07-28T01:17:08.840677Z","shell.execute_reply":"2022-07-28T01:17:08.889667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = list()\nfull_paths = list()\ntrain_cats_dir = list()\ntrain_dogs_dir = list()\n\nfor file_name in FILE_NAMES:\n    target = file_name.split(\".\")[0] # target name\n    full_path = os.path.join(IMAGE_FOLDER_PATH, file_name)\n    \n    if(target == \"dog\"):\n        train_dogs_dir.append(full_path)\n    if(target == \"cat\"):\n        train_cats_dir.append(full_path)\n    \n    full_paths.append(full_path)\n    targets.append(target)\n\ndataset = pd.DataFrame() # make dataframe\ndataset['image_path'] = full_paths # file path\ndataset['target'] = targets # file's target","metadata":{"execution":{"iopub.status.busy":"2022-07-28T01:17:19.094728Z","iopub.execute_input":"2022-07-28T01:17:19.095183Z","iopub.status.idle":"2022-07-28T01:17:19.221461Z","shell.execute_reply.started":"2022-07-28T01:17:19.095149Z","shell.execute_reply":"2022-07-28T01:17:19.220154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_train, dataset_test = train_test_split(dataset, test_size=0.5, random_state=seed)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T01:17:38.826366Z","iopub.execute_input":"2022-07-28T01:17:38.827022Z","iopub.status.idle":"2022-07-28T01:17:38.845431Z","shell.execute_reply.started":"2022-07-28T01:17:38.826966Z","shell.execute_reply":"2022-07-28T01:17:38.844416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#データ拡張\ntrain_datagen=ImageDataGenerator(\nrotation_range=15,\nrescale=1./255,\nzoom_range=0.2,\nchannel_shift_range=100,\nhorizontal_flip=True,\nwidth_shift_range=0.1,\nheight_shift_range=0.1)\n\n\ntrain_datagenerator=train_datagen.flow_from_dataframe(dataframe=dataset_train,\n                                                     x_col=\"image_path\",\n                                                     y_col=\"target\",\n                                                     target_size=(WIDTH, HEIGHT),\n                                                     class_mode=\"binary\",\n                                                     batch_size=100)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T01:17:48.807802Z","iopub.execute_input":"2022-07-28T01:17:48.808271Z","iopub.status.idle":"2022-07-28T01:17:49.017997Z","shell.execute_reply.started":"2022-07-28T01:17:48.808206Z","shell.execute_reply":"2022-07-28T01:17:49.016339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_datagen = ImageDataGenerator(rescale=1./255)\ntest_datagenerator=test_datagen.flow_from_dataframe(dataframe=dataset_test,\n                                                   x_col=\"image_path\",\n                                                   y_col=\"target\",\n                                                   target_size=(WIDTH, HEIGHT),\n                                                   class_mode=\"binary\",\n                                                   batch_size=100)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T01:18:02.074180Z","iopub.execute_input":"2022-07-28T01:18:02.074646Z","iopub.status.idle":"2022-07-28T01:18:02.259317Z","shell.execute_reply.started":"2022-07-28T01:18:02.074611Z","shell.execute_reply":"2022-07-28T01:18:02.257493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential() # implement model layer \nmodel.add(Conv2D(32, kernel_size=(3,3), input_shape=(WIDTH, HEIGHT, 3), activation='relu'))\nmodel.add(Conv2D(64, kernel_size=(3,3), activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size=2))\nmodel.add(Dropout(0.25))\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(1, activation='sigmoid'))\n\nmodel.compile(loss=\"binary_crossentropy\", optimizer='adam', metrics=['accuracy'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T01:18:10.906029Z","iopub.execute_input":"2022-07-28T01:18:10.907202Z","iopub.status.idle":"2022-07-28T01:18:11.816115Z","shell.execute_reply.started":"2022-07-28T01:18:10.907162Z","shell.execute_reply":"2022-07-28T01:18:11.814644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"History=model.fit(train_datagenerator,\n                       epochs=30,\n                       validation_data=test_datagenerator,\n                       validation_steps=dataset_test.shape[0]/150,\n                       steps_per_epoch=dataset_train.shape[0]/150)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T01:18:32.120555Z","iopub.execute_input":"2022-07-28T01:18:32.120971Z","iopub.status.idle":"2022-07-28T04:57:32.022584Z","shell.execute_reply.started":"2022-07-28T01:18:32.120941Z","shell.execute_reply":"2022-07-28T04:57:32.019917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = History.history['accuracy']\nval_acc = History.history['val_accuracy']\nloss = History.history['loss']\nval_loss = History.history['val_loss']\n\nepochs = range(len(acc))\n\nplt.plot(epochs, acc, 'bo', label='Training accuracy')\nplt.plot(epochs, val_acc, 'b', label='Validation accuracy')\nplt.title('Training and validation accuracy')\nplt.legend()\n\nplt.figure()\n\nplt.plot(epochs, loss, 'go', label='Training Loss')\nplt.plot(epochs, val_loss, 'g', label='Validation Loss')\nplt.title('Training and validation loss')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T05:35:10.232716Z","iopub.execute_input":"2022-07-28T05:35:10.233577Z","iopub.status.idle":"2022-07-28T05:35:10.770467Z","shell.execute_reply.started":"2022-07-28T05:35:10.233529Z","shell.execute_reply":"2022-07-28T05:35:10.769243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_loss, test_acc = model.evaluate(test_datagenerator, steps=len(test_datagenerator), verbose=1)\nprint('Loss: %.3f' % (test_loss * 100.0))\nprint('Accuracy: %.3f' % (test_acc * 100.0))","metadata":{"execution":{"iopub.status.busy":"2022-07-28T05:35:17.087837Z","iopub.execute_input":"2022-07-28T05:35:17.088335Z","iopub.status.idle":"2022-07-28T05:37:07.341528Z","shell.execute_reply.started":"2022-07-28T05:35:17.088294Z","shell.execute_reply":"2022-07-28T05:37:07.340348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport itertools\n\npredictions = model.predict(x=test_datagenerator, steps= len(test_datagenerator), verbose=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T05:44:28.929797Z","iopub.execute_input":"2022-07-28T05:44:28.930276Z","iopub.status.idle":"2022-07-28T05:46:19.658799Z","shell.execute_reply.started":"2022-07-28T05:44:28.930237Z","shell.execute_reply":"2022-07-28T05:46:19.657606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gen_image_label(directory):\n    ''' A generator that yields (label, id, jpg_filename) tuple.'''\n    for root, dirs, files in os.walk(directory):\n        for f in files:\n            _, ext = os.path.splitext(f)\n            if ext != '.jpg':\n                continue\n            basename = os.path.basename(f)\n            splits = basename.split('.')\n            if len(splits) == 3:\n                label, id_, ext = splits\n            else:\n                label = None\n                id_, ext = splits\n            fullname = os.path.join(root, f)\n            yield label, int(id_), fullname","metadata":{"execution":{"iopub.status.busy":"2022-07-28T05:46:40.234379Z","iopub.execute_input":"2022-07-28T05:46:40.234800Z","iopub.status.idle":"2022-07-28T05:46:40.244625Z","shell.execute_reply.started":"2022-07-28T05:46:40.234767Z","shell.execute_reply":"2022-07-28T05:46:40.243033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# テストデータのデータフレームを作成\ntest_data_dir = \"../working/test/\"\nlst = list(gen_image_label(test_data_dir))\ntest_df = pd.DataFrame(lst, columns=['label', 'id', 'filename'])\ntest_df = test_df.sort_values(by=['label', 'id'])\ntest_df['label_code'] = test_df.label.map({'cat':0, 'dog':1})\n\ntest_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T05:46:43.935171Z","iopub.execute_input":"2022-07-28T05:46:43.935692Z","iopub.status.idle":"2022-07-28T05:46:44.115802Z","shell.execute_reply.started":"2022-07-28T05:46:43.935648Z","shell.execute_reply":"2022-07-28T05:46:44.114253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# csvの作成\nresults = pd.DataFrame({'id': pd.Series(test_df.id.values[:predictions.shape[0]]),\n                        'label': pd.Series(predictions.T[0])})\nresults.to_csv('submission.csv', index=False)\nresults.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T05:46:53.443283Z","iopub.execute_input":"2022-07-28T05:46:53.443695Z","iopub.status.idle":"2022-07-28T05:46:53.503179Z","shell.execute_reply.started":"2022-07-28T05:46:53.443663Z","shell.execute_reply":"2022-07-28T05:46:53.501803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}