{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#使用するライブラリのインポート\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport cv2\nimport zipfile\nimport gc\nimport sys\nimport csv\n\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\n\n\n#機械学習に使用するライブラリ\nimport keras\nfrom tensorflow.keras import optimizers\nfrom keras.utils import np_utils\nfrom sklearn.model_selection import train_test_split\n\nfrom keras.models import Sequential\nfrom keras.models import model_from_json\nfrom keras.models import Model\nfrom keras.layers import Input, Activation, merge, Dense, Flatten, Dropout\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-21T15:26:28.980792Z","iopub.execute_input":"2022-07-21T15:26:28.982049Z","iopub.status.idle":"2022-07-21T15:26:34.926384Z","shell.execute_reply.started":"2022-07-21T15:26:28.981906Z","shell.execute_reply":"2022-07-21T15:26:34.925084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ファイルの入出力\nwith zipfile.ZipFile(\"../input/dogs-vs-cats-redux-kernels-edition/train.zip\",\"r\") as z:\n    z.extractall(\".\")\n    \nwith zipfile.ZipFile(\"../input/dogs-vs-cats-redux-kernels-edition/test.zip\",\"r\") as z:\n    z.extractall(\".\")\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:26:34.930637Z","iopub.execute_input":"2022-07-21T15:26:34.931924Z","iopub.status.idle":"2022-07-21T15:26:58.789579Z","shell.execute_reply.started":"2022-07-21T15:26:34.931883Z","shell.execute_reply":"2022-07-21T15:26:58.788363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#学習データのラベル付けを行う\ntrain_path = \"./train\"\ntest_path = \"./test\"\nt_data = []\nt_label = []\nct = 0\ndg = 0\n\nconv = lambda category : int(category == 'dog')\n\nfor p in os.listdir(train_path):\n    category = p.split(\".\")[0]\n    category = conv(category)\n    if category == 1:\n        dg = dg + 1\n        if dg < 4000:\n            img_array = cv2.imread(os.path.join(train_path,p))\n            new_img_array = cv2.resize(img_array, dsize=(224, 224))\n            t_data.append(new_img_array)\n            t_label.append(category)\n    else :\n        ct = ct + 1\n        if ct < 4000:\n            img_array = cv2.imread(os.path.join(train_path,p))\n            new_img_array = cv2.resize(img_array, dsize=(224, 224))\n            t_data.append(new_img_array)\n            t_label.append(category)\n            \nprint(ct,dg)\n    \n    \nt_data = np.array(t_data).reshape(-1, 224,224,3)\nt_data = t_data.astype('float32')\nt_data /= 255.0\n\nt_label= np.array(t_label)\n\ndel img_array,new_img_array\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:26:58.790829Z","iopub.execute_input":"2022-07-21T15:26:58.791210Z","iopub.status.idle":"2022-07-21T15:27:18.475801Z","shell.execute_reply.started":"2022-07-21T15:26:58.791173Z","shell.execute_reply":"2022-07-21T15:27:18.474276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_train, X_val, y_train, y_val = train_test_split(t_data, t_label, train_size=0.8, random_state=1)\n\n#del t_data,t_label\n#gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:27:18.479811Z","iopub.execute_input":"2022-07-21T15:27:18.480286Z","iopub.status.idle":"2022-07-21T15:27:18.485921Z","shell.execute_reply.started":"2022-07-21T15:27:18.480241Z","shell.execute_reply":"2022-07-21T15:27:18.484445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#学習モデルの作成\nn_classes = 2\ninput_tensor = Input(shape=(224,224,3))\n#モデルのインポート\n#最終層以外をインポートする\nbase_model = tf.keras.applications.vgg16.VGG16(input_shape = (224,224,3),weights='imagenet', input_tensor=input_tensor,include_top=False)\n#モデル作成\n#VGG16から生成される分類ベクトルを学習\ntop_model = Sequential()\ntop_model.add(Flatten(input_shape=base_model.output_shape[1:]))\ntop_model.add(Dense(256, activation='relu'))\ntop_model.add(Dropout(0.5))\ntop_model.add(Dense(2, activation='sigmoid'))\ntop_model.summary()\n#学習済みモデルと作成したモデルを結合\nmodel = Model(inputs=base_model.input, outputs=top_model(base_model.output))\n\n#学習済みモデルの一部学習層の学習をしないように設定\n#VGGの最終層と出力の全結合層以外の学習をしないように\n#VGGの最終層を学習させるfine-tuning\nfor layer in model.layers[:15]:\n    layer.trainable = False\n\n#モデルの作成\nprint('# layers=', len(model.layers))\n\nmodel.compile(loss='binary_crossentropy',optimizer=optimizers.SGD(lr=1e-4, momentum=0.9),metrics=['accuracy'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:27:18.487864Z","iopub.execute_input":"2022-07-21T15:27:18.488426Z","iopub.status.idle":"2022-07-21T15:27:21.207285Z","shell.execute_reply.started":"2022-07-21T15:27:18.488357Z","shell.execute_reply":"2022-07-21T15:27:21.206126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#レイヤーの確認\n#layer_names = [l.name for l in base_model.layers]\n#idx = layer_names.index('block_6_expand')\n#print(idx)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:27:21.208726Z","iopub.execute_input":"2022-07-21T15:27:21.209503Z","iopub.status.idle":"2022-07-21T15:27:21.215009Z","shell.execute_reply.started":"2022-07-21T15:27:21.209442Z","shell.execute_reply":"2022-07-21T15:27:21.213794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ラベルの設定\n#y_train = np_utils.to_categorical(y_train,n_classes)\n#y_val = np_utils.to_categorical(y_val,n_classes)\nt_label = np_utils.to_categorical(t_label,n_classes)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:27:21.216648Z","iopub.execute_input":"2022-07-21T15:27:21.217273Z","iopub.status.idle":"2022-07-21T15:27:21.224821Z","shell.execute_reply.started":"2022-07-21T15:27:21.217237Z","shell.execute_reply":"2022-07-21T15:27:21.223853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#コールバックの設定\n#es_cb = keras.callbacks.EarlyStopping(monitor='val_loss', patience=0, verbose=0, mode='auto')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:27:21.226371Z","iopub.execute_input":"2022-07-21T15:27:21.226986Z","iopub.status.idle":"2022-07-21T15:27:21.234678Z","shell.execute_reply.started":"2022-07-21T15:27:21.226937Z","shell.execute_reply":"2022-07-21T15:27:21.233531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#学習データで学習(学習データの一部を分離して最適なエポック数を探索する)\n#model.fit(X_train, y_train, batch_size=128, epochs=20, verbose=0, validation_data=(X_val, y_val), callbacks=[es_cb])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:27:21.235727Z","iopub.execute_input":"2022-07-21T15:27:21.235979Z","iopub.status.idle":"2022-07-21T15:27:21.246098Z","shell.execute_reply.started":"2022-07-21T15:27:21.235956Z","shell.execute_reply":"2022-07-21T15:27:21.245153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#学習データで学習\n#model.fit(X_train, y_train, epochs=30, batch_size=16)\nmodel.fit(t_data, t_label, epochs=20, batch_size=1)\n\ndel t_data,t_label\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:27:21.248905Z","iopub.execute_input":"2022-07-21T15:27:21.251288Z","iopub.status.idle":"2022-07-21T15:43:52.899507Z","shell.execute_reply.started":"2022-07-21T15:27:21.251261Z","shell.execute_reply":"2022-07-21T15:43:52.898437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#テストデータで確認\n#score = model.evaluate(X_val, y_val, batch_size=16)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:43:52.900994Z","iopub.execute_input":"2022-07-21T15:43:52.902196Z","iopub.status.idle":"2022-07-21T15:43:52.906565Z","shell.execute_reply.started":"2022-07-21T15:43:52.902156Z","shell.execute_reply":"2022-07-21T15:43:52.905589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import pickle\n#モデルの保存\n#model.save('cnn_dvsc.h5')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:43:52.908057Z","iopub.execute_input":"2022-07-21T15:43:52.909141Z","iopub.status.idle":"2022-07-21T15:43:52.917911Z","shell.execute_reply.started":"2022-07-21T15:43:52.909104Z","shell.execute_reply":"2022-07-21T15:43:52.916912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model = keras.models.load_model('./cnn_dvsc.h5')\n#提出用にデータを整形\ntest_path = \"./test\"\n#テストデータの整形\ns_data = []\ns_id = []\n\nfor p in os.listdir(test_path):\n    sid = p.split(\".\")[0]\n    img_array = cv2.imread(os.path.join(test_path,p))\n    new_img_array = cv2.resize(img_array, dsize=(224, 224))\n    s_data.append(new_img_array)\n    s_id.append(sid)\n    \ns_data = np.array(s_data).reshape(-1, 224,224,3)\ns_data = s_data.astype('float32')\ns_data /= 255.0\n        \n\n#やたらとメモリがリークするので明示的に解放\ndel img_array,new_img_array\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:43:52.919336Z","iopub.execute_input":"2022-07-21T15:43:52.919916Z","iopub.status.idle":"2022-07-21T15:44:27.473815Z","shell.execute_reply.started":"2022-07-21T15:43:52.919878Z","shell.execute_reply":"2022-07-21T15:44:27.472531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s_pred = []\nfor i,img in enumerate(s_data):\n    \n    img = img[None, ...]\n    result = model.predict(img)\n    \n    s_pred.append((result[0][1]*0.999999)+0.0000005)\n    #s_pred.append(result[0][1])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:44:27.476250Z","iopub.execute_input":"2022-07-21T15:44:27.476758Z","iopub.status.idle":"2022-07-21T15:54:14.632524Z","shell.execute_reply.started":"2022-07-21T15:44:27.476714Z","shell.execute_reply":"2022-07-21T15:54:14.631435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#提出用のデータ出力\nansdict = {'id':s_id,'label':s_pred}\ndf = pd.DataFrame(ansdict)\ndf.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T15:54:14.633938Z","iopub.execute_input":"2022-07-21T15:54:14.634292Z","iopub.status.idle":"2022-07-21T15:54:14.702917Z","shell.execute_reply.started":"2022-07-21T15:54:14.634256Z","shell.execute_reply":"2022-07-21T15:54:14.701993Z"},"trusted":true},"execution_count":null,"outputs":[]}]}