{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import os\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\n\nimport sklearn.metrics as metrics\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Используем keras из tensorflow\n\nkeras = tf.keras\nlayers = keras.layers\nimage = keras.preprocessing.image\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Некоторые настройки tensorflow нужно делать сразу\n\n# tf.debugging.set_log_device_placement(True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Обзор исходных данных\n\ninp_dir = '/kaggle/input'\n\nprint(\"    Содержимое inp_dir:\")\nprint(os.listdir(inp_dir))\nprint()\n\nhcd_dir = os.path.join(inp_dir, 'histopathologic-cancer-detection')\n\nprint(\"    Содержимое hcd_dir:\")\nprint(os.listdir(hcd_dir))\nprint()\n\ntrn_dir = os.path.join(hcd_dir, 'train')\ntst_dir = os.path.join(hcd_dir, 'test')\n\ntrn_files = os.listdir(trn_dir)\ntst_files = os.listdir(tst_dir)\n\ntrn_files.sort()\ntst_files.sort()\n\nprint(\"len(trn_files):\", len(trn_files))\nprint(\"len(tst_files):\", len(tst_files))\nprint()\n\nprint(\"    Имена файлов:\")\nprint(trn_files[:2])\nprint(tst_files[:2])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trn_df = pd.read_csv(os.path.join(hcd_dir, 'train_labels.csv'))\nprint(\"len(trn_df):\", len(trn_df))\ntrn_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Посмотрим на данные\n\nimg0 = image.load_img(os.path.join(trn_dir, trn_files[0]))\n\nprint(\"type(img0):\", type(img0))\nprint(\"size:\", img0.size)\nprint(\"mode:\", img0.mode)\n\nimg0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\n# Выборочно проверим по 100 файлов\n\nfor k in np.random.choice(len(trn_files), 100):\n    fn = os.path.join(trn_dir, trn_files[k])\n    img = image.load_img(fn)\n    assert img.size == (96, 96) and img.mode == \"RGB\"\n\nfor k in np.random.choice(len(tst_files), 100):\n    fn = os.path.join(tst_dir, tst_files[k])\n    img = image.load_img(fn)\n    assert img.size == (96, 96) and img.mode == \"RGB\"\n'''\nNone","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Сколько нужно RAM\n\narr0 = image.img_to_array(img0)\n\nprint(\"type(arr0):\", type(arr0))\nprint(\"shape:\", arr0.shape)\nprint(\"dtype:\", arr0.dtype)\nprint(\"size:\", arr0.size)\n\ntrn_mem = 4 * arr0.size * len(trn_files)\ntst_mem = 4 * arr0.size * len(tst_files)\n\nprint(\"Memory for all train samples in GB:\", trn_mem/2**30)\nprint(\"Memory for all test samples in GB: \", tst_mem/2**30)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\"\"\"\n# Так как нам выделено всего 16 GB RAM,\n# для обработки всех данных нужно потрудиться,\n# например, использовать генератор \n\nclass TestSequence(keras.utils.Sequence):\n    def init(self, ???):\n        ???\n    \n    def __len__(self):\n        ???\n\n    def __getitem__(self, idx):\n        ???\n\"\"\"\nNone","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Для простоты ограничимся 8GB\n\ndef make_trn_xy(prefix, df, num):\n    assert 4*27648*num < 8*2**30\n    np.random.seed(2020)  # Для воспроизводимости выборки\n    kf_box = np.random.choice(len(df), num, replace=False)\n    x_all = np.empty((num,96,96,3), dtype=np.float32)\n    y_all = np.empty((num,), dtype=np.float32)\n    for k0, kf in enumerate(kf_box):\n        fn = df['id'].values[kf] + '.tif'\n        x_all[k0] = make_array(os.path.join(prefix, fn))\n        y_all[k0] = df.iloc[kf]['label']\n    return x_all, y_all\n\ndef make_tst_x(prefix, fns):\n    num = len(fns)\n    x = np.empty((num,96,96,3), dtype=np.float32)\n    for k0, fn in enumerate(fns):\n        x[k0] = make_array(os.path.join(prefix, fn))\n    return x\n\ndef make_array(filename):\n    img = image.load_img(filename)\n    x = image.img_to_array(img)\n    assert x.shape == (96, 96, 3)\n    x = x / 255.0\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Модель\n# https://blog.keras.io/building-powerful-image-classification-models-using-very-little-data.html\n\nmodel = keras.Sequential()\n\nmodel.add(layers.Conv2D(32, (3, 3), input_shape=(96, 96, 3)))\nmodel.add(layers.Activation('relu'))\nmodel.add(layers.MaxPooling2D(pool_size=(2, 2)))\n\nmodel.add(layers.Conv2D(32, (3, 3)))\nmodel.add(layers.Activation('relu'))\nmodel.add(layers.MaxPooling2D(pool_size=(2, 2)))\n\nmodel.add(layers.Conv2D(64, (3, 3)))\nmodel.add(layers.Activation('relu'))\nmodel.add(layers.MaxPooling2D(pool_size=(2, 2)))\n\nmodel.add(layers.Flatten())  # this converts our 3D feature maps to 1D feature vectors\nmodel.add(layers.Dense(64))\nmodel.add(layers.Activation('relu'))\nmodel.add(layers.Dropout(0.5))\nmodel.add(layers.Dense(1))\nmodel.add(layers.Activation('sigmoid'))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(\n    loss='binary_crossentropy',\n    optimizer=keras.optimizers.Adam(),\n    metrics=['AUC']\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Создаем данные в RAM (дорогая операция)\n\nn_t, n_v = 32000, 8000\nx_all, y_all = make_trn_xy(trn_dir, trn_df, n_t+n_v)\nx_a = x_all[:n_t]\ny_a = y_all[:n_t]\nx_b = x_all[n_t:]\ny_b = y_all[n_t:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\n# Если нужно гарантированно вычислять на определенном устройстве\n\nwith tf.device('/CPU:0'):\n    h2 = model.fit(\n        x_a,\n        y_a,\n        epochs=20,\n    )\n'''\nNone","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"h = model.fit(\n    x_a,\n    y_a,\n    batch_size=64,\n    epochs=8,\n    validation_data=(x_b,y_b),\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"z0 = h.epoch\nz1 = h.history['AUC']\nz2 = h.history['val_AUC']\n\nplt.plot(z0, z1, 'ob')\nplt.plot(z0, z2, 'or')\n\nfor zz in zip(z0, z1, z2):\n    print(\"%2d  %5.3f  %5.3f\" % zz)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Напечатаем\n\nloss, auc = model.evaluate(\n    x_b,\n    y_b\n)\n# print(\"loss: %5.3f\" % loss)\nprint(\"AUC:  %5.3f\" % auc)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Построим ROC\n\np_b = model.predict(x_b)\n\nfpr, tpr, thr = metrics.roc_curve(y_b, p_b)\n\nplt.plot([0, 1], [0, 1], 'k--')\nplt.plot(fpr, tpr)\nplt.xlabel('FPR')\n\nNone","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Confusion matrix\n\ny_pred = np.round(p_b)\ncm = metrics.confusion_matrix(y_b, y_pred)\ntn, fp, fn, tp = cm.ravel()\nprint(\"TN:\", tn)\nprint(\"FP:\", fp)\nprint(\"FN:\", fn)\nprint(\"TP:\", tp)\ncm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Функция создания файла предсказаний\n\ndef make_pred_file(prefix, fns, m, output):\n    with open(output, 'w') as fout:\n        print('id,label', file=fout)\n        k0 = 0\n        k0_report = 0\n        while k0 < len(fns):\n            if k0 >= k0_report:\n                print(\"Start:\", k0)\n                k0_report += 4096\n            k1 = k0 + 64\n            b = fns[k0:k1]\n            x = make_tst_x(prefix, b)\n            p = m.predict(x)\n            for fn, v in zip(b, p):\n                ident = fn[:-4]\n                label = 0 if v < 0.5 else 1\n                print(\"%s,%d\" % (ident,label), file=fout)\n            k0 = k1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Создаем файл для загрузки\n\nmake_pred_file(tst_dir, tst_files, model, \"tmp.csv\")\ntmp_df = pd.read_csv('tmp.csv')\ntmp_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"Ссылка для скачивания:\n<a href=\"tmp.csv\">tmp.csv</a>"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Результаты\n# Private score: 0.7880\n# Public score:  0.8050","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}