{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport datetime\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nimport tensorflow as tf\nprint(\"TensorFlow version:\", tf.__version__)\nfrom tensorflow.keras import models, layers\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.optimizers import Adam\n\n# ignoring warnings\nimport warnings\nwarnings.simplefilter(\"ignore\")\n\nimport os, cv2, json\nfrom PIL import Image\n\nimport random\nimport gc # for garbage cleaning","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:12:37.142896Z","iopub.execute_input":"2022-03-27T09:12:37.143378Z","iopub.status.idle":"2022-03-27T09:12:43.845395Z","shell.execute_reply.started":"2022-03-27T09:12:37.143281Z","shell.execute_reply":"2022-03-27T09:12:43.844408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:    \n    # config\n    WORK_DIR = '../input/ultra-mnist/'\n    IMG_FOLDER = '../input/ultramnist-resized-512/'\n    BATCH_SIZE = 1\n    EPOCHS = 30\n    IMG_SIZE = 512\n    TARGET_SIZE = 28\n    # ResNet\n    RESNET_POOLING_AVERAGE = 'avg'\n    DENSE_LAYER_ACTIVATION = 'softmax'\n    OBJECTIVE_FUNCTION = 'categorical_crossentropy'\n    # Path\n    TRAIN_DIRECTORY = os.path.join(IMG_FOLDER,'train_img')","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:12:43.848236Z","iopub.execute_input":"2022-03-27T09:12:43.848435Z","iopub.status.idle":"2022-03-27T09:12:43.858079Z","shell.execute_reply.started":"2022-03-27T09:12:43.848411Z","shell.execute_reply":"2022-03-27T09:12:43.855024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed: int = 42) -> None:\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    tf.random.set_seed(seed)\n       \nseed_everything(42)","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:12:43.860841Z","iopub.execute_input":"2022-03-27T09:12:43.861075Z","iopub.status.idle":"2022-03-27T09:12:43.870813Z","shell.execute_reply.started":"2022-03-27T09:12:43.861026Z","shell.execute_reply":"2022-03-27T09:12:43.870149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('../input/ultra-mnist/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:12:43.873243Z","iopub.execute_input":"2022-03-27T09:12:43.873626Z","iopub.status.idle":"2022-03-27T09:12:43.908051Z","shell.execute_reply.started":"2022-03-27T09:12:43.873588Z","shell.execute_reply":"2022-03-27T09:12:43.907324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model():\n    conv_base = ResNet50(include_top=False,\n                     weights=\"imagenet\", pooling=CFG.RESNET_POOLING_AVERAGE)\n    model = conv_base.output\n    \n    model = layers.Dropout(.8)(model)\n    \n    model = layers.Dense(28, activation = \"softmax\")(model)\n    model = models.Model(conv_base.input, model)\n\n    model.compile(optimizer = Adam(lr = 0.001),\n                  loss = \"sparse_categorical_crossentropy\",\n                  metrics = [\"acc\"])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:12:43.909971Z","iopub.execute_input":"2022-03-27T09:12:43.910513Z","iopub.status.idle":"2022-03-27T09:12:43.918147Z","shell.execute_reply.started":"2022-03-27T09:12:43.910472Z","shell.execute_reply":"2022-03-27T09:12:43.917370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = get_model()\nmodel.load_weights('../input/keras-resnet-ultramnist-training/model_weights.h5')\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:12:43.921147Z","iopub.execute_input":"2022-03-27T09:12:43.921543Z","iopub.status.idle":"2022-03-27T09:12:50.083498Z","shell.execute_reply.started":"2022-03-27T09:12:43.921513Z","shell.execute_reply":"2022-03-27T09:12:50.082788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from tensorflow.keras.applications.resnet50 import preprocess_input\n# from keras.preprocessing.image import ImageDataGenerator\n# data_generator = ImageDataGenerator(preprocessing_function=preprocess_input)\n# test_generator = data_generator.flow_from_directory(\n#     directory = '../input/ultramnist-resized-512/test_img',\n#     target_size = (CFG.IMG_SIZE, CFG.IMG_SIZE),\n#     batch_size = CFG.BATCH_SIZE,\n#     class_mode = None,\n#     shuffle = False,\n#     seed = 42\n# )","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:12:50.084888Z","iopub.execute_input":"2022-03-27T09:12:50.085151Z","iopub.status.idle":"2022-03-27T09:12:50.090554Z","shell.execute_reply.started":"2022-03-27T09:12:50.085100Z","shell.execute_reply":"2022-03-27T09:12:50.088638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, shutil\nfrom distutils.dir_util import copy_tree\n\n#create directories\nif os.path.exists('test_folder'):\n    shutil.rmtree(r'test_folder') # Or reset the session in faster\nos.mkdir('test_folder')\nos.mkdir('test_folder/test_images')\n\n# copy subdirectory example\nfromDirectory = \"../input/ultramnist-resized-512/test_img\"\ntoDirectory = \"test_folder/test_images\"\n\ncopy_tree(fromDirectory, toDirectory, verbose=0);","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:12:50.091995Z","iopub.execute_input":"2022-03-27T09:12:50.092397Z","iopub.status.idle":"2022-03-27T09:15:21.571780Z","shell.execute_reply.started":"2022-03-27T09:12:50.092360Z","shell.execute_reply":"2022-03-27T09:15:21.570988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_datagen = ImageDataGenerator(rescale=1./255)\ntest_generator = test_datagen.flow_from_directory(directory='test_folder', \n                                                  seed=42, \n                                                  class_mode=None, \n                                                  target_size=(CFG.IMG_SIZE, CFG.IMG_SIZE), \n                                                  batch_size=CFG.BATCH_SIZE, \n                                                  shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:15:21.573139Z","iopub.execute_input":"2022-03-27T09:15:21.573418Z","iopub.status.idle":"2022-03-27T09:15:22.436708Z","shell.execute_reply.started":"2022-03-27T09:15:21.573381Z","shell.execute_reply":"2022-03-27T09:15:22.435856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#we need to use .reset() here otherwise\n#the other of predictions will be different\n#then the expected\ntest_generator.reset()\npreds = model.predict_generator(test_generator,verbose = 1)\npreds.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:15:22.439300Z","iopub.execute_input":"2022-03-27T09:15:22.439719Z","iopub.status.idle":"2022-03-27T09:23:45.579774Z","shell.execute_reply.started":"2022-03-27T09:15:22.439668Z","shell.execute_reply":"2022-03-27T09:23:45.579038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''delete the files on disk - otherwise the Kaggle kernel will throw an error'''\nshutil.rmtree('test_folder', ignore_errors=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:23:45.580912Z","iopub.execute_input":"2022-03-27T09:23:45.581190Z","iopub.status.idle":"2022-03-27T09:23:46.638582Z","shell.execute_reply.started":"2022-03-27T09:23:45.581147Z","shell.execute_reply":"2022-03-27T09:23:46.637688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_preds = preds.argmax(axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:23:46.642818Z","iopub.execute_input":"2022-03-27T09:23:46.643235Z","iopub.status.idle":"2022-03-27T09:23:46.648792Z","shell.execute_reply.started":"2022-03-27T09:23:46.643197Z","shell.execute_reply":"2022-03-27T09:23:46.648066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\"id\": [path.split('.')[0] for path in os.listdir(fromDirectory)], \"digit_sum\": final_preds})\nsubmission.to_csv(\"submission.csv\", index = False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:23:46.649907Z","iopub.execute_input":"2022-03-27T09:23:46.650851Z","iopub.status.idle":"2022-03-27T09:23:46.741722Z","shell.execute_reply.started":"2022-03-27T09:23:46.650814Z","shell.execute_reply":"2022-03-27T09:23:46.741004Z"},"trusted":true},"execution_count":null,"outputs":[]}]}