{"cells":[{"metadata":{"trusted":true,"_uuid":"47560b5ce3c43ab06f05970feb21bd3a94a5c865"},"cell_type":"code","source":"%matplotlib inline\nimport matplotlib.pyplot as plt\nfrom fastai.vision import *\nfrom fastai.metrics import accuracy\nfrom fastai.basic_data import *\nfrom skimage.util import montage\nimport pandas as pd\nfrom torch import optim\nimport re\n\nfrom utils import *","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"867f3040224a26ea525913afa2050c36f4b46ff3"},"cell_type":"code","source":"!git clone https://github.com/radekosmulski/whale\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bae60c08140f5259f06931840721e2f1424a6bb9"},"cell_type":"code","source":"import sys\n # Add directory holding utility functions to path to allow importing utility funcitons\n#sys.path.insert(0, '/kaggle/working/protein-atlas-fastai')\nsys.path.append('/kaggle/working/whale')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"887917667d90cc8fa445d9647780f44c237565f9"},"cell_type":"code","source":"from whale.utils import map5","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f8ede3e8e28d382f681e5c190031b8a8d124e0d9"},"cell_type":"markdown","source":"I take a curriculum approach to training here. I first expose the model to as many different images of whales as quickly as possible (no oversampling) and train on images resized to 224x224.\n\nI would like the conv layers to start picking up on features useful for identifying whales. For that, I want to show the model as rich of a dataset as possible.\n\nI then train on images resized to 448x448.\n\nFinally, I train on oversampled data. Here, the model will see some images more often than others but I am hoping that this will help alleviate the class imbalance in the training data."},{"metadata":{"trusted":true,"_uuid":"df56a68805ac1c279f016fc273468b0c45f5810a"},"cell_type":"code","source":"import fastai\nfrom fastprogress import force_console_behavior\nimport fastprogress\nfastprogress.fastprogress.NO_BAR = True\nmaster_bar, progress_bar = force_console_behavior()\nfastai.basic_train.master_bar, fastai.basic_train.progress_bar = master_bar, progress_bar","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"07710c0acc72991fe3e16b19f3748d960b3c4854"},"cell_type":"code","source":"from fastai import *\nfrom fastai.vision import *","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f5b2e28b310a4abc9ae380b8780f66f93954d8c8"},"cell_type":"code","source":"ls ../input","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"28516f55a00cf31d58ffcbb5e2e27f5abf089f50"},"cell_type":"code","source":"path = Path('../input/humpback-whale-identification/')\npath_test = Path('../input/humpback-whale-identification/test')\npath_train = Path('../input/humpback-whale-identification/train')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"61184c91a4bdd4e7994fb7c7d0c4910a0fdbfaf7"},"cell_type":"code","source":"df = pd.read_csv(path/'train.csv')#.sample(frac=0.05)\ndf.head()\nval_fns = {'69823499d.jpg'}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c12d77e73230032586493b7f87228a1e9155f076"},"cell_type":"code","source":"fn2label = {row[1].Image: row[1].Id for row in df.iterrows()}\npath2fn = lambda path: re.search('\\w*\\.jpg$', path).group(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"63b6d558fbbd4c03b1a01c856f7ac8aa7a785831"},"cell_type":"code","source":"name = f'res50-full-train'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fa95cf046c6b6ddc5b5f68df1bc78c4f39dfd757"},"cell_type":"code","source":"SZ = 224\nBS = 64\nNUM_WORKERS = 0\nSEED=0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7eb47507fe918b32851542410700c1bef3b33b7"},"cell_type":"code","source":"data = (\n    ImageItemList\n        .from_df(df[df.Id != 'new_whale'], '../input/humpback-whale-identification/train', cols=['Image'])\n        .split_by_valid_func(lambda path: path2fn(path) in val_fns)\n        .label_from_func(lambda path: fn2label[path2fn(path)])\n        .add_test(ImageItemList.from_folder('../input/humpback-whale-identification/test'))\n        .transform(get_transforms(do_flip=False), size=SZ, resize_method=ResizeMethod.SQUISH)\n        .databunch(bs=BS, num_workers=NUM_WORKERS, path='../input/humpback-whale-identification')\n        .normalize(imagenet_stats)\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c89d9854b0be058e87664af22a9cccb6654db631"},"cell_type":"code","source":"MODEL_PATH = \"/kaggle/working/\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8956a84c0fb27bf7ed1a70cfb28d183ae6da145f"},"cell_type":"code","source":"%%time\n\nlearn = create_cnn(data, models.resnet50, lin_ftrs=[2048], model_dir=MODEL_PATH)\nlearn.clip_grad();","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"79938f3700bd62726ad8ed557cd7532e750cb43a"},"cell_type":"code","source":"SZ = 224 * 2\nBS = 64 // 4\nNUM_WORKERS = 0\nSEED=0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"91a117e9d4f47589d021408ec70c96b5ecbab1c3"},"cell_type":"code","source":"data = (\n    ImageItemList\n        .from_df(df[df.Id != 'new_whale'], '../input/humpback-whale-identification/train', cols=['Image'])\n        .split_by_valid_func(lambda path: path2fn(path) in val_fns)\n        .label_from_func(lambda path: fn2label[path2fn(path)])\n        .add_test(ImageItemList.from_folder('../input/humpback-whale-identification/test'))\n        .transform(get_transforms(do_flip=False), size=SZ, resize_method=ResizeMethod.SQUISH)\n        .databunch(bs=BS, num_workers=NUM_WORKERS, path='../input/humpback-whale-identification')\n        .normalize(imagenet_stats)\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54feec1d1a171b2d17b380d6cf0e3e26af4fa9e5"},"cell_type":"code","source":"# with oversampling\ndf = pd.read_csv('../input/radek-whale-oversample/oversampled_train_and_val.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e176949783b0e1fbef06a55db9652ecc07038c5b"},"cell_type":"code","source":"data = (\n    ImageItemList\n        .from_df(df, '../input/humpback-whale-identification/train', cols=['Image'])\n        .split_by_valid_func(lambda path: path2fn(path) in val_fns)\n        .label_from_func(lambda path: fn2label[path2fn(path)])\n        .add_test(ImageItemList.from_folder('data/test'))\n        .transform(get_transforms(do_flip=False), size=SZ, resize_method=ResizeMethod.SQUISH)\n        .databunch(bs=BS, num_workers=NUM_WORKERS, path='data')\n        .normalize(imagenet_stats)\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"05164fc57478b8c307a10885f037d40121363a9f"},"cell_type":"code","source":"!cp ../input/radek-fast-ai-whale-full-stage5/res50-full-train-stage-5.pth /kaggle/working","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6051fb85c4ffb67a51d1c12686e66818363f92ef"},"cell_type":"code","source":"%%time\nlearn = create_cnn(data, models.resnet50, lin_ftrs=[2048], model_dir=MODEL_PATH)\nlearn.load(f'{name}-stage-5')\n\nlearn.unfreeze()\n\nmax_lr = 1e-3 / 4\nlrs = [max_lr/100, max_lr/10, max_lr]\n\nlearn.fit_one_cycle(3, lrs)\nlearn.save(f'{name}-stage-6')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"662d57fbfbadbaede5ea6b0640c287dfb0f0d79c"},"cell_type":"code","source":"!rm -rf /kaggle/working/whale","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"90c699119240317a624ee17cbd470dd2869c9e84"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1e4b38ed57aba67258d0919ba9d2938d67878084"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}