{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os.path as osp\nfrom glob import glob\nimport random\nimport time\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport plotly.express as px\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import accuracy_score\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.utils.data as data\nimport torch.optim as optim\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:19:34.255537Z","iopub.execute_input":"2023-04-10T00:19:34.255882Z","iopub.status.idle":"2023-04-10T00:19:43.700861Z","shell.execute_reply.started":"2023-04-10T00:19:34.255849Z","shell.execute_reply":"2023-04-10T00:19:43.699798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fix_seed(seed):\n    # random\n    random.seed(seed)\n    # Numpy\n    np.random.seed(seed)\n    # Pytorch\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n\n    \nSEED = 42\nfix_seed(SEED)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:19:51.648470Z","iopub.execute_input":"2023-04-10T00:19:51.649541Z","iopub.status.idle":"2023-04-10T00:19:51.659001Z","shell.execute_reply.started":"2023-04-10T00:19:51.649500Z","shell.execute_reply":"2023-04-10T00:19:51.657793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ndata_path = '/kaggle/input/state-farm-distracted-driver-detection/'\n\nimgs_list = pd.read_csv(data_path + 'driver_imgs_list.csv')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv')\n\nimport os\nimgs_list['file_path'] = imgs_list.apply(lambda x: os.path.join(data_path, 'imgs/train', x.classname, x.img), axis=1)\nimgs_list['class_num'] = imgs_list['classname'].map(lambda x: int(x[1]))\nimgs_list.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:19:52.299650Z","iopub.execute_input":"2023-04-10T00:19:52.300579Z","iopub.status.idle":"2023-04-10T00:19:53.045416Z","shell.execute_reply.started":"2023-04-10T00:19:52.300541Z","shell.execute_reply":"2023-04-10T00:19:53.044252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_map = {'c0': 'Safe driving', \n                'c1': 'Texting - right', \n                'c2': 'Talking on the phone - right', \n                'c3': 'Texting - left', \n                'c4': 'Talking on the phone - left', \n                'c5': 'Operating the radio', \n                'c6': 'Drinking', \n                'c7': 'Reaching behind', \n                'c8': 'Hair and makeup', \n                'c9': 'Talking to passenger'}","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:19:53.047679Z","iopub.execute_input":"2023-04-10T00:19:53.048432Z","iopub.status.idle":"2023-04-10T00:19:53.055064Z","shell.execute_reply.started":"2023-04-10T00:19:53.048392Z","shell.execute_reply":"2023-04-10T00:19:53.054013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 데이터 셋 정의","metadata":{}},{"cell_type":"code","source":"class Dataset(data.Dataset):\n    def __init__(self, df, phase, transform=None):\n        super().__init__()\n        self.df = df\n        self.phase = phase\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, index):\n        label = self.df.iloc[index]['class_num']\n        image_path = self.df.iloc[index]['file_path']\n        image = cv2.imread(image_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB).astype(np.float32)\n        image /= 255.0\n        \n        if self.transform is not None:\n            image = self.transform(self.phase, image)\n        return image, label","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:19:54.562984Z","iopub.execute_input":"2023-04-10T00:19:54.563978Z","iopub.status.idle":"2023-04-10T00:19:54.572467Z","shell.execute_reply.started":"2023-04-10T00:19:54.563937Z","shell.execute_reply":"2023-04-10T00:19:54.570961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataTransform():\n    def __init__(self):\n        self.data_transform = {\n            'train': A.Compose([\n                #A.Crop(x_min = 80, y_min = 30, x_max = 500, y_max = 450, p = 1.0),\n                A.Resize(224, 224),\n                A.Rotate(-10, 10, p=0.5),\n                A.RandomBrightnessContrast(brightness_limit=0.1, contrast_limit=0.1, p=0.3),\n                A.OneOf([A.Emboss(p=1), A.Sharpen(p=1), A.Blur(p=1)], p=0.3),\n                ToTensorV2() # 텐서 변환\n            ]),\n            'val': A.Compose([\n                #A.Crop(x_min = 80, y_min = 30, x_max = 500, y_max = 450, p = 1.0),\n                A.Resize(224, 224),\n                ToTensorV2()\n            ])\n        }\n\n    def __call__(self, phase, image):\n        # phase : 'train' or 'val'\n        transformed = self.data_transform[phase](image=image)\n        return transformed['image']   \n","metadata":{"execution":{"iopub.status.busy":"2023-04-10T01:01:54.254302Z","iopub.execute_input":"2023-04-10T01:01:54.255539Z","iopub.status.idle":"2023-04-10T01:01:54.264460Z","shell.execute_reply.started":"2023-04-10T01:01:54.255480Z","shell.execute_reply":"2023-04-10T01:01:54.263148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 데이터 분리\ndf_train, df_val = train_test_split(imgs_list, stratify=imgs_list['class_num'], test_size=0.2, random_state=SEED)\n\n# 데이터 셋 생성\ntrain_dataset = Dataset(df_train, phase=\"train\", transform=DataTransform())\nval_dataset = Dataset(df_val, phase=\"val\", transform=DataTransform())","metadata":{"execution":{"iopub.status.busy":"2023-04-10T01:01:57.584961Z","iopub.execute_input":"2023-04-10T01:01:57.586048Z","iopub.status.idle":"2023-04-10T01:01:57.607306Z","shell.execute_reply.started":"2023-04-10T01:01:57.585996Z","shell.execute_reply":"2023-04-10T01:01:57.605887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 데이터 변환 확인","metadata":{}},{"cell_type":"code","source":"# 변환된 이미지 확인\nplt.figure(figsize=(15, 24))\n\norder = list(range(len(train_dataset)))\nrandom.shuffle(order)\n\nfor idx, rand_num in enumerate(order[:24]):\n    image, label = train_dataset[rand_num]\n    plt.subplot(6, 4, idx + 1)\n    plt.imshow(image.permute(1, 2, 0))\n    plt.title(label) \n","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:17:26.820490Z","iopub.execute_input":"2023-04-10T00:17:26.820953Z","iopub.status.idle":"2023-04-10T00:17:32.370131Z","shell.execute_reply.started":"2023-04-10T00:17:26.820908Z","shell.execute_reply":"2023-04-10T00:17:32.368302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"!pip install efficientnet-pytorch","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:20:00.861600Z","iopub.execute_input":"2023-04-10T00:20:00.862060Z","iopub.status.idle":"2023-04-10T00:20:15.007437Z","shell.execute_reply.started":"2023-04-10T00:20:00.862015Z","shell.execute_reply":"2023-04-10T00:20:15.005934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from efficientnet_pytorch import EfficientNet\nmodel = EfficientNet.from_pretrained('efficientnet-b0', num_classes=10)\n# model = EfficientNet.from_pretrained('efficientnet-b7', num_classes=10)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:20:15.013560Z","iopub.execute_input":"2023-04-10T00:20:15.016506Z","iopub.status.idle":"2023-04-10T00:20:15.986318Z","shell.execute_reply.started":"2023-04-10T00:20:15.016456Z","shell.execute_reply":"2023-04-10T00:20:15.985314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_checkpoint(model, optimizer, scheduler, epoch, path):\n    print(path)\n    torch.save(\n        {'epoch': epoch,\n                'model': model.state_dict(),\n                'optimizer': optimizer.state_dict(),\n                'scheduler': scheduler.state_dict(), \n        }, path)\n\ndef load_checkpoint(model, optimizer, scheduler, path):\n    checkpoint = torch.load(path)\n    model.load_state_dict(checkpoint['model'])\n    optimizer.load_state_dict(checkpoint['optimizer'])\n    scheduler.load_state_dict(checkpoint['scheduler'])\n\ndef train_model(model, dataloaders_dict, criterion, scheduler, optimizer, device, num_epochs, save_path):\n    model.to(device)\n\n    best_val_loss = float('inf')\n    best_preds = None\n    \n    for epoch in range(num_epochs):\n\n        t_epoch_start = time.time()\n        epoch_train_loss = 0.0\n        epoch_val_loss = 0.0\n        preds = []\n        trues = []\n\n        print('-------------')\n        print(f'Epoch {epoch+1}/{num_epochs}')\n        print('-------------')\n\n        for phase in ['train', 'val']:\n            if phase == 'train':\n                model.train()  \n            else:\n                model.eval()   \n                print('-------------')\n                \n            for i, (images, labels) in enumerate(dataloaders_dict[phase]):\n                images = images.to(device)\n                labels = labels.to(device)\n\n                with torch.set_grad_enabled(phase == 'train'):\n                    outputs = model(images)\n                    loss = criterion(outputs, labels)\n                    \n                    if phase == 'train':\n                        loss.backward()  \n                        optimizer.step()\n                        optimizer.zero_grad() \n                        epoch_train_loss += loss.item()/len(dataloaders_dict[phase].dataset)\n                    else: # val\n                        preds += [outputs.detach().cpu().softmax(dim=1).numpy()]\n                        trues += [labels.detach().cpu()]\n                        epoch_val_loss += loss.item()/len(dataloaders_dict[phase].dataset)\n                    \n                    # 진행표시\n                    if i%10 == 0:\n                        print(f'[{phase}][{i+1}/{len(dataloaders_dict[phase])}] loss: {loss.item()/images.size(0): .4f}')\n        \n        if phase == 'train':\n            scheduler.step()\n            \n        # epoch 결과표시. (train_Loss, vla_Loss, 훈련시간. 정확도)\n        t_epoch_finish = time.time()\n        print('-------------')\n        print(f'epoch {epoch+1} epoch_train_Loss:{epoch_train_loss:.4f} epoch_val_loss:{epoch_val_loss:.4f} time: {t_epoch_finish - t_epoch_start:.4f} sec.')\n        print(f'epoch_val_acc: {accuracy_score(np.concatenate(trues), np.concatenate(preds).argmax(axis=1))}')\n        \n        # validation loss 가장 작은 모델을 저장.\n        if best_val_loss > epoch_val_loss:\n            best_preds = np.concatenate(preds)\n            best_val_loss = epoch_val_loss\n            save_checkpoint(model, optimizer, scheduler, epoch, save_path)\n            print(\"save model\")\n    return best_val_loss, best_preds\n\n# fold 1개 수행 함수\ndef run_one_fold(df_train, df_val, fold, device):\n    train_dataset = Dataset(df_train, phase=\"train\", transform=DataTransform())\n    val_dataset = Dataset(df_val, phase=\"val\", transform=DataTransform())   \n    train_dataloader = data.DataLoader(train_dataset, batch_size=args.batch_size, shuffle=True)\n    val_dataloader = data.DataLoader(val_dataset, batch_size=args.batch_size, shuffle=False)\n    dataloaders_dict = {\"train\": train_dataloader, \"val\": val_dataloader}\n\n    # 모델정의\n    model = EfficientNet.from_pretrained(args.model_name, num_classes=args.num_classes)\n    optimizer = optim.Adam(model.parameters(), lr=args.lr) \n    criterion = nn.CrossEntropyLoss() \n    scheduler = optim.lr_scheduler.ExponentialLR(optimizer, gamma=args.gamma) \n    \n    save_path = f\"{args.model_name}_fold_{fold}.pth\"\n    best_val_loss, best_preds = train_model(model, dataloaders_dict, criterion, scheduler, optimizer, device, num_epochs=args.epochs, save_path=save_path)\n    return best_val_loss, best_preds\n\n# kfold 함수\ndef run_k_fold(df):\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    print(\"사용장치：\", device)\n    \n    skf = StratifiedKFold(n_splits=args.folds, shuffle=True, random_state=SEED)\n    oof = pd.DataFrame(index=df.index)\n    for fold, (train_index, val_index) in enumerate(skf.split(df, df['class_num'])): # \n#    for fold, (train_index, val_index) in enumerate(skf.split(df, df['subject'])): \n        print(f'\\n\\nFOLD: {fold}')\n        print('-'*50)\n        df_train, df_val = df.loc[train_index], df.loc[val_index]\n        best_val_loss, best_preds = run_one_fold(df_train, df_val, fold, device)\n        oof.loc[val_index, class_map.keys()] = best_preds\n    return oof","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:20:15.991172Z","iopub.execute_input":"2023-04-10T00:20:15.993347Z","iopub.status.idle":"2023-04-10T00:20:16.025061Z","shell.execute_reply.started":"2023-04-10T00:20:15.993307Z","shell.execute_reply":"2023-04-10T00:20:16.023701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class args:\n    model_name = 'efficientnet-b0'\n    num_classes = 10\n    batch_size = 128\n    epochs = 10\n    folds = 5\n    lr = 1e-3\n    gamma = 0.98\n    debug = True\n    train = True","metadata":{"execution":{"iopub.status.busy":"2023-04-10T00:21:21.572500Z","iopub.execute_input":"2023-04-10T00:21:21.573556Z","iopub.status.idle":"2023-04-10T00:21:21.579690Z","shell.execute_reply.started":"2023-04-10T00:21:21.573514Z","shell.execute_reply":"2023-04-10T00:21:21.578139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train","metadata":{}},{"cell_type":"code","source":"train_dataloader = data.DataLoader(train_dataset, batch_size=args.batch_size, shuffle=True)\nval_dataloader = data.DataLoader(val_dataset, batch_size=args.batch_size, shuffle=False)\ndataloaders_dict = {\"train\": train_dataloader, \"val\": val_dataloader}\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(\"사용장치：\", device)\n\n# 모델정의\nmodel = EfficientNet.from_pretrained(args.model_name, num_classes=args.num_classes)\n#optimizer = optim.Adam(model.parameters(), lr=args.lr) \n#optimizer = torch.optim.AdamW(model.parameters(), lr=0.00006, weight_decay=0.0001)\noptimizer = optim.AdamW(model.parameters(), lr=0.0001, weight_decay=0.0001)\n\ncriterion = nn.CrossEntropyLoss() \nscheduler = optim.lr_scheduler.ExponentialLR(optimizer, gamma=args.gamma) \n     \nsave_path = f\"{args.model_name}_train.pth\"\nbest_val_loss, best_preds = train_model(model, dataloaders_dict, criterion, scheduler, optimizer, device, num_epochs=args.epochs, save_path=save_path)\n\n# best 0.9942028985507246","metadata":{"execution":{"iopub.status.busy":"2023-04-10T01:02:12.362167Z","iopub.execute_input":"2023-04-10T01:02:12.362888Z","iopub.status.idle":"2023-04-10T01:25:51.154930Z","shell.execute_reply.started":"2023-04-10T01:02:12.362848Z","shell.execute_reply":"2023-04-10T01:25:51.153194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"5-Fold","metadata":{}},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(\"사용장치：\", device)\n#best_val_loss, best_preds = run_one_fold(df_train, df_val, 0, device) \noof = run_k_fold(imgs_list)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T09:24:31.606108Z","iopub.execute_input":"2023-04-05T09:24:31.606545Z","iopub.status.idle":"2023-04-05T12:11:51.098186Z","shell.execute_reply.started":"2023-04-05T09:24:31.606507Z","shell.execute_reply":"2023-04-05T12:11:51.096964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pwd","metadata":{"execution":{"iopub.status.busy":"2023-04-05T06:42:53.039960Z","iopub.execute_input":"2023-04-05T06:42:53.040809Z","iopub.status.idle":"2023-04-05T06:42:54.064259Z","shell.execute_reply.started":"2023-04-05T06:42:53.040762Z","shell.execute_reply":"2023-04-05T06:42:54.062973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}