{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os.path as osp\nfrom glob import glob\nimport random\nimport time\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport plotly.express as px\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import accuracy_score\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.utils.data as data\nimport torch.optim as optim\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2","metadata":{"execution":{"iopub.status.busy":"2023-04-05T06:50:56.533257Z","iopub.execute_input":"2023-04-05T06:50:56.534056Z","iopub.status.idle":"2023-04-05T06:51:04.492109Z","shell.execute_reply.started":"2023-04-05T06:50:56.533991Z","shell.execute_reply":"2023-04-05T06:51:04.491055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path = '/kaggle/input/state-farm-distracted-driver-detection/'\n\nimgs_list = pd.read_csv(data_path + 'driver_imgs_list.csv')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv')\n\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T06:51:04.494188Z","iopub.execute_input":"2023-04-05T06:51:04.494961Z","iopub.status.idle":"2023-04-05T06:51:04.688421Z","shell.execute_reply.started":"2023-04-05T06:51:04.494928Z","shell.execute_reply":"2023-04-05T06:51:04.687290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_map = {'c0': 'Safe driving', \n                'c1': 'Texting - right', \n                'c2': 'Talking on the phone - right', \n                'c3': 'Texting - left', \n                'c4': 'Talking on the phone - left', \n                'c5': 'Operating the radio', \n                'c6': 'Drinking', \n                'c7': 'Reaching behind', \n                'c8': 'Hair and makeup', \n                'c9': 'Talking to passenger'}","metadata":{"execution":{"iopub.status.busy":"2023-04-05T06:51:04.689900Z","iopub.execute_input":"2023-04-05T06:51:04.690314Z","iopub.status.idle":"2023-04-05T06:51:04.697679Z","shell.execute_reply.started":"2023-04-05T06:51:04.690276Z","shell.execute_reply":"2023-04-05T06:51:04.695968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install efficientnet-pytorch","metadata":{"execution":{"iopub.status.busy":"2023-04-05T06:51:09.320141Z","iopub.execute_input":"2023-04-05T06:51:09.320882Z","iopub.status.idle":"2023-04-05T06:51:22.084772Z","shell.execute_reply.started":"2023-04-05T06:51:09.320842Z","shell.execute_reply":"2023-04-05T06:51:22.083548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from efficientnet_pytorch import EfficientNet\n# 사전 훈련된 efficientnet-b7 모델 불러오기\nmodel = EfficientNet.from_pretrained('efficientnet-b0', num_classes=10)\n# model = EfficientNet.from_pretrained('efficientnet-b7', num_classes=10)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T06:51:22.087974Z","iopub.execute_input":"2023-04-05T06:51:22.088636Z","iopub.status.idle":"2023-04-05T06:51:22.627282Z","shell.execute_reply.started":"2023-04-05T06:51:22.088586Z","shell.execute_reply":"2023-04-05T06:51:22.626102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 自分で実装\n# テストデータの推論を行う関数\ndef inference(model, dataloader, device):\n    model.to(device)\n    model.eval()\n    preds = []\n    for i, (images, labels) in enumerate(dataloader):\n        images = images.to(device)\n        with torch.no_grad():\n            outputs = model(images)\n        preds += [outputs.detach().cpu().softmax(dim=1).numpy()]\n        \n        if i%10 == 0:\n            print(f'[test][{i+1}/{len(dataloader)}]')\n        \n    preds = np.concatenate(preds)\n    return preds\n\n# k個のモデルに対して推論を行い，アンサンブル\ndef inference_k_fold(df_test):\n    test_dataset = Dataset(df_test, phase=\"val\", transform=DataTransform())\n    test_dataloader = data.DataLoader(\n        test_dataset, batch_size=args.batch_size, shuffle=False)\n\n    model = EfficientNet.from_pretrained(args.model_name, num_classes=args.num_classes)\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n    for fold in range(args.folds):\n        print(f'\\n\\nFOLD: {fold}')\n        print('-'*50)\n        \"/kaggle/input/DriverCassificationModel/\"\n        model.load_state_dict(torch.load(f\"/kaggle/input/efficientnet-b0-20230405/{args.model_name}_fold_{fold}.pth\")['model'])\n        # model.load_state_dict(torch.load(f\"{args.model_name}_fold_{fold}.pth\")['model'])\n        df_test.loc[:, class_map.keys()] += (inference(model, test_dataloader, device) / args.folds)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T06:52:32.584183Z","iopub.execute_input":"2023-04-05T06:52:32.584978Z","iopub.status.idle":"2023-04-05T06:52:32.595514Z","shell.execute_reply.started":"2023-04-05T06:52:32.584938Z","shell.execute_reply":"2023-04-05T06:52:32.594073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class args:\n    model_name = 'efficientnet-b0'\n    num_classes = 10\n    batch_size = 128\n    folds = 5\n    debug = True\n    train = True","metadata":{"execution":{"iopub.status.busy":"2023-04-05T06:52:32.929863Z","iopub.execute_input":"2023-04-05T06:52:32.930291Z","iopub.status.idle":"2023-04-05T06:52:32.936835Z","shell.execute_reply.started":"2023-04-05T06:52:32.930253Z","shell.execute_reply":"2023-04-05T06:52:32.935400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Dataset(data.Dataset):\n    def __init__(self, df, phase, transform=None):\n        super().__init__()\n        self.df = df\n        self.phase = phase\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, index):\n        label = self.df.iloc[index]['class_num']\n        image_path = self.df.iloc[index]['file_path']\n        image = cv2.imread(image_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB).astype(np.float32)\n        image /= 255.0\n        \n        if self.transform is not None:\n            image = self.transform(self.phase, image)\n        return image, label","metadata":{"execution":{"iopub.status.busy":"2023-04-05T06:54:12.295703Z","iopub.execute_input":"2023-04-05T06:54:12.296157Z","iopub.status.idle":"2023-04-05T06:54:12.304747Z","shell.execute_reply.started":"2023-04-05T06:54:12.296119Z","shell.execute_reply":"2023-04-05T06:54:12.303432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataTransform():\n    def __init__(self):\n        self.data_transform = {\n            'train': A.Compose([\n                A.Crop(x_min = 40, y_min = 0, x_max = 520, y_max = 480, p = 1.0),\n                A.Resize(224, 224),\n                A.Rotate(-10, 10, p=0.5),\n                A.RandomBrightnessContrast(brightness_limit=0.1, contrast_limit=0.1, p=0.3),\n                A.OneOf([A.Emboss(p=1), A.Sharpen(p=1), A.Blur(p=1)], p=0.3),\n                ToTensorV2() # 텐서 변환\n            ]),\n            'val': A.Compose([\n                A.Crop(x_min = 40, y_min = 0, x_max = 520, y_max = 480, p = 1.0),\n                A.Resize(224, 224),\n                ToTensorV2()\n            ])        }\n\n    def __call__(self, phase, image):\n        # phase : 'train' or 'val'\n        transformed = self.data_transform[phase](image=image)\n        return transformed['image']","metadata":{"execution":{"iopub.status.busy":"2023-04-05T06:54:12.643273Z","iopub.execute_input":"2023-04-05T06:54:12.644003Z","iopub.status.idle":"2023-04-05T06:54:12.653274Z","shell.execute_reply.started":"2023-04-05T06:54:12.643966Z","shell.execute_reply":"2023-04-05T06:54:12.651925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\ndf_test = pd.read_csv(osp.join(data_path, 'sample_submission.csv'))\n\ndf_test['file_path'] = df_test.apply(lambda row: osp.join(data_path, f'imgs/test/{row.img}'), axis=1)\ndf_test['class_num'] = 0\ndf_test.loc[:, class_map.keys()] = 0\n\n\ntest_dataset = Dataset(df_test, phase=\"val\", transform=DataTransform(\n        img_w=args.input_width, img_h=args.input_height, color_mean=args.color_mean, color_std=args.color_std))\ntest_dataloader = data.DataLoader(test_dataset, batch_size=args.batch_size, shuffle=False)\n\nmodel = EfficientNet.from_pretrained(args.model_name, num_classes=args.num_classes)\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nmodel.load_state_dict(torch.load(\"/kaggle/input/efficientnet-test/efficientnet-b0_fold_0.pth\")['model'])\ndf_test.loc[:, class_map.keys()] = inference(model, test_dataloader, device)\n\nresults = df_test.drop(['file_path', 'class_num'], axis=1)\nresults.iloc[:, 1:] = results.iloc[:, 1:].clip(0, 1)\nresults.to_csv('result.csv', index=False)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-04-05T06:54:13.079349Z","iopub.execute_input":"2023-04-05T06:54:13.080291Z","iopub.status.idle":"2023-04-05T06:54:13.090220Z","shell.execute_reply.started":"2023-04-05T06:54:13.080251Z","shell.execute_reply":"2023-04-05T06:54:13.089117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(osp.join(data_path, 'sample_submission.csv'))\n\ndf_test['file_path'] = df_test.apply(lambda row: osp.join(data_path, f'imgs/test/{row.img}'), axis=1)\ndf_test['class_num'] = 0\ndf_test.loc[:, class_map.keys()] = 0\n    \ninference_k_fold(df_test)\nresults = df_test.drop(['file_path', 'class_num'], axis=1)\nresults.iloc[:, 1:] = results.iloc[:, 1:].clip(0, 1)\nresults.to_csv('result.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-30T07:17:21.063638Z","iopub.execute_input":"2023-03-30T07:17:21.064401Z","iopub.status.idle":"2023-03-30T07:17:21.082168Z","shell.execute_reply.started":"2023-03-30T07:17:21.064362Z","shell.execute_reply":"2023-03-30T07:17:21.081057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}