{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":8780101,"sourceType":"datasetVersion","datasetId":5277490}],"dockerImageVersionId":29928,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\n\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\n\nfrom torchvision import datasets, transforms, models\nfrom torch.utils.data import DataLoader, Dataset\nimport torch.nn as nn\nfrom torchvision import transforms\nimport torch\nimport torchvision\n\n!pip install efficientnet_pytorch\nfrom efficientnet_pytorch import EfficientNet","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-25T07:39:08.763837Z","iopub.execute_input":"2024-06-25T07:39:08.764133Z","iopub.status.idle":"2024-06-25T07:39:21.548898Z","shell.execute_reply.started":"2024-06-25T07:39:08.764102Z","shell.execute_reply":"2024-06-25T07:39:21.547881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('/kaggle/input/gender-25062024')","metadata":{"execution":{"iopub.status.busy":"2024-06-25T07:39:21.551418Z","iopub.execute_input":"2024-06-25T07:39:21.551859Z","iopub.status.idle":"2024-06-25T07:39:21.564122Z","shell.execute_reply.started":"2024-06-25T07:39:21.551807Z","shell.execute_reply":"2024-06-25T07:39:21.563431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '/kaggle/input/gender-25062024'","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2024-06-25T07:56:14.54289Z","iopub.execute_input":"2024-06-25T07:56:14.54328Z","iopub.status.idle":"2024-06-25T07:56:14.547715Z","shell.execute_reply.started":"2024-06-25T07:56:14.54324Z","shell.execute_reply":"2024-06-25T07:56:14.546788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = pd.read_csv(\"/kaggle/input/gender-25062024/columns.csv\").columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-06-25T07:43:26.80221Z","iopub.execute_input":"2024-06-25T07:43:26.802534Z","iopub.status.idle":"2024-06-25T07:43:26.818145Z","shell.execute_reply.started":"2024-06-25T07:43:26.802505Z","shell.execute_reply":"2024-06-25T07:43:26.817326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train = pd.read_csv('/kaggle/input/gender-25062024/new_train.csv', names=columns)\nprint(data_train.shape)\ndata_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-25T07:48:07.082311Z","iopub.execute_input":"2024-06-25T07:48:07.082642Z","iopub.status.idle":"2024-06-25T07:48:07.141636Z","shell.execute_reply.started":"2024-06-25T07:48:07.082612Z","shell.execute_reply":"2024-06-25T07:48:07.14094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_f = data_train[columns[1]].sum()\ntotal_nam = data_train[columns[2]].sum()\ntotal_nu = data_train[columns[3]].sum()\nprint(columns[1], total_f)\nprint(columns[2], total_nam)\nprint(columns[3], total_nu)","metadata":{"execution":{"iopub.status.busy":"2024-06-25T07:48:57.468278Z","iopub.execute_input":"2024-06-25T07:48:57.468634Z","iopub.status.idle":"2024-06-25T07:48:57.482247Z","shell.execute_reply.started":"2024-06-25T07:48:57.4686Z","shell.execute_reply":"2024-06-25T07:48:57.48141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_valid = pd.read_csv('/kaggle/input/gender-25062024/new_valid.csv', names=columns)\nprint(data_valid.shape)\ndata_valid.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-25T07:49:25.367341Z","iopub.execute_input":"2024-06-25T07:49:25.367819Z","iopub.status.idle":"2024-06-25T07:49:25.42386Z","shell.execute_reply.started":"2024-06-25T07:49:25.367746Z","shell.execute_reply":"2024-06-25T07:49:25.423029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_f = data_valid[columns[1]].sum()\ntotal_nam = data_valid[columns[2]].sum()\ntotal_nu = data_valid[columns[3]].sum()\nprint(columns[1], total_f)\nprint(columns[2], total_nam)\nprint(columns[3], total_nu)","metadata":{"execution":{"iopub.status.busy":"2024-06-25T07:49:42.99802Z","iopub.execute_input":"2024-06-25T07:49:42.998386Z","iopub.status.idle":"2024-06-25T07:49:43.008803Z","shell.execute_reply.started":"2024-06-25T07:49:42.998349Z","shell.execute_reply":"2024-06-25T07:49:43.007348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df, val_df = train_test_split(df, stratify=df.target, test_size=0.10)\n# train_df.reset_index(drop=True, inplace=True)\n# val_df.reset_index(drop=True, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T07:52:44.603029Z","iopub.execute_input":"2024-06-20T07:52:44.603351Z","iopub.status.idle":"2024-06-20T07:52:44.617747Z","shell.execute_reply.started":"2024-06-20T07:52:44.603323Z","shell.execute_reply":"2024-06-20T07:52:44.616864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train.iloc[0:1, 1:]","metadata":{"execution":{"iopub.status.busy":"2024-06-25T08:14:05.408917Z","iopub.execute_input":"2024-06-25T08:14:05.409331Z","iopub.status.idle":"2024-06-25T08:14:05.418945Z","shell.execute_reply.started":"2024-06-25T08:14:05.409294Z","shell.execute_reply":"2024-06-25T08:14:05.418168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def default_image_loader(path):\n    return Image.open(path).convert('RGB')\n\nclass ImageDataset(Dataset):\n    def __init__(self, data_path, df, transform):\n        self.df = df\n        self.loader = default_image_loader\n        self.transform = transform\n        self.dir = data_path\n\n    def __getitem__(self, index):\n        image_name = self.df.file_path[index]\n        image = self.loader(os.path.join(self.dir, image_name))\n        image = self.transform(image)\n        \n        # Lấy dòng dữ liệu từ DataFrame với 3 cột cuối\n        label = self.df.iloc[index, 1:].astype(float).values\n        label = torch.tensor(label, dtype=torch.float)\n        return image, label\n            \n    def __len__(self):\n        return self.df.shape[0]","metadata":{"execution":{"iopub.status.busy":"2024-06-25T08:18:39.090927Z","iopub.execute_input":"2024-06-25T08:18:39.0913Z","iopub.status.idle":"2024-06-25T08:18:39.101729Z","shell.execute_reply.started":"2024-06-25T08:18:39.091268Z","shell.execute_reply":"2024-06-25T08:18:39.100678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"channels=3\nbatch_size=32\nimg_size = (256, 256)\nimg_shape=(img_size[0], img_size[1], channels)","metadata":{"execution":{"iopub.status.busy":"2024-06-25T08:36:33.835107Z","iopub.execute_input":"2024-06-25T08:36:33.835439Z","iopub.status.idle":"2024-06-25T08:36:33.839999Z","shell.execute_reply.started":"2024-06-25T08:36:33.83541Z","shell.execute_reply":"2024-06-25T08:36:33.839234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_transform = transforms.Compose([\n                              transforms.Resize(img_size),\n#                               transforms.RandomHorizontalFlip(),\n#                               transforms.RandomRotation(20),\n                              transforms.ToTensor(),\n#                               transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n                            ])\n\nval_transform = transforms.Compose([\n                              transforms.Resize(img_size),\n                              transforms.ToTensor(),\n#                               transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n                            ])","metadata":{"execution":{"iopub.status.busy":"2024-06-25T08:04:31.633297Z","iopub.execute_input":"2024-06-25T08:04:31.633667Z","iopub.status.idle":"2024-06-25T08:04:31.639448Z","shell.execute_reply.started":"2024-06-25T08:04:31.63363Z","shell.execute_reply":"2024-06-25T08:04:31.638544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_dir = os.path.join(PATH, 'data')\n\ntrain_dataset = ImageDataset(img_dir, data_train, train_transform)\nval_dataset = ImageDataset(img_dir, data_valid, val_transform)\n\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, num_workers=4, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=batch_size, num_workers=4)","metadata":{"execution":{"iopub.status.busy":"2024-06-25T08:36:36.499395Z","iopub.execute_input":"2024-06-25T08:36:36.499798Z","iopub.status.idle":"2024-06-25T08:36:36.507873Z","shell.execute_reply.started":"2024-06-25T08:36:36.499763Z","shell.execute_reply":"2024-06-25T08:36:36.506892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor i, l in train_loader:\n    print(l)\n    break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Model(nn.Module):\n    def __init__(self):\n        super(Model, self).__init__()\n        self.resnetmodel = EfficientNet.from_pretrained('efficientnet-b3')\n        \n        self.fc = nn.Sequential(nn.Linear(1000, 512), nn.ReLU(),\n                                  nn.Linear(512, 3))\n#                                 , nn.Sigmoid())\n        \n    def forward(self, x):\n        x = self.resnetmodel(x)\n        return self.fc(x)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model()","metadata":{"execution":{"iopub.status.busy":"2024-06-25T08:36:46.350204Z","iopub.execute_input":"2024-06-25T08:36:46.350575Z","iopub.status.idle":"2024-06-25T08:36:46.60633Z","shell.execute_reply.started":"2024-06-25T08:36:46.350527Z","shell.execute_reply":"2024-06-25T08:36:46.605347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nimport torch\nfrom torch.optim.optimizer import Optimizer, required\n\nclass RAdam(Optimizer):\n\n    def __init__(self, params, lr=1e-3, betas=(0.9, 0.999), eps=1e-8, weight_decay=0):\n        defaults = dict(lr=lr, betas=betas, eps=eps, weight_decay=weight_decay)\n        self.buffer = [[None, None, None] for ind in range(10)]\n        super(RAdam, self).__init__(params, defaults)\n\n    def __setstate__(self, state):\n        super(RAdam, self).__setstate__(state)\n\n    def step(self, closure=None):\n\n        loss = None\n        if closure is not None:\n            loss = closure()\n\n        for group in self.param_groups:\n\n            for p in group['params']:\n                if p.grad is None:\n                    continue\n                grad = p.grad.data.float()\n                if grad.is_sparse:\n                    raise RuntimeError('RAdam does not support sparse gradients')\n\n                p_data_fp32 = p.data.float()\n\n                state = self.state[p]\n\n                if len(state) == 0:\n                    state['step'] = 0\n                    state['exp_avg'] = torch.zeros_like(p_data_fp32)\n                    state['exp_avg_sq'] = torch.zeros_like(p_data_fp32)\n                else:\n                    state['exp_avg'] = state['exp_avg'].type_as(p_data_fp32)\n                    state['exp_avg_sq'] = state['exp_avg_sq'].type_as(p_data_fp32)\n\n                exp_avg, exp_avg_sq = state['exp_avg'], state['exp_avg_sq']\n                beta1, beta2 = group['betas']\n\n                exp_avg_sq.mul_(beta2).addcmul_(1 - beta2, grad, grad)\n                exp_avg.mul_(beta1).add_(1 - beta1, grad)\n\n                state['step'] += 1\n                buffered = self.buffer[int(state['step'] % 10)]\n                if state['step'] == buffered[0]:\n                    N_sma, step_size = buffered[1], buffered[2]\n                else:\n                    buffered[0] = state['step']\n                    beta2_t = beta2 ** state['step']\n                    N_sma_max = 2 / (1 - beta2) - 1\n                    N_sma = N_sma_max - 2 * state['step'] * beta2_t / (1 - beta2_t)\n                    buffered[1] = N_sma\n\n                    # more conservative since it's an approximated value\n                    if N_sma >= 5:\n                        step_size = group['lr'] * math.sqrt((1 - beta2_t) * (N_sma - 4) / (N_sma_max - 4) * (N_sma - 2) / N_sma * N_sma_max / (N_sma_max - 2)) / (1 - beta1 ** state['step'])\n                    else:\n                        step_size = group['lr'] / (1 - beta1 ** state['step'])\n                    buffered[2] = step_size\n\n                if group['weight_decay'] != 0:\n                    p_data_fp32.add_(-group['weight_decay'] * group['lr'], p_data_fp32)\n\n                # more conservative since it's an approximated value\n                if N_sma >= 5:                    \n                    denom = exp_avg_sq.sqrt().add_(group['eps'])\n                    p_data_fp32.addcdiv_(-step_size, exp_avg, denom)\n                else:\n                    p_data_fp32.add_(-step_size, exp_avg)\n\n                p.data.copy_(p_data_fp32)\n\n        return loss","metadata":{"execution":{"iopub.status.busy":"2024-06-25T08:36:48.576151Z","iopub.execute_input":"2024-06-25T08:36:48.576475Z","iopub.status.idle":"2024-06-25T08:36:48.603963Z","shell.execute_reply.started":"2024-06-25T08:36:48.576447Z","shell.execute_reply":"2024-06-25T08:36:48.603065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"criterion = nn.BCEWithLogitsLoss()\n# optimizer = torch.optim.Adam(model.parameters(), lr=0.001)\noptimizer = RAdam(model.parameters(), lr=0.001)\nscheduler = torch.optim.lr_scheduler.ExponentialLR(optimizer, 0.95)","metadata":{"execution":{"iopub.status.busy":"2024-06-25T08:36:52.353134Z","iopub.execute_input":"2024-06-25T08:36:52.353464Z","iopub.status.idle":"2024-06-25T08:36:52.362507Z","shell.execute_reply.started":"2024-06-25T08:36:52.353436Z","shell.execute_reply":"2024-06-25T08:36:52.36147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.cuda()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nmodel = model.cuda()\n\n# Wrap the model with DataParallel\nmodel = nn.DataParallel(model)\n\nfor epoch in range(10):  # loop over the dataset multiple times\n\n    running_loss = 0.0\n    model.train()\n    correct = 0\n    total = 0\n    for images, labels in tqdm(train_loader):\n        images = images.cuda()\n        labels = labels.cuda()\n        out = model(images)\n        labels = labels.unsqueeze(1).float()\n        labels = labels.view(-1, 3)\n        loss = criterion(out, labels)\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        running_loss += loss.item()\n        \n        total += labels.size(0)\n        out = torch.sigmoid(out)\n        correct += ((out > 0.6).int() == labels).sum().item()\n    \n    print(\"Epoch: {}, Loss: {}, Train Accuracy: {}\".format(epoch, running_loss, round(correct/total, 4)))\n    if epoch % 2 == 1:\n        scheduler.step()\n        \n    model.eval()\n    running_loss = 0\n    correct = 0\n    total = 0\n    \n    with torch.no_grad():\n        for images, labels in val_loader:\n            images = images.cuda()\n            labels = labels.cuda()\n            labels = labels.unsqueeze(1).float()\n            labels = labels.view(-1, 3)\n            out = model(images)\n            loss = criterion(out.data, labels)\n            \n            running_loss += loss.item()\n\n            total += labels.size(0)\n            out = torch.sigmoid(out)\n            correct += ((out > 0.6).int() == labels).sum().item()\n            \n    print(\"Epoch: {}, Loss: {}, Test Accuracy: {}\\n\".format(epoch, running_loss, round(correct/total, 4)))\n    \nprint('Finished Training')","metadata":{"execution":{"iopub.status.busy":"2024-06-25T08:53:53.253112Z","iopub.execute_input":"2024-06-25T08:53:53.253451Z","iopub.status.idle":"2024-06-25T08:53:54.38009Z","shell.execute_reply.started":"2024-06-25T08:53:53.253422Z","shell.execute_reply":"2024-06-25T08:53:54.377537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nfor epoch in range(10):  # loop over the dataset multiple times\n\n    running_loss = 0.0\n    model.train()\n    correct = 0\n    total = 0\n    for images, labels in tqdm(train_loader):\n#         print(\"---------------------------------\",labels)\n        images = images.cuda()\n        labels = labels.cuda()\n        out = model(images)\n        labels = labels.unsqueeze(1).float()\n        labels = labels.view(-1, 3)\n        loss = criterion(out, labels)\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        running_loss += loss.item()\n        \n#         _, predicted = torch.max(out.data, 1)\n        total += labels.size(0)\n        out = torch.sigmoid(out)\n        correct += ((out > 0.6).int() == labels).sum().item()\n    \n    print(\"Epoch: {}, Loss: {}, Train Accuracy: {}\".format(epoch, running_loss, round(correct/total, 4)))\n    if epoch % 2 == 1:\n        scheduler.step()\n        \n    model.eval()\n    running_loss = 0\n    correct = 0\n    total = 0\n    \n    with torch.no_grad():\n        for images, labels in val_loader:\n            images = images.cuda()\n            labels = labels.cuda()\n            labels = labels.unsqueeze(1).float()\n            \n            out = model(images)\n            loss = criterion(out.data, labels)\n            \n            running_loss += loss.item()\n\n            total += labels.size(0)\n            out = torch.sigmoid(out)\n            correct += ((out > 0.6).int() == labels).sum().item()\n            \n    print(\"Epoch: {}, Loss: {}, Test Accuracy: {}\\n\".format(epoch, running_loss, round(correct/total, 4)))\n    \nprint('Finished Training')","metadata":{"execution":{"iopub.status.busy":"2024-06-25T08:24:47.912721Z","iopub.execute_input":"2024-06-25T08:24:47.91311Z","iopub.status.idle":"2024-06-25T08:35:17.643333Z","shell.execute_reply.started":"2024-06-25T08:24:47.913078Z","shell.execute_reply":"2024-06-25T08:35:17.641542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')\ntest_df = test_df.reset_index(drop=True)\ntest_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-20T08:08:56.324753Z","iopub.execute_input":"2024-06-20T08:08:56.325201Z","iopub.status.idle":"2024-06-20T08:08:56.362554Z","shell.execute_reply.started":"2024-06-20T08:08:56.325158Z","shell.execute_reply":"2024-06-20T08:08:56.361723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-20T08:09:02.059111Z","iopub.execute_input":"2024-06-20T08:09:02.059487Z","iopub.status.idle":"2024-06-20T08:09:02.075341Z","shell.execute_reply.started":"2024-06-20T08:09:02.059454Z","shell.execute_reply":"2024-06-20T08:09:02.074469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_dir = os.path.join(PATH, 'test')\n\ntest_dataset = ImageDataset(img_dir, test_df, val_transform)\ntest_loader = DataLoader(test_dataset, batch_size=64, num_workers=8)","metadata":{"execution":{"iopub.status.busy":"2024-06-20T08:09:05.928022Z","iopub.execute_input":"2024-06-20T08:09:05.928373Z","iopub.status.idle":"2024-06-20T08:09:05.933232Z","shell.execute_reply.started":"2024-06-20T08:09:05.92834Z","shell.execute_reply":"2024-06-20T08:09:05.93223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for a, b in test_loader:\n#     print(b)\n    pass\n    break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.eval()\nres_id = []\nres_prob = []\n\nwith torch.no_grad():\n    for images, ids in test_loader:\n        images = images.cuda()\n\n        out = model(images)\n        predicted = torch.sigmoid(out)\n        \n        res_id += ids\n        res_prob += predicted.cpu().numpy().tolist()\n        ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res_prob = [x[0] for x in res_prob]\nsum(res_prob)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({\"image_name\":res_id, \"target\":res_prob})\nsub.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}