{
  "id": 178501,
  "title": "GPU not working",
  "url": "/competitions/landmark-recognition-2020/discussion/178501",
  "author_name": "Hitesh Somani",
  "post_date": "2020-08-30T07:41:32.837000",
  "votes": 1,
  "comment_count": 15,
  "views": 0,
  "content": "<pre><code>class LandmarksDatasetTrain(Dataset):\n    \"\"\"Landmarks dataset.\"\"\" \n\n    def __init__(self, landmarks_frame, root_dir, transform=None):\n        \"\"\"\n        Args:\n            csv_file (string): Path to the csv file with annotations.\n            root_dir (string): Directory with all the images.\n            transform (callable, optional): Optional transform to be applied\n                on a sample.\n        \"\"\"\n        self.landmarks_frame = landmarks_frame\n        self.root_dir = root_dir\n        self.transform = transform \n\n    def __len__(self):\n        return len(self.landmarks_frame)\n\n    def __getitem__(self, idx):\n        if torch.is_tensor(idx):\n            idx = idx.tolist()\n        img_name = os.path.join(self.root_dir,self.landmarks_frame.loc[idx, 'id'][0],self.landmarks_frame.loc[idx, 'id'][1], self.landmarks_frame.loc[idx, 'id'][2], self.landmarks_frame.loc[idx, 'id'])\n        img_name += \".jpg\"\n        image = Image.open(img_name)\n        landmarks = self.landmarks_frame.loc[idx, 'landmark_id']\n        sample = {'image': image, 'landmarks': landmarks}\n\n        if self.transform:\n            sample['image'] = self.transform(sample['image'])\n            sample['landmarks'] = torch.tensor(sample['landmarks'])\n\n        return sample\n\ndataset_train = LandmarksDatasetTrain(landmarks_frame = frame,\n                                      root_dir='/kaggle/input/landmark-recognition-2020/train',\n                                      transform=transform)\n\ntrain_loader = DataLoader(dataset_train, batch_size=4, shuffle=True, num_workers=4, drop_last=False)\n\nclass Net(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.fc1 = nn.Linear(CROP_SIZE*CROP_SIZE*3, 64)\n        self.fc2 = nn.Linear(64, 64)\n        self.fc3 = nn.Linear(64, 64)\n        self.fc4 = nn.Linear(64, frame['landmark_id'].nunique())\n\n    def forward(self, x):\n        x = F.relu(self.fc1(x))\n        x = F.relu(self.fc2(x))\n        x = F.relu(self.fc3(x))\n        x = self.fc4(x)\n        return F.log_softmax(x, dim=1)\n\nnet = Net()\nnet.to(device)\n\nfor epoch in range(3): \n    optimizer = optim.Adam(net.parameters(), lr=0.001)\n    for data in tqdm(train_loader):  \n        X = data['image'].to(device)  \n        y = data['landmarks'].to(device) \n        net.zero_grad()  \n        output = net(X.view(-1,CROP_SIZE*CROP_SIZE*3))  \n        loss = F.nll_loss(output, y) \n        loss.backward()  \n        optimizer.step() \n    print(loss)  \n</code></pre>\n<p>batch size is 4</p>\n<p><code>data['image']</code> and <code>data['landmarks']</code> are tensors <code>device = torch.device(\"cuda:0\")</code> and deep learning library I am using is <code>Pytorch</code> but GPU is still not working for me. Its usage shows 5% and total time for 1 epoch 3.5 to 4 hours </p>\n<p>Will be really helpful if someone points out my mistake. </p>\n<p>Also please see attached image might be helpful to debug</p>\n<p>This is my kaggle notebook link:<br>\n<a href=\"https://www.kaggle.com/hiteshsom/google-landmark-recognition\" target=\"_blank\">https://www.kaggle.com/hiteshsom/google-landmark-recognition</a></p>",
  "messages": [
    {
      "id": 991173,
      "postDate": "2020-08-30T07:41:32.837Z",
      "content": "<pre><code>class LandmarksDatasetTrain(Dataset):\n    \"\"\"Landmarks dataset.\"\"\" \n\n    def __init__(self, landmarks_frame, root_dir, transform=None):\n        \"\"\"\n        Args:\n            csv_file (string): Path to the csv file with annotations.\n            root_dir (string): Directory with all the images.\n            transform (callable, optional): Optional transform to be applied\n                on a sample.\n        \"\"\"\n        self.landmarks_frame = landmarks_frame\n        self.root_dir = root_dir\n        self.transform = transform \n\n    def __len__(self):\n        return len(self.landmarks_frame)\n\n    def __getitem__(self, idx):\n        if torch.is_tensor(idx):\n            idx = idx.tolist()\n        img_name = os.path.join(self.root_dir,self.landmarks_frame.loc[idx, 'id'][0],self.landmarks_frame.loc[idx, 'id'][1], self.landmarks_frame.loc[idx, 'id'][2], self.landmarks_frame.loc[idx, 'id'])\n        img_name += \".jpg\"\n        image = Image.open(img_name)\n        landmarks = self.landmarks_frame.loc[idx, 'landmark_id']\n        sample = {'image': image, 'landmarks': landmarks}\n\n        if self.transform:\n            sample['image'] = self.transform(sample['image'])\n            sample['landmarks'] = torch.tensor(sample['landmarks'])\n\n        return sample\n\ndataset_train = LandmarksDatasetTrain(landmarks_frame = frame,\n                                      root_dir='/kaggle/input/landmark-recognition-2020/train',\n                                      transform=transform)\n\ntrain_loader = DataLoader(dataset_train, batch_size=4, shuffle=True, num_workers=4, drop_last=False)\n\nclass Net(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.fc1 = nn.Linear(CROP_SIZE*CROP_SIZE*3, 64)\n        self.fc2 = nn.Linear(64, 64)\n        self.fc3 = nn.Linear(64, 64)\n        self.fc4 = nn.Linear(64, frame['landmark_id'].nunique())\n\n    def forward(self, x):\n        x = F.relu(self.fc1(x))\n        x = F.relu(self.fc2(x))\n        x = F.relu(self.fc3(x))\n        x = self.fc4(x)\n        return F.log_softmax(x, dim=1)\n\nnet = Net()\nnet.to(device)\n\nfor epoch in range(3): \n    optimizer = optim.Adam(net.parameters(), lr=0.001)\n    for data in tqdm(train_loader):  \n        X = data['image'].to(device)  \n        y = data['landmarks'].to(device) \n        net.zero_grad()  \n        output = net(X.view(-1,CROP_SIZE*CROP_SIZE*3))  \n        loss = F.nll_loss(output, y) \n        loss.backward()  \n        optimizer.step() \n    print(loss)  \n</code></pre>\n<p>batch size is 4</p>\n<p><code>data['image']</code> and <code>data['landmarks']</code> are tensors <code>device = torch.device(\"cuda:0\")</code> and deep learning library I am using is <code>Pytorch</code> but GPU is still not working for me. Its usage shows 5% and total time for 1 epoch 3.5 to 4 hours </p>\n<p>Will be really helpful if someone points out my mistake. </p>\n<p>Also please see attached image might be helpful to debug</p>\n<p>This is my kaggle notebook link:<br>\n<a href=\"https://www.kaggle.com/hiteshsom/google-landmark-recognition\" target=\"_blank\">https://www.kaggle.com/hiteshsom/google-landmark-recognition</a></p>",
      "rawMarkdown": "```\nclass LandmarksDatasetTrain(Dataset):\n    \"\"\"Landmarks dataset.\"\"\" \n\n    def __init__(self, landmarks_frame, root_dir, transform=None):\n        \"\"\"\n        Args:\n            csv_file (string): Path to the csv file with annotations.\n            root_dir (string): Directory with all the images.\n            transform (callable, optional): Optional transform to be applied\n                on a sample.\n        \"\"\"\n        self.landmarks_frame = landmarks_frame\n        self.root_dir = root_dir\n        self.transform = transform \n\n    def __len__(self):\n        return len(self.landmarks_frame)\n\n    def __getitem__(self, idx):\n        if torch.is_tensor(idx):\n            idx = idx.tolist()\n        img_name = os.path.join(self.root_dir,self.landmarks_frame.loc[idx, 'id'][0],self.landmarks_frame.loc[idx, 'id'][1], self.landmarks_frame.loc[idx, 'id'][2], self.landmarks_frame.loc[idx, 'id'])\n        img_name += \".jpg\"\n        image = Image.open(img_name)\n        landmarks = self.landmarks_frame.loc[idx, 'landmark_id']\n        sample = {'image': image, 'landmarks': landmarks}\n\n        if self.transform:\n            sample['image'] = self.transform(sample['image'])\n            sample['landmarks'] = torch.tensor(sample['landmarks'])\n\n        return sample\n\ndataset_train = LandmarksDatasetTrain(landmarks_frame = frame,\n                                      root_dir='/kaggle/input/landmark-recognition-2020/train',\n                                      transform=transform)\n\ntrain_loader = DataLoader(dataset_train, batch_size=4, shuffle=True, num_workers=4, drop_last=False)\n\nclass Net(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.fc1 = nn.Linear(CROP_SIZE*CROP_SIZE*3, 64)\n        self.fc2 = nn.Linear(64, 64)\n        self.fc3 = nn.Linear(64, 64)\n        self.fc4 = nn.Linear(64, frame['landmark_id'].nunique())\n\n    def forward(self, x):\n        x = F.relu(self.fc1(x))\n        x = F.relu(self.fc2(x))\n        x = F.relu(self.fc3(x))\n        x = self.fc4(x)\n        return F.log_softmax(x, dim=1)\n\nnet = Net()\nnet.to(device)\n\nfor epoch in range(3): \n    optimizer = optim.Adam(net.parameters(), lr=0.001)\n    for data in tqdm(train_loader):  \n        X = data['image'].to(device)  \n        y = data['landmarks'].to(device) \n        net.zero_grad()  \n        output = net(X.view(-1,CROP_SIZE*CROP_SIZE*3))  \n        loss = F.nll_loss(output, y) \n        loss.backward()  \n        optimizer.step() \n    print(loss)  \n\n```\nbatch size is 4\n\n`data['image']` and `data['landmarks']` are tensors `device = torch.device(\"cuda:0\")` and deep learning library I am using is `Pytorch` but GPU is still not working for me. Its usage shows 5% and total time for 1 epoch 3.5 to 4 hours \n\nWill be really helpful if someone points out my mistake. \n\nAlso please see attached image might be helpful to debug\n\nThis is my kaggle notebook link:\nhttps://www.kaggle.com/hiteshsom/google-landmark-recognition",
      "votes": 1
    },
    {
      "id": 1004246,
      "postDate": "2020-09-09T14:57:54.150Z",
      "content": "<p>Does below information help anyone to understand why GPU is not working ?</p>\n<pre><code>class LandmarksDatasetTrain(Dataset):\n    \"\"\"Landmarks dataset.\"\"\" \n\n    def __init__(self, landmarks_frame, root_dir, transform=None):\n        \"\"\"\n        Args:\n            csv_file (string): Path to the csv file with annotations.\n            root_dir (string): Directory with all the images.\n            transform (callable, optional): Optional transform to be applied\n                on a sample.\n        \"\"\"\n        self.landmarks_frame = landmarks_frame\n        self.root_dir = root_dir\n        self.transform = transform \n\n    def __len__(self):\n        return len(self.landmarks_frame)\n\n    def __getitem__(self, idx):\n        if torch.is_tensor(idx):\n            idx = idx.tolist()\n        img_name = os.path.join(self.root_dir,self.landmarks_frame.loc[idx, 'id'][0],self.landmarks_frame.loc[idx, 'id'][1], self.landmarks_frame.loc[idx, 'id'][2], self.landmarks_frame.loc[idx, 'id'])\n        img_name += \".jpg\"\n        image = Image.open(img_name)\n        landmarks = self.landmarks_frame.loc[idx, 'landmark_id']\n        sample = {'image': image, 'landmarks': landmarks}\n\n        if self.transform:\n            sample['image'] = self.transform(sample['image'])\n            sample['landmarks'] = torch.tensor(sample['landmarks'])\n\n        return sample\n\ndataset_train = LandmarksDatasetTrain(landmarks_frame = frame,\n                                      root_dir='/kaggle/input/landmark-recognition-2020/train',\n                                      transform=transform)\n\ntrain_loader = DataLoader(dataset_train, batch_size=4, shuffle=True, num_workers=4, drop_last=False)\n</code></pre>\n<p>Is loading data causing problem <a href=\"https://www.kaggle.com/sanjaydsb\" target=\"_blank\">@sanjaydsb</a> <a href=\"https://www.kaggle.com/heyytanay\" target=\"_blank\">@heyytanay</a> </p>",
      "rawMarkdown": "Does below information help anyone to understand why GPU is not working ?\n```\n\nclass LandmarksDatasetTrain(Dataset):\n    \"\"\"Landmarks dataset.\"\"\" \n\n    def __init__(self, landmarks_frame, root_dir, transform=None):\n        \"\"\"\n        Args:\n            csv_file (string): Path to the csv file with annotations.\n            root_dir (string): Directory with all the images.\n            transform (callable, optional): Optional transform to be applied\n                on a sample.\n        \"\"\"\n        self.landmarks_frame = landmarks_frame\n        self.root_dir = root_dir\n        self.transform = transform \n\n    def __len__(self):\n        return len(self.landmarks_frame)\n\n    def __getitem__(self, idx):\n        if torch.is_tensor(idx):\n            idx = idx.tolist()\n        img_name = os.path.join(self.root_dir,self.landmarks_frame.loc[idx, 'id'][0],self.landmarks_frame.loc[idx, 'id'][1], self.landmarks_frame.loc[idx, 'id'][2], self.landmarks_frame.loc[idx, 'id'])\n        img_name += \".jpg\"\n        image = Image.open(img_name)\n        landmarks = self.landmarks_frame.loc[idx, 'landmark_id']\n        sample = {'image': image, 'landmarks': landmarks}\n\n        if self.transform:\n            sample['image'] = self.transform(sample['image'])\n            sample['landmarks'] = torch.tensor(sample['landmarks'])\n\n        return sample\n\ndataset_train = LandmarksDatasetTrain(landmarks_frame = frame,\n                                      root_dir='/kaggle/input/landmark-recognition-2020/train',\n                                      transform=transform)\n\ntrain_loader = DataLoader(dataset_train, batch_size=4, shuffle=True, num_workers=4, drop_last=False)\n```\n\nIs loading data causing problem @sanjaydsb @heyytanay ",
      "replies": [
        {
          "id": 1004340,
          "postDate": "2020-09-09T16:39:46.570Z",
          "content": "<p>Step 1. GPU should be enabled as an accelerator<br>\nstep 2. The above code does not contain any info about the device you are using<br>\nstep 3. use model=model.to('cuda')<br>\nstep 4. use sample= sample.to('cuda')</p>\n<p>These steps will help kernel to start consuming ram form GPU. Please upvote if you find it helpful! </p>",
          "rawMarkdown": "Step 1. GPU should be enabled as an accelerator\nstep 2. The above code does not contain any info about the device you are using\nstep 3. use model=model.to('cuda')\nstep 4. use sample= sample.to('cuda')\n\nThese steps will help kernel to start consuming ram form GPU. Please upvote if you find it helpful! \n"
        },
        {
          "id": 1004759,
          "postDate": "2020-09-10T02:29:56.377Z",
          "content": "<p>Step 1: GPU is enabled as accelerator</p>\n<p>Step 2:<br>\n<code>torch.cuda.get_device_name()</code></p>\n<p>This gives 'Tesla P100-PCIE-16GB'</p>\n<p>Step 3:</p>\n<pre><code>if torch.cuda.is_available():\n    device = torch.device(\"cuda:0\")  # you can continue going on here, like cuda:1 cuda:2....etc. \n#     device = \"cuda\"\n    print(\"Running on the GPU\")\nelse:\n    device = torch.device(\"cpu\")\n#     device = \"cpu\"\n    print(\"Running on the CPU\")\n</code></pre>\n<p>So <code>.to(device)</code> should work. I tried .to('cuda') did not solve the problem.</p>\n<p>Step 4:<br>\nsample is dictionary object so it throws error if I do sample.to(device). Error that dict object does not have to() method </p>",
          "rawMarkdown": "Step 1: GPU is enabled as accelerator\n\nStep 2:\n```torch.cuda.get_device_name()```\n\nThis gives 'Tesla P100-PCIE-16GB'\n\nStep 3:\n```\nif torch.cuda.is_available():\n    device = torch.device(\"cuda:0\")  # you can continue going on here, like cuda:1 cuda:2....etc. \n#     device = \"cuda\"\n    print(\"Running on the GPU\")\nelse:\n    device = torch.device(\"cpu\")\n#     device = \"cpu\"\n    print(\"Running on the CPU\")\n```\n\nSo ```.to(device)``` should work. I tried .to('cuda') did not solve the problem.\n\nStep 4:\nsample is dictionary object so it throws error if I do sample.to(device). Error that dict object does not have to() method \n"
        }
      ]
    },
    {
      "id": 1001520,
      "postDate": "2020-09-07T11:39:48.637Z",
      "content": "<p>you have to make sure the accelerator selected is GPU and try reducing your batch size. Additionally try this device = \"cuda\" if torch.cuda.is_available() else \"cpu\"</p>",
      "rawMarkdown": "you have to make sure the accelerator selected is GPU and try reducing your batch size. Additionally try this device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n",
      "replies": [
        {
          "id": 1001537,
          "postDate": "2020-09-07T11:50:37.643Z",
          "content": "<p>Reducing batch size from 32 to 8 did increase the number of it/s from 3.5 to 12it/s but did not start GPU and also the total time for 1 loop is still 3.5 hrs.</p>\n<p>GPU is on as such but the usage is just 5%</p>\n<p>while CPU usage shows 185%</p>",
          "rawMarkdown": "Reducing batch size from 32 to 8 did increase the number of it/s from 3.5 to 12it/s but did not start GPU and also the total time for 1 loop is still 3.5 hrs.\n\nGPU is on as such but the usage is just 5%\n\nwhile CPU usage shows 185%",
          "votes": 1
        },
        {
          "id": 1001568,
          "postDate": "2020-09-07T12:14:54.053Z",
          "content": "<p>try using model.cuda()</p>",
          "rawMarkdown": "try using model.cuda()\n"
        },
        {
          "id": 1001596,
          "postDate": "2020-09-07T12:31:56.040Z",
          "rawMarkdown": "",
          "isDeleted": true
        },
        {
          "id": 1001597,
          "postDate": "2020-09-07T12:31:56.040Z",
          "rawMarkdown": "",
          "isDeleted": true
        },
        {
          "id": 1001595,
          "postDate": "2020-09-07T12:31:56.040Z",
          "content": "<p>Tried <code>model.cuda()</code> did not improve the total time or it/s</p>",
          "rawMarkdown": "Tried ```model.cuda()``` did not improve the total time or it/s"
        },
        {
          "id": 1001598,
          "postDate": "2020-09-07T12:31:56.397Z",
          "rawMarkdown": "",
          "isDeleted": true
        },
        {
          "id": 1001616,
          "postDate": "2020-09-07T12:39:41.597Z",
          "content": "<p>use optimizer.zero_grad() instead net.zero_grad() <a href=\"https://www.kaggle.com/hiteshsom\" target=\"_blank\">@hiteshsom</a> </p>",
          "rawMarkdown": "use optimizer.zero_grad() instead net.zero_grad() @hiteshsom "
        },
        {
          "id": 1004180,
          "postDate": "2020-09-09T14:13:42.773Z",
          "content": "<p>Still no improvement for me.</p>",
          "rawMarkdown": "Still no improvement for me."
        }
      ]
    },
    {
      "id": 991183,
      "postDate": "2020-08-30T07:55:01.120Z",
      "content": "<p>How is the training going on? I mean does the training process looks fast compared to when you run your kernel on CPU?<br>\nAlso, try this:<br>\n<code>device = \"cuda\" if torch.cuda.is_available() else \"cpu\"</code></p>",
      "rawMarkdown": "How is the training going on? I mean does the training process looks fast compared to when you run your kernel on CPU?\nAlso, try this:\n`device = \"cuda\" if torch.cuda.is_available() else \"cpu\"`",
      "replies": [
        {
          "id": 1001521,
          "postDate": "2020-09-07T11:40:41.403Z",
          "content": "<p>tqdm says <br>\n3.5 it/s approx<br>\nbatch size 32</p>\n<p>total time to complete one training loop is around 3.5 hour approx</p>",
          "rawMarkdown": "tqdm says \n3.5 it/s approx\nbatch size 32\n\ntotal time to complete one training loop is around 3.5 hour approx"
        },
        {
          "id": 1001529,
          "postDate": "2020-09-07T11:45:06.463Z",
          "content": "<p>I tried <code>device = \"cuda\" if torch.cuda.is_available() else \"cpu\"</code> but did not start GPU</p>",
          "rawMarkdown": "I tried ```device = \"cuda\" if torch.cuda.is_available() else \"cpu\"``` but did not start GPU"
        }
      ]
    }
  ],
  "comments": [
    {
      "id": 1004246,
      "author_name": "Hitesh Somani",
      "author_url": "",
      "post_date": "2020-09-09T14:57:54.150000",
      "content": "<p>Does below information help anyone to understand why GPU is not working ?</p>\n<pre><code>class LandmarksDatasetTrain(Dataset):\n    \"\"\"Landmarks dataset.\"\"\" \n\n    def __init__(self, landmarks_frame, root_dir, transform=None):\n        \"\"\"\n        Args:\n            csv_file (string): Path to the csv file with annotations.\n            root_dir (string): Directory with all the images.\n            transform (callable, optional): Optional transform to be applied\n                on a sample.\n        \"\"\"\n        self.landmarks_frame = landmarks_frame\n        self.root_dir = root_dir\n        self.transform = transform \n\n    def __len__(self):\n        return len(self.landmarks_frame)\n\n    def __getitem__(self, idx):\n        if torch.is_tensor(idx):\n            idx = idx.tolist()\n        img_name = os.path.join(self.root_dir,self.landmarks_frame.loc[idx, 'id'][0],self.landmarks_frame.loc[idx, 'id'][1], self.landmarks_frame.loc[idx, 'id'][2], self.landmarks_frame.loc[idx, 'id'])\n        img_name += \".jpg\"\n        image = Image.open(img_name)\n        landmarks = self.landmarks_frame.loc[idx, 'landmark_id']\n        sample = {'image': image, 'landmarks': landmarks}\n\n        if self.transform:\n            sample['image'] = self.transform(sample['image'])\n            sample['landmarks'] = torch.tensor(sample['landmarks'])\n\n        return sample\n\ndataset_train = LandmarksDatasetTrain(landmarks_frame = frame,\n                                      root_dir='/kaggle/input/landmark-recognition-2020/train',\n                                      transform=transform)\n\ntrain_loader = DataLoader(dataset_train, batch_size=4, shuffle=True, num_workers=4, drop_last=False)\n</code></pre>\n<p>Is loading data causing problem <a href=\"https://www.kaggle.com/sanjaydsb\" target=\"_blank\">@sanjaydsb</a> <a href=\"https://www.kaggle.com/heyytanay\" target=\"_blank\">@heyytanay</a> </p>",
      "votes": 0,
      "replies": [
        {
          "id": 1004340,
          "author_name": "sanjay bhargav danyamraju",
          "author_url": "",
          "post_date": "2020-09-09T16:39:46.570000",
          "content": "<p>Step 1. GPU should be enabled as an accelerator<br>\nstep 2. The above code does not contain any info about the device you are using<br>\nstep 3. use model=model.to('cuda')<br>\nstep 4. use sample= sample.to('cuda')</p>\n<p>These steps will help kernel to start consuming ram form GPU. Please upvote if you find it helpful! </p>",
          "votes": 0,
          "replies": []
        },
        {
          "id": 1004759,
          "author_name": "Hitesh Somani",
          "author_url": "",
          "post_date": "2020-09-10T02:29:56.377000",
          "content": "<p>Step 1: GPU is enabled as accelerator</p>\n<p>Step 2:<br>\n<code>torch.cuda.get_device_name()</code></p>\n<p>This gives 'Tesla P100-PCIE-16GB'</p>\n<p>Step 3:</p>\n<pre><code>if torch.cuda.is_available():\n    device = torch.device(\"cuda:0\")  # you can continue going on here, like cuda:1 cuda:2....etc. \n#     device = \"cuda\"\n    print(\"Running on the GPU\")\nelse:\n    device = torch.device(\"cpu\")\n#     device = \"cpu\"\n    print(\"Running on the CPU\")\n</code></pre>\n<p>So <code>.to(device)</code> should work. I tried .to('cuda') did not solve the problem.</p>\n<p>Step 4:<br>\nsample is dictionary object so it throws error if I do sample.to(device). Error that dict object does not have to() method </p>",
          "votes": 0,
          "replies": []
        }
      ]
    },
    {
      "id": 1001520,
      "author_name": "sanjay bhargav danyamraju",
      "author_url": "",
      "post_date": "2020-09-07T11:39:48.637000",
      "content": "<p>you have to make sure the accelerator selected is GPU and try reducing your batch size. Additionally try this device = \"cuda\" if torch.cuda.is_available() else \"cpu\"</p>",
      "votes": 0,
      "replies": [
        {
          "id": 1001537,
          "author_name": "Hitesh Somani",
          "author_url": "",
          "post_date": "2020-09-07T11:50:37.643000",
          "content": "<p>Reducing batch size from 32 to 8 did increase the number of it/s from 3.5 to 12it/s but did not start GPU and also the total time for 1 loop is still 3.5 hrs.</p>\n<p>GPU is on as such but the usage is just 5%</p>\n<p>while CPU usage shows 185%</p>",
          "votes": 1,
          "replies": []
        },
        {
          "id": 1001568,
          "author_name": "sanjay bhargav danyamraju",
          "author_url": "",
          "post_date": "2020-09-07T12:14:54.053000",
          "content": "<p>try using model.cuda()</p>",
          "votes": 0,
          "replies": []
        },
        {
          "id": 1001596,
          "author_name": "",
          "author_url": "",
          "post_date": "2020-09-07T12:31:56.040000",
          "content": "",
          "votes": 0,
          "replies": []
        },
        {
          "id": 1001597,
          "author_name": "",
          "author_url": "",
          "post_date": "2020-09-07T12:31:56.040000",
          "content": "",
          "votes": 0,
          "replies": []
        },
        {
          "id": 1001595,
          "author_name": "Hitesh Somani",
          "author_url": "",
          "post_date": "2020-09-07T12:31:56.040000",
          "content": "<p>Tried <code>model.cuda()</code> did not improve the total time or it/s</p>",
          "votes": 0,
          "replies": []
        },
        {
          "id": 1001598,
          "author_name": "",
          "author_url": "",
          "post_date": "2020-09-07T12:31:56.397000",
          "content": "",
          "votes": 0,
          "replies": []
        },
        {
          "id": 1001616,
          "author_name": "sanjay bhargav danyamraju",
          "author_url": "",
          "post_date": "2020-09-07T12:39:41.597000",
          "content": "<p>use optimizer.zero_grad() instead net.zero_grad() <a href=\"https://www.kaggle.com/hiteshsom\" target=\"_blank\">@hiteshsom</a> </p>",
          "votes": 0,
          "replies": []
        },
        {
          "id": 1004180,
          "author_name": "Hitesh Somani",
          "author_url": "",
          "post_date": "2020-09-09T14:13:42.773000",
          "content": "<p>Still no improvement for me.</p>",
          "votes": 0,
          "replies": []
        }
      ]
    },
    {
      "id": 991183,
      "author_name": "Tanay Mehta",
      "author_url": "",
      "post_date": "2020-08-30T07:55:01.120000",
      "content": "<p>How is the training going on? I mean does the training process looks fast compared to when you run your kernel on CPU?<br>\nAlso, try this:<br>\n<code>device = \"cuda\" if torch.cuda.is_available() else \"cpu\"</code></p>",
      "votes": 0,
      "replies": [
        {
          "id": 1001521,
          "author_name": "Hitesh Somani",
          "author_url": "",
          "post_date": "2020-09-07T11:40:41.403000",
          "content": "<p>tqdm says <br>\n3.5 it/s approx<br>\nbatch size 32</p>\n<p>total time to complete one training loop is around 3.5 hour approx</p>",
          "votes": 0,
          "replies": []
        },
        {
          "id": 1001529,
          "author_name": "Hitesh Somani",
          "author_url": "",
          "post_date": "2020-09-07T11:45:06.463000",
          "content": "<p>I tried <code>device = \"cuda\" if torch.cuda.is_available() else \"cpu\"</code> but did not start GPU</p>",
          "votes": 0,
          "replies": []
        }
      ]
    }
  ],
  "raw_markdown_by_id": {
    "991173": "```\nclass LandmarksDatasetTrain(Dataset):\n    \"\"\"Landmarks dataset.\"\"\" \n\n    def __init__(self, landmarks_frame, root_dir, transform=None):\n        \"\"\"\n        Args:\n            csv_file (string): Path to the csv file with annotations.\n            root_dir (string): Directory with all the images.\n            transform (callable, optional): Optional transform to be applied\n                on a sample.\n        \"\"\"\n        self.landmarks_frame = landmarks_frame\n        self.root_dir = root_dir\n        self.transform = transform \n\n    def __len__(self):\n        return len(self.landmarks_frame)\n\n    def __getitem__(self, idx):\n        if torch.is_tensor(idx):\n            idx = idx.tolist()\n        img_name = os.path.join(self.root_dir,self.landmarks_frame.loc[idx, 'id'][0],self.landmarks_frame.loc[idx, 'id'][1], self.landmarks_frame.loc[idx, 'id'][2], self.landmarks_frame.loc[idx, 'id'])\n        img_name += \".jpg\"\n        image = Image.open(img_name)\n        landmarks = self.landmarks_frame.loc[idx, 'landmark_id']\n        sample = {'image': image, 'landmarks': landmarks}\n\n        if self.transform:\n            sample['image'] = self.transform(sample['image'])\n            sample['landmarks'] = torch.tensor(sample['landmarks'])\n\n        return sample\n\ndataset_train = LandmarksDatasetTrain(landmarks_frame = frame,\n                                      root_dir='/kaggle/input/landmark-recognition-2020/train',\n                                      transform=transform)\n\ntrain_loader = DataLoader(dataset_train, batch_size=4, shuffle=True, num_workers=4, drop_last=False)\n\nclass Net(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.fc1 = nn.Linear(CROP_SIZE*CROP_SIZE*3, 64)\n        self.fc2 = nn.Linear(64, 64)\n        self.fc3 = nn.Linear(64, 64)\n        self.fc4 = nn.Linear(64, frame['landmark_id'].nunique())\n\n    def forward(self, x):\n        x = F.relu(self.fc1(x))\n        x = F.relu(self.fc2(x))\n        x = F.relu(self.fc3(x))\n        x = self.fc4(x)\n        return F.log_softmax(x, dim=1)\n\nnet = Net()\nnet.to(device)\n\nfor epoch in range(3): \n    optimizer = optim.Adam(net.parameters(), lr=0.001)\n    for data in tqdm(train_loader):  \n        X = data['image'].to(device)  \n        y = data['landmarks'].to(device) \n        net.zero_grad()  \n        output = net(X.view(-1,CROP_SIZE*CROP_SIZE*3))  \n        loss = F.nll_loss(output, y) \n        loss.backward()  \n        optimizer.step() \n    print(loss)  \n\n```\nbatch size is 4\n\n`data['image']` and `data['landmarks']` are tensors `device = torch.device(\"cuda:0\")` and deep learning library I am using is `Pytorch` but GPU is still not working for me. Its usage shows 5% and total time for 1 epoch 3.5 to 4 hours \n\nWill be really helpful if someone points out my mistake. \n\nAlso please see attached image might be helpful to debug\n\nThis is my kaggle notebook link:\nhttps://www.kaggle.com/hiteshsom/google-landmark-recognition",
    "1004246": "Does below information help anyone to understand why GPU is not working ?\n```\n\nclass LandmarksDatasetTrain(Dataset):\n    \"\"\"Landmarks dataset.\"\"\" \n\n    def __init__(self, landmarks_frame, root_dir, transform=None):\n        \"\"\"\n        Args:\n            csv_file (string): Path to the csv file with annotations.\n            root_dir (string): Directory with all the images.\n            transform (callable, optional): Optional transform to be applied\n                on a sample.\n        \"\"\"\n        self.landmarks_frame = landmarks_frame\n        self.root_dir = root_dir\n        self.transform = transform \n\n    def __len__(self):\n        return len(self.landmarks_frame)\n\n    def __getitem__(self, idx):\n        if torch.is_tensor(idx):\n            idx = idx.tolist()\n        img_name = os.path.join(self.root_dir,self.landmarks_frame.loc[idx, 'id'][0],self.landmarks_frame.loc[idx, 'id'][1], self.landmarks_frame.loc[idx, 'id'][2], self.landmarks_frame.loc[idx, 'id'])\n        img_name += \".jpg\"\n        image = Image.open(img_name)\n        landmarks = self.landmarks_frame.loc[idx, 'landmark_id']\n        sample = {'image': image, 'landmarks': landmarks}\n\n        if self.transform:\n            sample['image'] = self.transform(sample['image'])\n            sample['landmarks'] = torch.tensor(sample['landmarks'])\n\n        return sample\n\ndataset_train = LandmarksDatasetTrain(landmarks_frame = frame,\n                                      root_dir='/kaggle/input/landmark-recognition-2020/train',\n                                      transform=transform)\n\ntrain_loader = DataLoader(dataset_train, batch_size=4, shuffle=True, num_workers=4, drop_last=False)\n```\n\nIs loading data causing problem @sanjaydsb @heyytanay ",
    "1001520": "you have to make sure the accelerator selected is GPU and try reducing your batch size. Additionally try this device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n",
    "991183": "How is the training going on? I mean does the training process looks fast compared to when you run your kernel on CPU?\nAlso, try this:\n`device = \"cuda\" if torch.cuda.is_available() else \"cpu\"`"
  }
}