{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n\n%matplotlib inline\nimport os\nimport time\nimport tqdm\nimport torch\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nimport matplotlib.pyplot as plt\nfrom torchvision import models, transforms\nfrom torch.utils.data import TensorDataset,DataLoader , Dataset\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import f1_score\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-02T08:21:38.724313Z","iopub.execute_input":"2022-08-02T08:21:38.725102Z","iopub.status.idle":"2022-08-02T08:21:41.721089Z","shell.execute_reply.started":"2022-08-02T08:21:38.724993Z","shell.execute_reply":"2022-08-02T08:21:41.719769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = '../input/human-protein-atlas-image-classification/train/'\ntest_path = '../input/human-protein-atlas-image-classification/test/'\n","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:22:00.795716Z","iopub.execute_input":"2022-08-02T08:22:00.796347Z","iopub.status.idle":"2022-08-02T08:22:00.801240Z","shell.execute_reply.started":"2022-08-02T08:22:00.796311Z","shell.execute_reply":"2022-08-02T08:22:00.800141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"name_label_dict = {\n0:  'Nucleoplasm',\n1:  'Nuclear membrane',\n2:  'Nucleoli',   \n3:  'Nucleoli fibrillar center',\n4:  'Nuclear speckles',\n5:  'Nuclear bodies',\n6:  'Endoplasmic reticulum',   \n7:  'Golgi apparatus',\n8:  'Peroxisomes',\n9:  'Endosomes',\n10:  'Lysosomes',\n11:  'Intermediate filaments',\n12:  'Actin filaments',\n13:  'Focal adhesion sites',   \n14:  'Microtubules',\n15:  'Microtubule ends',  \n16:  'Cytokinetic bridge',   \n17:  'Mitotic spindle',\n18:  'Microtubule organizing center',  \n19:  'Centrosome',\n20:  'Lipid droplets',\n21:  'Plasma membrane',   \n22:  'Cell junctions', \n23:  'Mitochondria',\n24:  'Aggresome',\n25:  'Cytosol',\n26:  'Cytoplasmic bodies',   \n27:  'Rods & rings' }","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:22:06.017845Z","iopub.execute_input":"2022-08-02T08:22:06.018206Z","iopub.status.idle":"2022-08-02T08:22:06.024955Z","shell.execute_reply.started":"2022-08-02T08:22:06.018175Z","shell.execute_reply":"2022-08-02T08:22:06.023810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv('../input/human-protein-atlas-image-classification/train.csv').set_index('Id')\ntrain_csv['Target'] = [[int(i) for i in s.split()] for s in train_csv['Target']]\ntrain_csv.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:22:08.728004Z","iopub.execute_input":"2022-08-02T08:22:08.728518Z","iopub.status.idle":"2022-08-02T08:22:09.154115Z","shell.execute_reply.started":"2022-08-02T08:22:08.728467Z","shell.execute_reply":"2022-08-02T08:22:09.153123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = np.concatenate(train_csv['Target'].values).ravel()\nlabels","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:22:18.682352Z","iopub.execute_input":"2022-08-02T08:22:18.682712Z","iopub.status.idle":"2022-08-02T08:22:18.719691Z","shell.execute_reply.started":"2022-08-02T08:22:18.682678Z","shell.execute_reply":"2022-08-02T08:22:18.718717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dublicat = []\nfor c in range(len(name_label_dict)):\n    count = (labels == c).astype('uint8').sum()\n    if count < 100:\n        dublicat.append(c)\ndublicat    ","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:22:20.888894Z","iopub.execute_input":"2022-08-02T08:22:20.889462Z","iopub.status.idle":"2022-08-02T08:22:20.901609Z","shell.execute_reply.started":"2022-08-02T08:22:20.889428Z","shell.execute_reply":"2022-08-02T08:22:20.900446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,id in enumerate(train_csv.index):\n    for l in train_csv.loc[id]['Target']:\n        if l in dublicat: \n            train_csv = train_csv.append([train_csv.loc[id]]*3)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:22:26.955076Z","iopub.execute_input":"2022-08-02T08:22:26.955528Z","iopub.status.idle":"2022-08-02T08:22:49.216810Z","shell.execute_reply.started":"2022-08-02T08:22:26.955497Z","shell.execute_reply":"2022-08-02T08:22:49.215671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vector = []\nfor i,id in enumerate(train_csv.index):\n    vector.append((np.eye(len(name_label_dict))[train_csv.iloc[i]['Target']]))\ntrain_csv['vector'] = vector    \ntrain_csv.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:22:52.501846Z","iopub.execute_input":"2022-08-02T08:22:52.502418Z","iopub.status.idle":"2022-08-02T08:22:55.459661Z","shell.execute_reply.started":"2022-08-02T08:22:52.502381Z","shell.execute_reply":"2022-08-02T08:22:55.458706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataGenerator(Dataset):\n    def __init__(self,df,path):\n        self.df = df\n        self.path = path\n    \n    def __getitem__(self,idx):\n        img_name = self.df.index[idx]\n        label_vector = self.df.iloc[idx]['vector']\n        label_vector = np.sum(label_vector,axis=0)\n        img_path = os.path.join(self.path,img_name)\n        r = np.array(Image.open(img_path+'_red.png'))\n        g = np.array(Image.open(img_path+'_green.png'))\n        b = np.array(Image.open(img_path+'_blue.png'))\n        y = np.array(Image.open(img_path+'_yellow.png'))\n        img = np.stack([r,g,b,y],axis=-1)\n        img = cv2.resize(img , (299,299))\n        img = np.array(img)/255\n        img = torch.from_numpy(img).permute(2,0,1)\n        label_vector = torch.from_numpy(label_vector)\n        return img , label_vector\n    \n    def __len__(self):\n        return len(self.df)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:23:04.343604Z","iopub.execute_input":"2022-08-02T08:23:04.343976Z","iopub.status.idle":"2022-08-02T08:23:04.355851Z","shell.execute_reply.started":"2022-08-02T08:23:04.343943Z","shell.execute_reply":"2022-08-02T08:23:04.354870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"use_gpu = torch.cuda.is_available()\nif use_gpu:\n    print('GPU is available!')\n    device = \"cuda\"\n    pinMem = True\nelse:\n    print('GPU is not available!')\n    device = \"cpu\"\n    pinMem = False","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:23:05.683119Z","iopub.execute_input":"2022-08-02T08:23:05.683843Z","iopub.status.idle":"2022-08-02T08:23:05.748885Z","shell.execute_reply.started":"2022-08-02T08:23:05.683802Z","shell.execute_reply":"2022-08-02T08:23:05.747681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df , val_df = train_test_split(train_csv,test_size=0.2,random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:23:07.135864Z","iopub.execute_input":"2022-08-02T08:23:07.136487Z","iopub.status.idle":"2022-08-02T08:23:07.151147Z","shell.execute_reply.started":"2022-08-02T08:23:07.136452Z","shell.execute_reply":"2022-08-02T08:23:07.150167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainpath = '../input/human-protein-atlas-image-classification/train'\ntrainDataset = DataGenerator(train_df ,trainpath )\nvalDataset = DataGenerator(val_df ,trainpath )\nbatch_size = 64\n\ntrainDataLoader = DataLoader(trainDataset , batch_size = batch_size , shuffle = True, num_workers=2,pin_memory =True)\nvalDataLoader = DataLoader(valDataset, batch_size = batch_size, shuffle = True , num_workers = 2, pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:23:12.772294Z","iopub.execute_input":"2022-08-02T08:23:12.772653Z","iopub.status.idle":"2022-08-02T08:23:12.778642Z","shell.execute_reply.started":"2022-08-02T08:23:12.772622Z","shell.execute_reply":"2022-08-02T08:23:12.777405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = models.inception_v3(pretrained = True)\nprint(model)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:23:15.402424Z","iopub.execute_input":"2022-08-02T08:23:15.402828Z","iopub.status.idle":"2022-08-02T08:23:22.383729Z","shell.execute_reply.started":"2022-08-02T08:23:15.402774Z","shell.execute_reply":"2022-08-02T08:23:22.382434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.AuxLogits.fc = nn.Sequential(nn.Linear(768,1024,bias=True),\n                         nn.Linear(1024,28,bias=True))\n\nmodel.fc = nn.Sequential(\n                          nn.BatchNorm1d(2048, eps=1e-05, momentum=0.1, affine=True, track_running_stats=True),\n                          nn.Dropout(p=0.25),\n                          nn.Linear(in_features=2048, out_features=2048, bias=True),\n                          nn.ReLU(),\n                          nn.BatchNorm1d(2048, eps=1e-05, momentum=0.1, affine=True, track_running_stats=True),\n                          nn.Dropout(p=0.5),\n                          nn.Linear(in_features=2048, out_features=28, bias=True),\n                         )\nmodel = model.to(device)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:23:32.689231Z","iopub.execute_input":"2022-08-02T08:23:32.689859Z","iopub.status.idle":"2022-08-02T08:23:35.491592Z","shell.execute_reply.started":"2022-08-02T08:23:32.689822Z","shell.execute_reply":"2022-08-02T08:23:35.490643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This loss combines a Sigmoid layer and the BCELoss in one single class. This version is more numerically stable than using a plain Sigmoid followed by a BCELoss as, by combining the operations into one layer","metadata":{}},{"cell_type":"code","source":"criterion = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam(model.parameters() , lr = 1e-4)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:23:38.240818Z","iopub.execute_input":"2022-08-02T08:23:38.241480Z","iopub.status.idle":"2022-08-02T08:23:38.249022Z","shell.execute_reply.started":"2022-08-02T08:23:38.241438Z","shell.execute_reply":"2022-08-02T08:23:38.247829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def f1_metric(y_true , y_pred):\n    tp = (y_pred * y_true).sum(axis=0)\n    fp = ((1-y_true)*y_pred).sum(axis=0)\n    fn = (y_true*(1-y_pred)).sum(axis=0)\n    epsilon = 1e-7\n    \n    precision = tp / (tp+fp+epsilon)\n    recall = tp / (tp+fn+epsilon)\n    \n    f1 = (2*precision*recall) / (precision+recall+epsilon)\n    \n    return f1.mean()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:23:42.577534Z","iopub.execute_input":"2022-08-02T08:23:42.577907Z","iopub.status.idle":"2022-08-02T08:23:42.584997Z","shell.execute_reply.started":"2022-08-02T08:23:42.577874Z","shell.execute_reply":"2022-08-02T08:23:42.583429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_state_dict(torch.load('../input/inceptionv3/inception_v3Pre_fcO_28adam.pt'))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"iterations = 5\ntrainLoss = []\nvalLoss = []\nvalAcc = []\nf1 = []\nf1_epoch = []\nstartTime = time.time()\n\nfor epoch in range(iterations):\n    runningLoss = 0\n    epochstart = time.time()\n    model.train()\n    for data in tqdm.notebook.tqdm(trainDataLoader):\n        inputs , labels = data\n        inputs , labels = inputs.float().to(device) , labels.float().to(device)\n        outputs , aux_outputs = model(inputs)\n        loss = criterion(outputs, labels)\n        aux_loss = criterion(aux_outputs , labels)\n        total_loss = loss + aux_loss\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n        runningLoss += total_loss.item()\n    avgTrainLoss = runningLoss/(len(train_df)/batch_size)\n    trainLoss.append(avgTrainLoss)\n    \n    model.eval()\n    val_runningLoss = 0\n    running_correct = 0\n    with torch.no_grad():\n        for data in tqdm.notebook.tqdm(valDataLoader):\n            inputs , labels = data\n            inputs , labels = inputs.float().to(device) , labels.float().to(device)\n            outputs = model(inputs)\n            loss = criterion(outputs, labels)\n            val_runningLoss += loss.item()\n            predicted = (outputs.data > 0.0)\n            labels = labels.data.cpu().numpy()\n            predicted = predicted.cpu().numpy().astype('float')\n            running_correct += (predicted == labels).sum()            \n            f1.append(f1_metric(labels , predicted))\n    avgValLoss = val_runningLoss/(len(val_df)/batch_size)  \n    avgValAcc = (float(running_correct)*100/len(val_df))/28\n    valAcc.append(avgValAcc)\n    valLoss.append(avgValLoss)\n    f1_score = np.mean(f1)\n    f1_epoch.append(np.mean(f1))\n    fig1= plt.figure(1)\n    plt.plot(range(epoch+1),trainLoss,'r--',label='train')\n    plt.plot(range(epoch+1),valLoss,'g--',label= 'valid')\n    if epoch == 0:\n        plt.legend(loc = 'upper left')\n        plt.xlabel('Epochs')\n        plt.ylabel('Losses')\n    fig2 = plt.figure(2)\n    plt.plot(range(epoch+1),f1_epoch,label='f1_score')\n    if epoch == 0:\n        plt.legend(loc = 'upper left')\n        plt.xlabel('Epochs')\n        plt.ylabel('f1 score')\n    epochEnd = time.time() - epochstart\n    print('At Iteration: {:.0f} /{:.0f}  ;  Training Loss: {:.6f} ; Validation Loss: {:.6f} ; Validation Acc: {:.3f} ; f1 score: {:.4f} ; Time consumed: {:.0f}m {:.0f}s '\\\n          .format(epoch + 1,iterations,avgTrainLoss,avgValLoss,avgValAcc,f1_score,epochEnd//60,epochEnd%60))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:25:58.731438Z","iopub.execute_input":"2022-08-02T08:25:58.731821Z","iopub.status.idle":"2022-08-02T09:28:47.707910Z","shell.execute_reply.started":"2022-08-02T08:25:58.731774Z","shell.execute_reply":"2022-08-02T09:28:47.706718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.save(model.state_dict(), 'inception_v3Pre_fcO_28adam.pt')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:15:52.622078Z","iopub.execute_input":"2022-08-01T18:15:52.622450Z","iopub.status.idle":"2022-08-01T18:15:52.979290Z","shell.execute_reply.started":"2022-08-01T18:15:52.622418Z","shell.execute_reply":"2022-08-01T18:15:52.978273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path = '../input/human-protein-atlas-image-classification/test'\ntest_df = pd.read_csv('../input/human-protein-atlas-image-classification/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:30:42.394264Z","iopub.execute_input":"2022-08-02T09:30:42.394723Z","iopub.status.idle":"2022-08-02T09:30:42.431068Z","shell.execute_reply.started":"2022-08-02T09:30:42.394680Z","shell.execute_reply":"2022-08-02T09:30:42.430098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:30:47.491227Z","iopub.execute_input":"2022-08-02T09:30:47.491741Z","iopub.status.idle":"2022-08-02T09:30:47.509868Z","shell.execute_reply.started":"2022-08-02T09:30:47.491705Z","shell.execute_reply":"2022-08-02T09:30:47.508553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TestGenerator(Dataset):\n    def __init__(self,df,path):\n        self.df = df\n        self.path = path\n    \n    def __getitem__(self,idx):\n        img_name = self.df.iloc[idx]['Id']\n        img_path = os.path.join(self.path,img_name)\n        r = np.array(Image.open(img_path+'_red.png'))\n        g = np.array(Image.open(img_path+'_green.png'))\n        b = np.array(Image.open(img_path+'_blue.png'))\n        y = np.array(Image.open(img_path+'_yellow.png'))\n        img = np.stack([r,g,b,y],axis=-1)\n        img = cv2.resize(img , (299,299))\n        img = np.array(img)/255\n        img = torch.from_numpy(img).permute(2,0,1)\n        return img , img_name\n    \n    def __len__(self):\n        return len(self.df)    ","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:52:52.439156Z","iopub.execute_input":"2022-08-02T09:52:52.440085Z","iopub.status.idle":"2022-08-02T09:52:52.448551Z","shell.execute_reply.started":"2022-08-02T09:52:52.440046Z","shell.execute_reply":"2022-08-02T09:52:52.446939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testDataset = TestGenerator(test_df , test_path)\ntestDataLoader = DataLoader(testDataset, batch_size = 16, shuffle = False , num_workers = 2, pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:52:57.526724Z","iopub.execute_input":"2022-08-02T09:52:57.527842Z","iopub.status.idle":"2022-08-02T09:52:57.533012Z","shell.execute_reply.started":"2022-08-02T09:52:57.527771Z","shell.execute_reply":"2022-08-02T09:52:57.531884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_vector = []\nmodel.eval()\nfor idx in tqdm.notebook.tqdm(test_df['Id']):\n    img_name = idx\n    img_path = os.path.join(test_path,img_name)\n    r = np.array(Image.open(img_path+'_red.png'))\n    g = np.array(Image.open(img_path+'_green.png'))\n    b = np.array(Image.open(img_path+'_blue.png'))\n    y = np.array(Image.open(img_path+'_yellow.png'))\n    img = np.stack([r,g,b,y],axis=-1)\n    img = cv2.resize(img , (299,299))\n    img = np.array(img)/255\n    inputs = torch.from_numpy(img).permute(2,0,1).unsqueeze(0)    \n    inputs = inputs.float().to(device)\n    outputs = model(inputs)\n    preds = (outputs.data >= 0.2).cpu().numpy().astype('uint8')\n    pred_vector.append(preds)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:05:23.387417Z","iopub.execute_input":"2022-08-02T10:05:23.388101Z","iopub.status.idle":"2022-08-02T10:11:22.751748Z","shell.execute_reply.started":"2022-08-02T10:05:23.388065Z","shell.execute_reply":"2022-08-02T10:11:22.750705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['Predicted'] = pred_vector\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:11:26.708026Z","iopub.execute_input":"2022-08-02T10:11:26.708403Z","iopub.status.idle":"2022-08-02T10:11:26.726335Z","shell.execute_reply.started":"2022-08-02T10:11:26.708371Z","shell.execute_reply":"2022-08-02T10:11:26.723568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.to_csv('submission.csv' , index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:11:36.302269Z","iopub.execute_input":"2022-08-02T10:11:36.303029Z","iopub.status.idle":"2022-08-02T10:11:37.386982Z","shell.execute_reply.started":"2022-08-02T10:11:36.302991Z","shell.execute_reply":"2022-08-02T10:11:37.385989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}