{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Analyze your model performance by confusion matrix"},{"metadata":{"trusted":true},"cell_type":"code","source":"output_dirs = ['../input/20t-efficientnet-b3-cutmix-tta',\n               '../input/22t-efficientnet-b4-cutmix-tta',\n               '../input/33t-seresnext50',\n               '../input/34t-efficientnet-b5',\n               '../input/35t-vit16']\n\noof_dirs = ['../input/20i-efficientnet-b3-later-cutmix-tta',\n            '../input/22i-efficientnet-b4-cutmix-tta',\n            '../input/33i-seresnext50',\n            '../input/34i-efficientnet-b5',\n            '../input/35i-vit16']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# output_dirs = ['../input/20t-efficientnet-b3-cutmix-tta',\n#                '../input/22t-efficientnet-b4-cutmix-tta',\n#                '../input/24t-efficientnet-b3-cutmix-tta-randombrightness',\n#                '../input/33t-seresnext50',\n#                '../input/34t-efficientnet-b5',\n#                '../input/35t-vit16']\n\n# output_dirs = ['../input/20t-efficientnet-b3-cutmix-tta',\n#                '../input/22t-efficientnet-b4-cutmix-tta',\n#                '../input/33t-seresnext50',\n#                '../input/34t-efficientnet-b5',\n#                '../input/35t-vit16']\n\n# output_dirs = ['../input/06t-efficientnet-b4-ns-512',\n#                '../input/12t-efficientnet-b5-cutout',\n#                '../input/14t-seresnext50',\n#                '../input/20t-efficientnet-b3-cutmix-tta',\n#                '../input/22t-efficientnet-b4-cutmix-tta',\n#                '../input/31t-vit-baseline']\n\n# output_dirs = ['../input/06t-efficientnet-b4-ns-512',\n#                '../input/12t-efficientnet-b5-cutout',\n#                '../input/14t-seresnext50',\n#                '../input/16t-efficientnet-b3-cutmix',\n#                '../input/20t-efficientnet-b3-cutmix-tta',\n#                '../input/21t-efficientnet-b3-later-cutmix-tta',\n#                '../input/22t-efficientnet-b4-cutmix-tta',\n#                '../input/23t-efficientnet-b4-cutmix-tta-batch512',\n#                '../input/24t-efficientnet-b3-cutmix-tta-randombrightness',\n#                '../input/26t-efficientnet-b3-cutmix-augmix-tta',\n#                '../input/27t-efficientnet-b4-cutmix-tta',\n#                '../input/28t-efficientnet-b3-cutmix-light-tta']\n\n# output_dirs = ['../input/06t-efficientnet-b4-ns-512',\n#                '../input/12t-efficientnet-b5-cutout',\n#                '../input/14t-seresnext50',\n#                '../input/20t-efficientnet-b3-cutmix-tta',\n#                '../input/22t-efficientnet-b4-cutmix-tta']\n\n# output_dirs = ['../input/20t-efficientnet-b3-cutmix-tta',\n#                '../input/22t-efficientnet-b4-cutmix-tta']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TTA_LIST = [\n    'CenterCrop-Normalize-ToTensorV2',\n    'CenterCrop-Transpose-Normalize-ToTensorV2',\n#     'CenterCrop-HorizontalFlip-Normalize-ToTensorV2',\n#     'CenterCrop-VerticalFlip-Normalize-ToTensorV2',\n    'Resize-Normalize-ToTensorV2',\n    'Resize-Transpose-Normalize-ToTensorV2',\n#     'Resize-VerticalFlip-Normalize-ToTensorV2',\n#     'Resize-HorizontalFlip-Normalize-ToTensorV2'\n]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\ntrain","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# # inference results\n# oof_list = [pd.read_csv(d+'/oof_df.csv') for d in output_dirs]\n\n# oof_list[0].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\n\noof_list = []\nfor d in oof_dirs:\n    oof = pd.read_csv(os.path.join(d, TTA_LIST[0]+'.csv'), usecols=['image_id','label','fold'])\n    tta_proba = np.zeros((len(oof), 5), dtype=np.float)\n    for tta in TTA_LIST:\n        tta_proba += pd.read_csv(os.path.join(d, tta+'.csv'), usecols=['0','1','2','3','4']).values / len(TTA_LIST)\n    oof[['0','1','2','3','4']] = tta_proba\n    oof['preds'] = tta_proba.argmax(1)\n    oof_list.append(oof)\n\noof_list[0].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_df_pred(oof_list):\n    df_pred = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\n    for model_path, oof in zip(output_dirs, oof_list):\n        model_name = model_path.split('/')[-1].split('-')[0]\n        df_pred = df_pred.merge(oof[['image_id','preds']], how='left', on=\"image_id\")\n        df_pred.rename(columns={'preds':model_name}, inplace=True)\n    return df_pred","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_pred = get_df_pred(oof_list)\ndf_pred.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from itertools import combinations_with_replacement\n\nfrom sklearn.metrics import accuracy_score\n\n\ndef score_cor_heatmap(df_pred : pd.DataFrame, metric=accuracy_score):\n    cols = df_pred.columns\n    df_corr = pd.DataFrame(index=cols, columns=cols, dtype=np.float)\n    for i, j in combinations_with_replacement(cols, 2):\n        val = metric(df_pred[i], df_pred[j])\n        df_corr.loc[i,j] = val\n        df_corr.loc[j,i] = val\n    return df_corr","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(25, 25)) \ndf_corr = score_cor_heatmap(df_pred.iloc[:,1:])\nsns.heatmap(df_corr, square=True, annot=True, fmt=\".3f\")\n# df_corr.values","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# bi-logitloss"},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"import torch\n\ndef log_t(u, t):\n    \"\"\"Compute log_t for `u'.\"\"\"\n    if t==1.0:\n        return u.log()\n    else:\n        return (u.pow(1.0 - t) - 1.0) / (1.0 - t)\n\ndef exp_t(u, t):\n    \"\"\"Compute exp_t for `u'.\"\"\"\n    if t==1:\n        return u.exp()\n    else:\n        return (1.0 + (1.0-t)*u).relu().pow(1.0 / (1.0 - t))\n\ndef compute_normalization_fixed_point(activations, t, num_iters):\n\n    \"\"\"Returns the normalization value for each example (t > 1.0).\n    Args:\n      activations: A multi-dimensional tensor with last dimension `num_classes`.\n      t: Temperature 2 (> 1.0 for tail heaviness).\n      num_iters: Number of iterations to run the method.\n    Return: A tensor of same shape as activation with the last dimension being 1.\n    \"\"\"\n    mu, _ = torch.max(activations, -1, keepdim=True)\n    normalized_activations_step_0 = activations - mu\n\n    normalized_activations = normalized_activations_step_0\n\n    for _ in range(num_iters):\n        logt_partition = torch.sum(\n                exp_t(normalized_activations, t), -1, keepdim=True)\n        normalized_activations = normalized_activations_step_0 * \\\n                logt_partition.pow(1.0-t)\n\n    logt_partition = torch.sum(\n            exp_t(normalized_activations, t), -1, keepdim=True)\n    normalization_constants = - log_t(1.0 / logt_partition, t) + mu\n\n    return normalization_constants\n\ndef compute_normalization_binary_search(activations, t, num_iters):\n\n    \"\"\"Returns the normalization value for each example (t < 1.0).\n    Args:\n      activations: A multi-dimensional tensor with last dimension `num_classes`.\n      t: Temperature 2 (< 1.0 for finite support).\n      num_iters: Number of iterations to run the method.\n    Return: A tensor of same rank as activation with the last dimension being 1.\n    \"\"\"\n\n    mu, _ = torch.max(activations, -1, keepdim=True)\n    normalized_activations = activations - mu\n\n    effective_dim = \\\n        torch.sum(\n                (normalized_activations > -1.0 / (1.0-t)).to(torch.int32),\n            dim=-1, keepdim=True).to(activations.dtype)\n\n    shape_partition = activations.shape[:-1] + (1,)\n    lower = torch.zeros(shape_partition, dtype=activations.dtype, device=activations.device)\n    upper = -log_t(1.0/effective_dim, t) * torch.ones_like(lower)\n\n    for _ in range(num_iters):\n        logt_partition = (upper + lower)/2.0\n        sum_probs = torch.sum(\n                exp_t(normalized_activations - logt_partition, t),\n                dim=-1, keepdim=True)\n        update = (sum_probs < 1.0).to(activations.dtype)\n        lower = torch.reshape(\n                lower * update + (1.0-update) * logt_partition,\n                shape_partition)\n        upper = torch.reshape(\n                upper * (1.0 - update) + update * logt_partition,\n                shape_partition)\n\n    logt_partition = (upper + lower)/2.0\n    return logt_partition + mu\n\nclass ComputeNormalization(torch.autograd.Function):\n    \"\"\"\n    Class implementing custom backward pass for compute_normalization. See compute_normalization.\n    \"\"\"\n    @staticmethod\n    def forward(ctx, activations, t, num_iters):\n        if t < 1.0:\n            normalization_constants = compute_normalization_binary_search(activations, t, num_iters)\n        else:\n            normalization_constants = compute_normalization_fixed_point(activations, t, num_iters)\n\n        ctx.save_for_backward(activations, normalization_constants)\n        ctx.t=t\n        return normalization_constants\n\n    @staticmethod\n    def backward(ctx, grad_output):\n        activations, normalization_constants = ctx.saved_tensors\n        t = ctx.t\n        normalized_activations = activations - normalization_constants \n        probabilities = exp_t(normalized_activations, t)\n        escorts = probabilities.pow(t)\n        escorts = escorts / escorts.sum(dim=-1, keepdim=True)\n        grad_input = escorts * grad_output\n        \n        return grad_input, None, None\n\ndef compute_normalization(activations, t, num_iters=5):\n    \"\"\"Returns the normalization value for each example. \n    Backward pass is implemented.\n    Args:\n      activations: A multi-dimensional tensor with last dimension `num_classes`.\n      t: Temperature 2 (> 1.0 for tail heaviness, < 1.0 for finite support).\n      num_iters: Number of iterations to run the method.\n    Return: A tensor of same rank as activation with the last dimension being 1.\n    \"\"\"\n    return ComputeNormalization.apply(activations, t, num_iters)\n\ndef tempered_sigmoid(activations, t, num_iters = 5):\n    \"\"\"Tempered sigmoid function.\n    Args:\n      activations: Activations for the positive class for binary classification.\n      t: Temperature tensor > 0.0.\n      num_iters: Number of iterations to run the method.\n    Returns:\n      A probabilities tensor.\n    \"\"\"\n    internal_activations = torch.stack([activations,\n        torch.zeros_like(activations)],\n        dim=-1)\n    internal_probabilities = tempered_softmax(internal_activations, t, num_iters)\n    return internal_probabilities[..., 0]\n\n\ndef tempered_softmax(activations, t, num_iters=5):\n    \"\"\"Tempered softmax function.\n    Args:\n      activations: A multi-dimensional tensor with last dimension `num_classes`.\n      t: Temperature > 1.0.\n      num_iters: Number of iterations to run the method.\n    Returns:\n      A probabilities tensor.\n    \"\"\"\n    if t == 1.0:\n        return activations.softmax(dim=-1)\n\n    normalization_constants = compute_normalization(activations, t, num_iters)\n    return exp_t(activations - normalization_constants, t)\n\ndef bi_tempered_binary_logistic_loss(activations,\n        labels,\n        t1,\n        t2,\n        label_smoothing = 0.0,\n        num_iters=5,\n        reduction='mean'):\n\n    \"\"\"Bi-Tempered binary logistic loss.\n    Args:\n      activations: A tensor containing activations for class 1.\n      labels: A tensor with shape as activations, containing probabilities for class 1\n      t1: Temperature 1 (< 1.0 for boundedness).\n      t2: Temperature 2 (> 1.0 for tail heaviness, < 1.0 for finite support).\n      label_smoothing: Label smoothing\n      num_iters: Number of iterations to run the method.\n    Returns:\n      A loss tensor.\n    \"\"\"\n    internal_activations = torch.stack([activations,\n        torch.zeros_like(activations)],\n        dim=-1)\n    internal_labels = torch.stack([labels.to(activations.dtype),\n        1.0 - labels.to(activations.dtype)],\n        dim=-1)\n    return bi_tempered_logistic_loss(internal_activations, \n            internal_labels,\n            t1,\n            t2,\n            label_smoothing = label_smoothing,\n            num_iters = num_iters,\n            reduction = reduction)\n\ndef bi_tempered_logistic_loss(activations,\n        labels,\n        t1,\n        t2,\n        label_smoothing=0.0,\n        num_iters=5,\n        reduction = 'mean'):\n\n    \"\"\"Bi-Tempered Logistic Loss.\n    Args:\n      activations: A multi-dimensional tensor with last dimension `num_classes`.\n      labels: A tensor with shape and dtype as activations (onehot), \n        or a long tensor of one dimension less than activations (pytorch standard)\n      t1: Temperature 1 (< 1.0 for boundedness).\n      t2: Temperature 2 (> 1.0 for tail heaviness, < 1.0 for finite support).\n      label_smoothing: Label smoothing parameter between [0, 1). Default 0.0.\n      num_iters: Number of iterations to run the method. Default 5.\n      reduction: ``'none'`` | ``'mean'`` | ``'sum'``. Default ``'mean'``.\n        ``'none'``: No reduction is applied, return shape is shape of\n        activations without the last dimension.\n        ``'mean'``: Loss is averaged over minibatch. Return shape (1,)\n        ``'sum'``: Loss is summed over minibatch. Return shape (1,)\n    Returns:\n      A loss tensor.\n    \"\"\"\n\n    if len(labels.shape)<len(activations.shape): #not one-hot\n        labels_onehot = torch.zeros_like(activations)\n        labels_onehot.scatter_(1, labels[..., None], 1)\n    else:\n        labels_onehot = labels\n\n    if label_smoothing > 0:\n        num_classes = labels_onehot.shape[-1]\n        labels_onehot = ( 1 - label_smoothing * num_classes / (num_classes - 1) ) \\\n                * labels_onehot + \\\n                label_smoothing / (num_classes - 1)\n\n    probabilities = tempered_softmax(activations, t2, num_iters)\n\n    loss_values = labels_onehot * log_t(labels_onehot + 1e-10, t1) \\\n            - labels_onehot * log_t(probabilities, t1) \\\n            - labels_onehot.pow(2.0 - t1) / (2.0 - t1) \\\n            + probabilities.pow(2.0 - t1) / (2.0 - t1)\n    loss_values = loss_values.sum(dim = -1) #sum over classes\n\n    if reduction == 'none':\n        return loss_values\n    if reduction == 'sum':\n        return loss_values.sum()\n    if reduction == 'mean':\n        return loss_values.mean()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# weight optimization"},{"metadata":{"trusted":true},"cell_type":"code","source":"oof_proba_list = []\nfor oof in oof_list:\n    df = train.merge(oof[[\"image_id\", \"0\",\"1\",\"2\",\"3\",\"4\"]], on='image_id', how=\"left\")\n    oof_proba_list.append(df[[\"0\",\"1\",\"2\",\"3\",\"4\"]].values)\n    \nlabels = train[\"label\"].values\ntrain = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from scipy.optimize import minimize, Bounds, LinearConstraint\n\nzeros = np.zeros(len(output_dirs))\nones = np.ones(len(output_dirs))               \nbounds = Bounds(zeros, ones)\nconstraint = LinearConstraint(ones, [1], [1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"LOOP  = 5\nNUM_FOLDS = 5\n\n# ==============================================================================\n# minimize\n# ==============================================================================\nfrom scipy.optimize import minimize, Bounds, LinearConstraint\nfrom sklearn.model_selection import KFold\nfrom tqdm import tqdm\n\n\nweights = np.ones(len(output_dirs)) / len(output_dirs)\n# 制約条件\nzeros = np.zeros(len(output_dirs))\nones = np.ones(len(output_dirs))              \nbounds = Bounds(zeros, ones)\nconstraint = LinearConstraint(ones, [1], [1])\n\n# def get_score(weights: np.array, train_idx, oofs, labels):\n#     oof_ = oof.copy()\n\n#     for i in range(oofs.shape[1]):\n#         oof_[train_idx, i] *= weights[i]\n\n#     return utils.validate(oof_, False)\n\n\ndef get_score(weights: np.array, train_idx, oofs, labels):\n        blend = np.zeros_like(oofs[0][train_idx, :])\n        \n        for oof, weight in zip(oofs, weights):\n            blend += weight * oof[train_idx, :]\n            \n        blend_tensor = torch.from_numpy(blend.astype(np.float32)).clone()\n        labels_tensor = torch.from_numpy(labels.astype(np.long)).clone()\n        return bi_tempered_logistic_loss(blend_tensor, labels_tensor[train_idx], t1=0.2, t2=1.)\n\n    \n    \nweights_list = []\nfor i in tqdm(range(LOOP)):\n    kf = KFold(n_splits=NUM_FOLDS, shuffle=True, random_state=i)\n    for fold, (train_idx, valid_idx) in enumerate(kf.split(train)):\n        res = minimize(get_score, weights,\n                       args=(train_idx, oof_proba_list, labels),\n                       method=\"Nelder-Mead\",\n                       tol=1e-6)\n\n        print(\"score:\", res.fun)\n        print(\"weights:\", res.x)\n        weights_list.append(res.x)\n\n\nbest_weights = np.array(weights_list).mean(0)\nbest_weights /= np.sum(best_weights)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"best_weights","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"stack_proba = np.zeros_like(oof_proba_list[0])\nfor i in range(len(oof_proba_list)):\n    stack_proba += oof_proba_list[i] * best_weights[i]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i, weight in zip(range(len(oof_proba_list)), best_weights):\n    model_name = output_dirs[i].split('/')[-1].split('-')[0]\n    single_score = accuracy_score(oof_proba_list[i].argmax(1), labels)\n    print(f'{model_name}: {single_score:.3f}, {weight:.3f}')\nprint(accuracy_score(stack_proba.argmax(1), labels))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"weights_dict = dict(zip(output_dirs, best_weights))\n\nimport json\nwith open('best_weights.json', 'w') as f:\n    json.dump(weights_dict, f, indent=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}