{"cells":[{"metadata":{},"cell_type":"markdown","source":"# TL;DR\n\n### I made utility function to get classes and accuracy rate\n### I wish this will be useful for your modeling"},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport random\nimport torch\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import train_test_split\n\nSEED = 1129\n\ndef seed_everything(seed=1129):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\nseed_everything(SEED)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# load competition data\n\nclass_map = pd.read_csv(\"../input/bengaliai-cv19/class_map.csv\")\ntrain = pd.read_csv(\"../input/bengaliai-cv19/train.csv\")\n\ny = train[[\"vowel_diacritic\", \"grapheme_root\", \"consonant_diacritic\"]]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# get validation data which you used\n\ntrain_idx, val_idx = train_test_split(train.index.tolist(), test_size=0.2, random_state=SEED, stratify=train[\"grapheme_root\"])\n\ny_val = y.values[val_idx].T\ny_val.shape, y_val","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# load your validation preds data\n\ny_pred = np.load('/kaggle/input/bengali-valid-preds/bengali_valid_preds.npy')\ny_pred.shape, y_pred","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_mislabel_and_probs(y_pred_df, y_test_df, class_map, component_type, threshold):\n    # Compute confusion matrix\n    cnf_matrix = confusion_matrix(y_test_df, y_pred_df)\n    \n    # normalization\n    cm = cnf_matrix.astype('float') / cnf_matrix.sum(axis=1)[:, np.newaxis]\n    \n    class_names = list(y_pred_df[0].unique())\n    \n    matrix_map = dict([(n, c) for n, c in enumerate(class_names)])\n    res = [[dict([c for c in class_map[class_map['component_type']==component_type] \\\n           [['label', 'component']].values])[matrix_map[np.argmax(c)]], \n            matrix_map[np.argmax(c)],\n            max(c)] for c in cm if max(c) < threshold]\n    return res","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# vowel_diacritic\ny_test_df = pd.DataFrame(y_val[0]) # V\ny_pred_df = pd.DataFrame(y_pred[0]) # V\n\nget_mislabel_and_probs(y_pred_df, y_test_df, class_map, 'vowel_diacritic', 0.99)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 🔼 above means your model predict 'ৈ' with probability 0.9858053650908216"},{"metadata":{"trusted":true},"cell_type":"code","source":"# grapheme_root\ny_test_df = pd.DataFrame(y_val[1]) # G\ny_pred_df = pd.DataFrame(y_pred[1]) # G\n\nget_mislabel_and_probs(y_pred_df, y_test_df, class_map, 'grapheme_root', 0.9)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 🔼 My model is poor at predicting 'দ্দ'\n\n### I have to tuning more."},{"metadata":{"trusted":true},"cell_type":"code","source":"# consonant_diacritic\ny_test_df = pd.DataFrame(y_val[2]) # C\ny_pred_df = pd.DataFrame(y_pred[2]) # C\n\nget_mislabel_and_probs(y_pred_df, y_test_df, class_map, 'consonant_diacritic', 0.99)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### EOF\n\n### Thank you!"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}