{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Preamble\nSome basic imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom pathlib import Path\n\nfiles_csv = []\nfiles_parquet = []  \nfiles_other = []\nmain_dir = \"/kaggle/input\"\nfor dirname, _, filenames in os.walk(main_dir):\n    for filename in filenames:\n        abs_filename = os.path.join(dirname, filename)\n        if Path(filename).suffix == \".csv\":\n            files_csv.append(abs_filename)\n        elif Path(filename).suffix == \".parquet\":\n            files_parquet.append(abs_filename)\n        else:\n            files_other.append(abs_filename)\n\nprint(f\"csv files:{files_csv}\")\nprint(f\"number of parquet files: {len(files_parquet)}\")\nprint(f\"number of other files: {len(files_other)}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-24T02:46:06.500918Z","iopub.execute_input":"2024-10-24T02:46:06.501393Z","iopub.status.idle":"2024-10-24T02:46:07.550243Z","shell.execute_reply.started":"2024-10-24T02:46:06.501351Z","shell.execute_reply":"2024-10-24T02:46:07.548961Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Metric Definition\nIn this section, we re-implement the quadratic weighted kappa loss","metadata":{}},{"cell_type":"code","source":"import torch\nfrom typing import Optional\n\ndef quadratic_weighted_kappa_matrices(prediction: torch.Tensor, ground_truth: torch.Tensor, N: Optional[int] = None) -> (torch.Tensor, torch.Tensor, torch.Tensor):\n    if N is None:\n        N = len(ground_truth.unique())\n    # Matrix O\n    O = torch.zeros(N, N, dtype=int)\n    masks = ground_truth == torch.arange(N).view(-1, 1)\n    for ii in range(N):\n        mask = masks[ii,]\n        predicted_values_masked = prediction[mask]\n        idx = predicted_values_masked.unique()\n        values = predicted_values_masked.bincount()\n        # catch case of exact predictions\n        if len(idx) == 1 and len(values) > 1:\n            values = values[idx]\n        O[ii, idx] = values\n    O = O / O.sum()\n    # Matrix W\n    i = torch.arange(N).unsqueeze(1)  # Shape (n, 1)\n    j = torch.arange(N).unsqueeze(0)  # Shape (1, n)\n    W = (i - j) ** 2 / (N - 1)**2\n    # Matrix E\n    hist_outcomes = ground_truth.bincount().unsqueeze(1)\n    pred_bincount = prediction.bincount()\n    # handle edge case of a constant prediction\n    if len(pred_bincount) < N:\n        pred_bincount = torch.concat([pred_bincount, torch.zeros(len(hist_outcomes)-len(pred_bincount))])\n    hist_predicted = pred_bincount.unsqueeze(0)\n    E = hist_outcomes * hist_predicted\n    E = E / E.sum()\n    return O, W, E, N\n\ndef quadratic_weighted_kappa_loss(prediction: torch.Tensor, ground_truth: torch.Tensor, N: Optional[int] = None) -> float:\n    O, W, E, N = quadratic_weighted_kappa_matrices(prediction=prediction, ground_truth=ground_truth, N=N)\n    assert O.shape == (N, N)\n    assert W.shape == (N, N)\n    assert E.shape == (N, N)\n    return 1.0 - (W*O).sum().item() / (W*E).sum().item()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T02:46:07.552703Z","iopub.execute_input":"2024-10-24T02:46:07.553183Z","iopub.status.idle":"2024-10-24T02:46:07.565711Z","shell.execute_reply.started":"2024-10-24T02:46:07.553132Z","shell.execute_reply":"2024-10-24T02:46:07.564237Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Metric analysis","metadata":{}},{"cell_type":"markdown","source":"## Basic observations","metadata":{}},{"cell_type":"code","source":"# Randomly samples a ground truth and prediction vectors\nprediction = torch.randint(0, 3, (100,))\nground_truth = torch.randint(0, 3, (100,))","metadata":{"execution":{"iopub.status.busy":"2024-10-24T02:46:07.567047Z","iopub.execute_input":"2024-10-24T02:46:07.567428Z","iopub.status.idle":"2024-10-24T02:46:07.581237Z","shell.execute_reply.started":"2024-10-24T02:46:07.567392Z","shell.execute_reply":"2024-10-24T02:46:07.580228Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"O, W, E, N = quadratic_weighted_kappa_matrices(prediction, ground_truth)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-24T02:46:07.584299Z","iopub.execute_input":"2024-10-24T02:46:07.585073Z","iopub.status.idle":"2024-10-24T02:46:07.592323Z","shell.execute_reply.started":"2024-10-24T02:46:07.585023Z","shell.execute_reply":"2024-10-24T02:46:07.591256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Matrix O is simply the confusion matrix (normalized):\nfrom sklearn.metrics import confusion_matrix\nconfusion_matrix(ground_truth.numpy(), prediction.numpy(), normalize='all'), O","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-24T02:46:07.593746Z","iopub.execute_input":"2024-10-24T02:46:07.594218Z","iopub.status.idle":"2024-10-24T02:46:07.606184Z","shell.execute_reply.started":"2024-10-24T02:46:07.594181Z","shell.execute_reply":"2024-10-24T02:46:07.605088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# W is the weight-penalities\n# - It always has 0's along the diagonal, i.e., no penality for the correct prediction\n# - And the penaly increases as you move away from the diagonal, i.e, \n# the farther you are from the ground truth the worst you do (which is not the case with a metric like accuracy)\nW","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-24T02:46:07.607857Z","iopub.execute_input":"2024-10-24T02:46:07.608315Z","iopub.status.idle":"2024-10-24T02:46:07.620206Z","shell.execute_reply.started":"2024-10-24T02:46:07.608266Z","shell.execute_reply":"2024-10-24T02:46:07.619169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"E","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-24T02:46:07.621773Z","iopub.execute_input":"2024-10-24T02:46:07.622264Z","iopub.status.idle":"2024-10-24T02:46:07.630163Z","shell.execute_reply.started":"2024-10-24T02:46:07.622215Z","shell.execute_reply":"2024-10-24T02:46:07.629188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Perfect prediction yields a value of 1.0, which the theoretical max:\nquadratic_weighted_kappa_loss(ground_truth, ground_truth)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T02:46:07.631660Z","iopub.execute_input":"2024-10-24T02:46:07.632712Z","iopub.status.idle":"2024-10-24T02:46:07.640232Z","shell.execute_reply.started":"2024-10-24T02:46:07.632662Z","shell.execute_reply":"2024-10-24T02:46:07.639220Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# And a constant prediction will always produce a score of 0.0, since prediction and ground truth are then uncorrelated\nquadratic_weighted_kappa_loss(prediction=torch.ones(len(ground_truth)).to(int), ground_truth=ground_truth)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-24T02:46:07.641733Z","iopub.execute_input":"2024-10-24T02:46:07.642295Z","iopub.status.idle":"2024-10-24T02:46:07.651196Z","shell.execute_reply.started":"2024-10-24T02:46:07.642246Z","shell.execute_reply":"2024-10-24T02:46:07.650179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# compare against sklearn's kappa score, which represents the same when using the quadratic weights\nfrom sklearn.metrics import cohen_kappa_score\ncohen_kappa_score(ground_truth.numpy(), ground_truth.numpy(), weights='quadratic'), cohen_kappa_score(ground_truth.numpy(), np.ones(len(ground_truth)), weights='quadratic')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-24T02:46:42.892583Z","iopub.execute_input":"2024-10-24T02:46:42.892991Z","iopub.status.idle":"2024-10-24T02:46:42.904905Z","shell.execute_reply.started":"2024-10-24T02:46:42.892953Z","shell.execute_reply":"2024-10-24T02:46:42.903596Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Numerical experiments","metadata":{}},{"cell_type":"markdown","source":"### Variation with the distance from the ground truth\nLet's look at a scenario where all predictions are wrong, but they're shifted by a given integer compared to the ground truth. As we increase that shift, kappa will vary smoothly. Whereas the accuracy will always be 0.0.","metadata":{}},{"cell_type":"code","source":"ground_truth = torch.randint(0, 20, (1000,))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-24T02:46:07.669874Z","iopub.execute_input":"2024-10-24T02:46:07.670370Z","iopub.status.idle":"2024-10-24T02:46:07.681005Z","shell.execute_reply.started":"2024-10-24T02:46:07.670299Z","shell.execute_reply":"2024-10-24T02:46:07.679924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scores = []\nfor shift in range(0, 11):\n    prediction = torch.where(ground_truth >= 20 - shift, ground_truth - shift, ground_truth + shift)\n    kappa_score = quadratic_weighted_kappa_loss(prediction=prediction, ground_truth=ground_truth, N=20)\n    O, W, E, N = quadratic_weighted_kappa_matrices(prediction=prediction, ground_truth=ground_truth, N=20)\n    acc = (ground_truth == prediction).to(int).mean(dtype=float)\n    scores.append([shift, kappa_score, acc.item(), 1.0 - (W*O).sum().item()])\n    print(f\"shift={shift}, kappa={kappa_score:.2f}, accuracy={acc}\")\n\nscores = pd.DataFrame(scores, columns=[\"shift\", \"kappa\", \"accuracy\", \"OW\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-24T02:46:07.682760Z","iopub.execute_input":"2024-10-24T02:46:07.683249Z","iopub.status.idle":"2024-10-24T02:46:07.732305Z","shell.execute_reply.started":"2024-10-24T02:46:07.683197Z","shell.execute_reply":"2024-10-24T02:46:07.731277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scores.set_index('shift').plot(marker='o')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-24T02:46:07.733527Z","iopub.execute_input":"2024-10-24T02:46:07.733851Z","iopub.status.idle":"2024-10-24T02:46:08.046552Z","shell.execute_reply.started":"2024-10-24T02:46:07.733816Z","shell.execute_reply":"2024-10-24T02:46:08.045318Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Random predictions\nShow histogram of the scores when doing random predictions on the training set","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-24T02:46:08.047922Z","iopub.execute_input":"2024-10-24T02:46:08.048294Z","iopub.status.idle":"2024-10-24T02:46:08.086949Z","shell.execute_reply.started":"2024-10-24T02:46:08.048257Z","shell.execute_reply":"2024-10-24T02:46:08.085912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sii = df_train['sii'].dropna().astype(int)\ntrain_sii.hist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-24T02:46:08.088243Z","iopub.execute_input":"2024-10-24T02:46:08.088579Z","iopub.status.idle":"2024-10-24T02:46:08.366984Z","shell.execute_reply.started":"2024-10-24T02:46:08.088545Z","shell.execute_reply":"2024-10-24T02:46:08.365838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# sample for the true distribution and plot the histogram of the weighted kappa scores\n# As we randomly sample, the resulting scores are centered around 0.0\nimport matplotlib.pyplot as plt\nn_samples = 1000\nidx_samples = np.random.randint(0, len(train_sii), (len(train_sii), n_samples))\nkappa_scores = []\nfor ii in range(n_samples):\n    pred = train_sii.iloc[idx_samples[:, ii]]\n    kappa_scores.append(cohen_kappa_score(pred.to_numpy(), train_sii.to_numpy(), weights='quadratic'))\nplt.hist(kappa_scores, bins=20);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-24T02:47:42.362179Z","iopub.execute_input":"2024-10-24T02:47:42.362591Z","iopub.status.idle":"2024-10-24T02:47:43.981994Z","shell.execute_reply.started":"2024-10-24T02:47:42.362555Z","shell.execute_reply":"2024-10-24T02:47:43.980798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}