{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**I’m glad that many people are trying my idea. I noticed that some parts of my original program were misleading, so I decided to create a concise version here.**\n\nMy original blog post (Japanese)  \nhttps://zenn.dev/jackthekaggler/articles/cf988ca341e34ed83034","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-27T00:40:00.017034Z","iopub.execute_input":"2024-09-27T00:40:00.017635Z","iopub.status.idle":"2024-09-27T00:40:00.024011Z","shell.execute_reply.started":"2024-09-27T00:40:00.017572Z","shell.execute_reply":"2024-09-27T00:40:00.022487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_DIR = \"/kaggle/input/child-mind-institute-problematic-internet-use/\"","metadata":{"execution":{"iopub.status.busy":"2024-09-27T00:40:00.026518Z","iopub.execute_input":"2024-09-27T00:40:00.027008Z","iopub.status.idle":"2024-09-27T00:40:00.036980Z","shell.execute_reply.started":"2024-09-27T00:40:00.026951Z","shell.execute_reply":"2024-09-27T00:40:00.035590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(INPUT_DIR + \"train.csv\")\ndf_test = pd.read_csv(INPUT_DIR + \"test.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-09-27T00:40:00.038333Z","iopub.execute_input":"2024-09-27T00:40:00.038939Z","iopub.status.idle":"2024-09-27T00:40:00.111803Z","shell.execute_reply.started":"2024-09-27T00:40:00.038885Z","shell.execute_reply":"2024-09-27T00:40:00.110578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.dropna(subset=[\"sii\"]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-27T00:40:00.114999Z","iopub.execute_input":"2024-09-27T00:40:00.115420Z","iopub.status.idle":"2024-09-27T00:40:00.126633Z","shell.execute_reply.started":"2024-09-27T00:40:00.115376Z","shell.execute_reply":"2024-09-27T00:40:00.125204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_cols = list(set(df.columns) - set(df_test.columns))\n\nX = df.drop([\"id\"] + drop_cols, axis=1)\ny = df[\"sii\"].astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-09-27T00:40:00.128545Z","iopub.execute_input":"2024-09-27T00:40:00.128935Z","iopub.status.idle":"2024-09-27T00:40:00.137988Z","shell.execute_reply.started":"2024-09-27T00:40:00.128882Z","shell.execute_reply":"2024-09-27T00:40:00.136588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = X.columns[X.dtypes==\"object\"].tolist()\nif cat_cols:\n    X[cat_cols] = X[cat_cols].astype(\"category\")","metadata":{"execution":{"iopub.status.busy":"2024-09-27T00:40:00.139536Z","iopub.execute_input":"2024-09-27T00:40:00.139902Z","iopub.status.idle":"2024-09-27T00:40:00.162701Z","shell.execute_reply.started":"2024-09-27T00:40:00.139864Z","shell.execute_reply":"2024-09-27T00:40:00.161703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = y.mean()\nb = y.var(ddof=0)\n\ny_min = y.min()\ny_max = y.max()","metadata":{"execution":{"iopub.status.busy":"2024-09-27T00:40:00.164156Z","iopub.execute_input":"2024-09-27T00:40:00.164571Z","iopub.status.idle":"2024-09-27T00:40:00.170894Z","shell.execute_reply.started":"2024-09-27T00:40:00.164521Z","shell.execute_reply":"2024-09-27T00:40:00.169671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(preds, data):\n    y_true = data.get_label()\n    y_pred = preds.clip(y_min, y_max).round()\n    qwk = cohen_kappa_score(y_true, y_pred, weights=\"quadratic\")\n    return 'QWK', qwk, True\n\n\ndef qwk_obj(preds, dtrain):\n    labels = dtrain.get_label()\n    preds = preds.clip(y_min, y_max)\n    f = 1/2 * np.sum((preds - labels)**2)\n    g = 1/2 * np.sum((preds - a)**2 + b)\n    df = preds - labels\n    dg = preds - a\n    grad = (df/g - f*dg/g**2)*len(labels)\n    hess = np.ones(len(labels))\n    return grad, hess","metadata":{"execution":{"iopub.status.busy":"2024-09-27T00:40:00.172313Z","iopub.execute_input":"2024-09-27T00:40:00.172715Z","iopub.status.idle":"2024-09-27T00:40:00.184337Z","shell.execute_reply.started":"2024-09-27T00:40:00.172675Z","shell.execute_reply":"2024-09-27T00:40:00.183057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {\n    \"objective\": qwk_obj,\n    \"metric\": \"None\",\n    \"verbosity\": -1,\n    \"learning_rate\": 0.01,\n    \"num_leaves\": 16,\n    \"feature_fraction\": 0.5\n}\n\n# The initial score of the model is the key parameter.\n# I found that the mean value of the target is a bad choice for this data.\ninit_score = 2.0","metadata":{"execution":{"iopub.status.busy":"2024-09-27T00:40:00.187868Z","iopub.execute_input":"2024-09-27T00:40:00.188270Z","iopub.status.idle":"2024-09-27T00:40:00.195949Z","shell.execute_reply.started":"2024-09-27T00:40:00.188228Z","shell.execute_reply":"2024-09-27T00:40:00.194893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=0)\nfolds = [(idx_train, idx_valid) for idx_train, idx_valid in skf.split(X, y)]\n\nmodels = lgb.cv(\n    params=params,\n    train_set=lgb.Dataset(X, y, init_score=[init_score]*len(X)),\n    num_boost_round=10000,\n    folds=folds,\n    feval=quadratic_weighted_kappa,\n    callbacks=[\n        lgb.early_stopping(stopping_rounds=100, verbose=True),\n        lgb.log_evaluation(100)\n    ],\n    return_cvbooster=True\n)[\"cvbooster\"].boosters\n\npreds_oof = np.zeros(len(df))\nfor model, (idx_train, idx_valid) in zip(models, folds):\n    preds_oof[idx_valid] = model.predict(X.iloc[idx_valid]) + init_score","metadata":{"execution":{"iopub.status.busy":"2024-09-27T00:40:00.197315Z","iopub.execute_input":"2024-09-27T00:40:00.197704Z","iopub.status.idle":"2024-09-27T00:40:26.700827Z","shell.execute_reply.started":"2024-09-27T00:40:00.197664Z","shell.execute_reply":"2024-09-27T00:40:26.699512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# without threshold optimization\npreds_oof = preds_oof.clip(y_min, y_max).round()\nqwk = cohen_kappa_score(y, preds_oof, weights=\"quadratic\")\nprint(\"QWK:\", qwk)","metadata":{"execution":{"iopub.status.busy":"2024-09-27T00:40:26.702487Z","iopub.execute_input":"2024-09-27T00:40:26.702908Z","iopub.status.idle":"2024-09-27T00:40:26.712743Z","shell.execute_reply.started":"2024-09-27T00:40:26.702863Z","shell.execute_reply":"2024-09-27T00:40:26.711542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Early stopping was applied during validation, so this score might be optimistic.","metadata":{}},{"cell_type":"code","source":"def extended_cohen_kappa_score(labels, preds):\n    f = np.sum((preds - labels)**2)\n    g = np.sum((preds - a) ** 2 + b)\n    return 1 - f / g \n\nextended_cohen_kappa_score(y, preds_oof)","metadata":{"execution":{"iopub.status.busy":"2024-09-27T00:40:37.876617Z","iopub.execute_input":"2024-09-27T00:40:37.877486Z","iopub.status.idle":"2024-09-27T00:40:37.890288Z","shell.execute_reply.started":"2024-09-27T00:40:37.877441Z","shell.execute_reply":"2024-09-27T00:40:37.889026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You can see that the above function can compute QWK exactly, given the parameters $a$ (target mean) and $b$ (target variance). This function is also applicable to continuous predictions and is differentiable. The custom objective function implemented above is a constant multiple of the gradient of this function.","metadata":{}}]}