{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.metrics import log_loss\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.linear_model import LogisticRegression\n\nimport matplotlib.pyplot as plt\nimport pickle\nimport random\nfrom tqdm import tqdm\nfrom bayes_opt import BayesianOptimization\nfrom collections import Counter","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-28T08:14:41.301628Z","iopub.execute_input":"2023-06-28T08:14:41.302028Z","iopub.status.idle":"2023-06-28T08:14:41.310964Z","shell.execute_reply.started":"2023-06-28T08:14:41.301996Z","shell.execute_reply":"2023-06-28T08:14:41.310074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def single (oof_1, name=\"\"):\n\n    true = oof_1.copy()\n    for k in range(18):\n        # GET TRUE LABELS\n        true[k] = targets[f\"q{k+1}\"].values\n\n    # FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\n    scores = []; thresholds = []\n    best_score = 0; best_threshold = 0\n\n    for threshold in np.arange(0.5,0.81,0.005):\n        preds = (oof_1.values.reshape((-1))>threshold).astype('int')\n        m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n        scores.append(m)\n        thresholds.append(threshold)\n        if m>best_score:\n            best_score = m\n            best_threshold = threshold\n\n    # COMPUTE F1 SCORE OVERALL\n    p =  (oof_1.values.reshape((-1))>best_threshold).astype('int')\n    t = true.values.reshape((-1))\n    \n    m = f1_score(t, p, average='macro')\n    print(f'{name} F1: {m:.5f}, threshold: {best_threshold:.3f}, shape: {true.shape}')    \n    return p , t, best_threshold\n\ndef compose ( oofs ):\n    oof = oofs[0].copy()\n    for n, oof_ in enumerate(oofs):         \n        if n > 0:\n            oof += oof_\n    oof /= len(oofs)\n    return oof\n\ndef merge (oofs, name=\"\"):\n\n    # PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\n    oof = compose(oofs)\n\n    true = oof.copy()\n    for k in range(18):\n        # GET TRUE LABELS\n        true[k] = targets[f\"q{k+1}\"].values\n\n    \n    # FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\n    scores = []; thresholds = []\n    best_score = 0; best_threshold = 0\n\n\n\n    for threshold in np.arange(0.5,0.81,0.005):\n \n        preds = (oof.values.reshape((-1))>threshold).astype('int')\n        m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n        scores.append(m)\n        thresholds.append(threshold)\n        if m>best_score:\n            best_score = m\n            best_threshold = threshold\n\n    # COMPUTE F1 SCORE OVERALL\n    \n    p =  (oof.values.reshape((-1))>best_threshold).astype('int')\n    t = true.values.reshape((-1))\n    m = f1_score(t, p, average='macro')\n    print(f'{name} F1: {m:.5f}, threshold: {best_threshold:.3f}, shape: {true.shape}')    \n    return p\n\ndef wmerge (a,b, name=\"\"):\n\n    true = a.copy()\n    for k in range(18):\n        # GET TRUE LABELS\n        true[k] = targets[f\"q{k+1}\"].values\n\n    # FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\n    scores = []; thresholds = []\n    best_score = 0; best_threshold = 0; best_p = 0\n\n\n    for p in np.arange(0.0,1.0,0.05):\n        for threshold in np.arange(0.5,0.81,0.005):\n            oof = p*a + (1-p)*b\n            preds = (oof.values.reshape((-1))>threshold).astype('int')\n            m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n            scores.append(m)\n            thresholds.append(threshold)\n            if m>best_score:\n                best_score = m\n                best_threshold = threshold\n                best_p =p\n                \n    # COMPUTE F1 SCORE OVERALL\n    oof = (best_p)*a + (1-best_p)*b    \n    pred =  (oof.values.reshape((-1))>best_threshold).astype('int')\n    t = true.values.reshape((-1))\n    m = f1_score(t, pred, average='macro')\n    print(f'{name} F1: {m:.5f}, threshold: {best_threshold:.3f}, p:{best_p:.3f}')    \n    return pred\n\n\ndef wmerge_3 (a,b,c, name=\"\"):\n\n    true = a.copy()\n    for k in range(18):\n        # GET TRUE LABELS\n        true[k] = targets[f\"q{k+1}\"].values\n\n\n    # FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\n    scores = []; thresholds = []\n    best_score = 0; best_threshold = 0; best_w1 = 0; best_w2 = 0\n\n    for w1 in np.arange(0.1,0.9,0.05):\n        for w2 in np.arange(0.1,0.9,0.05):\n            if w1 + w2 < 1:\n                for threshold in np.arange(0.5,0.81,0.005):\n                    oof = w1*a + w2*b + (1 -w1 - w2)*c\n                    preds = (oof.values.reshape((-1))>threshold).astype('int')\n                    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n                    scores.append(m)\n                    thresholds.append(threshold)\n                    if m>best_score:\n                        best_score = m\n                        best_threshold = threshold\n                        best_w1 =w1\n                        best_w2 =w2\n                        print( f\"({best_score:.4f}, {best_w1:.3f}, {best_w2:.3f})\", end=\",\")\n    print()\n                \n    # COMPUTE F1 SCORE OVERALL\n    oof = best_w1*a + best_w2*b + (1 -best_w1 - best_w2)*c   \n    pred =  (oof.values.reshape((-1))>best_threshold).astype('int')\n    t = true.values.reshape((-1))\n    m = f1_score(t, pred, average='macro')\n    print(f'{name} F1: {m:.5f}, threshold: {best_threshold:.3f}, w1:{best_w1:.3f}, w2:{best_w2:.3f}')    \n    return pred\n\ndef voting (ps, t, name=\"\"):\n    \n    p = ps[0].copy()\n    \n    for n, p_ in enumerate(ps):\n        if n > 0:\n            p += p_\n    p = p/len(ps)\n    p = (p>0.5).astype('int')\n    m = f1_score(t, p, average='macro')\n    print(f'{name} F1: {m:.5f}')","metadata":{"execution":{"iopub.status.busy":"2023-06-28T08:14:41.669469Z","iopub.execute_input":"2023-06-28T08:14:41.670206Z","iopub.status.idle":"2023-06-28T08:14:41.696115Z","shell.execute_reply.started":"2023-06-28T08:14:41.670169Z","shell.execute_reply":"2023-06-28T08:14:41.694967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mg = pd.read_parquet(\"/kaggle/input/psp2-eda-possibile-leak-2-sessioni/multiple_games.parquet\")\nmg_sids = mg[\"session_id\"].unique()\n\ntargets = pd.read_parquet(f\"/kaggle/input/psp2-fe-04b-gt-08-fillnull/labels.parquet\")\ntargets = targets.query(\"session_id not in @mg_sids\")\nALL_USERS =  targets[\"session_id\"].values","metadata":{"execution":{"iopub.status.busy":"2023-06-28T08:14:42.131140Z","iopub.execute_input":"2023-06-28T08:14:42.131526Z","iopub.status.idle":"2023-06-28T08:14:42.174552Z","shell.execute_reply.started":"2023-06-28T08:14:42.131498Z","shell.execute_reply":"2023-06-28T08:14:42.171950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\ndef find_oof_csv_files(directory):\n    oof_csv_files = []\n    for root, dirs, files in os.walk(directory):\n        for file in files:\n            if file == \"oof.csv\":\n                oof_csv_files.append(os.path.join(root, file))\n    return oof_csv_files\n\noof_files = {path.split(\"/\")[2]: path for path in find_oof_csv_files(\"../input\")}","metadata":{"execution":{"iopub.status.busy":"2023-06-28T08:14:42.721054Z","iopub.execute_input":"2023-06-28T08:14:42.721445Z","iopub.status.idle":"2023-06-28T08:14:42.776143Z","shell.execute_reply.started":"2023-06-28T08:14:42.721418Z","shell.execute_reply":"2023-06-28T08:14:42.775200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = dict()\n\nfor model, path in tqdm(oof_files.items()):\n    oof = pd.read_csv(path)\n    oof.index = ALL_USERS\n    oof.columns = range(18)\n    p, t, best_threshold = single(oof, name=model)\n    data[model] = [oof, p, t, best_threshold]","metadata":{"execution":{"iopub.status.busy":"2023-06-28T08:14:43.309477Z","iopub.execute_input":"2023-06-28T08:14:43.309827Z","iopub.status.idle":"2023-06-28T08:19:15.101360Z","shell.execute_reply.started":"2023-06-28T08:14:43.309800Z","shell.execute_reply":"2023-06-28T08:19:15.097500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true = data[list(data.keys())[0]][0].copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    true[k] = targets[f\"q{k+1}\"].values","metadata":{"execution":{"iopub.status.busy":"2023-06-28T08:19:15.103023Z","iopub.execute_input":"2023-06-28T08:19:15.104002Z","iopub.status.idle":"2023-06-28T08:19:15.117802Z","shell.execute_reply.started":"2023-06-28T08:19:15.103951Z","shell.execute_reply":"2023-06-28T08:19:15.116533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_best_threshold(y_true, y_prob, best_score, best_threshold):\n    for threshold in np.arange(0.5,0.81,0.005):\n        y_pred = np.ravel(y_prob > threshold).astype(int)\n        tmp_score = f1_score(y_true=y_true, y_pred=y_pred, average='macro')\n        if tmp_score > best_score:\n            best_score = tmp_score\n            best_threshold = threshold\n    return best_score, best_threshold\n\ndef build_ensemble(ensemble, data):\n    if len(ensemble) == 0:\n        built = None\n    else:\n        for k, p in enumerate(ensemble):\n            if k==0:\n                built = data[p][0].copy()\n            else:\n                built += data[p][0]\n            built /= len(ensemble)\n    return built\n\ndef build_weighted_ensemble(ensemble, data, thresholds):\n    if len(ensemble) == 0:\n        built = None\n        weights = None\n    elif len(ensemble) == 1:\n        built = get_multhreshold_preds(data[ensemble[0]][0], thresholds[ensemble[0]])\n        weights = [np.ones(1) for k in range(18)]\n    else:\n        folds = targets[\"fold\"]\n        weights = [None for k in range(18)]\n        for k in range(18):\n            X = pd.DataFrame(np.vstack([data[model][0].iloc[:,k] for model in ensemble]).T,\n                     index = targets.session_id, columns=ensemble)\n            y = targets.set_index(\"session_id\").iloc[:,k]\n            lr = LogisticRegression(penalty=None, C=1.0, fit_intercept=False)\n            lr.fit(X, y)\n            weights[k] = list(np.exp(lr.coef_[0]) / np.sum(np.exp(lr.coef_[0])))\n        for k, model in enumerate(ensemble):\n            if k==0:\n                built = get_multhreshold_preds(data[model][0], thresholds[model]) * np.array([w[k] for j,w in enumerate(weights)])\n            else:\n                built += get_multhreshold_preds(data[model][0], thresholds[model]) * np.array([w[k] for j,w in enumerate(weights)])\n    return built, weights\n\ndef compare_dataframe_with_dict(dataframe, comparison_dict):\n    # Create an empty dataframe to store the comparison results\n    comparison_df = pd.DataFrame()\n    for col in dataframe.columns:\n        compare_value = comparison_dict.get(col, None)\n        if compare_value is not None:\n            comparison_result = dataframe[col] > compare_value\n            comparison_df[col] = comparison_result\n    return comparison_df.astype(int)\n\ndef optimize_multithreshold(ensemble, targets, best_threshold, rounds=1):\n    best_thresholds = {i:best_threshold for i in range(18)}\n    y_true = np.ravel(targets.set_index(\"session_id\").drop(\"fold\", axis=1))\n    y_pred = np.ravel(compare_dataframe_with_dict(ensemble, best_thresholds))\n    best_score = f1_score(y_true=y_true, y_pred=y_pred, average='macro')\n    \n    for k in range(rounds):\n        for col in range(18):\n            previous_value = best_thresholds[col]\n            improvements = []\n            for threshold in np.arange(0.5, 0.81, 0.005):\n                best_thresholds[col] = threshold\n                y_pred = np.ravel(compare_dataframe_with_dict(ensemble, best_thresholds))\n                score = f1_score(y_true=y_true, y_pred=y_pred, average='macro')\n                if score > best_score:\n                    improvements.append((threshold, score))\n            if len(improvements) == 0:\n                best_thresholds[col] = previous_value\n            else:\n                improvements = sorted(improvements, key=lambda x: x[1], reverse=True)\n                best_thresholds[col] = improvements[0][0]\n    y_true = np.ravel(targets.set_index(\"session_id\").drop(\"fold\", axis=1))\n    y_pred = np.ravel(compare_dataframe_with_dict(ensemble, best_thresholds))\n    best_score = f1_score(y_true=y_true, y_pred=y_pred, average='macro')\n    return best_score, best_thresholds\n\ndef cv_evaluation(y_true, y_pred, cv=targets.fold.values):\n    scores = list()\n    for i in np.unique(cv):\n        cv_vec = np.ravel([cv!=i for z in range(18)])\n        scores.append(f1_score(y_true=y_true[cv_vec!=i], y_pred=y_pred[cv_vec!=i], average='macro'))\n    return np.mean(scores)\n\ndef get_multhreshold_preds(y_prob, thresholds):\n    y_pred = np.zeros(y_prob.shape)\n    for i in range(18):\n        y_pred[:, i] = (y_prob.values[:, i] > thresholds[i])\n    return pd.DataFrame(y_pred, columns=y_prob.columns, index=y_prob.index)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:58:15.822726Z","iopub.execute_input":"2023-06-28T09:58:15.823236Z","iopub.status.idle":"2023-06-28T09:58:15.851816Z","shell.execute_reply.started":"2023-06-28T09:58:15.823201Z","shell.execute_reply":"2023-06-28T09:58:15.851035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_single_thresholds = dict()\ny_true = np.ravel(targets.set_index(\"session_id\").drop(\"fold\", axis=1))\n\nfor model in tqdm(data):\n    best_threshold = 0.625\n    y_pred = np.ravel(data[model][0] > best_threshold).astype(int)\n    y_prob = data[model][0]\n    best_score = cv_evaluation(y_true, y_pred, cv=targets.fold.values)\n    best_score, best_thresholds = find_best_threshold(y_true, y_prob, best_score, best_threshold)\n    best_score, best_thresholds = optimize_multithreshold(y_prob, targets, best_thresholds, rounds=1)\n    #best_thresholds = {t:best_thresholds for t in range(18)}\n    best_single_thresholds[model] = best_thresholds","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:32:04.181337Z","iopub.execute_input":"2023-06-28T09:32:04.181725Z","iopub.status.idle":"2023-06-28T09:36:45.821437Z","shell.execute_reply.started":"2023-06-28T09:32:04.181694Z","shell.execute_reply":"2023-06-28T09:36:45.820536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# optimization","metadata":{}},{"cell_type":"code","source":"#np.random.seed(1)\n\nensemble = []\nensemble_score = 0.0\nensemble_threshold = 0.5\nensemble_weights = None\ntemperature = 0.65\ncoolant = 0.05\nrounds = 100\n\nfor r in range(rounds):\n    print(f\"round {r+1}\")\n    # rebuild ensemble\n    en, weights = build_weighted_ensemble(ensemble, data, best_single_thresholds)\n    \n    # try adding something\n    addings = []\n    for item in tqdm(list(data.keys())):\n        en2, weights2 = build_weighted_ensemble(ensemble + [item], data, best_single_thresholds)\n        best_threshold = 0.625\n        y_true = np.ravel(targets.set_index(\"session_id\").drop(\"fold\", axis=1))\n        y_pred = np.ravel(en2 > best_threshold).astype(int)\n        best_score = cv_evaluation(y_true, y_pred, cv=targets.fold.values)\n        \n        y_true = np.ravel(targets.set_index(\"session_id\").drop(\"fold\", axis=1))\n        y_prob = en2\n        best_score, best_thresholds = find_best_threshold(y_true, y_prob, best_score, best_threshold)\n        #best_score, best_thresholds = optimize_multithreshold(y_prob, targets, best_thresholds, rounds=1)\n        best_score = cv_evaluation(y_true, np.ravel(y_prob) > best_thresholds, cv=targets.fold.values)\n        addings.append((item, best_score, best_thresholds, weights2))\n    addings = sorted(addings, key=lambda x: x[1], reverse=True)\n    improvements = [items for items in addings if items[1] > ensemble_score]\n    print(f\"{len(improvements)} additions can improve the ensemble\", end = \" | \")\n    if len(improvements) > 0:\n        if random.random() <= temperature:\n            chosen_element = random.choices(list(range(len(improvements))))\n            a,b,c,w = improvements[chosen_element[0]]\n            print(\"random choice happening\")\n        else:\n            a,b,c,w = improvements[0]\n            print(\"choosing the best one\")\n        new_ensemble = ensemble + [a]\n        new_ensemble_score = b\n        new_ensemble_threshold = c\n        new_ensemble_weights = w\n        print(f\"{new_ensemble} | {new_ensemble_score:.5f} with threshold {new_ensemble_threshold}\")\n    else:\n        new_ensemble = ensemble\n        new_ensemble_score = ensemble_score\n        new_ensemble_threshold = ensemble_threshold\n        new_ensemble_weights = ensemble_weights\n    \n    # try removing something\n    if len(new_ensemble) > 1:\n        removings = list()\n        for i in tqdm(range(len(new_ensemble))):\n            removed_element = new_ensemble.pop(i)\n            if len([True for d in removings if d[0]==removed_element])==0:\n                for replacement in list(data.keys()):\n                    if replacement != removed_element:\n                        y_true = np.ravel(targets.set_index(\"session_id\").drop(\"fold\", axis=1))\n                        y_prob, weights2 = build_weighted_ensemble(new_ensemble + [replacement], data, best_single_thresholds)\n                        best_score, best_thresholds = find_best_threshold(y_true, y_prob, 0.0, 0.625)\n                        #best_score, best_thresholds = optimize_multithreshold(y_prob, targets, best_thresholds, rounds=1)\n                        best_score = cv_evaluation(y_true, np.ravel(y_prob) > best_thresholds, cv=targets.fold.values)\n                        removings.append((removed_element, best_score, best_thresholds, replacement, weights2))\n            new_ensemble.insert(i, removed_element)\n        improvements = [items for items in removings if items[1] > new_ensemble_score]\n        print(f\"{len(improvements)} removings can improve the ensemble\", end=\" | \")\n        if len(improvements) > 0:\n            if random.random() <= temperature:\n                chosen_element = random.choices(list(range(len(improvements))))\n                a,b,c,d,w = improvements[chosen_element[0]]\n                print(\"random choice happening\")\n            else:\n                a,b,c,d,w = improvements[0]\n                print(\"choosing the best one\")\n            new_ensemble.pop(new_ensemble.index(a))\n            new_ensemble += [d]\n            new_ensemble_score = b\n            new_ensemble_threshold = c\n            new_ensemble_weights = w\n            print(f\"{new_ensemble} | {new_ensemble_score:.5f} with threshold {new_ensemble_threshold}\")\n    \n    # cooling\n    if random.random() <= temperature:\n        temperature = max([0, temperature - coolant])\n    print(f\"temperature is now {temperature}\")\n    \n    # stopping criteria\n    if new_ensemble_score == ensemble_score or r == (rounds - 1):\n        print(\"*** no improvements | Optimizing complete ***\")\n        break\n    else:\n        ensemble = new_ensemble\n        ensemble_score = new_ensemble_score\n        ensemble_threshold = new_ensemble_threshold\n        ensemble_weights = new_ensemble_weights\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:58:22.170412Z","iopub.execute_input":"2023-06-28T09:58:22.170788Z","iopub.status.idle":"2023-06-28T10:37:10.009724Z","shell.execute_reply.started":"2023-06-28T09:58:22.170760Z","shell.execute_reply":"2023-06-28T10:37:10.008952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(ensemble)\nprint()\nprint(ensemble_weights)\nprint()\nprint(f\"\\nf1 score: {ensemble_score:.5f}\")\nprint(f\"best threshold: \\n{ensemble_threshold}\")\n# REF f1_score: 0.7010924028845662","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:38:52.400719Z","iopub.execute_input":"2023-06-28T10:38:52.401147Z","iopub.status.idle":"2023-06-28T10:38:52.407474Z","shell.execute_reply.started":"2023-06-28T10:38:52.401116Z","shell.execute_reply":"2023-06-28T10:38:52.406477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"{k: best_single_thresholds[model] for k, model in enumerate(ensemble)}","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:39:08.608054Z","iopub.execute_input":"2023-06-28T10:39:08.608440Z","iopub.status.idle":"2023-06-28T10:39:08.622785Z","shell.execute_reply.started":"2023-06-28T10:39:08.608411Z","shell.execute_reply":"2023-06-28T10:39:08.621332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}