{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport os\nfrom deap.tools.mutation import mutPolynomialBounded\nfrom deap.tools.crossover import cxSimulatedBinaryBounded\n\nsns.set_style(\"darkgrid\", {\"grid.color\": \".6\", \"grid.linestyle\": \":\"})","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-07T18:01:24.696011Z","iopub.execute_input":"2022-07-07T18:01:24.697007Z","iopub.status.idle":"2022-07-07T18:01:24.705888Z","shell.execute_reply.started":"2022-07-07T18:01:24.696951Z","shell.execute_reply":"2022-07-07T18:01:24.704562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"00\"></a>\n# <p style=\"background-color:#002663;height: 60px;text-align: center;vertical-align: middle;line-height: 60px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> Introduction </p>\n<div style=\"font-family: courier; font-size:18px\">\n\n    This is an implemention of a genetic algorithm (GA) to optimize the parameters of an ensemble. Given the OOFs file from the crossvalidation training, the GA will try to optimize the contribution of each model to increase the final ensemble metric. \n    \n    In this scenario, the population of the GA will consist of N individuals. Each individual will be a vector of M weigths, one for  each model. The final population will be a matrix of (N,M) size. Hence the fitness metric will be the AMEX metric of the ensemble model with the given weights.\n    \n    To aid the ensemble optimization, the GA is used in a recursive loop to drop a model if it's contribution, it's weigth found in a single optimization, is above a given threshold.\n    \n","metadata":{}},{"cell_type":"markdown","source":" <div style=\"font-family: courier; font-size:18px\">\n Genetic Algorithm Main attributes: \n    <li> Population: (NxM) random initialization of Weigths [0, 100]\n    <li> Crossover: simulated binary crossover (bounded) (SBX)\n    <li> Mutation: Polynomial Mutation (bounded) (PM)\n    <li> Selection: (1) Random Selection or (2) Roullet Selection \n   ","metadata":{}},{"cell_type":"markdown","source":"<a id=\"00\"></a>\n# <p style=\"background-color:#002663;height: 60px;text-align: center;vertical-align: middle;line-height: 60px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> Functions </p>","metadata":{}},{"cell_type":"code","source":"def roullet_selection(metric, n = 2):\n    f = metric/metric.sum()\n    f = f.cumsum()\n    prob = np.random.random(n)\n    parents = []\n    for i in prob:\n        parents.append(np.argwhere(f>i)[0][0])\n    return parents\n\ndef random_selection(indices):\n    index = np.random.randint(len(indices))\n    p1 = indices.pop(index)\n    index = np.random.randint(len(indices))\n    p2 = indices.pop(index)\n    return p1, p2\n\ndef generate_population(n,m):\n    return np.random.uniform(0,100, (n,m))\n\ndef fob(x,y, pop):\n    result = []\n    for i in pop:\n        i = i.reshape(-1,1)\n        probas_weighted = np.sum(np.dot(x,i), axis = 1)/i.sum(axis = 0)\n\n        probas_weighted = np.round(probas_weighted,3)\n        metrics = amex_metric_np(y, probas_weighted )\n        result.append(metrics)\n\n    return result\n\n\ndef GA_OPTIMIZER(x, TRUE, n, gen, p_m, model_list_ = [], n_child = 10, selc_random = 1):\n    _, m = x.shape\n    metrics = {'best_':[],'hist_metric':[]}\n\n    pop = generate_population(n,m)\n    have_mut = 0\n\n    # hist_metric =[]\n    # best_ = []\n\n    metric = np.array(fob(x,TRUE,pop))\n\n    for e in range(gen):\n        indices = list(np.arange(n))\n        # child = np.array([], dtype=np.int64).reshape(int(n/2),m)\n        child = []\n\n\n        for i in range(int(n_child/2)):\n\n            if selc_random:\n                p1, p2 = random_selection(indices)\n            else:\n                p1, p2 = roullet_selection(metric)\n        \n            \n            # c1, c2 = np.array(SBX_with_repair(pop[p1],pop[p2],0.5, a = 0, b = 100))\n            c1, c2 = cxSimulatedBinaryBounded(pop[p1],pop[p2],0.5,  0, 100)\n\n            if np.random.rand() > p_m:\n                have_mut +=1\n                #    c1 = PLM_with_repair(c1, eta = 0.5, a = 0, b = 100)\n                c1 = mutPolynomialBounded(c1, eta = 0.5, low = 0, up = 100,indpb = 1)[0]\n\n            child.append(c1)\n            child.append(c2)\n\n            \n            \n        child = np.array(child)\n\n        metric_child = np.array(fob(x,TRUE,child))\n\n        metric = np.hstack([metric, metric_child])\n        pop = np.vstack([pop, child])\n        \n        args = np.argsort(metric)[-n:]\n        pop = pop[args,:]\n        metric = metric[args]\n        \n        metrics['best_'].append(max(metric))\n        metrics['hist_metric'].append(np.mean(metric))\n        print(f'Population Mean of AMEX in Generation {e+1}/{gen}: '\\\n              ,np.round(metrics[\"hist_metric\"][-1],4),f'- Best Metric: {np.round(metrics[\"best_\"][-1],4)}'\\\n              ,f'- Pop std: {np.round(np.std(metric),6)}' ,f'Mutations: {have_mut} ({have_mut/(n/2)*100}%)')\n        have_mut = 0\n\n\n\n    # pop = pop[np.argsort(metric)]\n\n    w = (pop[-1]/pop[-1].sum())\n\n    if False:\n        print('\\n')\n        print('#%'*20)\n        print('modelos_lista_ = ', model_list_)\n        print('w = ', list(w))\n        print('#%'*20)\n        print('\\n')\n\n    fig, ax = plt.subplots(figsize = (12,8))\n    ax.plot(metrics['hist_metric'], label = 'Pop Mean', marker = 'o' )\n    ax.plot(metrics['best_'], label = 'Best Solution', marker = 'o')\n    ax.legend()\n    ax.set_xlabel('Generations')\n    ax.set_ylabel('AMEX')\n    plt.show()\n\n    return model_list_, w, metrics\n\n\ndef amex_metric_np( target: np.ndarray,preds: np.ndarray) -> float:\n    indices = np.argsort(preds)[::-1]\n    preds, target = preds[indices], target[indices]\n\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_mask = cum_norm_weight <= 0.04\n    d = np.sum(target[four_pct_mask]) / np.sum(target)\n\n    weighted_target = target * weight\n    lorentz = (weighted_target / weighted_target.sum()).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    n_pos = np.sum(target)\n    n_neg = target.shape[0] - n_pos\n    gini_max = 10 * n_neg * (n_pos + 20 * n_neg - 19) / (n_pos + 20 * n_neg)\n\n    g = gini / gini_max\n    return 0.5 * (g + d)\n    \n\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T17:03:06.455495Z","iopub.execute_input":"2022-07-07T17:03:06.455865Z","iopub.status.idle":"2022-07-07T17:03:06.48672Z","shell.execute_reply.started":"2022-07-07T17:03:06.455836Z","shell.execute_reply":"2022-07-07T17:03:06.485294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"00\"></a>\n# <p style=\"background-color:#002663;height: 60px;text-align: center;vertical-align: middle;line-height: 60px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> Run Optimization </p>","metadata":{}},{"cell_type":"code","source":"class CFG():\n    # Threshold of contribution to drop a model\n    thr = 0.05\n    # Max number of runs (Calls GA optimizer)\n    runs = 15\n    #Size of parameters population\n    n = 20\n    #Number of generations\n    gen = 40\n    # Number of children in crossover\n    n_child = n\n    #probability of mutation\n    p_m = 0.9\n    #Max tries to reduce number of models \n    patience = 3\n    # Parent Selection Mode: True = Random, False = Roullet \n    selc_random  = True\n    #Path of OOFs\n    path_oofs_folder = '../input/oofs-amex-ex/content/oofs'","metadata":{"execution":{"iopub.status.busy":"2022-07-07T18:50:49.103685Z","iopub.execute_input":"2022-07-07T18:50:49.104064Z","iopub.status.idle":"2022-07-07T18:50:49.109901Z","shell.execute_reply.started":"2022-07-07T18:50:49.104035Z","shell.execute_reply":"2022-07-07T18:50:49.108918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"font-family: courier; font-size:20px\">\n<li>Read the OOF files from trained models","metadata":{}},{"cell_type":"code","source":"folder_models_ex = CFG.path_oofs_folder\n\nprint(folder_models_ex)\noofs_dict = {}\nfor ki, i in enumerate([i for i in os.listdir(folder_models_ex) if i.endswith(\"oof_complete.csv\")]):\n    model_name = f'MODEL_{ki}'\n    oofs_dict[model_name] = pd.read_csv(folder_models_ex+f'/{i}').drop('Unnamed: 0',axis = 1)\n    print(model_name)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T18:26:04.596221Z","iopub.execute_input":"2022-07-07T18:26:04.596621Z","iopub.status.idle":"2022-07-07T18:26:14.317271Z","shell.execute_reply.started":"2022-07-07T18:26:04.596588Z","shell.execute_reply":"2022-07-07T18:26:14.316272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"font-family: courier; font-size:18px\">\n<li>Create Matrix X and check Metric of OOF from the given models","metadata":{}},{"cell_type":"code","source":"x = np.zeros(( len(oofs_dict[list(oofs_dict.keys())[0]]),len(oofs_dict)))\nfor k, model_dict in enumerate((oofs_dict.keys())):\n    x[:,k] = oofs_dict[model_dict].sort_values(by = 'customer_ID_').oof.values\n    \n\nTRUE = pd.read_csv('../input/amex-default-prediction/train_labels.csv').sort_values(by = 'customer_ID')['target'].values\n\nlist_of_models = []\nall = []\n# y_true = pd.read_csv(path_labels)\nfor k, (model, df_oof) in enumerate(oofs_dict.items()):\n    amex_m = amex_metric_np( TRUE, df_oof.sort_values(by = 'customer_ID_').oof.values )\n    all.append(amex_m)\n    list_of_models.append(model)\n    print('Model %i - %s - has OOF AMEX = %.4f'%(k,model,amex_m))\n\nprint(f'#Summary#\\nMean of metrics: {np.mean(all)} - Best: {np.max(all)} - Worst: {np.min(all)} ')\nm = [np.argmax(all)]; w = []\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T18:26:14.32003Z","iopub.execute_input":"2022-07-07T18:26:14.320804Z","iopub.status.idle":"2022-07-07T18:26:39.50832Z","shell.execute_reply.started":"2022-07-07T18:26:14.320757Z","shell.execute_reply":"2022-07-07T18:26:39.506918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"font-family: courier; font-size:20px\">\n<li>Start Optimization","metadata":{}},{"cell_type":"code","source":"x_ite = x.copy()\nmodel_list_ = list_of_models.copy()\nmodels_removed = []\nmetrics_run =  []\nresults = {}\nold = 0 \nst = 0\n\nfor i in range(CFG.runs):\n\n    print(f'Runs {i+1}/{CFG.runs} - Total Models Avaiable {len(model_list_)}')\n\n    model_list_, w, metrics_r= GA_OPTIMIZER(x_ite, TRUE, CFG.n, CFG.gen, CFG.p_m, model_list_, n_child = CFG.n_child, selc_random = CFG.selc_random)\n    metrics_run.append(metrics_r)\n\n    results[f'run_{i}_best'] = (model_list_, w)\n\n    if i+1 < CFG.runs:\n        to_remove = np.argwhere(w<CFG.thr).flatten()\n        x_ite = np.delete(x_ite,to_remove , axis = 1)\n\n        print(f'Models below threshold ({CFG.thr *100}%) of contribution to be removed in this run: {to_remove}')\n        if to_remove.size > 0:\n            model_list_ = [model_list_[i] for i in range(len(model_list_)) if i not in to_remove ]\n\n    \n    if  len(model_list_) == old:\n        st +=1\n        print(f'No Number of Models reduction - {old} - {st}/{CFG.patience}...')\n\n    if st >= CFG.patience:\n        print('Stoping...')\n        break\n\n    old = len(model_list_)\n    print(f'Removed Models: {models_removed}')\n\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T18:27:53.682157Z","iopub.execute_input":"2022-07-07T18:27:53.682931Z","iopub.status.idle":"2022-07-07T18:49:37.039775Z","shell.execute_reply.started":"2022-07-07T18:27:53.682886Z","shell.execute_reply":"2022-07-07T18:49:37.038528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"00\"></a>\n# <p style=\"background-color:#002663;height: 60px;text-align: center;vertical-align: middle;line-height: 60px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> Results </p>","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(ncols = 2, figsize = (18,8))\nfor i, k in enumerate(metrics_run):\n    # ax.plot(k['hist_metric'], label = f'Pop Mean_{i}' )\n    ax[0].plot(k['best_'], label = f'(RUN{i}) Best_Sol Models: ({len(results[\"run_\"+str(i)+\"_best\"][0])})', marker = 'o')\n    ax[0].legend()\n    ax[0].set_xlabel('Generations')\n    ax[0].set_ylabel('AMEX')\n\n    ax[1].plot(k['hist_metric'], label = f'(RUN{i}) Pop_MEAN Models: ({len(results[\"run_\"+str(i)+\"_best\"][0])})', marker = 'o')\n    ax[1].legend()\n    ax[1].set_xlabel('Generations')\n    ax[1].set_ylabel('AMEX')\n    # plt.show()\n\n\nfor b, (k, i) in zip(metrics_run,results.items()):\n    print(k,' Best Amex Metric ',b['best_'][-1])\n    print(f'modelos_lista_ = {i[0]}\\nw={i[1]}\\n')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T18:25:02.765847Z","iopub.execute_input":"2022-07-07T18:25:02.7663Z","iopub.status.idle":"2022-07-07T18:25:03.36996Z","shell.execute_reply.started":"2022-07-07T18:25:02.766264Z","shell.execute_reply":"2022-07-07T18:25:03.368719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}