{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":1428760,"sourceType":"datasetVersion","datasetId":836753}],"dockerImageVersionId":29994,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# How To Ensemble OOF\nIn this notebook, we learn how to use `forward selection` to ensemble OOF. First build lots of models using the same KFolds (i.e. use same `seed`). Next save all the oof files as `oof_XX.csv` and submission files as `sub_XX.csv` where the oof and submission share the same `XX` number. Then save them in a Kaggle dataset and run the code below.\n\nThe ensemble begins with the model of highest oof AUC. Next each other model is added one by one to see which additional model increases ensemble AUC the most. The best additional model is kept and the process is repeated until the ensemble AUC doesn't increase.","metadata":{}},{"cell_type":"markdown","source":"# Read OOF Files\nWhen i get more time, I will compete this table to describe all 39 models in this notebook. For now here are the ones that get selected:\n\n| k | CV | LB | read size | crop size | effNet | ext data | upsample | misc | name |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |\n| 1 | 0.910 | 0.950 | 384 | 384 | B6 | 2018 | no |  | oof_100 |\n| 3 | 0.916 | 0.946 | 384 | 384 | B345 | no | no |  | oof_108 |\n| 8 | 0.935 | 0.949 | 768 | 512 | B7 | 2018 | 1,1,1,1 |  | oof_113 |\n| 10 | 0.920 | 0.941 | 512 | 384 | B5 | 2019 2018 | 10,0,0,0 |  | oof_117 |\n| 12 | 0.935 | 0.937 | 768 | 512 | B6 | 2019 2018 | 3,3,0,0 |  | oof_120 |\n| 21 | 0.933 | 0.950 | 1024 | 512 | B6 | 2018 | 2,2,2,2 |  | oof_30 |\n| 26 | 0.927 | 0.942 | 768 | 384 | B4 | 2018 | no |  | oof_385 |\n| 37 | 0.936 | 0.956 | 512 | 384 | B5 | 2018 | 1,1,1,1 |  | oof_67 |\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd, numpy as np, os\nfrom sklearn.metrics import roc_auc_score\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-28T07:07:29.229768Z","iopub.execute_input":"2024-12-28T07:07:29.230149Z","iopub.status.idle":"2024-12-28T07:07:30.038652Z","shell.execute_reply.started":"2024-12-28T07:07:29.230121Z","shell.execute_reply":"2024-12-28T07:07:30.035506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PATH = '../input/melanoma-oof-and-sub/'\nFILES = os.listdir(PATH)\n\nOOF = np.sort( [f for f in FILES if 'oof' in f] )\nOOF_CSV = [pd.read_csv(PATH+k) for k in OOF]\n\nprint('We have %i oof files...'%len(OOF))\nprint(); print(OOF)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T07:07:30.040910Z","iopub.execute_input":"2024-12-28T07:07:30.041317Z","iopub.status.idle":"2024-12-28T07:07:31.719833Z","shell.execute_reply.started":"2024-12-28T07:07:30.041279Z","shell.execute_reply":"2024-12-28T07:07:31.718946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x = np.zeros(( len(OOF_CSV[0]),len(OOF) ))\nfor k in range(len(OOF)):\n    x[:,k] = OOF_CSV[k].pred.values\n    \nTRUE = OOF_CSV[0].target.values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T07:09:50.964046Z","iopub.execute_input":"2024-12-28T07:09:50.964423Z","iopub.status.idle":"2024-12-28T07:09:50.996516Z","shell.execute_reply.started":"2024-12-28T07:09:50.964392Z","shell.execute_reply":"2024-12-28T07:09:50.995580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all = []\nfor k in range(x.shape[1]):\n    auc = roc_auc_score(OOF_CSV[0].target,x[:,k])\n    all.append(auc)\n    print('Model %i has OOF AUC = %.4f'%(k,auc))\n    \nm = [np.argmax(all)]; w = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T07:11:42.716762Z","iopub.execute_input":"2024-12-28T07:11:42.717124Z","iopub.status.idle":"2024-12-28T07:11:43.185976Z","shell.execute_reply.started":"2024-12-28T07:11:42.717094Z","shell.execute_reply":"2024-12-28T07:11:43.184868Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Build OOF Ensemble. Maximize CV Score","metadata":{}},{"cell_type":"code","source":"old = np.max(all); \n\nRES = 200; \nPATIENCE = 10; \nTOL = 0.0003\nDUPLICATES = False\n\n# start with the best model \nprint('Ensemble AUC = %.4f by beginning with model %i'%(old,m[0]))\nprint()\n\nfor kk in range(len(OOF)): # This loop iterates, attempting to add one model to the ensemble in each iteration. \n    \n    # BUILD CURRENT ENSEMBLE\n    md = x[:,m[0]] # md is the best model\n    for i,k in enumerate(m[1:]): # This loop iterates through the models already added to the ensemble (excluding the first/best one).\n        md = w[i]*x[:,k] + (1-w[i])*md # Weighted average\n        \n    # FIND MODEL TO ADD\n    mx = 0; mx_k = 0; mx_w = 0 # mx is best score, mx_k is its index, mx_w is weight\n    print('Searching for best model to add... ')\n    \n    # TRY ADDING EACH MODEL\n    for k in range(x.shape[1]):\n        print(k,', ',end='')\n        if not DUPLICATES and (k in m): continue \n            \n        # EVALUATE ADDING MODEL K WITH WEIGHTS W\n        bst_j = 0; bst = 0; ct = 0\n        for j in range(RES): # This loop iterates through the discrete weight values \n            tmp = j/RES*x[:,k] + (1-j/RES)*md\n            auc = roc_auc_score(TRUE,tmp)\n            if auc>bst:\n                bst = auc\n                bst_j = j/RES\n            else: ct += 1\n            if ct>PATIENCE: break\n        if bst>mx:\n            mx = bst\n            mx_k = k\n            mx_w = bst_j\n            \n    # STOP IF INCREASE IS LESS THAN TOL\n    inc = mx-old\n    if inc<=TOL: \n        print(); print('No increase. Stopping.')\n        break\n        \n    # DISPLAY RESULTS\n    print(); #print(kk,mx,mx_k,mx_w,'%.5f'%inc)\n    print('Ensemble AUC = %.4f after adding model %i with weight %.3f. Increase of %.4f'%(mx,mx_k,mx_w,inc))\n    print()\n    \n    old = mx; m.append(mx_k); w.append(mx_w)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T07:22:09.650813Z","iopub.execute_input":"2024-12-28T07:22:09.651197Z","iopub.status.idle":"2024-12-28T07:23:45.012649Z","shell.execute_reply.started":"2024-12-28T07:22:09.651167Z","shell.execute_reply":"2024-12-28T07:23:45.011632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('We are using models',m)\nprint('with weights',w)\nprint('and achieve ensemble AUC = %.4f'%old)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T07:34:42.387616Z","iopub.execute_input":"2024-12-28T07:34:42.387931Z","iopub.status.idle":"2024-12-28T07:34:42.393618Z","shell.execute_reply.started":"2024-12-28T07:34:42.387905Z","shell.execute_reply":"2024-12-28T07:34:42.392812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"md = x[:,m[0]]\nfor i,k in enumerate(m[1:]):\n    md = w[i]*x[:,k] + (1-w[i])*md\nplt.hist(md,bins=100)\nplt.title('Ensemble OOF predictions')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T07:35:24.531946Z","iopub.execute_input":"2024-12-28T07:35:24.532296Z","iopub.status.idle":"2024-12-28T07:35:24.855396Z","shell.execute_reply.started":"2024-12-28T07:35:24.532265Z","shell.execute_reply":"2024-12-28T07:35:24.854403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = OOF_CSV[0].copy()\ndf.pred = md\ndf.to_csv('ensemble_oof.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T07:35:47.221874Z","iopub.execute_input":"2024-12-28T07:35:47.222236Z","iopub.status.idle":"2024-12-28T07:35:47.612186Z","shell.execute_reply.started":"2024-12-28T07:35:47.222207Z","shell.execute_reply":"2024-12-28T07:35:47.611300Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load SUB Files","metadata":{}},{"cell_type":"code","source":"SUB = np.sort( [f for f in FILES if 'sub' in f] )\nSUB_CSV = [pd.read_csv(PATH+k) for k in SUB]\n\nprint('We have %i submission files...'%len(SUB))\nprint(); print(SUB)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T07:35:48.664783Z","iopub.execute_input":"2024-12-28T07:35:48.665159Z","iopub.status.idle":"2024-12-28T07:35:49.306581Z","shell.execute_reply.started":"2024-12-28T07:35:48.665123Z","shell.execute_reply":"2024-12-28T07:35:49.305750Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# VERFIY THAT SUBMISSION FILES MATCH OOF FILES\na = np.array( [ int( x.split('_')[1].split('.')[0]) for x in SUB ] )\nb = np.array( [ int( x.split('_')[1].split('.')[0]) for x in OOF ] )\nif len(a)!=len(b):\n    print('ERROR submission files dont match oof files')\nelse:\n    for k in range(len(a)):\n        if a[k]!=b[k]: print('ERROR submission files dont match oof files')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T07:35:54.824850Z","iopub.execute_input":"2024-12-28T07:35:54.825241Z","iopub.status.idle":"2024-12-28T07:35:54.832018Z","shell.execute_reply.started":"2024-12-28T07:35:54.825204Z","shell.execute_reply":"2024-12-28T07:35:54.831146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = np.zeros(( len(SUB_CSV[0]),len(SUB) ))\nfor k in range(len(SUB)):\n    y[:,k] = SUB_CSV[k].target.values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T07:35:58.980448Z","iopub.execute_input":"2024-12-28T07:35:58.980907Z","iopub.status.idle":"2024-12-28T07:35:59.002070Z","shell.execute_reply.started":"2024-12-28T07:35:58.980860Z","shell.execute_reply":"2024-12-28T07:35:59.000771Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Build SUB Ensemble","metadata":{}},{"cell_type":"code","source":"md2 = y[:,m[0]]\nfor i,k in enumerate(m[1:]):\n    md2 = w[i]*y[:,k] + (1-w[i])*md2\nplt.hist(md2,bins=100)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T07:36:03.945944Z","iopub.execute_input":"2024-12-28T07:36:03.946320Z","iopub.status.idle":"2024-12-28T07:36:04.306172Z","shell.execute_reply.started":"2024-12-28T07:36:03.946284Z","shell.execute_reply":"2024-12-28T07:36:04.305279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = SUB_CSV[0].copy()\ndf.target = md2\ndf.to_csv('ensemble_sub.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T07:36:06.641247Z","iopub.execute_input":"2024-12-28T07:36:06.641608Z","iopub.status.idle":"2024-12-28T07:36:06.684630Z","shell.execute_reply.started":"2024-12-28T07:36:06.641575Z","shell.execute_reply":"2024-12-28T07:36:06.683866Z"}},"outputs":[],"execution_count":null}]}