{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Objective\n\nThis notebook is not intended to give a high leaderboard or show an awesome blend. \n\nIn this notebook i will use 2 public submissions that score 0.940 and 0.936 and 3 personal submissions. Blending and don't knowing the cv and the cv strategy of this models is a horrible idea because they are probably overtiffing. \n\nThe main idea of this notebook is to talk about the predictions distribution. The predictions have different distribution so combining them in the form x1*w1 + x2*w2 + .... + xn*wn is not recommended at all. Receiver Operating Characteristic area under the curve is sensible to this distribution. For this we should first rank each of this vector and then blend the predictions in the form x1*w1 + x2*w2 + .... + xn*wn. This is just an example of how you should do that.","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"sub1 = pd.read_csv('../input/melanoma-dif-sub/pl_0.936.csv')\nsub2 = pd.read_csv('../input/melanoma-dif-sub/pl_0.940.csv')\nsub3 = pd.read_csv('../input/melanoma-dif-sub/sub_EfficientNetB2_384.csv')\nsub4 = pd.read_csv('../input/melanoma-dif-sub/sub_EfficientNetB3_384.csv')\nsub5 = pd.read_csv('../input/melanoma-dif-sub/sub_EfficientNetB3_384_v2.csv')\n\n# lets rank each prediction and then divide it by its max value to we have our predictions between 0 and 1\ndef rank_data(sub):\n    sub['target'] = sub['target'].rank() / sub['target'].rank().max()\n    return sub\n\nsub1 = rank_data(sub1)\nsub2 = rank_data(sub2)\nsub3 = rank_data(sub3)\nsub4 = rank_data(sub4)\nsub5 = rank_data(sub5)\nsub1.columns = ['image_name', 'target1']\nsub2.columns = ['image_name', 'target2']\nsub3.columns = ['image_name', 'target3']\nsub4.columns = ['image_name', 'target4']\nsub5.columns = ['image_name', 'target5']\n\nf_sub = sub1.merge(sub2, on = 'image_name').merge(sub3, on = 'image_name').merge(sub4, on = 'image_name').merge(sub5, on = 'image_name')\nf_sub['target'] = f_sub['target1'] * 0.3 + f_sub['target2'] * 0.3 + f_sub['target3'] * 0.05 + f_sub['target4'] * 0.3 + f_sub['target5'] * 0.05\nf_sub = f_sub[['image_name', 'target']]\nf_sub.to_csv('blend_sub.csv', index = False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}