{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about  ? \n\nWe take available submits from the top public kernel and compute the average correlations. \n\nhttps://www.kaggle.com/code/jjleesunny/0-583-ensemble-submition-op by JJLEE \n\n#### Conclusion: better blend predictions with correlations not more than 0.75, and say 0.3,0.5,0.7\n\n#### Motivation: \n\nTypically blend gives better results as more diverse predictions we have, though that is not straightforward.\nAny way we can look on correlations to get some inituition.\n\n#### Logic:\n\nWe take an example of the top public blend. And compute correlation for those predictions which are available (not in private datasets). \nSince that public blend works quite good it may serve as a kind of indication what correlation range might be suitable. \n\n#### Details: \n\nEspecially for multi-target tasks - not that much clear how to define correlations between predictions.\n\nHere we take a simple route which might serve as a kind of indication - compute correlation for each columns and average these values.\n\nThe resulting  matrix is the following:\n\n\n|            | op2_603 | op2_720 | submission_df | OP2_607 |\n|------------|---------|---------|--------------|---------|\n| op2_603    | 1.00    | 0.31    | 0.49         | 0.75    |\n| op2_720    | 0.31    | 1.00    | 0.26         | 0.29    |\n| submission_df | 0.49    | 0.26    | 1.00         | 0.56    |\n| OP2_607    | 0.75    | 0.29    | 0.56         | 1.00    |\n\n\n#### PS \n\nThe original blend is : \n\nsubmission[col] = (import1[col] *0.2) + (import2[col] *0.12) + (import3[col] *0.2) + (import5[col] *0.15) + (prediction[col] *0.33)\n\nAvailable are: import1 (op2_603) , import2 (op2_720) , import3 (submission_df) , import4 (OP2_607), while import5 and prediction are in private data not available. \n\n\nTechnical notes:\n\n    Version 1,2 - we take abs of correlations before averaging, but later versions - NO abs. There is NO big change in result.\n    NO abs seems to be more reasonable since blend is typically with positive coeffficents. And negatively correlated targets in blend is indication of difference of submits,\n    and that is what we need - how different submits are.\n    \n","metadata":{}},{"cell_type":"code","source":"import warnings # suppress warnings\nwarnings.filterwarnings('ignore')\n#:::::::::::::::::::::::::::::::::::\nimport os\nimport gc\nimport glob\nimport random\nimport numpy as np \nimport pandas as pd\nimport seaborn as sns\nfrom tqdm import tqdm\nfrom scipy import stats\nfrom pathlib import Path\nfrom itertools import groupby\n\nimport time\nt0start = time.time() \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport pandas as pd\nimport tensorflow as tf\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n#:::::::::::::::::::::::::::::::::::\nimport matplotlib.pyplot as plt\nimport plotly.figure_factory as ff\nimport plotly.express as px\n%matplotlib inline\n!ls ../input/*","metadata":{"_kg_hide-input":true,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2023-11-03T11:00:54.804597Z","iopub.execute_input":"2023-11-03T11:00:54.805237Z","iopub.status.idle":"2023-11-03T11:01:11.856045Z","shell.execute_reply.started":"2023-11-03T11:00:54.805202Z","shell.execute_reply":"2023-11-03T11:01:11.854751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"# prediction = pd.read_csv('/kaggle/input/mlp-mf-onehot-baseline-op-5/submission (35).csv')\n\nimport1 = pd.read_csv('../input/op2-603/op2_603.csv', index_col='id')\nimport1.shape\n\nimport2 = pd.read_csv('../input/op2-720/op2_720.csv', index_col='id')\nimport2.shape\n\nimport3 = pd.read_csv('../input/op2-604/submission_df.csv', index_col='id')\nimport3.shape\n\nimport4 = pd.read_csv('../input/op2-607/OP2_607.csv', index_col='id')\nimport4.shape\n\n# import5 = pd.read_csv('/kaggle/input/cf-sim/CF_submission_Multi_Sim.csv')\n# import5.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-03T11:01:11.858296Z","iopub.execute_input":"2023-11-03T11:01:11.860690Z","iopub.status.idle":"2023-11-03T11:01:36.123915Z","shell.execute_reply.started":"2023-11-03T11:01:11.860635Z","shell.execute_reply":"2023-11-03T11:01:36.122900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_df = [ import1, import2, import3, import4]\nlist_ids = ['op2_603', 'op2_720', 'submission_df',  'OP2_607' ]","metadata":{"execution":{"iopub.status.busy":"2023-11-03T11:01:36.125502Z","iopub.execute_input":"2023-11-03T11:01:36.126061Z","iopub.status.idle":"2023-11-03T11:01:36.130453Z","shell.execute_reply.started":"2023-11-03T11:01:36.126022Z","shell.execute_reply":"2023-11-03T11:01:36.129543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fast compute correlation - use standard scaler\n\nCorrelation of two vector v1, v2 - just mean component-wise product (v1*v2) divided by norms of v1, and v2.\n\nSo if make standard scaling - norms become 1 and we do not need to divide - just do np.mean(v1*v2 )","metadata":{}},{"cell_type":"code","source":"%%time\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\n\nlist_np = []\nfor df in list_df:\n    d = scaler.fit_transform(df)\n    list_np.append(d)","metadata":{"execution":{"iopub.status.busy":"2023-11-03T11:01:36.132714Z","iopub.execute_input":"2023-11-03T11:01:36.133270Z","iopub.status.idle":"2023-11-03T11:01:38.423930Z","shell.execute_reply.started":"2023-11-03T11:01:36.133238Z","shell.execute_reply":"2023-11-03T11:01:38.422524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Example ","metadata":{}},{"cell_type":"code","source":"d0 = list_np[0]\nd1 = list_np[1]\nv = np.mean( d0*d1, axis = 0 )\nplt.hist(v)\nprint( pd.Series(v).describe() )\nc = np.mean( np.abs(v) )\nprint(c)","metadata":{"execution":{"iopub.status.busy":"2023-11-03T11:01:38.425572Z","iopub.execute_input":"2023-11-03T11:01:38.426018Z","iopub.status.idle":"2023-11-03T11:01:38.823438Z","shell.execute_reply.started":"2023-11-03T11:01:38.425982Z","shell.execute_reply":"2023-11-03T11:01:38.822177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Main computation ","metadata":{"execution":{"iopub.status.busy":"2023-11-02T09:44:28.989802Z","iopub.execute_input":"2023-11-02T09:44:28.990202Z","iopub.status.idle":"2023-11-02T09:44:29.342341Z","shell.execute_reply.started":"2023-11-02T09:44:28.990170Z","shell.execute_reply":"2023-11-02T09:44:29.341229Z"}}},{"cell_type":"code","source":"%%time\n\nfig = plt.figure(figsize = (20,10))\ndf_stat = pd.DataFrame()\ndf_corr_averages_for_submits = pd.DataFrame()\nfor i0,d0 in  enumerate(list_np):\n    for i1,d1 in  enumerate(list_np):\n        v = np.mean( d0*d1, axis = 0 )\n        if i0<i1:\n            plt.hist(v, label = str(i0)+ ' vs ' + str(i1))\n        dt = pd.Series(v).describe().to_frame()\n        dt.columns = [str(i0)+ ' vs  ' + str(i1)]\n        df_stat = pd.concat( [df_stat, dt] , axis = 1)\n        #print(  )\n        if 1:\n            c = np.mean( (v) )\n        else:\n            # Versions 1,2 used \"abs\", but later seems to be more reasonable - NO abs - since we blend with positive weights typically\n            c = np.mean( np.abs(v) )\n        print(c)\n        df_corr_averages_for_submits.loc[i0,i1] = c\n        \ndisplay(df_stat)        \nplt.legend(fontsize = 20 )\nplt.grid()\nplt.title('Distribution of correlations (over genes) for each prediction pair ', fontsize = 20)\nplt.show()        \ndf_corr_averages_for_submits.columns = list_ids\ndf_corr_averages_for_submits.index = list_ids\n\ndf_corr_averages_for_submits.round(2)        \n","metadata":{"execution":{"iopub.status.busy":"2023-11-03T11:01:38.824987Z","iopub.execute_input":"2023-11-03T11:01:38.825331Z","iopub.status.idle":"2023-11-03T11:01:39.928812Z","shell.execute_reply.started":"2023-11-03T11:01:38.825301Z","shell.execute_reply":"2023-11-03T11:01:39.927557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final correlation between submits ","metadata":{}},{"cell_type":"code","source":"display( df_corr_averages_for_submits.round(2)         )","metadata":{"execution":{"iopub.status.busy":"2023-11-03T11:01:39.930730Z","iopub.execute_input":"2023-11-03T11:01:39.931717Z","iopub.status.idle":"2023-11-03T11:01:39.946974Z","shell.execute_reply.started":"2023-11-03T11:01:39.931642Z","shell.execute_reply":"2023-11-03T11:01:39.945874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}