{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.13"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"},{"sourceId":12462463,"sourceType":"datasetVersion","datasetId":7861436},{"sourceId":12503273,"sourceType":"datasetVersion","datasetId":7870836},{"sourceId":12512096,"sourceType":"datasetVersion","datasetId":7897419},{"sourceId":12522757,"sourceType":"datasetVersion","datasetId":7904568},{"sourceId":12548879,"sourceType":"datasetVersion","datasetId":7909478}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":374.271766,"end_time":"2025-07-16T09:01:49.809559","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-07-16T08:55:35.537793","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"🔍 What The Code Does\n- Reads multiple CSV submission files - Loads prediction files from specified paths\n- Merges all predictions on ID column - Combines all models into one dataframe\n- Calculates prediction disagreement - Computes abs(max - min) for each row\n- Ranks models by prediction values - Sorts models from highest to lowest predictions\n- Applies conditional weighting - Uses different weight sets based on disagreement ranges\n- Adjusts weights by model ranking - Boosts/reduces weights based on model position in ranking\n- Creates ascending/descending ensembles - Runs blending twice with opposite sorting\n- Combines final prediction - Weighted average of asc/desc results (0.35 desc + 0.65 asc)\n- Saves final submission file - Outputs blended predictions to CSV","metadata":{}},{"cell_type":"code","source":"import pandas as pd","metadata":{"papermill":{"duration":1.879293,"end_time":"2025-07-16T08:55:42.280345","exception":false,"start_time":"2025-07-16T08:55:40.401052","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T03:59:51.483790Z","iopub.execute_input":"2025-07-23T03:59:51.484140Z","iopub.status.idle":"2025-07-23T03:59:51.954052Z","shell.execute_reply.started":"2025-07-23T03:59:51.484109Z","shell.execute_reply":"2025-07-23T03:59:51.953095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\ndef iBlend(path_to_ds, file_short_names, sls):\n\n    def tida(sls):\n        def read_subm(sls,i):  # Reads multiple CSV submission files - Loads prediction files from specified paths \n            tnm = sls[\"subm\"][i][\"name\"]\n            FiN = sls[\"path\"] + tnm + \".csv\"\n            df = pd.read_csv(FiN)\n            return df.rename(columns={'prediction': tnm})  # CHANGED: 'prediction' instead of 'target'\n        \n        # Merges all predictions on ID column - Combines all models into one dataframe\n        dfs_subm = [read_subm(sls,i) for i in range(len(sls[\"subm\"]))]  \n        df_subms = pd.merge(dfs_subm[0],  dfs_subm[1], on=['ID'])\n        for i in range(2, len(sls[\"subm\"])): \n            df_subms = pd.merge(df_subms, dfs_subm[i], on=['ID'])\n            \n        cols = [col for col in df_subms.columns if col != \"ID\"]\n        short_name_cols = [c.replace(sls[\"prefix\"], '') for c in cols]\n        \n        weights1,corrects1  = [subm['weight'] for subm in sls[\"subm\"]], [wt for wt in sls[\"subwts\"] ]\n        weights2,corrects2  = [subm['weight'] for subm in sls[\"subm2\"]],[wt for wt in sls[\"subwts2\"]]\n        weights3,corrects3  = [subm['weight'] for subm in sls[\"subm3\"]],[wt for wt in sls[\"subwts3\"]]\n        weights4,corrects4  = [subm['weight'] for subm in sls[\"subm4\"]],[wt for wt in sls[\"subwts4\"]]\n        weights5,corrects5  = [subm['weight'] for subm in sls[\"subm5\"]],[wt for wt in sls[\"subwts5\"]]\n        \n        # Ranks models by prediction values - Sorts models from highest to lowest predictions\n        def alls(x, cs=cols):\n            tes = {c: x[c] for c in cs}.items()\n            subms_sorted = [\n              t[0].replace(sls[\"prefix\"], '')\n              for t in sorted(tes,key=lambda k:k[1],reverse=True if sls[\"sort\"]=='desc' else False)]\n            return subms_sorted\n        \n        def correct(x, cs=cols, \n                    w1=weights1, cw1=corrects1, \n                    w2=weights2, cw2=corrects2,\n                    w3=weights3, cw3=corrects3,\n                    w4=weights4, cw4=corrects4,\n                    w5=weights5, cw5=corrects5,\n                   ):\n            ic = [x['alls'].index(c) for c in short_name_cols]  #Adjusts weights by model ranking - Boosts/reduces weights based on model position in ranking\n\n            mxm = x['abs(mx-m)']\n\n            # Applies conditional weighting - Uses different weight sets based on disagreement ranges\n            if   0.00 < mxm <= 0.50:\n                cS = [x[cols[j]] * (w1[j] + cw1[ic[j]]) for j in range(len(cols))]\n            elif 0.50 < mxm <= 1.00:\n                cS = [x[cols[j]] * (w2[j] + cw2[ic[j]]) for j in range(len(cols))]\n            elif 1.00 < mxm <= 1.50:\n                cS = [x[cols[j]] * (w3[j] + cw3[ic[j]]) for j in range(len(cols))]\n            elif 1.50 < mxm <= 2.00:\n                cS = [x[cols[j]] * (w4[j] + cw4[ic[j]]) for j in range(len(cols))]\n            else:\n                cS = [x[cols[j]] * (w5[j] + cw5[ic[j]]) for j in range(len(cols))]\n            return sum(cS)\n        \n        # Calculates prediction disagreement - Computes abs(max - min) for each row\n        def amxm(x, cs=cols):\n            list_values = x[cs].to_list()\n            mxm = abs(max(list_values)-min(list_values))\n            return mxm\n\n        df_subms['abs(mx-m)']   = df_subms.apply(lambda x: amxm   (x), axis=1)\n        \n        df_subms['alls']        = df_subms.apply(lambda x: alls   (x), axis=1)\n        df_subms[sls[\"target\"]] = df_subms.apply(lambda x: correct(x), axis=1)\n        \n        schema_rename = { old_nc:new_shnc for old_nc, new_shnc in zip(cols, short_name_cols) }\n        \n        df_subms = df_subms.rename(columns=schema_rename)\n        df_subms = df_subms.rename(columns={sls[\"target\"]:\"ensemble\"})\n        \n        df_subms.insert(loc=1, column=' _ ', value=['   '] * sls[\"q_rows\"])\n        \n        df_subms[' _ '] = df_subms[' _ '].astype(str)\n        pd.set_option('display.max_rows',100)\n        pd.set_option('display.float_format', '{:.4f}'.format)\n        vcols = ['ID'] + [' _ '] + short_name_cols + [' _ '] + ['abs(mx-m)'] + [' _ '] + ['alls'] + [' _ '] + ['ensemble']\n        df_subms = df_subms[vcols]\n        display(df_subms.head(8))\n        pd.set_option('display.float_format', '{:.7f}'.format)\n        df_subms = df_subms.rename(columns={\"ensemble\":sls[\"target\"]})\n        return df_subms\n\n    sample_subm = pd.read_csv(path_to_ds + file_short_names[1] + \".csv\")\n\n    def ensemble_tida(sls,submission=sample_subm):    #Creates ascending/descending ensembles - Runs blending twice with opposite sorting\n        sls['sort'] = 'desc'\n        dfs = tida(sls)\n        dfD = dfs[['ID', sls['target']]]\n        dfD.to_csv(f'tida_desc.csv', index=False)\n        sls['sort'] = 'asc'\n        dfs = tida(sls)\n        dfA = dfs[['ID', sls['target']]]\n        dfA.to_csv(f'tida_asc.csv',  index=False)\n        target,d,a = sls['target'],sls['desc'],sls['asc']\n        submission[target] = dfD[target] * d + a * dfA[target]  #Combines final prediction - Weighted average of asc/desc results (0.35 desc + 0.65 asc)\n        return submission\n\n    submission = ensemble_tida(sls)\n    \n    return submission\n\n# Keep everything exactly as you had it\npath_to_ds ='/kaggle/input/21-juli-2025-drw/submission '\nfile_short_names = ['0.95109','0.95004', '0.95002', '0.94857', '0.90222']\n\nparams_14 = {\n      'path'   : path_to_ds,                                 \n      'sort'   : \"dynamic\",\n      'target' : \"prediction\",\n      'q_rows' : 538_150,\n      'prefix' : \"subm_\",\n      'desc'   : 0.35,\n      'asc'    : 0.65,\n      'subwts' : [+0.015, +0.002, -0.002, -0.005, -0.010],      # LB=?\n      'subm'   : [\n         { 'name':file_short_names[0],'weight':0.84, },\n         { 'name':file_short_names[1],'weight':0.05, },\n         { 'name':file_short_names[2],'weight':0.05, },\n         { 'name':file_short_names[3],'weight':0.05, },\n         { 'name':file_short_names[4],'weight':0.01, },\n      ],\n      'subwts2' : [+0.020, +0.002, -0.002, -0.007, -0.013],     # LB=?\n      'subm2'   : [\n         { 'name':file_short_names[0],'weight':0.83, },\n         { 'name':file_short_names[1],'weight':0.053,},\n         { 'name':file_short_names[2],'weight':0.053,},\n         { 'name':file_short_names[3],'weight':0.054,},\n         { 'name':file_short_names[4],'weight':0.010,},\n      ],\n      'subwts3' : [+0.025, +0.002, -0.002, -0.010, -0.015],     # LB=?\n      'subm3'   : [\n         { 'name':file_short_names[0],'weight':0.82, },\n         { 'name':file_short_names[1],'weight':0.057,},\n         { 'name':file_short_names[2],'weight':0.057,},\n         { 'name':file_short_names[3],'weight':0.057,},\n         { 'name':file_short_names[4],'weight':0.010,},\n      ],\n      'subwts4' : [+0.030, +0.002, -0.002, -0.010, -0.020],     # LB=?    \n      'subm4'   : [\n         { 'name':file_short_names[0],'weight':0.79, },\n         { 'name':file_short_names[1],'weight':0.07, },\n         { 'name':file_short_names[2],'weight':0.07, },\n         { 'name':file_short_names[3],'weight':0.07, },\n         { 'name':file_short_names[4],'weight':0.00, },\n      ],\n      'subwts5' : [+0.035, +0.002, -0.002, -0.012, -0.023],     # LB=?   \n      'subm5'   : [\n         { 'name':file_short_names[0],'weight':0.78, },\n         { 'name':file_short_names[1],'weight':0.07, },\n         { 'name':file_short_names[2],'weight':0.07, },\n         { 'name':file_short_names[3],'weight':0.07, },\n         { 'name':file_short_names[4],'weight':0.01, },\n      ],\n    }\n\nparams = params_14\ndf = iBlend ( path_to_ds, file_short_names, params )\ndf.to_csv('submission.csv', index=False) \ndisplay(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T04:31:19.816011Z","iopub.execute_input":"2025-07-23T04:31:19.816763Z","iopub.status.idle":"2025-07-23T04:36:41.560918Z","shell.execute_reply.started":"2025-07-23T04:31:19.816722Z","shell.execute_reply":"2025-07-23T04:36:41.560221Z"}},"outputs":[],"execution_count":null}]}