{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"},{"sourceId":12570279,"sourceType":"datasetVersion","datasetId":7937103}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-24T20:02:23.859980Z","iopub.execute_input":"2025-07-24T20:02:23.860325Z","iopub.status.idle":"2025-07-24T20:02:23.864610Z","shell.execute_reply.started":"2025-07-24T20:02:23.860298Z","shell.execute_reply":"2025-07-24T20:02:23.863735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def iBlend(path_to_ds, file_short_names, sls):\n\n    def tida(sls):\n        \n        def read_subm(sls,i):\n            tnm = sls[\"subm\"][i][\"name\"]\n            FiN = sls[\"path\"] + tnm + \".csv\"\n            return pd.read_csv(FiN).rename(columns={'target':tnm, sls[\"target\"]:tnm})\n        \n        dfs_subm = [read_subm(sls,i) for i in range(len(sls[\"subm\"]))]\n        df_subms = pd.merge(dfs_subm[0],  dfs_subm[1], on=['ID'])\n        \n        for i in range(2, len(sls[\"subm\"])): \n            df_subms = pd.merge(df_subms, dfs_subm[i], on=['ID'])\n            \n        cols = [col for col in df_subms.columns if col != \"ID\"]\n        short_name_cols = [c.replace(sls[\"prefix\"], '') for c in cols]\n        corrects = [wt for wt in sls[\"subwts\"]]\n        weights = [subm['weight'] for subm in sls[\"subm\"]]\n        \n        def alls(x, cs=cols):\n            tes = {c: x[c] for c in cs}.items()\n            subms_sorted = [\n              t[0].replace(sls[\"prefix\"], '')\n              for t in sorted(tes,key=lambda k:k[1],reverse=True if sls[\"sort\"]=='desc' else False)]\n            return subms_sorted\n        \n        def correct(x, cs=cols, w=weights, cw=corrects):\n            ic = [x['alls'].index(c) for c in short_name_cols]\n            cS = [x[cols[j]] * (w[j] + cw[ic[j]]) for j in range(len(cols))]\n            return sum(cS)\n        \n        df_subms['alls']        = df_subms.apply(lambda x: alls   (x), axis=1)\n        df_subms[sls[\"target\"]] = df_subms.apply(lambda x: correct(x), axis=1)\n        \n        schema_rename = { old_nc:new_shnc for old_nc, new_shnc in zip(cols, short_name_cols) }\n        \n        df_subms = df_subms.rename(columns=schema_rename)\n        df_subms = df_subms.rename(columns={sls[\"target\"]:\"ensemble\"})\n        \n        df_subms.insert(loc=1, column=' _ ', value=['   '] * sls[\"q_rows\"])\n        \n        df_subms[' _ '] = df_subms[' _ '].astype(str)\n        pd.set_option('display.max_rows',100)\n        pd.set_option('display.float_format', '{:.4f}'.format)\n        vcols = ['ID'] + [' _ '] + short_name_cols + [' _ '] + ['alls'] + [' _ '] + ['ensemble']\n        df_subms = df_subms[vcols]\n        display(df_subms.head(7))\n        pd.set_option('display.float_format', '{:.7f}'.format)\n        df_subms = df_subms.rename(columns={\"ensemble\":sls[\"target\"]})\n        \n        return df_subms\n        \n\n    sample_subm = pd.read_csv(path_to_ds + file_short_names[1] + \".csv\")\n\n    \n    def ensemble_tida(sls,submission=sample_subm):   \n        sls['sort'] = 'desc'\n        dfs = tida(sls)\n        dfD = dfs[['ID', sls['target']]]\n        dfD.to_csv(f'tida_desc.csv', index=False)\n        sls['sort'] = 'asc'\n        dfs = tida(sls)\n        dfA = dfs[['ID', sls['target']]]\n        dfA.to_csv(f'tida_asc.csv',  index=False)\n        target,d,a = sls['target'],sls['desc'],sls['asc']\n        submission[target] = dfD[target] * d + a * dfA[target]\n        return submission\n\n    submission = ensemble_tida(sls)\n    \n    return submission\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T20:02:30.048116Z","iopub.execute_input":"2025-07-24T20:02:30.048440Z","iopub.status.idle":"2025-07-24T20:02:30.063292Z","shell.execute_reply.started":"2025-07-24T20:02:30.048417Z","shell.execute_reply":"2025-07-24T20:02:30.062524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path_to_ds ='/kaggle/input/24-juli-2025-drw/submission '\n\nfile_short_names = ['0.95167','0.95166','0.95165','0.95164','0.95163']\nparams = {\n      'path'   : path_to_ds,                                 \n      'sort'   : \"asc/desc\",\n      'target' : \"prediction\",\n      'q_rows' : 538_150,\n      'prefix' : \"subm_\",\n      'desc'   : 0.700003,\n      'asc'    : 0.300007,\n    \n      'subwts' : [+0.000004, +0.000002, -0.000001, -0.000002, -0.000003], # 0.95167\n      'subm'   : [\n         { 'name':file_short_names[0],'weight':0.99995, },\n         { 'name':file_short_names[1],'weight':0.00002, },\n         { 'name':file_short_names[2],'weight':0.00001, },\n         { 'name':file_short_names[3],'weight':0.00001, },\n         { 'name':file_short_names[4],'weight':0.00001, },\n      ],\n    }\n\n\nfile_short_names = ['0.95167','0.95166','0.95165','0.95164','0.95163']\nparams = {\n      'path'   : path_to_ds,                                 \n      'sort'   : \"asc/desc\",\n      'target' : \"prediction\",\n      'q_rows' : 538_150,\n      'prefix' : \"subm_\",\n      'desc'   : 0.20,\n      'asc'    : 0.50,\n    \n      'subwts' : [+0.01, +0.02, +0.03, +0.04, +0.05],                    # 0.92359\n      'subm'   : [\n         { 'name':file_short_names[0],'weight':0.20, },\n         { 'name':file_short_names[1],'weight':0.20, },\n         { 'name':file_short_names[2],'weight':0.20, },\n         { 'name':file_short_names[3],'weight':0.20, },\n         { 'name':file_short_names[4],'weight':0.20, },\n      ],\n    }\n\n\nfile_short_names = ['0.95167','0.95166','0.95165','0.95164','0.95163']\nparams = {\n      'path'   : path_to_ds,                                 \n      'sort'   : \"asc/desc\",\n      'target' : \"prediction\",\n      'q_rows' : 538_150,\n      'prefix' : \"subm_\",\n      'desc'   : 0.50,\n      'asc'    : 0.50,\n    \n      'subwts' : [+0.01, +0.02, +0.03, +0.04, +0.05],                    # 0.92358\n      'subm'   : [\n         { 'name':file_short_names[0],'weight':0.20, },\n         { 'name':file_short_names[1],'weight':0.20, },\n         { 'name':file_short_names[2],'weight':0.20, },\n         { 'name':file_short_names[3],'weight':0.20, },\n         { 'name':file_short_names[4],'weight':0.20, },\n      ],\n    }\n\nfile_short_names = ['0.95167','0.95166','0.95165','0.95164','0.95163']\nparams = {\n      'path'   : path_to_ds,                                 \n      'sort'   : \"asc/desc\",\n      'target' : \"prediction\",\n      'q_rows' : 538_150,\n      'prefix' : \"subm_\",\n      'desc'   : 0.50,\n      'asc'    : 0.50,\n    \n      'subwts' : [-0.01, -0.02, -0.03, -0.04, -0.05],                    # 0.92358\n      'subm'   : [\n         { 'name':file_short_names[0],'weight':0.20, },\n         { 'name':file_short_names[1],'weight':0.20, },\n         { 'name':file_short_names[2],'weight':0.20, },\n         { 'name':file_short_names[3],'weight':0.20, },\n         { 'name':file_short_names[4],'weight':0.20, },\n      ],\n    }\n\nfile_short_names = ['0.90006','0.95109','0.95129','0.95164','0.95167']\nparams = {\n      'path'   : path_to_ds,                                 \n      'sort'   : \"asc/desc\",\n      'target' : \"prediction\",\n      'q_rows' : 538_150,\n      'prefix' : \"subm_\",\n      'desc'   : 0.30,\n      'asc'    : 0.70,\n    \n      'subwts' : [+0.03, +0.01, 0.0, -0.01, -0.03], \n      'subm'   : [\n         { 'name':file_short_names[0],'weight':0.01, },\n         { 'name':file_short_names[1],'weight':0.02, },\n         { 'name':file_short_names[2],'weight':0.03, },\n         { 'name':file_short_names[3],'weight':0.04, },\n         { 'name':file_short_names[4],'weight':0.90, },\n      ],\n    }\n\nfile_short_names = ['0.90006','0.95129','0.95164','0.95167'] #'0.95109',\nparams = {\n      'path'   : path_to_ds,                                 \n      'sort'   : \"asc/desc\",\n      'target' : \"prediction\",\n      'q_rows' : 538_150,\n      'prefix' : \"subm_\",\n      'desc'   : 0.29,\n      'asc'    : 0.71,\n    \n      'subwts' : [+0.03, -0.005, 0.010, -0.015], #, -0.03], \n      'subm'   : [\n         { 'name':file_short_names[0],'weight':0.01 },\n         { 'name':file_short_names[1],'weight':0.02 },\n         { 'name':file_short_names[2],'weight':0.03 },\n         { 'name':file_short_names[3],'weight':0.94 },\n         #{ 'name':file_short_names[4],'weight':0.880, },\n      ],\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T00:13:05.564073Z","iopub.execute_input":"2025-07-27T00:13:05.565033Z","iopub.status.idle":"2025-07-27T00:13:05.579871Z","shell.execute_reply.started":"2025-07-27T00:13:05.565002Z","shell.execute_reply":"2025-07-27T00:13:05.578782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = iBlend ( path_to_ds, file_short_names, params )\n\ndf.to_csv('submission.csv', index=False)\n\ndisplay(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T20:03:03.264374Z","iopub.execute_input":"2025-07-24T20:03:03.264749Z","iopub.status.idle":"2025-07-24T20:03:50.677268Z","shell.execute_reply.started":"2025-07-24T20:03:03.264722Z","shell.execute_reply":"2025-07-24T20:03:50.676477Z"}},"outputs":[],"execution_count":null}]}