{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"},{"sourceId":12462463,"sourceType":"datasetVersion","datasetId":7861436},{"sourceId":12492376,"sourceType":"datasetVersion","datasetId":7883469},{"sourceId":12503273,"sourceType":"datasetVersion","datasetId":7870836},{"sourceId":12558997,"sourceType":"datasetVersion","datasetId":7909478}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Part 1.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2025-07-24T05:07:52.712124Z","iopub.execute_input":"2025-07-24T05:07:52.712470Z","iopub.status.idle":"2025-07-24T05:13:37.916815Z","shell.execute_reply.started":"2025-07-24T05:07:52.712445Z","shell.execute_reply":"2025-07-24T05:13:37.915989Z"}}},{"cell_type":"code","source":"import pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T08:12:50.544425Z","iopub.execute_input":"2025-07-24T08:12:50.544788Z","iopub.status.idle":"2025-07-24T08:12:50.550250Z","shell.execute_reply.started":"2025-07-24T08:12:50.544728Z","shell.execute_reply":"2025-07-24T08:12:50.549186Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"You bring in pandas, the core library for reading CSVs, merging DataFrames, computing new columns, and overall table-based data work. Almost every step of this ensemble relies on pandas’ DataFrame APIs.","metadata":{}},{"cell_type":"markdown","source":"# Part 2.","metadata":{}},{"cell_type":"code","source":"# Where your individual model outputs live\npath_to_ds = '/kaggle/input/21-juli-2025-drw/submission '\n\n# The basenames of each CSV (without \".csv\" or folder path)\nfile_short_names = [\n    '0.95109',\n    '0.95004',\n    '0.95002',\n    '0.94857',\n    '0.90222'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T08:12:50.552213Z","iopub.execute_input":"2025-07-24T08:12:50.552636Z","iopub.status.idle":"2025-07-24T08:12:50.585821Z","shell.execute_reply.started":"2025-07-24T08:12:50.552598Z","shell.execute_reply":"2025-07-24T08:12:50.583992Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* path_to_ds points to the folder containing your five model-output files (e.g. “0.95109.csv”, etc.).\n\n* file_short_names lists those filenames (minus extension), so later code can loop over them.","metadata":{}},{"cell_type":"markdown","source":"# Part 3.","metadata":{}},{"cell_type":"code","source":"# Ensemble configuration: blend ratios, static weights, and per-tier corrections\nparams = {\n    'path':   path_to_ds,\n    'sort':   \"dynamic\",     # placeholder; will be set to 'desc' or 'asc'\n    'target': \"prediction\",  # name of the final output column\n    'q_rows': 538_150,       # number of test rows (for display spacing)\n\n    'prefix': \"subm_\",       # prefix your code adds when renaming columns\n\n    # Final blend weights for the two passes:\n    'desc': 0.32,  # weight for the descending-sort pass\n    'asc':  0.68,  # weight for the ascending-sort pass\n\n    # Tier 1: if max–min spread in [0.00, 0.50]\n    'subwts':   [+0.015, +0.002, -0.002, -0.005, -0.010],\n    'subm': [\n        {'name': file_short_names[0], 'weight': 0.86},\n        {'name': file_short_names[1], 'weight': 0.04},\n        {'name': file_short_names[2], 'weight': 0.04},\n        {'name': file_short_names[3], 'weight': 0.05},\n        {'name': file_short_names[4], 'weight': 0.01},\n    ],\n\n    # Tier 2: if spread in (0.50, 1.00]\n    'subwts2':  [+0.020, +0.002, -0.002, -0.007, -0.013],\n    'subm2': [\n        {'name': file_short_names[0], 'weight': 0.83},\n        {'name': file_short_names[1], 'weight': 0.053},\n        {'name': file_short_names[2], 'weight': 0.053},\n        {'name': file_short_names[3], 'weight': 0.054},\n        {'name': file_short_names[4], 'weight': 0.010},\n    ],\n\n    # Tier 3: if spread in (1.00, 1.50]\n    'subwts3':  [+0.025, +0.002, -0.002, -0.010, -0.015],\n    'subm3': [\n        {'name': file_short_names[0], 'weight': 0.82},\n        {'name': file_short_names[1], 'weight': 0.057},\n        {'name': file_short_names[2], 'weight': 0.057},\n        {'name': file_short_names[3], 'weight': 0.057},\n        {'name': file_short_names[4], 'weight': 0.010},\n    ],\n\n    # Tier 4: if spread in (1.50, 2.00]\n    'subwts4':  [+0.030, +0.002, -0.002, -0.010, -0.020],\n    'subm4': [\n        {'name': file_short_names[0], 'weight': 0.79},\n        {'name': file_short_names[1], 'weight': 0.07},\n        {'name': file_short_names[2], 'weight': 0.07},\n        {'name': file_short_names[3], 'weight': 0.07},\n        {'name': file_short_names[4], 'weight': 0.00},\n    ],\n\n    # Tier 5: if spread > 2.00\n    'subwts5':  [+0.035, +0.002, -0.002, -0.012, -0.023],\n    'subm5': [\n        {'name': file_short_names[0], 'weight': 0.78},\n        {'name': file_short_names[1], 'weight': 0.07},\n        {'name': file_short_names[2], 'weight': 0.07},\n        {'name': file_short_names[3], 'weight': 0.07},\n        {'name': file_short_names[4], 'weight': 0.01},\n    ],\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T08:12:50.588616Z","iopub.execute_input":"2025-07-24T08:12:50.589001Z","iopub.status.idle":"2025-07-24T08:12:50.620390Z","shell.execute_reply.started":"2025-07-24T08:12:50.588972Z","shell.execute_reply":"2025-07-24T08:12:50.619260Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This dictionary encodes all of your ensemble logic:\n\n* desc / asc: how to blend the two sorting-based passes.\n\n* Five tiers (subm/subwts through subm5/subwts5):\n\n* Static weights (submN) reflecting each model’s baseline contribution.\n\n* Correction vectors (subwtsN) that tweak those weights based on a model’s rank within that row.\n","metadata":{}},{"cell_type":"markdown","source":"# Part 4.","metadata":{}},{"cell_type":"code","source":"def iBlend(path_to_ds, file_short_names, sls):\n    import pandas as pd\n\n    def tida(sls):\n        # 1) Read & rename individual submissions\n        def read_subm(sls, i):\n            fn = sls[\"path\"] + sls[\"subm\"][i][\"name\"] + \".csv\"\n            return pd.read_csv(fn).rename(\n                columns={'target': sls[\"subm\"][i][\"name\"],\n                         sls[\"target\"]: sls[\"subm\"][i][\"name\"]}\n            )\n\n        # Merge on ID\n        dfs = [read_subm(sls, i) for i in range(len(sls[\"subm\"]))]\n        df = dfs[0].merge(dfs[1], on=\"ID\")\n        for d in dfs[2:]:\n            df = df.merge(d, on=\"ID\")\n\n        # 2) Prepare column lists\n        cols = [c for c in df if c != \"ID\"]\n        short_cols = [c.replace(sls[\"prefix\"], \"\") for c in cols]\n\n        # 3) Extract static weights & corrections for each tier\n        weights = [\n            [m[\"weight\"] for m in sls[k]]\n            for k in (\"subm\", \"subm2\", \"subm3\", \"subm4\", \"subm5\")\n        ]\n        corrections = [\n            sls[k] for k in (\"subwts\", \"subwts2\", \"subwts3\", \"subwts4\", \"subwts5\")\n        ]\n\n        # 4) Helpers: spread, ranking, weighted sum\n        def spread(x): return abs(x[cols].max() - x[cols].min())\n\n        def ranking(x):\n            items = x[cols].items()\n            rev = (sls[\"sort\"] == \"desc\")\n            sorted_names = [n for n, _ in sorted(items, key=lambda p: p[1], reverse=rev)]\n            return [n.replace(sls[\"prefix\"], \"\") for n in sorted_names]\n\n        def apply_weights(x):\n            sp = x[\"spread\"]\n            tier = min(int(sp // 0.5), 4)  # 0→tier1, 1→tier2, …, 4→tier5\n            ws, cs = weights[tier], corrections[tier]\n            ranks = x[\"ranks\"]\n            return sum(\n                x[cols[j]] * (ws[j] + cs[ranks[j]])\n                for j in range(len(cols))\n            )\n\n        # 5) Compute\n        df[\"spread\"] = df.apply(spread, axis=1)\n        df[\"ranks\"]  = df.apply(ranking, axis=1).apply(\n            lambda lst: [lst.index(c) for c in short_cols]\n        )\n        df[sls[\"target\"]] = df.apply(apply_weights, axis=1)\n\n        # 6) Return only ID + final prediction\n        return df[[\"ID\", sls[\"target\"]]]\n\n    # 7) Two‐pass blend: desc then asc\n    sample = pd.read_csv(path_to_ds + file_short_names[1] + \".csv\")\n    def ensemble_tida(sls):\n        sls[\"sort\"] = \"desc\"\n        d1 = tida(sls)\n        d1.to_csv(\"tida_desc.csv\", index=False)\n\n        sls[\"sort\"] = \"asc\"\n        d2 = tida(sls)\n        d2.to_csv(\"tida_asc.csv\", index=False)\n\n        sample[sls[\"target\"]] = d1[sls[\"target\"]] * sls[\"desc\"] + d2[sls[\"target\"]] * sls[\"asc\"]\n        return sample\n\n    return ensemble_tida(sls)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T08:12:50.621465Z","iopub.execute_input":"2025-07-24T08:12:50.621799Z","iopub.status.idle":"2025-07-24T08:12:50.654903Z","shell.execute_reply.started":"2025-07-24T08:12:50.621768Z","shell.execute_reply":"2025-07-24T08:12:50.653571Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This single function bundles everything:\n\n1. tida reads & merges all your individual CSVs.\n\n2. It computes each row’s spread (max–min) and a ranking of models.\n\n3. It picks one of five (weights, corrections) tiers based on that spread.\n\n4. It builds a weighted sum for the ensemble prediction.\n\n5. ensemble_tida runs tida twice—with sort='desc' and sort='asc'—and linearly blends the results by your final desc/asc ratios.","metadata":{}},{"cell_type":"markdown","source":"# Part 5.","metadata":{}},{"cell_type":"code","source":"# Run the ensemble and save your final submission.csv\ndf = iBlend(path_to_ds, file_short_names, params)\ndf.to_csv('submission1.csv', index=False)\ndisplay(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T08:12:50.753117Z","iopub.execute_input":"2025-07-24T08:12:50.753496Z","iopub.status.idle":"2025-07-24T08:30:12.747388Z","shell.execute_reply.started":"2025-07-24T08:12:50.753464Z","shell.execute_reply":"2025-07-24T08:30:12.746385Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Calls your iBlend pipeline with the folder path, filename stems, and parameter dict.\n\n* Writes out submission.csv in the correct kaggle format (ID,prediction).\n\n* Displays the head of the DataFrame so you can visually confirm everything aligned as expected.","metadata":{}},{"cell_type":"markdown","source":"# Unsupervised PCA meta‐ensemble","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\n\n# 1) List the directories holding your submission CSVs\ndirs = [\n    '/kaggle/input/convergence-drw',\n    '/kaggle/input/13-juli-2025-drw',\n    '/kaggle/input/15-juli-2025-drw',\n    '/kaggle/input/21-juli-2025-drw'\n]\n\n# 2) Gather all .csv files from those directories\nsubmission_files = []\nfor d in dirs:\n    for fname in os.listdir(d):\n        if fname.endswith('.csv'):\n            submission_files.append(os.path.join(d, fname))\nsubmission_files = sorted(submission_files)  # sort for reproducibility\n\n# 3) Load the first submission to get 'ID' and a reference scale\nfirst_df   = pd.read_csv(submission_files[0])\ndf         = pd.DataFrame({'ID': first_df['ID']})\nfirst_name = os.path.splitext(os.path.basename(submission_files[0]))[0]\ndf[first_name] = first_df['prediction']\n\n# 4) Load each of the other submissions into its own column\nfor path in submission_files[1:]:\n    tmp = pd.read_csv(path)\n    col = os.path.splitext(os.path.basename(path))[0]\n    df[col] = tmp['prediction'].values\n\n# 5) Build the prediction matrix (rows × n_submissions)\npred_mat = df.drop(columns='ID').values\n\n# 6) Standardize and extract the first principal component\nscaler = StandardScaler()\npred_std = scaler.fit_transform(pred_mat)\npca     = PCA(n_components=1, random_state=42)\npc1     = pca.fit_transform(pred_std).ravel()\n\n# 7) Rescale PC1 to the min/max of the first submission\norig = df[first_name]\npc1_scaled = (pc1 - pc1.min()) / (pc1.max() - pc1.min()) \\\n             * (orig.max() - orig.min()) + orig.min()\n\n# 8) Use PC1 as your final prediction and save\ndf['prediction'] = pc1_scaled\nsubmission = df[['ID','prediction']]\nsubmission.to_csv('submission.csv', index=False)\ndisplay(submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T08:30:12.748772Z","iopub.execute_input":"2025-07-24T08:30:12.749075Z","iopub.status.idle":"2025-07-24T08:30:16.065725Z","shell.execute_reply.started":"2025-07-24T08:30:12.749047Z","shell.execute_reply":"2025-07-24T08:30:16.064551Z"}},"outputs":[],"execution_count":null}]}