{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":39763,"databundleVersionId":11756775,"sourceType":"competition"},{"sourceId":12344635,"sourceType":"datasetVersion","datasetId":7782246},{"sourceId":12344474,"sourceType":"datasetVersion","datasetId":7782151}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Reference:\nhttps://www.kaggle.com/code/atom1231/ensemble-csv-for-more-files","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\ndef ensemble_submissions(file_paths, output_dir='.', weights=None):\n    \"\"\"\n    將多個 submission CSV 檔案進行整合，可選擇計算加權平均。\n\n    Args:\n        file_paths (list): 一個包含所有 submission.csv 檔案路徑的列表。\n        output_dir (str): 輸出結果檔案的儲存目錄。\n        weights (list, optional): 一個與 file_paths 對應的權重列表。\n                                  若為 None，則計算簡單平均。預設為 None。\n    \"\"\"\n    if not file_paths:\n        print(\"錯誤：檔案路徑列表是空的。\")\n        return\n\n    # --- 新增：權重驗證 ---\n    if weights is not None:\n        if len(weights) != len(file_paths):\n            print(f\"錯誤：權重列表的長度 ({len(weights)}) 與檔案列表的長度 ({len(file_paths)}) 不符。\")\n            return\n        print(f\"開始進行加權整合，使用的權重為: {weights}\")\n    else:\n        print(f\"開始進行簡單平均整合（未提供權重）。\")\n\n    # --- 步驟 1: 讀取所有 CSV 檔案，並將 'oid_ypos' 設為索引 ---\n    try:\n        print(\"正在讀取 CSV 檔案...\")\n        dataframes = [pd.read_csv(f, index_col='oid_ypos') for f in file_paths]\n    except FileNotFoundError as e:\n        print(f\"錯誤：找不到檔案 -> {e}。請檢查您的檔案路徑是否正確。\")\n        return\n    except KeyError:\n        print(\"錯誤：某個檔案中找不到 'oid_ypos' 欄位。請確保所有 CSV 都有此欄位。\")\n        return\n\n    # --- 步驟 2: 資料對齊與堆疊 ---\n    print(\"正在對齊資料...\")\n    master_index = dataframes[0].index\n    master_columns = dataframes[0].columns\n    aligned_data = [df.reindex(master_index).values for df in dataframes]\n    data_array = np.stack(aligned_data, axis=0).transpose(1, 2, 0)\n    print(\"資料堆疊完成。 陣列維度 (列數, 欄數, 檔案數):\", data_array.shape)\n\n    # --- 步驟 3: 計算平均值 (加權或簡單) 與中位數 ---\n    \n    # 計算平均值\n    if weights is not None:\n        print(\"正在計算加權平均值...\")\n        # 使用 np.average 進行加權計算\n        mean_data = np.average(data_array, axis=2, weights=weights)\n        mean_output_filename = 'ensemble_weighted_mean.csv'\n    else:\n        print(\"正在計算簡單平均值...\")\n        mean_data = np.mean(data_array, axis=2)\n        mean_output_filename = 'ensemble_mean.csv'\n        \n    # 計算中位數 (中位數沒有加權的概念，維持原樣)\n    print(\"正在計算中位數...\")\n    median_data = np.median(data_array, axis=2)\n\n    # --- 步驟 4: 建立結果 DataFrame 並儲存為 CSV 檔案 ---\n    # 平均值結果\n    df_mean = pd.DataFrame(mean_data, index=master_index, columns=master_columns)\n    mean_output_path = os.path.join(output_dir, 'submission.csv')\n    df_mean.to_csv(mean_output_path)\n    print(f\"平均值整合結果已儲存至: {mean_output_path}\")\n\n    # # 中位數結果\n    # df_median = pd.DataFrame(median_data, index=master_index, columns=master_columns)\n\n\nsubmission_files = [\n    \"/kaggle/input/yale-fwi-best-submission-csv-file/CV07286_TRY8V16_TRY8V22ensTRY8V21_Top4Methods.csv\",\n    \"/kaggle/input/best-yale-submission/submission.csv\",\n]\n\n\n#file_weights = [0.5, 0.5]\nfile_weights = [0.4, 0.6]\n\nensemble_submissions(submission_files, weights=file_weights)\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}