{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"},{"sourceId":12344637,"sourceType":"datasetVersion","datasetId":7782248},{"sourceId":12344641,"sourceType":"datasetVersion","datasetId":7782251},{"sourceId":12344845,"sourceType":"datasetVersion","datasetId":7782399},{"sourceId":12345095,"sourceType":"datasetVersion","datasetId":7782540},{"sourceId":12345105,"sourceType":"datasetVersion","datasetId":7782547},{"sourceId":12345113,"sourceType":"datasetVersion","datasetId":7782554}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Nothing special in this notebook (yet). I've been exploring strategies offline and have found a promising path of ensembling the widely used XGB framework with a 50-120 variable MLP NN architecture with high drop out and noise injection.**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# File paths\nfile1_path = '/kaggle/input/drw-submission-test/submission_ensemble.csv'\nfile2_path = '/kaggle/input/offline-drw-nn/submission_c9f4be847068_0.447745.csv'\nfile3_path = '/kaggle/input/greedy21/submission_54ab3041cda338de.csv'\nfile4_path = '/kaggle/input/105-feature-node/submission_node.csv'\nfile5_path = '/kaggle/input/105mlp/submission (1).csv'\nfile6_path = '/kaggle/input/mldl-nn-arch-search/submission_ensemble(1).csv'\n\n# Weights (must sum to 1.0)\nweight1 = 0.89  # Weight for submission_ensemble.csv\nweight2 = 0.01  # Weight for submission_c9f4be847068_0.447745.csv\nweight3 = 0.07  # Weight for submission_54ab3041cda338de.csv\nweight4 = 0.01  # Weight for submission_node.csv\nweight5 = 0.01  # Weight for submission (1).csv\nweight6 = 0.01  # Weight for submission_ensemble(1).csv\n\n# Verify weights sum to 1\nprint(f\"Sum of weights: {weight1 + weight2 + weight3 + weight4 + weight5 + weight6}\")\n\n# Load the CSV files\ndf1 = pd.read_csv(file1_path)\ndf2 = pd.read_csv(file2_path)\ndf3 = pd.read_csv(file3_path)\ndf4 = pd.read_csv(file4_path)\ndf5 = pd.read_csv(file5_path)\ndf6 = pd.read_csv(file6_path)\n\n# Check if all dataframes have the same structure\nprint(f\"Shape of file 1: {df1.shape}\")\nprint(f\"Shape of file 2: {df2.shape}\")\nprint(f\"Shape of file 3: {df3.shape}\")\nprint(f\"Shape of file 4: {df4.shape}\")\nprint(f\"Shape of file 5: {df5.shape}\")\nprint(f\"Shape of file 6: {df6.shape}\")\nprint(f\"Columns in file 1: {df1.columns.tolist()}\")\nprint(f\"Columns in file 2: {df2.columns.tolist()}\")\nprint(f\"Columns in file 3: {df3.columns.tolist()}\")\nprint(f\"Columns in file 4: {df4.columns.tolist()}\")\nprint(f\"Columns in file 5: {df5.columns.tolist()}\")\nprint(f\"Columns in file 6: {df6.columns.tolist()}\")\n\n# Assuming all files have the same structure with an ID column and prediction columns\n# Identify the ID column (usually the first column)\nid_col = df1.columns[0]\n\n# Identify numeric columns to average (all columns except the ID column)\nnumeric_cols = [col for col in df1.columns if col != id_col]\n\n# Create a copy of the first dataframe to store results\nresult_df = df1.copy()\n\n# Apply weighted average to numeric columns\nfor col in numeric_cols:\n    result_df[col] = (df1[col] * weight1 + \n                      df2[col] * weight2 + \n                      df3[col] * weight3 + \n                      df4[col] * weight4 + \n                      df5[col] * weight5 + \n                      df6[col] * weight6)\n\n# Save the result\noutput_path = 'weighted_average_submission.csv'\nresult_df.to_csv(output_path, index=False)\nprint(f\"\\nWeighted average saved to: {output_path}\")\n\n# Display first few rows of the result\nprint(\"\\nFirst 5 rows of the weighted average:\")\nprint(result_df.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}