{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"},{"sourceId":12503854,"sourceType":"datasetVersion","datasetId":7891541}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Load the datasets\ncsv1 = pd.read_csv('/kaggle/input/datass/submission_prophet_enhanced.csv')\ncsv2 = pd.read_csv('/kaggle/input/datass/submission.csv')\ncsv3 = pd.read_csv('/kaggle/input/datass/submission_prophet_enhanced (1).csv')\ncsv4 = pd.read_csv('/kaggle/input/datass/submission_prophet_enhanced (2).csv')\n\n# Extract the first column (IDs) and headers\nid_column = csv1.iloc[:, 0]  # The ID column\nheaders = csv1.columns       # The headers\n\n# Extract the categorical parts (excluding the first column)\ncategorical1 = csv1.iloc[:, 1:]\ncategorical2 = csv2.iloc[:, 1:]\ncategorical3 = csv3.iloc[:, 1:]\ncategorical4 = csv4.iloc[:, 1:]\n\n# Initialize an empty DataFrame for results\nweighted_results = pd.DataFrame()\n\n# Weights\nweight_csv1 = 6  # 60%\nweight_others = 4  # remaining 40% across csv2, csv3, csv4 (each ~13.33%)\n\nfor col in categorical1.columns:\n    # Repeat csv1 predictions 6 times (60% weight)\n    weighted_preds = [categorical1[col]] * weight_csv1\n    # Repeat other predictions once each (~13.3% each)\n    weighted_preds += [categorical2[col], categorical3[col], categorical4[col]]\n    \n    # Concatenate predictions\n    combined = pd.concat(weighted_preds, axis=1)\n    # Take the mode per row\n    weighted_results[col] = combined.mode(axis=1)[0]\n\n# Reinsert the ID column\nweighted_results.insert(0, headers[0], id_column)\n\n# Save the result\nweighted_results.to_csv('submission.csv', index=False)\n\n# Compare with csv1\ndifferences = csv1.compare(weighted_results)\nprint(differences)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T04:44:34.081223Z","iopub.execute_input":"2025-07-18T04:44:34.081555Z","iopub.status.idle":"2025-07-18T04:45:49.361288Z","shell.execute_reply.started":"2025-07-18T04:44:34.081531Z","shell.execute_reply":"2025-07-18T04:45:49.360386Z"}},"outputs":[],"execution_count":null}]}