{"cells":[{"metadata":{},"cell_type":"markdown","source":"This comp's metric is different from others and must be well understood in order to design good bagging schemes. One of it's characteristics is that it **does not depend** on the absolute values of the predicted scores. Only their order is relevant. Hence, to bag serveal models, we just need to sort the scores and use their ranks.\n\nIn this work, I'll be proposing an order based bagging scheme that seems to perform very well. You can try it yourself !"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd, numpy as np\nimport os","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"paths = [\n    \"../input/rfcx-best-performing-public-kernels/kkiller_inference-tpu-rfcx-audio-detection-fast_0861.csv\",\n    \"../input/rfcx-best-performing-public-kernels/submission_khoongweihao_0845.csv\",\n#     \"../input/rfcx-best-performing-public-kernels/submission_mekhdigakhramanian_0824.csv\",\n]\n\nweights = np.array([0.6, 0.4])\nsum(weights)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv(paths[0]).sort_values(\"recording_id\").reset_index(drop=True)\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cols = [f\"s{i}\" for i in range(24)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"scores = []\nfor path in paths:\n    df = pd.read_csv(path).sort_values(\"recording_id\").reset_index(drop=True)\n    score = np.empty((len(df), 24))\n    o = df[cols].values.argsort(1)\n    score[np.arange(len(df))[:, None], o] = np.arange(24)[None]\n    scores.append(score)\nscores = np.stack(scores)\nscores.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_score = np.sum(scores*weights[:, None, None], 0)\nprint(sub_score.shape)\nsub_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = pd.DataFrame(sub_score, columns=cols)\nsub[\"recording_id\"] = df[\"recording_id\"]\nsub = sub[[\"recording_id\"] + cols]\nprint(sub.shape)\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}