{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"I used @radek1's approach for voting ensemble: https://www.kaggle.com/code/radek1/2-methods-how-to-ensemble-predictions","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"# Loading the data","metadata":{}},{"cell_type":"code","source":"!pip install polars","metadata":{"execution":{"iopub.status.busy":"2023-01-08T13:27:06.643467Z","iopub.execute_input":"2023-01-08T13:27:06.645347Z","iopub.status.idle":"2023-01-08T13:27:20.755121Z","shell.execute_reply.started":"2023-01-08T13:27:06.645266Z","shell.execute_reply":"2023-01-08T13:27:20.753479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl\npaths_raw = ['/kaggle/input/candidate-rerank-model-lb-0-575/submission.csv', # 0.575\n         '/kaggle/input/otto-pipeline2-lb-0-576/submission.csv', # 0.576\n         '/kaggle/input/otto-tuning-candidate-rerank-model-lb-0-577/submission.csv', # 0.577\n         '/kaggle/input/otto-fast-cpu-end-to-end-pipeline/submission.csv',\n         '/kaggle/input/otto-recommender/submission.csv',\n         '/kaggle/input/candidate-rerank-model-lb-0-575/submission.csv',\n         '/kaggle/input/otto-pipeline2-lb-0-576/submission.csv',         \n         '/kaggle/input/otto-fast-handcrafted-model-recall-20/submission.csv',         \n         '/kaggle/input/otto-multi-objective-recommender-system/submission.csv',         \n         '/kaggle/input/faster-with-co-visitation/submission.csv',         \n         '/kaggle/input/duplicate-fix-otto-tuning-pipeline2-lb-0-577/submission.csv',         \n         '/kaggle/input/0-578-ensemble-of-public-notebooks/submission.csv',\n#          '',         \n#          '',         \n#          '',         \n        ]\nprint(len(paths_raw))\n\nuse_p = [9, 10, 11]\npaths = [paths_raw[i] for i in use_p]\npaths","metadata":{"execution":{"iopub.status.busy":"2023-01-08T13:27:20.757841Z","iopub.execute_input":"2023-01-08T13:27:20.758261Z","iopub.status.idle":"2023-01-08T13:27:20.772063Z","shell.execute_reply.started":"2023-01-08T13:27:20.758221Z","shell.execute_reply":"2023-01-08T13:27:20.770606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_sub(path, weight=1): # by default let us assing the weight of 1 to predictions from each submission, this will be akin to a standard vote ensemble\n    '''a helper function for loading and preprocessing submissions'''\n    return (\n        pl.read_csv(path)\n            .with_column(pl.col('labels').str.split(by=' '))\n            .with_column(pl.lit(weight).alias('vote'))\n            .explode('labels')\n            .rename({'labels': 'aid'})\n            .with_column(pl.col('aid').cast(pl.UInt32)) # we are casting the `aids` to `Int32`! memory management is super important to ensure we don't run out of resources\n            .with_column(pl.col('vote').cast(pl.UInt8))\n    )","metadata":{"execution":{"iopub.status.busy":"2023-01-08T13:27:20.773278Z","iopub.execute_input":"2023-01-08T13:27:20.773713Z","iopub.status.idle":"2023-01-08T13:27:20.792329Z","shell.execute_reply.started":"2023-01-08T13:27:20.773678Z","shell.execute_reply":"2023-01-08T13:27:20.79114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs = [read_sub(path) for path in paths]\nsubs[0].head()","metadata":{"execution":{"iopub.status.busy":"2023-01-08T13:27:20.795389Z","iopub.execute_input":"2023-01-08T13:27:20.796612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs[1].head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = subs[0].join(subs[1], how='outer', on=['session_type', 'aid']).head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs = subs[0].join(subs[1], how='outer', on=['session_type', 'aid']).join(subs[2], how='outer', on=['session_type', 'aid'], suffix='_right2')\nsubs.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs = (subs\n    .fill_null(0)\n    .with_column((pl.col('vote') + pl.col('vote_right') + pl.col('vote_right2')).alias('vote_sum'))\n    .drop(['vote', 'vote_right', 'vote_right2'])\n    .sort(by='vote_sum')\n    .reverse()\n)\n\nsubs.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npreds = subs.groupby('session_type').agg([\n    pl.col('aid').head(20).alias('labels')\n])\n\npreds = preds.with_column(pl.col('labels').apply(lambda lst: ' '.join([str(aid) for aid in lst])))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.write_csv('submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}