{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"tmp = [4, 7, 8, 10] # 1-13\nweights = [1.]*len(tmp) # diff weight also\nprint(tmp, weights)\n\nassert len(tmp)==len(weights)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T08:32:30.867069Z","iopub.execute_input":"2023-01-23T08:32:30.8675Z","iopub.status.idle":"2023-01-23T08:32:30.873416Z","shell.execute_reply.started":"2023-01-23T08:32:30.867468Z","shell.execute_reply":"2023-01-23T08:32:30.872337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I used @radek1's approach for voting ensemble: https://www.kaggle.com/code/radek1/2-methods-how-to-ensemble-predictions","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"# Loading the data","metadata":{}},{"cell_type":"code","source":"!pip install polars","metadata":{"execution":{"iopub.status.busy":"2023-01-23T08:32:30.878496Z","iopub.execute_input":"2023-01-23T08:32:30.878753Z","iopub.status.idle":"2023-01-23T08:32:40.116596Z","shell.execute_reply.started":"2023-01-23T08:32:30.878729Z","shell.execute_reply":"2023-01-23T08:32:40.115347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl\nimport numpy as np\npaths = ['/kaggle/input/otto-fast-cpu-end-to-end-pipeline/submission.csv', # 0.575\n         '/kaggle/input/otto-recommender/submission.csv', # 0.576\n         '/kaggle/input/otto-tuning-candidate-rerank-model-lb-0-577/submission.csv', # 0.577\n         '/kaggle/input/candidate-rerank-model-lb-0-575/submission.csv',\n         '/kaggle/input/candidate-rerank-model-lb-0-574/submission.csv',         \n         '/kaggle/input/otto-lb-0-574-fast-framework/submission.csv',\n         '/kaggle/input/otto-pipeline2-lb-0-576/submission.csv',         \n         '/kaggle/input/otto-fast-handcrafted-model-recall-20/submission.csv',\n         '/kaggle/input/otto-multi-objective-recommender-system/submission.csv',\n         '/kaggle/input/faster-with-co-visitation/submission.csv',\n         '/kaggle/input/otto-tuning-candidate-rerank-model-lb-0-577/submission.csv',\n         '/kaggle/input/duplicate-fix-otto-tuning-pipeline2-lb-0-577/submission.csv',\n         '/kaggle/input/k/karakasatarik/0-578-ensemble-of-public-notebooks/submission.csv',\n        ]\nprint(len(paths))\npaths = [paths[p-1] for p in tmp]\nprint(paths)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T08:32:40.118414Z","iopub.execute_input":"2023-01-23T08:32:40.118768Z","iopub.status.idle":"2023-01-23T08:32:40.125828Z","shell.execute_reply.started":"2023-01-23T08:32:40.118737Z","shell.execute_reply":"2023-01-23T08:32:40.124839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_sub(path, weight=1): # by default let us assing the weight of 1 to predictions from each submission, this will be akin to a standard vote ensemble\n    '''a helper function for loading and preprocessing submissions'''\n    return (\n        pl.read_csv(path)\n            .with_column(pl.col('labels').str.split(by=' '))\n            .with_column(pl.lit(weight).alias('vote'))\n            .explode('labels')\n            .rename({'labels': 'aid'})\n            .with_column(pl.col('aid').cast(pl.UInt32)) # we are casting the `aids` to `Int32`! memory management is super important to ensure we don't run out of resources\n            .with_column(pl.col('vote').cast(pl.UInt8))\n    )","metadata":{"execution":{"iopub.status.busy":"2023-01-23T08:32:40.126837Z","iopub.execute_input":"2023-01-23T08:32:40.127068Z","iopub.status.idle":"2023-01-23T08:32:40.138917Z","shell.execute_reply.started":"2023-01-23T08:32:40.127046Z","shell.execute_reply":"2023-01-23T08:32:40.138288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for path,weight in zip(paths, weights):\n# subs = [read_sub(path) for path in paths]\n\nsubs = [read_sub(path,weight) for path,weight in zip(paths, weights)]\nsubs[0].head()","metadata":{"execution":{"iopub.status.busy":"2023-01-23T08:32:40.14047Z","iopub.execute_input":"2023-01-23T08:32:40.140884Z","iopub.status.idle":"2023-01-23T08:34:01.170802Z","shell.execute_reply.started":"2023-01-23T08:32:40.140856Z","shell.execute_reply":"2023-01-23T08:34:01.169989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs[2].head()","metadata":{"execution":{"iopub.status.busy":"2023-01-23T08:34:01.172087Z","iopub.execute_input":"2023-01-23T08:34:01.173041Z","iopub.status.idle":"2023-01-23T08:34:01.180932Z","shell.execute_reply.started":"2023-01-23T08:34:01.173007Z","shell.execute_reply":"2023-01-23T08:34:01.179606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subs[0].head().join(subs[1].head(), how='outer', on=['session_type', 'aid'])\n","metadata":{"execution":{"iopub.status.busy":"2023-01-23T08:34:01.182363Z","iopub.execute_input":"2023-01-23T08:34:01.18266Z","iopub.status.idle":"2023-01-23T08:34:01.193054Z","shell.execute_reply.started":"2023-01-23T08:34:01.182633Z","shell.execute_reply":"2023-01-23T08:34:01.192301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntmp_df = subs[0]\nfor i in np.arange(1, len(tmp)):\n    tmp_df = tmp_df.join(subs[i], how='outer', on=['session_type', 'aid'], suffix=f'_right{i}')\ntmp_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-23T08:34:01.194107Z","iopub.execute_input":"2023-01-23T08:34:01.194378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subs_ = subs[0].join(subs[1], how='outer', on=['session_type', 'aid']).join(subs[2], how='outer', on=['session_type', 'aid'], suffix='_right2')\n# subs.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_df.columns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs = tmp_df\ndel tmp_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subs.fill_null(0).with_column((pl.col('vote') + pl.col('vote_right1') + pl.col('vote_right2')).alias('vote_sum'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# pd.DataFrame(subs['vote', 'vote_right1', 'vote_right2', 'vote_right3']).apply(lambda x:x.sum(),axis =1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subs.with_columns(pd.DataFrame(subs[['vote', 'vote_right1', 'vote_right2', 'vote_right3']]).apply(lambda x:x.sum(),axis =0))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subs = subs.fill_null(0)\n# # df[\"sum\"] =df.apply(lambda x:x.sum(),axis =1)\n\n# subs['vote_sum'] = pd.DataFrame(subs[['vote', 'vote_right1', 'vote_right2', 'vote_right3']]).apply(lambda x:x.sum(),axis =1)\n# subs.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pl.col('vote') + pl.col('vote_right') + pl.col('vote_right2')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs_ = subs.fill_null(0)\ndel subs\nsubs_","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col = ['vote_sum0' if c=='vote' else c for c in subs_.columns]\nsubs_.columns = col\nsubs_","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in np.arange(1,len(tmp)):\n    print(i)\n    subs_ = (subs_.with_column((pl.col(f'vote_sum{i-1}') + pl.col(f'vote_right{i}')).alias(f'vote_sum{i}')))\n    subs_ = subs_.drop([f'vote_sum{i-1}', f'vote_right{i}'])\nsubs_","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_res = subs_[['session_type', 'aid', f'vote_sum{len(tmp)-1}']].sort(by=f'vote_sum{len(tmp)-1}').reverse()\n\ndel subs_\ntmp_res","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subs = (subs\n#     .fill_null(0)\n#     .with_column((pl.col('vote') + pl.col('vote_right1') + pl.col('vote_right2')).alias('vote_sum'))\n#     .drop(['vote', 'vote_right', 'vote_right2'])\n#     .sort(by='vote_sum')\n#     .reverse()\n# )\n\n# subs.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subs = (subs\n#     .fill_null(0)\n#     .with_column((pl.col('vote') + pl.col('vote_right') + pl.col('vote_right2')).alias('vote_sum'))\n#     .drop(['vote', 'vote_right', 'vote_right2'])\n#     .sort(by='vote_sum')\n#     .reverse()\n# )\n\n# subs.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n%%time\npreds = tmp_res.groupby('session_type').agg([\n    pl.col('aid').head(20).alias('labels')\n])\n\npreds = preds.with_column(pl.col('labels').apply(lambda lst: ' '.join([str(aid) for aid in lst])))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# preds = subs.groupby('session_type').agg([\n#     pl.col('aid').head(20).alias('labels')\n# ])\n\n# preds = preds.with_column(pl.col('labels').apply(lambda lst: ' '.join([str(aid) for aid in lst])))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.write_csv('submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}