{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport glob","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T13:22:36.076885Z","iopub.execute_input":"2022-07-27T13:22:36.077317Z","iopub.status.idle":"2022-07-27T13:22:36.106287Z","shell.execute_reply.started":"2022-07-27T13:22:36.077233Z","shell.execute_reply":"2022-07-27T13:22:36.105197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths = [x for x in glob.glob('../input/*/*.csv') if 'amex-default-prediction' not in x]\ndfs = [pd.read_csv(x) for x in paths]\ndfs = [x.sort_values(by='customer_ID') for x in dfs]","metadata":{"execution":{"iopub.status.busy":"2022-07-27T13:22:36.108348Z","iopub.execute_input":"2022-07-27T13:22:36.108669Z","iopub.status.idle":"2022-07-27T13:22:49.269096Z","shell.execute_reply.started":"2022-07-27T13:22:36.108640Z","shell.execute_reply":"2022-07-27T13:22:49.268022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for df in dfs:\n    df['prediction'] = np.clip(df['prediction'], 0, 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T13:22:49.270469Z","iopub.execute_input":"2022-07-27T13:22:49.270904Z","iopub.status.idle":"2022-07-27T13:22:49.337926Z","shell.execute_reply.started":"2022-07-27T13:22:49.270867Z","shell.execute_reply":"2022-07-27T13:22:49.336881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsubmit['prediction'] = 0\n\nfor df in dfs:\n    submit['prediction'] += df['prediction']\n    \nsubmit['prediction'] /= 3\n\nsubmit.to_csv('mean_submission.csv', index=None)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T13:22:49.340799Z","iopub.execute_input":"2022-07-27T13:22:49.341851Z","iopub.status.idle":"2022-07-27T13:22:56.522249Z","shell.execute_reply.started":"2022-07-27T13:22:49.341811Z","shell.execute_reply":"2022-07-27T13:22:56.521064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsubmit['prediction'] = 0\n\nfrom scipy.stats import rankdata\nfor df in dfs:\n    submit['prediction'] += rankdata(df['prediction'])/df.shape[0]\n    \nsubmit['prediction'] /= 3\n\nsubmit.to_csv('rank_submission.csv', index=None)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T13:22:56.525857Z","iopub.execute_input":"2022-07-27T13:22:56.527011Z","iopub.status.idle":"2022-07-27T13:23:04.063965Z","shell.execute_reply.started":"2022-07-27T13:22:56.526972Z","shell.execute_reply":"2022-07-27T13:23:04.062792Z"},"trusted":true},"execution_count":null,"outputs":[]}]}