{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **American Express - Default Prediction**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport glob","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:17:40.560965Z","iopub.execute_input":"2022-07-24T13:17:40.561376Z","iopub.status.idle":"2022-07-24T13:17:40.566455Z","shell.execute_reply.started":"2022-07-24T13:17:40.561340Z","shell.execute_reply":"2022-07-24T13:17:40.565374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths = [x for x in glob.glob('../input/*/*.csv') if 'amex-default-prediction' not in x]\ndfs = [pd.read_csv(x) for x in paths]\ndfs = [x.sort_values(by='customer_ID') for x in dfs]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:17:40.569057Z","iopub.execute_input":"2022-07-24T13:17:40.570051Z","iopub.status.idle":"2022-07-24T13:17:47.867380Z","shell.execute_reply.started":"2022-07-24T13:17:40.569990Z","shell.execute_reply":"2022-07-24T13:17:47.866162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for df in dfs:\n    df['prediction'] = np.clip(df['prediction'], 0, 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:17:47.868711Z","iopub.execute_input":"2022-07-24T13:17:47.869080Z","iopub.status.idle":"2022-07-24T13:17:47.958881Z","shell.execute_reply.started":"2022-07-24T13:17:47.869047Z","shell.execute_reply":"2022-07-24T13:17:47.957519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsubmit['prediction'] = 0\n\nfor df in dfs:\n    submit['prediction'] += df['prediction']\n    \nsubmit['prediction'] /= 4\n\nsubmit.to_csv('mean_submission.csv', index=None)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:17:47.960558Z","iopub.execute_input":"2022-07-24T13:17:47.961037Z","iopub.status.idle":"2022-07-24T13:17:52.190983Z","shell.execute_reply.started":"2022-07-24T13:17:47.961001Z","shell.execute_reply":"2022-07-24T13:17:52.190159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsubmit['prediction'] = 0\n\nfrom scipy.stats import rankdata\nfor df in dfs:\n    submit['prediction'] += rankdata(df['prediction'])/df.shape[0]\n    \nsubmit['prediction'] /= 4\n\nsubmit.to_csv('rank_submission.csv', index=None)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:17:52.193070Z","iopub.execute_input":"2022-07-24T13:17:52.193630Z","iopub.status.idle":"2022-07-24T13:17:57.320920Z","shell.execute_reply.started":"2022-07-24T13:17:52.193599Z","shell.execute_reply":"2022-07-24T13:17:57.319646Z"},"trusted":true},"execution_count":null,"outputs":[]}]}