{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pickle\nimport pandas as pd\nfrom sklearn.linear_model import LogisticRegression","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-11T05:42:18.519514Z","iopub.execute_input":"2023-02-11T05:42:18.520279Z","iopub.status.idle":"2023-02-11T05:42:19.569901Z","shell.execute_reply.started":"2023-02-11T05:42:18.520175Z","shell.execute_reply":"2023-02-11T05:42:19.568927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_1 = pd.read_feather('../input/amex-imputed-and-1hot-encoded/X_test_1.ftr')\nX_test_1 = X_test_1.set_index('customer_ID')","metadata":{"execution":{"iopub.status.busy":"2023-02-11T05:42:19.574298Z","iopub.execute_input":"2023-02-11T05:42:19.574756Z","iopub.status.idle":"2023-02-11T05:42:39.623111Z","shell.execute_reply.started":"2023-02-11T05:42:19.574722Z","shell.execute_reply":"2023-02-11T05:42:39.621948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_2 = pd.read_feather('../input/amex-imputed-and-1hot-encoded/X_test_2.ftr')\nX_test_2 = X_test_2.set_index('customer_ID')","metadata":{"execution":{"iopub.status.busy":"2023-02-11T05:42:39.624254Z","iopub.execute_input":"2023-02-11T05:42:39.624534Z","iopub.status.idle":"2023-02-11T05:43:01.714079Z","shell.execute_reply.started":"2023-02-11T05:42:39.624507Z","shell.execute_reply":"2023-02-11T05:43:01.712596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = pickle.load(open('/kaggle/input/test-dtc-model/logistic_regression_model.sav', 'rb'))\npreds_1 = pd.DataFrame(model.predict_proba(X_test_1)[:, 1], index=X_test_1.index, columns=['prediction'])\npreds_2 = pd.DataFrame(model.predict_proba(X_test_2)[:, 1], index=X_test_2.index, columns=['prediction'])","metadata":{"execution":{"iopub.status.busy":"2023-02-11T05:43:09.456771Z","iopub.execute_input":"2023-02-11T05:43:09.457146Z","iopub.status.idle":"2023-02-11T05:43:22.643075Z","shell.execute_reply.started":"2023-02-11T05:43:09.457117Z","shell.execute_reply":"2023-02-11T05:43:22.641985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.concat([preds_1, preds_2])\n\n# predictions only need to be for each customer\nsubmission = submission.groupby('customer_ID').agg(['last'])\nsubmission.columns = submission.columns.droplevel(1)\n\n# predictions need to be doubles \nsubmission['prediction'] = submission['prediction'].astype('double')\n\n# index needs to be removed from submission csv\nsubmission = submission.reset_index()\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-02-11T05:43:22.644675Z","iopub.execute_input":"2023-02-11T05:43:22.645176Z","iopub.status.idle":"2023-02-11T05:43:27.240656Z","shell.execute_reply.started":"2023-02-11T05:43:22.645143Z","shell.execute_reply":"2023-02-11T05:43:27.239841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}