{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# AMEX - ProfileReport EDA\n\nAs you get more experienced, the question becomes how to get the most results for minimum effort.\n\nIntroducing DataPrep\n- https://dataprep.ai/","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-23T18:18:26.180302Z","iopub.execute_input":"2022-05-23T18:18:26.180717Z","iopub.status.idle":"2022-05-23T18:18:26.212894Z","shell.execute_reply.started":"2022-05-23T18:18:26.180611Z","shell.execute_reply":"2022-05-23T18:18:26.211473Z"}}},{"cell_type":"code","source":" %%html\n<!-- Is there a better way than hard coding the height? -->\n<style> \niframe { height: 550em; }\n</style>","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-06-02T16:52:05.632434Z","iopub.execute_input":"2022-06-02T16:52:05.633034Z","iopub.status.idle":"2022-06-02T16:52:05.641849Z","shell.execute_reply.started":"2022-06-02T16:52:05.632996Z","shell.execute_reply":"2022-06-02T16:52:05.641077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !yes | pip uninstall flask pandas_profiling 2>&1 > /dev/null\n# !      pip   install flask pandas_profiling 2>&1 > /dev/null\n!pip install dataprep 2>&1 > /dev/null","metadata":{"execution":{"iopub.status.busy":"2022-06-02T16:52:05.690948Z","iopub.execute_input":"2022-06-02T16:52:05.691566Z","iopub.status.idle":"2022-06-02T16:53:02.710297Z","shell.execute_reply.started":"2022-06-02T16:52:05.691528Z","shell.execute_reply":"2022-06-02T16:53:02.708932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!find ../input/ -type f -name '*'","metadata":{"execution":{"iopub.status.busy":"2022-06-02T16:53:02.71328Z","iopub.execute_input":"2022-06-02T16:53:02.713838Z","iopub.status.idle":"2022-06-02T16:53:03.590487Z","shell.execute_reply.started":"2022-06-02T16:53:02.713783Z","shell.execute_reply":"2022-06-02T16:53:03.589236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom glob import glob\n\npd.options.display.max_rows = 6\npd.options.display.max_columns = 999","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def enhance(df):   \n    return df\n\n# train_df = pd.read_csv('../input/amex-default-prediction/train_data.csv')  # too big for RAM\ntrain_df = pd.read_feather('../input/amexfeather/train_data.ftr')\ntrain_df = enhance(train_df)\ntrain_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dataprep \nfrom dataprep.eda import plot, create_report, plot_correlation, plot_missing\n\ndataprep.eda.plot(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-06-02T16:53:23.065929Z","iopub.execute_input":"2022-06-02T16:53:23.066611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataprep.eda.create_report(train_df).show_browser()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataprep.eda.plot_correlation(train_df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataprep.eda.plot_missing(train_df)","metadata":{},"execution_count":null,"outputs":[]}]}