{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Input data analysis","metadata":{}},{"cell_type":"code","source":"# import libraries\nimport pandas as pd\nimport numpy as np","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import data - small dataframe","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/amex-default-prediction/test_data.csv', nrows=10000)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-25T19:02:25.476631Z","iopub.execute_input":"2022-05-25T19:02:25.477169Z","iopub.status.idle":"2022-05-25T19:02:26.321151Z","shell.execute_reply.started":"2022-05-25T19:02:25.477118Z","shell.execute_reply":"2022-05-25T19:02:26.320171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-05-25T19:02:29.316191Z","iopub.execute_input":"2022-05-25T19:02:29.317469Z","iopub.status.idle":"2022-05-25T19:02:29.337325Z","shell.execute_reply.started":"2022-05-25T19:02:29.317384Z","shell.execute_reply":"2022-05-25T19:02:29.335941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Finding the columns with float and int datatype","metadata":{}},{"cell_type":"code","source":"float_cols = [x for x in df.columns if df[x].dtype == 'float64']\nint_cols = [x for x in df.columns if df[x].dtype == 'int64']\nlen(float_cols), print(int_cols)","metadata":{"execution":{"iopub.status.busy":"2022-05-25T18:46:57.397494Z","iopub.execute_input":"2022-05-25T18:46:57.398730Z","iopub.status.idle":"2022-05-25T18:46:57.412358Z","shell.execute_reply.started":"2022-05-25T18:46:57.398665Z","shell.execute_reply":"2022-05-25T18:46:57.410931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check for rows in train and test","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv('../input/amex-default-prediction/train_data.csv',usecols=['B_31'])\ntrain_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-25T19:31:52.793818Z","iopub.execute_input":"2022-05-25T19:31:52.795728Z","iopub.status.idle":"2022-05-25T19:35:29.960358Z","shell.execute_reply.started":"2022-05-25T19:31:52.795645Z","shell.execute_reply":"2022-05-25T19:35:29.959090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# deleting the variable to reduce memory usage\ndel(train_data)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv('../input/amex-default-prediction/test_data.csv',usecols=['B_31'])\ntest_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-25T18:48:11.591961Z","iopub.execute_input":"2022-05-25T18:48:11.593084Z","iopub.status.idle":"2022-05-25T18:55:39.785924Z","shell.execute_reply.started":"2022-05-25T18:48:11.593024Z","shell.execute_reply":"2022-05-25T18:55:39.784640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# deleting the variable to reduce memory usage\ndel(test_data)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get Min - Max values for float datatype columns","metadata":{}},{"cell_type":"code","source":"# Dictionary to store values during import\nmin_max_dict = dict()\nfor x in float_cols:\n    min_max_dict[x] = dict()\n    min_max_dict[x]['min'] = None\n    min_max_dict[x]['max'] = None","metadata":{"execution":{"iopub.status.busy":"2022-05-25T18:12:14.746236Z","iopub.execute_input":"2022-05-25T18:12:14.747776Z","iopub.status.idle":"2022-05-25T18:12:14.754902Z","shell.execute_reply.started":"2022-05-25T18:12:14.747710Z","shell.execute_reply":"2022-05-25T18:12:14.753658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to get min and max values in each columns of given dataframe\ndef get_min_max(df):\n    for c in df.columns:\n        values = df[c].dropna().values\n        min_value = None if len(values) == 0 else np.min(values)\n        max_value = None if len(values) == 0 else np.max(values)\n        if min_max_dict[c]['min'] != None:\n            if min_value < min_max_dict[c]['min']:\n                min_max_dict[c]['min'] = min_value\n        else:\n            min_max_dict[c]['min'] = min_value\n        if min_max_dict[c]['max'] != None:\n            if max_value > min_max_dict[c]['max']:\n                min_max_dict[c]['max'] = max_value\n        else:\n            min_max_dict[c]['max'] = max_value","metadata":{"execution":{"iopub.status.busy":"2022-05-25T18:12:17.040374Z","iopub.execute_input":"2022-05-25T18:12:17.040884Z","iopub.status.idle":"2022-05-25T18:12:17.049897Z","shell.execute_reply.started":"2022-05-25T18:12:17.040847Z","shell.execute_reply":"2022-05-25T18:12:17.049032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the dataset in small chunks\nchunksize = 1000000 #rows subset from large dataframe\nfor chunk in pd.read_csv('../input/amex-default-prediction/train_data.csv',usecols=float_cols, chunksize=chunksize):\n    get_min_max(chunk)","metadata":{"execution":{"iopub.status.busy":"2022-05-25T18:13:29.412236Z","iopub.execute_input":"2022-05-25T18:13:29.412872Z","iopub.status.idle":"2022-05-25T18:19:03.249661Z","shell.execute_reply.started":"2022-05-25T18:13:29.412833Z","shell.execute_reply":"2022-05-25T18:19:03.248340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# min-max values of each column\nmin_max_dict","metadata":{"execution":{"iopub.status.busy":"2022-05-25T18:19:22.190293Z","iopub.execute_input":"2022-05-25T18:19:22.190746Z","iopub.status.idle":"2022-05-25T18:19:22.216299Z","shell.execute_reply.started":"2022-05-25T18:19:22.190702Z","shell.execute_reply":"2022-05-25T18:19:22.215369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# assigning data type to float16 while reading the data\ndata_types = dict()\nfor x in float_cols:\n    data_types[x] = np.float16","metadata":{"execution":{"iopub.status.busy":"2022-05-25T18:37:51.714216Z","iopub.execute_input":"2022-05-25T18:37:51.715590Z","iopub.status.idle":"2022-05-25T18:37:51.721760Z","shell.execute_reply.started":"2022-05-25T18:37:51.715517Z","shell.execute_reply":"2022-05-25T18:37:51.720767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import data with modified data types\ndata = pd.read_csv('../input/amex-default-prediction/train_data.csv',dtype=data_types)\ndata.info()","metadata":{"execution":{"iopub.status.busy":"2022-05-25T18:37:58.873225Z","iopub.execute_input":"2022-05-25T18:37:58.873619Z","iopub.status.idle":"2022-05-25T18:43:53.093118Z","shell.execute_reply.started":"2022-05-25T18:37:58.873579Z","shell.execute_reply":"2022-05-25T18:43:53.091927Z"},"trusted":true},"execution_count":null,"outputs":[]}]}