{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Introduction**\n\nThis notebook is intended to demonstrate the power of **Vaex**.  \n  \nVaex is very fast and easy to use with large datasets as compares to other libraries like Dask.  \n  \nResulting HDF5 is 3.94GB as compared to 15.2GB original CSV file (75% compression).  \n\nThe API of **vaex.ml** stays close to that of **scikit-learn**, while providing better performance and the ability to efficiently perform operations on data that is larger than the available RAM.","metadata":{}},{"cell_type":"markdown","source":"# **Setup**","metadata":{}},{"cell_type":"code","source":"import vaex\nvaex.multithreading.thread_count_default = 8\nimport vaex.ml","metadata":{"execution":{"iopub.status.busy":"2022-07-10T12:37:13.256564Z","iopub.execute_input":"2022-07-10T12:37:13.256986Z","iopub.status.idle":"2022-07-10T12:37:15.565466Z","shell.execute_reply.started":"2022-07-10T12:37:13.256904Z","shell.execute_reply":"2022-07-10T12:37:15.564275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_floats(ddf):\n    for c in ddf.columns:\n        if ddf[c].dtype == 'float64':\n            ddf[c] = ddf[c].astype('float32')\n    return ddf\n\ndef fill_and_convert_floats(ddf):\n    for c in ddf.columns:\n        if ddf[c].dtype == 'float64':\n            ddf[c] = ddf[c].fillna(0.0).astype('float32')\n    return ddf\n\ndef convert_floats_to_ints(ddf):\n    feature_names = ['R_2','R_3','R_4','R_5','R_8','R_9','R_10','R_11','R_13','R_15','R_16','R_17','R_18','R_19','R_20','R_21','R_22','R_23','R_24','R_25']\n    for c in feature_names:\n        ddf[c] = ddf[c].fillna(0.0).astype('int32')\n    return ddf","metadata":{"execution":{"iopub.status.busy":"2022-07-10T12:37:20.13653Z","iopub.execute_input":"2022-07-10T12:37:20.137243Z","iopub.status.idle":"2022-07-10T12:37:20.146342Z","shell.execute_reply.started":"2022-07-10T12:37:20.137204Z","shell.execute_reply":"2022-07-10T12:37:20.145159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Convert CSV to HDF5**","metadata":{}},{"cell_type":"code","source":"#uncomment this code only when you want to experience conversion\n#this step requires around 8GB of your notebook's output folder\n\n#Output - 12 HDF5 files with 754MB each\n\nfor i, df in enumerate(vaex.from_csv('../input/amex-default-prediction/train_data.csv', chunk_size=500_000)):\n    df.export_hdf5(f'./train_{i:02}.hdf5')\n    break #uncomment this to convert all","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-10T12:37:25.851672Z","iopub.execute_input":"2022-07-10T12:37:25.852047Z","iopub.status.idle":"2022-07-10T12:38:03.120491Z","shell.execute_reply.started":"2022-07-10T12:37:25.852015Z","shell.execute_reply":"2022-07-10T12:38:03.11933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Read Multiple HDF5 Files**","metadata":{}},{"cell_type":"code","source":"df = vaex.open('./train*.hdf5')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T12:38:14.642247Z","iopub.execute_input":"2022-07-10T12:38:14.642671Z","iopub.status.idle":"2022-07-10T12:38:15.904731Z","shell.execute_reply.started":"2022-07-10T12:38:14.642636Z","shell.execute_reply":"2022-07-10T12:38:15.903431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Encode Categorical Coulmns**","metadata":{}},{"cell_type":"code","source":"label_encoder = vaex.ml.LabelEncoder(features=['customer_ID', 'D_64', 'D_63'])\ndf = label_encoder.fit_transform(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T12:39:27.333593Z","iopub.execute_input":"2022-07-10T12:39:27.334036Z","iopub.status.idle":"2022-07-10T12:39:28.207482Z","shell.execute_reply.started":"2022-07-10T12:39:27.333998Z","shell.execute_reply":"2022-07-10T12:39:28.206151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customer = df[['customer_ID','label_encoded_customer_ID']]","metadata":{"execution":{"iopub.status.busy":"2022-07-10T12:39:30.608527Z","iopub.execute_input":"2022-07-10T12:39:30.608882Z","iopub.status.idle":"2022-07-10T12:39:30.616426Z","shell.execute_reply.started":"2022-07-10T12:39:30.608853Z","shell.execute_reply":"2022-07-10T12:39:30.615157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['customer_ID', 'D_64', 'D_63'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T12:39:32.870626Z","iopub.execute_input":"2022-07-10T12:39:32.871007Z","iopub.status.idle":"2022-07-10T12:39:33.166204Z","shell.execute_reply.started":"2022-07-10T12:39:32.870978Z","shell.execute_reply":"2022-07-10T12:39:33.165347Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Convert Floats to Ints**","metadata":{}},{"cell_type":"code","source":"df['S_2'] = df['S_2'].str.replace('-','').astype('int64')\ndf['R_26'] = df['R_26'].astype('int16')\ndf = fill_and_convert_floats(df)\ndf = convert_floats_to_ints(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T13:04:29.56987Z","iopub.execute_input":"2022-07-10T13:04:29.570665Z","iopub.status.idle":"2022-07-10T13:04:31.48082Z","shell.execute_reply.started":"2022-07-10T13:04:29.570621Z","shell.execute_reply":"2022-07-10T13:04:31.479497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Save Mapping - Customer ID - Original and Encoded Values**","metadata":{}},{"cell_type":"code","source":"df_customer.export_hdf5('./train_customer_map.hdf5')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T13:05:49.301547Z","iopub.execute_input":"2022-07-10T13:05:49.301972Z","iopub.status.idle":"2022-07-10T13:05:49.522503Z","shell.execute_reply.started":"2022-07-10T13:05:49.301941Z","shell.execute_reply":"2022-07-10T13:05:49.521379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Save HDF5**","metadata":{}},{"cell_type":"code","source":"df.export_hdf5('./train.hdf5')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T13:06:35.912711Z","iopub.execute_input":"2022-07-10T13:06:35.9137Z","iopub.status.idle":"2022-07-10T13:06:43.270807Z","shell.execute_reply.started":"2022-07-10T13:06:35.91365Z","shell.execute_reply":"2022-07-10T13:06:43.269102Z"},"trusted":true},"execution_count":null,"outputs":[]}]}