{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Welcome All\n\nI developed a new module to easily clean dataframes, and want to test it on a big dataframe.\n\nIf you need to see the full package documentation, [please click here](https://naelaqel.com/clean_df/).","metadata":{}},{"cell_type":"code","source":"# install clean_df\n!pip install clean_df","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-03-04T15:20:52.729973Z","iopub.execute_input":"2023-03-04T15:20:52.730802Z","iopub.status.idle":"2023-03-04T15:21:06.347376Z","shell.execute_reply.started":"2023-03-04T15:20:52.730763Z","shell.execute_reply":"2023-03-04T15:21:06.346108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import modules\nimport pandas as pd\nfrom clean_df import CleanDataFrame\n\n# load traing dataframe\ndf_train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\n\n# see info\ndf_train.info()","metadata":{"execution":{"iopub.status.busy":"2023-03-04T15:21:06.349228Z","iopub.execute_input":"2023-03-04T15:21:06.349559Z","iopub.status.idle":"2023-03-04T15:22:21.337420Z","shell.execute_reply.started":"2023-03-04T15:21:06.349527Z","shell.execute_reply":"2023-03-04T15:22:21.336440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We** have 19 columns, with more than 13M columns, the size is above 2GB.","metadata":{}},{"cell_type":"code","source":"# initialize CleanDataFrame to df_train\ncdf = CleanDataFrame(df_train, max_num_cat=20)","metadata":{"execution":{"iopub.status.busy":"2023-03-04T15:22:21.339065Z","iopub.execute_input":"2023-03-04T15:22:21.339766Z","iopub.status.idle":"2023-03-04T15:23:44.353974Z","shell.execute_reply.started":"2023-03-04T15:22:21.339713Z","shell.execute_reply":"2023-03-04T15:23:44.352978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To not delete any **missing values**, we will `call` clean method with `min_missing_ratio=1` and `drop_nan=False`.","metadata":{}},{"cell_type":"code","source":"# call clean\ncdf.clean(min_missing_ratio=1, drop_nan=False)\n\n# call optimize\ncdf.optimize()","metadata":{"execution":{"iopub.status.busy":"2023-03-04T15:23:44.356228Z","iopub.execute_input":"2023-03-04T15:23:44.356932Z","iopub.status.idle":"2023-03-04T15:23:51.444699Z","shell.execute_reply.started":"2023-03-04T15:23:44.356893Z","shell.execute_reply":"2023-03-04T15:23:51.443735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Here** optimize will convert the data types (if the columns does not have any missing values).","metadata":{}},{"cell_type":"code","source":"# check info\ncdf.df.info()","metadata":{"execution":{"iopub.status.busy":"2023-03-04T15:23:51.446280Z","iopub.execute_input":"2023-03-04T15:23:51.446959Z","iopub.status.idle":"2023-03-04T15:23:51.462179Z","shell.execute_reply.started":"2023-03-04T15:23:51.446923Z","shell.execute_reply":"2023-03-04T15:23:51.461034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Above** we can see that around 50% of memory without affecting any values, still we can do more process and optimize values after dealing with missing.","metadata":{}}]}