{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Created by yunsuxiaozi 2024/3/14\n\n### There are differences in data dtypes between the train and test data in this competition, which has caused many program errors for the participants. In order to facilitate data processing, I have compiled the data types of all the data in the training data here.\n","metadata":{}},{"cell_type":"code","source":"#这个notebook就是检查一下train_file和test_file中哪些列的特征不一致.\nimport polars as pl#和pandas类似,但是处理大型数据集有更好的性能.\nimport pandas as pd#导入csv文件的库\nimport os#与操作系统进行交互的库\nimport gc#垃圾回收模块","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path=\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/\"\ntest_path=\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/\"\ntrain_files=os.listdir(train_path)\ntest_files=os.listdir(test_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colname2dtype={}\nfor file in train_files:\n    csv_file=pl.read_csv(train_path+file)\n    for col in csv_file.columns:\n        if col not in colname2dtype:\n            colname2dtype[col]=csv_file[col].dtype\n    print(f\"len(colname2dtype):{len(colname2dtype)},len(train_files):{len(train_files)}\")\n    del csv_file\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\ncolname2dtype    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"error_cnt=0\nfor file in test_files:\n    csv_file=pl.read_csv(test_path+file)\n    for col in csv_file.columns:\n        if (col in colname2dtype) and (csv_file[col].dtype!=colname2dtype[col]):\n            print(f\"train and test file {col}.dtype is inconsistent.train:{colname2dtype[col]},test:{csv_file[col].dtype}\")\n            error_cnt+=1\n        if col not in colname2dtype:\n            print(f\"{col} in train file but not in test file\")\nprint(f\"error_cnt:{error_cnt}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(list(colname2dtype.items()), columns=['Column', 'DataType'])\ndf.to_csv(\"colname2dtype.csv\",index=None)\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}