{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"The dataset comprises numerous large CSV and parquet files, collectively encompassing over 450 variables. Given the scale and complexity of the data, I thought it would help streamline the development process by cataloging the variables present in each file, aggregating this information into a comprehensive table.\n\nHopefully this will help others make sense of the large number of possible features as well.","metadata":{}},{"cell_type":"code","source":"import polars as pl\nimport os\n\n\ndata_path = '/kaggle/input/home-credit-credit-risk-model-stability/'\nfeature_def = pl.read_csv(data_path + 'feature_definitions.csv')\n\n# Get a list of all CSV files in the directory\ncsv_files = [f for f in os.listdir(data_path + 'csv_files/train') if f.endswith('.csv')]\n\n# Create an empty dictionary to hold the feature-file mapping\nfeature_files = {}\n\n# Loop through the list of CSV files\nfor csv_file in csv_files:\n    df = pl.read_csv(f'{data_path}/csv_files/train/{csv_file}', n_rows=1)\n    \n    features = df.columns\n    \n    for feature in features:\n        if feature in feature_files:\n            feature_files[feature].append(csv_file)\n        else:\n            feature_files[feature] = [csv_file]\n\n# Convert the dictionary to a DataFrame\nfeature_files_df = pl.DataFrame({\n    'Variable': list(feature_files.keys()),\n    'Files': [', '.join(files) for files in feature_files.values()] \n})\n\n# Merge this DataFrame with the feature definition DataFrame\nresult = feature_def.join(feature_files_df, on='Variable', how='left')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-18T21:43:35.163589Z","iopub.execute_input":"2024-02-18T21:43:35.164066Z","iopub.status.idle":"2024-02-18T21:43:35.249435Z","shell.execute_reply.started":"2024-02-18T21:43:35.164028Z","shell.execute_reply":"2024-02-18T21:43:35.248069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-02-18T21:43:35.252158Z","iopub.execute_input":"2024-02-18T21:43:35.254034Z","iopub.status.idle":"2024-02-18T21:43:35.264911Z","shell.execute_reply.started":"2024-02-18T21:43:35.253943Z","shell.execute_reply":"2024-02-18T21:43:35.263592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result.write_csv('feature_files.csv')","metadata":{"execution":{"iopub.status.busy":"2024-02-18T21:43:35.266490Z","iopub.execute_input":"2024-02-18T21:43:35.267701Z","iopub.status.idle":"2024-02-18T21:43:35.277125Z","shell.execute_reply.started":"2024-02-18T21:43:35.267661Z","shell.execute_reply":"2024-02-18T21:43:35.275808Z"},"trusted":true},"execution_count":null,"outputs":[]}]}