{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import polars as pl\n\n# 1. Lazy scan of the CSV files (creates a query plan without loading data into RAM)\ntrain_path = \"/kaggle/input/amex-default-prediction/train_data.csv\"\nlabels_path = \"/kaggle/input/amex-default-prediction/train_labels.csv\"\n\n# Using scan_csv for memory-efficient lazy evaluation\ntrain_lazy = pl.scan_csv(train_path)\nprint('Train data scan initialized...')\n\nlabels_lazy = pl.scan_csv(labels_path)\nprint('Labels scan initialized...')\n\n# 2. Memory optimization (Downcasting numeric types)\n# Converting Float64 to Float32 and Int64 to Int32 to reduce memory footprint by 50%\ntrain_lazy = train_lazy.select([\n    pl.col(pl.Float64).cast(pl.Float32),\n    pl.col(pl.Int64).cast(pl.Int32),\n    pl.exclude([pl.Float64, pl.Int64]) # Retains customer_ID and date columns\n])\nprint('Train data types optimized.')\n\n# 3. Label optimization\n# Target is binary (0/1), so Int8 is more than enough\nlabels_lazy = labels_lazy.select([\n    pl.col(\"customer_ID\"),\n    pl.col(\"target\").cast(pl.Int8)\n])\nprint('Label types optimized.')\n\n# 4. Join datasets on customer_ID\n# This performs a Left Join, broadcasting the label to every timestamped row of a customer\nfull_train_lazy = train_lazy.join(labels_lazy, on=\"customer_ID\", how=\"left\")\nprint('Join operation added to the graph.')\n\n# 5. Execution and Export\n# The .collect() method triggers the actual computation (Eager execution)\nprint('Processing and writing to Parquet... This may take a few minutes.')\n\nfull_train_lazy.collect().write_parquet(\"train_full_with_labels.parquet\")\n\nprint(\"Success! You now have a single compact Parquet file with features and target.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T05:29:57.242654Z","iopub.execute_input":"2026-01-10T05:29:57.243014Z","iopub.status.idle":"2026-01-10T05:34:49.980623Z","shell.execute_reply.started":"2026-01-10T05:29:57.242981Z","shell.execute_reply":"2026-01-10T05:34:49.979188Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null}]}