{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"},{"sourceId":12405352,"sourceType":"datasetVersion","datasetId":7823194},{"sourceId":12409307,"sourceType":"datasetVersion","datasetId":7826020}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#!/usr/bin/env python3\n\nimport pandas as pd\nimport numpy as np\nimport pickle\nimport gc\nimport os\nimport json\nfrom typing import Optional\nimport warnings\nfrom datetime import datetime\n\nwarnings.filterwarnings('ignore')\n\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\ndef clean_memory():\n    \"\"\"Force garbage collection to free up RAM\"\"\"\n    gc.collect()\n\ndef process_parquet_file_chunked(file_path: str, chunk_size: int = 100000, \n                                target_column: str = \"label\") -> pd.DataFrame:\n    \"\"\"Process a large parquet file in chunks to optimize memory usage while preserving timestamp index\"\"\"\n    print(f\"📊 Processing {file_path} in chunks of {chunk_size:,} rows...\")\n    \n    # Read a small sample first to get basic info\n    try:\n        df_sample = pd.read_parquet(file_path, engine='pyarrow')\n        total_rows = len(df_sample)\n        columns = df_sample.columns.tolist()\n        \n        # Check if index is timestamp\n        index_info = {\n            'name': df_sample.index.name,\n            'dtype': str(df_sample.index.dtype),\n            'is_datetime': pd.api.types.is_datetime64_any_dtype(df_sample.index),\n            'first_value': df_sample.index[0] if len(df_sample) > 0 else None,\n            'last_value': df_sample.index[-1] if len(df_sample) > 0 else None\n        }\n        \n        print(f\"   📈 File has {total_rows:,} rows and {len(columns)} columns\")\n        print(f\"   📅 Index info: {index_info['name']} ({index_info['dtype']}), datetime: {index_info['is_datetime']}\")\n        if index_info['is_datetime']:\n            print(f\"   📅 Time range: {index_info['first_value']} to {index_info['last_value']}\")\n        \n        # If file is small enough, process all at once\n        if total_rows <= chunk_size:\n            print(f\"   📈 File is small enough, processing all {total_rows:,} rows at once\")\n            df_optimized = reduce_mem_usage(df_sample)\n            clean_memory()\n            return df_optimized\n        \n        del df_sample\n        clean_memory()\n        \n        print(f\"   📈 Processing in {(total_rows + chunk_size - 1) // chunk_size} chunks\")\n        \n        # Process in chunks using pyarrow for efficiency\n        import pyarrow.parquet as pq\n        parquet_file = pq.ParquetFile(file_path)\n        \n        chunks = []\n        processed_rows = 0\n        \n        for batch in parquet_file.iter_batches(batch_size=chunk_size):\n            # Convert batch to pandas DataFrame - this preserves the index\n            chunk_df = batch.to_pandas()\n            chunk_rows = len(chunk_df)\n            \n            # Verify index preservation\n            if processed_rows == 0:  # First chunk\n                print(f\"   📅 Index preserved: {chunk_df.index.name} ({chunk_df.index.dtype})\")\n                if pd.api.types.is_datetime64_any_dtype(chunk_df.index):\n                    print(f\"   📅 First chunk time range: {chunk_df.index[0]} to {chunk_df.index[-1]}\")\n            \n            # Apply memory optimization\n            print(f\"   🔧 Optimizing chunk {len(chunks) + 1}: {chunk_rows:,} rows\")\n            chunk_optimized = reduce_mem_usage(chunk_df)\n            chunks.append(chunk_optimized)\n            \n            processed_rows += chunk_rows\n            print(f\"   ✅ Processed {processed_rows:,}/{total_rows:,} rows ({processed_rows/total_rows*100:.1f}%)\")\n            \n            # Clean up\n            del chunk_df\n            clean_memory()\n        \n        # Combine all chunks - DO NOT use ignore_index=True to preserve timestamp index\n        print(\"   🔄 Combining all chunks while preserving index...\")\n        df_final = pd.concat(chunks, ignore_index=False)  # Keep original index\n        \n        # Verify final index\n        print(f\"   📅 Final index: {df_final.index.name} ({df_final.index.dtype})\")\n        if pd.api.types.is_datetime64_any_dtype(df_final.index):\n            print(f\"   📅 Final time range: {df_final.index[0]} to {df_final.index[-1]}\")\n            print(f\"   ✅ Index is sorted: {df_final.index.is_monotonic_increasing}\")\n        \n        # Clean up chunks\n        del chunks\n        clean_memory()\n        \n        print(f\"   ✅ Final dataframe: {df_final.shape}\")\n        return df_final\n        \n    except Exception as e:\n        print(f\"   ❌ Error processing file: {e}\")\n        # Fallback to pandas chunked reading\n        return process_parquet_pandas_chunked(file_path, chunk_size)\n\ndef process_parquet_pandas_chunked(file_path: str, chunk_size: int = 100000) -> pd.DataFrame:\n    \"\"\"Fallback method using pandas chunked reading while preserving timestamp index\"\"\"\n    print(f\"   🔄 Using pandas fallback method for {file_path}\")\n    \n    # Read entire file first (pandas doesn't support chunked parquet reading directly)\n    df = pd.read_parquet(file_path, engine='pyarrow')\n    total_rows = len(df)\n    \n    # Check index info\n    print(f\"   📅 Original index: {df.index.name} ({df.index.dtype}), datetime: {pd.api.types.is_datetime64_any_dtype(df.index)}\")\n    if pd.api.types.is_datetime64_any_dtype(df.index):\n        print(f\"   📅 Time range: {df.index[0]} to {df.index[-1]}\")\n        print(f\"   ✅ Index is sorted: {df.index.is_monotonic_increasing}\")\n    \n    if total_rows <= chunk_size:\n        print(f\"   📈 Processing all {total_rows:,} rows at once\")\n        df_optimized = reduce_mem_usage(df)\n        clean_memory()\n        return df_optimized\n    \n    print(f\"   📈 Processing {total_rows:,} rows in chunks\")\n    \n    # Process in chunks\n    chunks = []\n    for i in range(0, total_rows, chunk_size):\n        end_idx = min(i + chunk_size, total_rows)\n        # Use iloc to preserve index structure\n        chunk = df.iloc[i:end_idx].copy()\n        \n        print(f\"   🔧 Optimizing chunk: rows {i:,} to {end_idx:,}\")\n        chunk_optimized = reduce_mem_usage(chunk)\n        chunks.append(chunk_optimized)\n        \n        del chunk\n        clean_memory()\n    \n    # Clean up original dataframe\n    del df\n    clean_memory()\n    \n    # Combine chunks - preserve original index\n    print(\"   🔄 Combining chunks while preserving index...\")\n    df_final = pd.concat(chunks, ignore_index=False)  # Keep original index\n    \n    # Verify final index\n    print(f\"   📅 Final index: {df_final.index.name} ({df_final.index.dtype})\")\n    if pd.api.types.is_datetime64_any_dtype(df_final.index):\n        print(f\"   📅 Final time range: {df_final.index[0]} to {df_final.index[-1]}\")\n        print(f\"   ✅ Index is sorted: {df_final.index.is_monotonic_increasing}\")\n    \n    del chunks\n    clean_memory()\n    \n    return df_final\n\ndef save_optimized_data(df: pd.DataFrame, filename: str, metadata: dict):\n    \"\"\"Save optimized dataframe with metadata including index information\"\"\"\n    print(f\"💾 Saving optimized data to {filename}...\")\n    \n    # Capture index information\n    index_info = {\n        'name': df.index.name,\n        'dtype': str(df.index.dtype),\n        'is_datetime': pd.api.types.is_datetime64_any_dtype(df.index),\n        'is_sorted': df.index.is_monotonic_increasing if pd.api.types.is_datetime64_any_dtype(df.index) else None,\n        'first_value': df.index[0] if len(df) > 0 else None,\n        'last_value': df.index[-1] if len(df) > 0 else None,\n        'size': len(df.index)\n    }\n    \n    # Prepare data package\n    data_package = {\n        'data': df,\n        'metadata': metadata,\n        'shape': df.shape,\n        'columns': df.columns.tolist(),\n        'dtypes': {col: str(dtype) for col, dtype in df.dtypes.items()},\n        'index_info': index_info,\n        'memory_usage_mb': df.memory_usage(deep=True).sum() / 1024**2,\n        'created_at': datetime.now().isoformat()\n    }\n    \n    # Save with highest compression\n    with open(filename, 'wb') as f:\n        pickle.dump(data_package, f, protocol=pickle.HIGHEST_PROTOCOL)\n    \n    # Get file size\n    file_size = os.path.getsize(filename) / 1024**2  # MB\n    print(f\"   ✅ Saved {filename} ({file_size:.2f} MB)\")\n    print(f\"   📊 Data shape: {df.shape}\")\n    print(f\"   📅 Index: {index_info['name']} ({index_info['dtype']}), datetime: {index_info['is_datetime']}\")\n    if index_info['is_datetime']:\n        print(f\"   📅 Time range: {index_info['first_value']} to {index_info['last_value']}\")\n        print(f\"   ✅ Index sorted: {index_info['is_sorted']}\")\n    print(f\"   🧠 Memory usage: {data_package['memory_usage_mb']:.2f} MB\")\n    \n    return file_size\n\ndef analyze_data_quality(df: pd.DataFrame, name: str) -> dict:\n    \"\"\"Analyze data quality and return statistics\"\"\"\n    print(f\"🔍 Analyzing data quality for {name}...\")\n    \n    stats = {\n        'name': name,\n        'shape': df.shape,\n        'memory_usage_mb': df.memory_usage(deep=True).sum() / 1024**2,\n        'null_counts': df.isnull().sum().to_dict(),\n        'null_percentages': (df.isnull().sum() / len(df) * 100).to_dict(),\n        'dtypes': {col: str(dtype) for col, dtype in df.dtypes.items()},\n        'numeric_summary': {},\n        'categorical_summary': {}\n    }\n    \n    # Analyze numeric columns\n    numeric_cols = df.select_dtypes(include=[np.number]).columns\n    if len(numeric_cols) > 0:\n        numeric_summary = df[numeric_cols].describe()\n        stats['numeric_summary'] = {\n            col: {\n                'count': float(numeric_summary.loc['count', col]),\n                'mean': float(numeric_summary.loc['mean', col]),\n                'std': float(numeric_summary.loc['std', col]),\n                'min': float(numeric_summary.loc['min', col]),\n                'max': float(numeric_summary.loc['max', col])\n            } for col in numeric_cols\n        }\n    \n    # Analyze categorical columns\n    categorical_cols = df.select_dtypes(include=['category', 'object']).columns\n    if len(categorical_cols) > 0:\n        for col in categorical_cols:\n            unique_count = df[col].nunique()\n            stats['categorical_summary'][col] = {\n                'unique_count': int(unique_count),\n                'most_common': df[col].value_counts().head(5).to_dict() if unique_count <= 1000 else {}\n            }\n    \n    print(f\"   ✅ Analysis complete for {name}\")\n    return stats\n\ndef main():\n    \"\"\"Main function to create memory-optimized versions of full datasets\"\"\"\n    print(\"🚀 FULL DATASET MEMORY OPTIMIZATION\")\n    print(\"=\" * 80)\n    print(\"📁 Processing ENTIRE train.parquet and test.parquet files\")\n    print(\"🔧 Applying advanced memory optimization techniques\")\n    print(\"💾 Saving as compressed pickle files with metadata\")\n    print(\"=\" * 80)\n    \n    # Configuration\n    TRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    TEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    TARGET_COLUMN = \"label\"\n    CHUNK_SIZE = 100000  # Process in 100k row chunks\n    \n    # Output files\n    TRAIN_OUTPUT = \"train_full_optimized.pkl\"\n    TEST_OUTPUT = \"test_full_optimized.pkl\"\n    SUMMARY_OUTPUT = \"full_optimization_summary.json\"\n    \n    # Check if input files exist\n    if not os.path.exists(TRAIN_PATH):\n        print(f\"❌ Error: {TRAIN_PATH} not found!\")\n        return\n    \n    if not os.path.exists(TEST_PATH):\n        print(f\"❌ Error: {TEST_PATH} not found!\")\n        return\n    \n    results = {}\n    \n    try:\n        # Step 1: Process train data\n        print(f\"\\n📊 Step 1: Processing FULL train data\")\n        print(f\"   📁 Source: {TRAIN_PATH}\")\n        \n        train_df = process_parquet_file_chunked(TRAIN_PATH, CHUNK_SIZE, TARGET_COLUMN)\n        \n        # Analyze train data\n        train_stats = analyze_data_quality(train_df, \"train\")\n        \n        # Save train data\n        train_metadata = {\n            'source_file': TRAIN_PATH,\n            'target_column': TARGET_COLUMN,\n            'processing_method': 'chunked_memory_optimization',\n            'chunk_size': CHUNK_SIZE,\n            'optimization_applied': True\n        }\n        \n        train_file_size = save_optimized_data(train_df, TRAIN_OUTPUT, train_metadata)\n        results['train'] = {\n            'file_size_mb': train_file_size,\n            'shape': train_df.shape,\n            'stats': train_stats\n        }\n        \n        # Clean up train data\n        del train_df\n        clean_memory()\n        \n        # Step 2: Process test data\n        print(f\"\\n📊 Step 2: Processing FULL test data\")\n        print(f\"   📁 Source: {TEST_PATH}\")\n        \n        test_df = process_parquet_file_chunked(TEST_PATH, CHUNK_SIZE, TARGET_COLUMN)\n        \n        # Analyze test data\n        test_stats = analyze_data_quality(test_df, \"test\")\n        \n        # Save test data\n        test_metadata = {\n            'source_file': TEST_PATH,\n            'target_column': TARGET_COLUMN,\n            'processing_method': 'chunked_memory_optimization',\n            'chunk_size': CHUNK_SIZE,\n            'optimization_applied': True\n        }\n        \n        test_file_size = save_optimized_data(test_df, TEST_OUTPUT, test_metadata)\n        results['test'] = {\n            'file_size_mb': test_file_size,\n            'shape': test_df.shape,\n            'stats': test_stats\n        }\n        \n        # Clean up test data\n        del test_df\n        clean_memory()\n        \n        # Step 3: Create comprehensive summary\n        print(f\"\\n📋 Step 3: Creating comprehensive summary\")\n        \n        summary = {\n            'processing_info': {\n                'created_at': datetime.now().isoformat(),\n                'source_files': {\n                    'train': TRAIN_PATH,\n                    'test': TEST_PATH\n                },\n                'output_files': {\n                    'train': TRAIN_OUTPUT,\n                    'test': TEST_OUTPUT\n                },\n                'chunk_size': CHUNK_SIZE,\n                'target_column': TARGET_COLUMN\n            },\n            'optimization_results': {\n                'train': {\n                    'original_file': TRAIN_PATH,\n                    'optimized_file': TRAIN_OUTPUT,\n                    'file_size_mb': train_file_size,\n                    'shape': results['train']['shape'],\n                    'memory_usage_mb': results['train']['stats']['memory_usage_mb']\n                },\n                'test': {\n                    'original_file': TEST_PATH,\n                    'optimized_file': TEST_OUTPUT,\n                    'file_size_mb': test_file_size,\n                    'shape': results['test']['shape'],\n                    'memory_usage_mb': results['test']['stats']['memory_usage_mb']\n                }\n            },\n            'data_quality': {\n                'train': train_stats,\n                'test': test_stats\n            },\n            'usage_instructions': {\n                'loading': \"Use load_full_optimized_data() function to load the data\",\n                'format': \"Each pickle file contains: {'data': DataFrame, 'metadata': dict, 'shape': tuple, 'columns': list, 'dtypes': dict, 'index_info': dict, 'memory_usage_mb': float, 'created_at': str}\",\n                'compatibility': \"Compatible with pandas and all standard ML libraries\",\n                'index_preservation': \"Original timestamp index from parquet files is preserved and sorted\",\n                'rolling_features': \"Timestamp index allows proper rolling window calculations\"\n            }\n        }\n        \n        # Save summary\n        with open(SUMMARY_OUTPUT, \"w\") as f:\n            json.dump(summary, f, indent=2, default=str)\n        \n        print(f\"   ✅ Summary saved to {SUMMARY_OUTPUT}\")\n        \n        # Final memory cleanup\n        clean_memory()\n        \n        print(f\"\\n{'='*80}\")\n        print(\"🎉 FULL DATASET OPTIMIZATION COMPLETED SUCCESSFULLY!\")\n        print(f\"{'='*80}\")\n        print(f\"📊 Train data: {results['train']['shape'][0]:,} rows × {results['train']['shape'][1]} features\")\n        print(f\"📊 Test data:  {results['test']['shape'][0]:,} rows × {results['test']['shape'][1]} features\")\n        print(f\"💾 File sizes:\")\n        print(f\"   📁 {TRAIN_OUTPUT}: {train_file_size:.2f} MB\")\n        print(f\"   📁 {TEST_OUTPUT}: {test_file_size:.2f} MB\")\n        print(f\"🧠 Memory usage:\")\n        print(f\"   📁 Train: {results['train']['stats']['memory_usage_mb']:.2f} MB\")\n        print(f\"   📁 Test:  {results['test']['stats']['memory_usage_mb']:.2f} MB\")\n        print(f\"📋 Summary: {SUMMARY_OUTPUT}\")\n        print(f\"\\n🔧 Use load_full_optimized_data() to load the optimized datasets\")\n        \n    except Exception as e:\n        print(f\"❌ Error during processing: {e}\")\n        import traceback\n        traceback.print_exc()\n    \n    finally:\n        # Final cleanup\n        clean_memory()\n\ndef load_full_optimized_data(train_file: str = \"train_full_optimized.pkl\", \n                           test_file: str = \"test_full_optimized.pkl\"):\n    \"\"\"Utility function to load the optimized full datasets with index information\"\"\"\n    print(\"📂 Loading full optimized datasets...\")\n    \n    try:\n        # Load train data\n        with open(train_file, 'rb') as f:\n            train_package = pickle.load(f)\n        \n        # Load test data  \n        with open(test_file, 'rb') as f:\n            test_package = pickle.load(f)\n        \n        train_df = train_package['data']\n        test_df = test_package['data']\n        \n        print(f\"✅ Loaded train data: {train_df.shape[0]:,} rows × {train_df.shape[1]} features\")\n        print(f\"✅ Loaded test data: {test_df.shape[0]:,} rows × {test_df.shape[1]} features\")\n        print(f\"🧠 Train memory: {train_package['memory_usage_mb']:.2f} MB\")\n        print(f\"🧠 Test memory: {test_package['memory_usage_mb']:.2f} MB\")\n        print(f\"📅 Created: {train_package.get('created_at', 'Unknown')}\")\n        \n        # Display index information if available\n        if 'index_info' in train_package:\n            train_idx = train_package['index_info']\n            print(f\"📅 Train index: {train_idx['name']} ({train_idx['dtype']}), datetime: {train_idx['is_datetime']}\")\n            if train_idx['is_datetime']:\n                print(f\"   📅 Train time range: {train_idx['first_value']} to {train_idx['last_value']}\")\n                print(f\"   ✅ Train index sorted: {train_idx['is_sorted']}\")\n        \n        if 'index_info' in test_package:\n            test_idx = test_package['index_info']\n            print(f\"📅 Test index: {test_idx['name']} ({test_idx['dtype']}), datetime: {test_idx['is_datetime']}\")\n            if test_idx['is_datetime']:\n                print(f\"   📅 Test time range: {test_idx['first_value']} to {test_idx['last_value']}\")\n                print(f\"   ✅ Test index sorted: {test_idx['is_sorted']}\")\n        \n        return {\n            'train': {\n                'data': train_df,\n                'metadata': train_package['metadata'],\n                'index_info': train_package.get('index_info', {})\n            },\n            'test': {\n                'data': test_df,\n                'metadata': test_package['metadata'],\n                'index_info': test_package.get('index_info', {})\n            }\n        }\n    \n    except Exception as e:\n        print(f\"❌ Error loading data: {e}\")\n        return None\n\ndef check_optimized_data_info(train_file: str = \"train_full_optimized.pkl\", \n                              test_file: str = \"test_full_optimized.pkl\"):\n    \"\"\"Check information about optimized datasets without loading the full data\"\"\"\n    print(\"🔍 Checking optimized dataset information...\")\n    \n    try:\n        files_info = {}\n        \n        for name, file_path in [(\"train\", train_file), (\"test\", test_file)]:\n            if os.path.exists(file_path):\n                with open(file_path, 'rb') as f:\n                    package = pickle.load(f)\n                \n                # Extract key information without loading the full dataframe\n                info = {\n                    'file_size_mb': os.path.getsize(file_path) / 1024**2,\n                    'shape': package.get('shape', 'Unknown'),\n                    'memory_usage_mb': package.get('memory_usage_mb', 'Unknown'),\n                    'created_at': package.get('created_at', 'Unknown'),\n                    'columns_count': len(package.get('columns', [])),\n                    'index_info': package.get('index_info', {}),\n                    'metadata': package.get('metadata', {})\n                }\n                \n                files_info[name] = info\n                \n                print(f\"\\n📊 {name.upper()} DATASET:\")\n                print(f\"   📁 File: {file_path} ({info['file_size_mb']:.2f} MB)\")\n                print(f\"   📊 Shape: {info['shape']}\")\n                print(f\"   🧠 Memory: {info['memory_usage_mb']:.2f} MB\")\n                print(f\"   📅 Created: {info['created_at']}\")\n                \n                if info['index_info']:\n                    idx = info['index_info']\n                    print(f\"   📅 Index: {idx.get('name', 'None')} ({idx.get('dtype', 'Unknown')})\")\n                    print(f\"   📅 Datetime index: {idx.get('is_datetime', False)}\")\n                    if idx.get('is_datetime'):\n                        print(f\"   📅 Time range: {idx.get('first_value')} to {idx.get('last_value')}\")\n                        print(f\"   ✅ Sorted: {idx.get('is_sorted', 'Unknown')}\")\n                \n            else:\n                print(f\"\\n❌ {name.upper()} file not found: {file_path}\")\n                files_info[name] = None\n        \n        return files_info\n    \n    except Exception as e:\n        print(f\"❌ Error checking data info: {e}\")\n        return None\n\nif __name__ == \"__main__\":\n    main() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-11T16:42:55.400842Z","iopub.execute_input":"2025-07-11T16:42:55.401118Z"}},"outputs":[],"execution_count":null}]}