{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!/usr/bin/env python\n# -*- coding: utf-8 -*-\n\n\"\"\"\nDRW Crypto Market Prediction - TPOT Iterative Feature Engineering\nThis implementation uses TPOT on data chunks to identify robust features\nand transformations that consistently perform well across samples.\n\"\"\"\n\n# Install required packages\n!pip install tpot -q\n!pip install lightgbm -q\n!pip install xgboost -q\n!pip install deap -q\n!pip install update_checker -q\n!pip install stopit -q\n\nimport pandas as pd\nimport numpy as np\nfrom tpot import TPOTRegressor\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, RobustScaler\nfrom sklearn.decomposition import PCA, FastICA\nfrom sklearn.feature_selection import SelectKBest, f_regression, mutual_info_regression\nfrom sklearn.ensemble import RandomForestRegressor\nfrom scipy.stats import pearsonr\nimport warnings\nimport json\nfrom datetime import datetime\nimport gc\nimport os\nfrom collections import defaultdict, Counter\nimport re\nimport pickle\n\nwarnings.filterwarnings('ignore')\n\nclass CFG:\n    \"\"\"Configuration for TPOT iterative approach\"\"\"\n    # File paths\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    \n    # TPOT settings\n    n_bootstraps = 8  # Number of bootstrap samples to run\n    chunk_size = 10000  # Size of each chunk for TPOT\n    tpot_generations = 5  # Generations per TPOT run\n    tpot_population_size = 20  # Population size per TPOT run\n    tpot_time_limit = 600  # 10 minutes per TPOT run\n    \n    # Feature settings\n    max_features_per_chunk = 50  # Limit features per chunk to manage memory\n    min_feature_frequency = 0.5  # Feature must appear in 50% of bootstraps\n    top_transformations = 10  # Number of top transformations to keep\n    \n    # Model settings\n    final_model_time = 1800  # 30 minutes for final model\n    random_seed = 42\n    n_jobs = -1  # Use all cores\n    \n    # Checkpointing\n    checkpoint_dir = \"./tpot_checkpoints\"\n    results_file = \"tpot_bootstrap_results.pkl\"\n\ndef ensure_checkpoint_dir():\n    \"\"\"Create checkpoint directory if it doesn't exist\"\"\"\n    if not os.path.exists(CFG.checkpoint_dir):\n        os.makedirs(CFG.checkpoint_dir)\n\ndef save_checkpoint(data, filename):\n    \"\"\"Save checkpoint data\"\"\"\n    ensure_checkpoint_dir()\n    filepath = os.path.join(CFG.checkpoint_dir, filename)\n    with open(filepath, 'wb') as f:\n        pickle.dump(data, f)\n    print(f\"Checkpoint saved: {filename}\")\n\ndef load_checkpoint(filename):\n    \"\"\"Load checkpoint if exists\"\"\"\n    filepath = os.path.join(CFG.checkpoint_dir, filename)\n    if os.path.exists(filepath):\n        with open(filepath, 'rb') as f:\n            data = pickle.load(f)\n        print(f\"Checkpoint loaded: {filename}\")\n        return data\n    return None\n\ndef extract_pipeline_components(exported_pipeline):\n    \"\"\"Extract transformation steps from TPOT exported pipeline\"\"\"\n    components = {\n        'preprocessors': [],\n        'selectors': [],\n        'transformations': [],\n        'model': None\n    }\n    \n    # Common TPOT components patterns\n    preprocessor_patterns = [\n        r'StandardScaler',\n        r'MinMaxScaler',\n        r'RobustScaler',\n        r'Normalizer'\n    ]\n    \n    selector_patterns = [\n        r'SelectKBest',\n        r'SelectPercentile',\n        r'VarianceThreshold',\n        r'SelectFromModel'\n    ]\n    \n    transformation_patterns = [\n        r'PCA',\n        r'FastICA',\n        r'PolynomialFeatures',\n        r'Nystroem',\n        r'RBFSampler'\n    ]\n    \n    # Extract components\n    for pattern in preprocessor_patterns:\n        if re.search(pattern, exported_pipeline):\n            components['preprocessors'].append(pattern)\n    \n    for pattern in selector_patterns:\n        if re.search(pattern, exported_pipeline):\n            # Try to extract parameters\n            match = re.search(f'{pattern}\\([^)]*\\)', exported_pipeline)\n            if match:\n                components['selectors'].append(match.group(0))\n            else:\n                components['selectors'].append(pattern)\n    \n    for pattern in transformation_patterns:\n        if re.search(pattern, exported_pipeline):\n            match = re.search(f'{pattern}\\([^)]*\\)', exported_pipeline)\n            if match:\n                components['transformations'].append(match.group(0))\n            else:\n                components['transformations'].append(pattern)\n    \n    # Extract model\n    model_patterns = [\n        r'XGBRegressor',\n        r'LGBMRegressor',\n        r'RandomForestRegressor',\n        r'GradientBoostingRegressor',\n        r'ElasticNetCV',\n        r'LassoLarsCV'\n    ]\n    \n    for pattern in model_patterns:\n        if re.search(pattern, exported_pipeline):\n            components['model'] = pattern\n            break\n    \n    return components\n\ndef create_base_features(df):\n    \"\"\"Create market microstructure features\"\"\"\n    features_created = []\n    \n    # Safe division helper\n    def safe_divide(a, b, fill_value=0):\n        return np.where(b != 0, a / b, fill_value)\n    \n    if 'bid_qty' in df.columns and 'ask_qty' in df.columns:\n        df['bid_ask_spread'] = df['ask_qty'] - df['bid_qty']\n        df['bid_ask_imbalance'] = safe_divide(\n            df['bid_qty'] - df['ask_qty'],\n            df['bid_qty'] + df['ask_qty']\n        )\n        df['total_depth'] = df['bid_qty'] + df['ask_qty']\n        features_created.extend(['bid_ask_spread', 'bid_ask_imbalance', 'total_depth'])\n    \n    if 'buy_qty' in df.columns and 'sell_qty' in df.columns:\n        df['order_imbalance'] = safe_divide(\n            df['buy_qty'] - df['sell_qty'],\n            df['buy_qty'] + df['sell_qty']\n        )\n        df['order_flow'] = df['buy_qty'] - df['sell_qty']\n        df['trade_intensity'] = df['buy_qty'] + df['sell_qty']\n        features_created.extend(['order_imbalance', 'order_flow', 'trade_intensity'])\n    \n    if 'volume' in df.columns:\n        df['log_volume'] = np.log1p(df['volume'])\n        features_created.append('log_volume')\n    \n    return features_created\n\ndef select_feature_subset(df, features, n_features):\n    \"\"\"Select top features using mutual information\"\"\"\n    print(f\"Selecting top {n_features} features from {len(features)}...\")\n    \n    X = df[features]\n    y = df['label']\n    \n    # Handle missing values\n    X = X.fillna(X.median())\n    \n    # Calculate mutual information scores\n    mi_scores = mutual_info_regression(X, y, random_state=CFG.random_seed)\n    \n    # Get top features\n    feature_scores = pd.DataFrame({\n        'feature': features,\n        'score': mi_scores\n    }).sort_values('score', ascending=False)\n    \n    top_features = feature_scores.head(n_features)['feature'].tolist()\n    \n    return top_features, feature_scores\n\ndef run_tpot_on_chunk(X_train, y_train, X_val, y_val, chunk_id):\n    \"\"\"Run TPOT on a data chunk\"\"\"\n    print(f\"\\nRunning TPOT on chunk {chunk_id}...\")\n    \n    tpot = TPOTRegressor(\n        generations=CFG.tpot_generations,\n        population_size=CFG.tpot_population_size,\n        scoring='r2',\n        cv=3,\n        random_state=CFG.random_seed,\n        n_jobs=CFG.n_jobs,\n        max_time_mins=CFG.tpot_time_limit // 60,\n        max_eval_time_mins=5,\n        verbosity=2,\n        config_dict={\n            # Limit to lightweight operations for memory efficiency\n            'sklearn.preprocessing.StandardScaler': {},\n            'sklearn.preprocessing.MinMaxScaler': {},\n            'sklearn.preprocessing.RobustScaler': {},\n            'sklearn.decomposition.PCA': {\n                'n_components': [5, 10, 20, 30]\n            },\n            'sklearn.decomposition.FastICA': {\n                'n_components': [5, 10, 20]\n            },\n            'sklearn.feature_selection.SelectKBest': {\n                'k': [10, 20, 30, 40]\n            },\n            'sklearn.feature_selection.SelectPercentile': {\n                'percentile': [10, 25, 50, 75]\n            },\n            'lightgbm.LGBMRegressor': {\n                'n_estimators': [100],\n                'num_leaves': [31, 63],\n                'learning_rate': [0.05, 0.1],\n                'feature_fraction': [0.8],\n                'boosting_type': ['gbdt']\n            },\n            'xgboost.XGBRegressor': {\n                'n_estimators': [100],\n                'max_depth': [3, 5, 7],\n                'learning_rate': [0.05, 0.1],\n                'subsample': [0.8]\n            }\n        }\n    )\n    \n    # Fit TPOT\n    tpot.fit(X_train, y_train)\n    \n    # Get validation score\n    val_score = tpot.score(X_val, y_val)\n    \n    # Export pipeline\n    pipeline_string = tpot.export()\n    \n    # Extract components\n    components = extract_pipeline_components(pipeline_string)\n    \n    # Get feature importances if possible\n    feature_importances = {}\n    try:\n        if hasattr(tpot.fitted_pipeline_, 'steps'):\n            final_estimator = tpot.fitted_pipeline_.steps[-1][1]\n            if hasattr(final_estimator, 'feature_importances_'):\n                # Get feature names after transformations\n                feature_names = list(range(X_val.shape[1]))\n                importances = final_estimator.feature_importances_\n                feature_importances = dict(zip(feature_names, importances))\n    except:\n        pass\n    \n    results = {\n        'chunk_id': chunk_id,\n        'val_score': val_score,\n        'pipeline': pipeline_string,\n        'components': components,\n        'feature_importances': feature_importances,\n        'best_params': tpot.fitted_pipeline_\n    }\n    \n    # Save checkpoint\n    save_checkpoint(results, f'tpot_chunk_{chunk_id}.pkl')\n    \n    return results\n\ndef consolidate_results(all_results):\n    \"\"\"Consolidate results from all bootstrap runs\"\"\"\n    print(\"\\nConsolidating results from all bootstraps...\")\n    \n    # Track component frequencies\n    preprocessor_counts = Counter()\n    selector_counts = Counter()\n    transformation_counts = Counter()\n    model_counts = Counter()\n    \n    # Track scores\n    scores = []\n    \n    for result in all_results:\n        components = result['components']\n        scores.append(result['val_score'])\n        \n        for prep in components['preprocessors']:\n            preprocessor_counts[prep] += 1\n        \n        for sel in components['selectors']:\n            selector_counts[sel] += 1\n        \n        for trans in components['transformations']:\n            transformation_counts[trans] += 1\n        \n        if components['model']:\n            model_counts[components['model']] += 1\n    \n    # Calculate frequencies\n    n_bootstraps = len(all_results)\n    \n    consolidated = {\n        'mean_score': np.mean(scores),\n        'std_score': np.std(scores),\n        'preprocessors': {\n            k: v/n_bootstraps for k, v in preprocessor_counts.most_common()\n        },\n        'selectors': {\n            k: v/n_bootstraps for k, v in selector_counts.most_common()\n        },\n        'transformations': {\n            k: v/n_bootstraps for k, v in transformation_counts.most_common()\n        },\n        'models': {\n            k: v/n_bootstraps for k, v in model_counts.most_common()\n        },\n        'all_scores': scores\n    }\n    \n    return consolidated\n\ndef build_robust_pipeline(consolidated_results, min_frequency=0.5):\n    \"\"\"Build pipeline using most frequent successful components\"\"\"\n    print(\"\\nBuilding robust pipeline from consolidated results...\")\n    \n    pipeline_steps = []\n    \n    # Add most frequent preprocessor\n    for prep, freq in consolidated_results['preprocessors'].items():\n        if freq >= min_frequency:\n            print(f\"Adding preprocessor: {prep} (frequency: {freq:.2f})\")\n            if 'StandardScaler' in prep:\n                pipeline_steps.append(('scaler', StandardScaler()))\n            elif 'RobustScaler' in prep:\n                pipeline_steps.append(('scaler', RobustScaler()))\n            elif 'MinMaxScaler' in prep:\n                pipeline_steps.append(('scaler', MinMaxScaler()))\n            break\n    \n    # Add transformations\n    for trans, freq in consolidated_results['transformations'].items():\n        if freq >= min_frequency:\n            print(f\"Adding transformation: {trans} (frequency: {freq:.2f})\")\n            if 'PCA' in trans:\n                # Extract n_components if specified\n                n_components = 20  # default\n                match = re.search(r'n_components=(\\d+)', trans)\n                if match:\n                    n_components = int(match.group(1))\n                pipeline_steps.append(('pca', PCA(n_components=n_components)))\n            break\n    \n    # Add feature selection\n    for sel, freq in consolidated_results['selectors'].items():\n        if freq >= min_frequency:\n            print(f\"Adding selector: {sel} (frequency: {freq:.2f})\")\n            if 'SelectKBest' in sel:\n                k = 30  # default\n                match = re.search(r'k=(\\d+)', sel)\n                if match:\n                    k = int(match.group(1))\n                pipeline_steps.append(('selector', SelectKBest(f_regression, k=k)))\n            break\n    \n    return pipeline_steps\n\ndef apply_robust_transformations(X, pipeline_steps):\n    \"\"\"Apply the robust pipeline transformations\"\"\"\n    X_transformed = X.copy()\n    \n    for name, transformer in pipeline_steps:\n        print(f\"Applying {name}...\")\n        X_transformed = transformer.fit_transform(X_transformed)\n        \n        # Convert back to DataFrame if needed\n        if hasattr(X_transformed, 'shape') and len(X_transformed.shape) == 2:\n            if not isinstance(X_transformed, pd.DataFrame):\n                X_transformed = pd.DataFrame(\n                    X_transformed, \n                    columns=[f'feature_{i}' for i in range(X_transformed.shape[1])]\n                )\n    \n    return X_transformed\n\ndef main():\n    \"\"\"Main execution pipeline\"\"\"\n    timestamp = datetime.now().strftime(\"%Y%m%d_%H%M%S\")\n    \n    print(\"=\"*60)\n    print(\"DRW CRYPTO PREDICTION - TPOT ITERATIVE APPROACH\")\n    print(f\"Timestamp: {timestamp}\")\n    print(\"=\"*60)\n    \n    # Load data\n    print(\"\\nLoading data...\")\n    train_data = pd.read_parquet(CFG.train_path)\n    test_data = pd.read_parquet(CFG.test_path)\n    sample_submission = pd.read_csv(CFG.sample_sub_path)\n    \n    print(f\"Train shape: {train_data.shape}\")\n    print(f\"Test shape: {test_data.shape}\")\n    \n    # Create base features\n    base_features = create_base_features(train_data)\n    test_base_features = create_base_features(test_data)\n    \n    # Get all features\n    feature_cols = [col for col in train_data.columns \n                   if col not in ['label', 'timestamp']]\n    \n    print(f\"Total features: {len(feature_cols)}\")\n    \n    # Check for existing results\n    existing_results = load_checkpoint(CFG.results_file)\n    if existing_results:\n        all_results = existing_results\n        start_bootstrap = len(all_results)\n        print(f\"Resuming from bootstrap {start_bootstrap}\")\n    else:\n        all_results = []\n        start_bootstrap = 0\n    \n    # Run TPOT on bootstrap samples\n    for i in range(start_bootstrap, CFG.n_bootstraps):\n        print(f\"\\n{'='*50}\")\n        print(f\"Bootstrap iteration {i+1}/{CFG.n_bootstraps}\")\n        print('='*50)\n        \n        # Sample data\n        sample_indices = np.random.choice(\n            len(train_data), \n            size=min(CFG.chunk_size, len(train_data)), \n            replace=False\n        )\n        chunk_data = train_data.iloc[sample_indices].copy()\n        \n        # Select feature subset for this chunk\n        selected_features, feature_scores = select_feature_subset(\n            chunk_data, \n            feature_cols, \n            CFG.max_features_per_chunk\n        )\n        \n        # Prepare data\n        X_chunk = chunk_data[selected_features]\n        y_chunk = chunk_data['label']\n        \n        # Split for validation\n        X_train, X_val, y_train, y_val = train_test_split(\n            X_chunk, y_chunk, \n            test_size=0.2, \n            random_state=CFG.random_seed + i\n        )\n        \n        # Run TPOT\n        try:\n            result = run_tpot_on_chunk(X_train, y_train, X_val, y_val, i)\n            result['selected_features'] = selected_features\n            all_results.append(result)\n            \n            # Save intermediate results\n            save_checkpoint(all_results, CFG.results_file)\n            \n        except Exception as e:\n            print(f\"Error in bootstrap {i}: {e}\")\n            continue\n        \n        # Clean up memory\n        gc.collect()\n    \n    # Consolidate results\n    consolidated = consolidate_results(all_results)\n    \n    print(\"\\n\" + \"=\"*50)\n    print(\"CONSOLIDATED RESULTS\")\n    print(\"=\"*50)\n    print(f\"Mean validation score: {consolidated['mean_score']:.6f} \"\n          f\"(±{consolidated['std_score']:.6f})\")\n    \n    print(\"\\nMost frequent components:\")\n    print(f\"Preprocessors: {list(consolidated['preprocessors'].keys())[:3]}\")\n    print(f\"Transformations: {list(consolidated['transformations'].keys())[:3]}\")\n    print(f\"Selectors: {list(consolidated['selectors'].keys())[:3]}\")\n    print(f\"Models: {list(consolidated['models'].keys())[:3]}\")\n    \n    # Build robust pipeline\n    pipeline_steps = build_robust_pipeline(\n        consolidated, \n        min_frequency=CFG.min_feature_frequency\n    )\n    \n    # Apply to full dataset\n    print(\"\\nApplying robust transformations to full dataset...\")\n    \n    # Use all features for final training\n    X_train_full = train_data[feature_cols]\n    y_train = train_data['label']\n    X_test_full = test_data[feature_cols]\n    \n    # Apply transformations\n    if pipeline_steps:\n        # Fit on train and transform both sets\n        X_train_transformed = X_train_full.copy()\n        X_test_transformed = X_test_full.copy()\n        \n        for name, transformer in pipeline_steps:\n            X_train_transformed = transformer.fit_transform(X_train_transformed)\n            X_test_transformed = transformer.transform(X_test_transformed)\n            \n            # Maintain DataFrame structure\n            if not isinstance(X_train_transformed, pd.DataFrame):\n                n_features = X_train_transformed.shape[1]\n                feature_names = [f'{name}_feature_{i}' for i in range(n_features)]\n                X_train_transformed = pd.DataFrame(X_train_transformed, columns=feature_names)\n                X_test_transformed = pd.DataFrame(X_test_transformed, columns=feature_names)\n    else:\n        X_train_transformed = X_train_full\n        X_test_transformed = X_test_full\n    \n    # Train final model using most successful approach\n    print(\"\\nTraining final model...\")\n    \n    # Determine best model type from results\n    if consolidated['models']:\n        best_model_type = list(consolidated['models'].keys())[0]\n        print(f\"Using most frequent model type: {best_model_type}\")\n        \n        if 'LGBM' in best_model_type:\n            final_model = lgb.LGBMRegressor(\n                n_estimators=500,\n                num_leaves=63,\n                learning_rate=0.05,\n                feature_fraction=0.8,\n                bagging_fraction=0.8,\n                bagging_freq=5,\n                random_state=CFG.random_seed,\n                n_jobs=CFG.n_jobs\n            )\n        elif 'XGB' in best_model_type:\n            import xgboost as xgb\n            final_model = xgb.XGBRegressor(\n                n_estimators=500,\n                max_depth=6,\n                learning_rate=0.05,\n                subsample=0.8,\n                random_state=CFG.random_seed,\n                n_jobs=CFG.n_jobs\n            )\n        else:\n            final_model = RandomForestRegressor(\n                n_estimators=500,\n                max_depth=10,\n                random_state=CFG.random_seed,\n                n_jobs=CFG.n_jobs\n            )\n    else:\n        # Default to LightGBM\n        final_model = lgb.LGBMRegressor(\n            n_estimators=500,\n            random_state=CFG.random_seed,\n            n_jobs=CFG.n_jobs\n        )\n    \n    # Fit final model\n    final_model.fit(X_train_transformed, y_train)\n    \n    # Make predictions\n    test_predictions = final_model.predict(X_test_transformed)\n    \n    # Calculate CV score\n    print(\"\\nCalculating cross-validation score...\")\n    kf = KFold(n_splits=5, shuffle=True, random_state=CFG.random_seed)\n    cv_scores = []\n    \n    for fold, (train_idx, val_idx) in enumerate(kf.split(X_train_transformed)):\n        X_fold_train = X_train_transformed.iloc[train_idx] if isinstance(X_train_transformed, pd.DataFrame) else X_train_transformed[train_idx]\n        y_fold_train = y_train.iloc[train_idx]\n        X_fold_val = X_train_transformed.iloc[val_idx] if isinstance(X_train_transformed, pd.DataFrame) else X_train_transformed[val_idx]\n        y_fold_val = y_train.iloc[val_idx]\n        \n        fold_model = final_model.__class__(**final_model.get_params())\n        fold_model.fit(X_fold_train, y_fold_train)\n        fold_pred = fold_model.predict(X_fold_val)\n        \n        fold_score = pearsonr(y_fold_val, fold_pred)[0]\n        cv_scores.append(fold_score)\n        print(f\"  Fold {fold + 1}: {fold_score:.6f}\")\n    \n    print(f\"\\nFinal CV Score: {np.mean(cv_scores):.6f} (±{np.std(cv_scores):.6f})\")\n    \n    # Save predictions\n    submission = sample_submission.copy()\n    submission['prediction'] = test_predictions\n    submission_filename = f\"submission_tpot_{timestamp}.csv\"\n    submission.to_csv(submission_filename, index=False)\n    \n    print(f\"\\nSaved predictions to {submission_filename}\")\n    \n    # Save detailed report\n    report = {\n        'timestamp': timestamp,\n        'n_bootstraps': CFG.n_bootstraps,\n        'chunk_size': CFG.chunk_size,\n        'consolidated_results': consolidated,\n        'pipeline_steps': [str(step) for step in pipeline_steps],\n        'final_model': str(final_model),\n        'cv_scores': cv_scores,\n        'cv_mean': np.mean(cv_scores),\n        'cv_std': np.std(cv_scores)\n    }\n    \n    with open(f'tpot_report_{timestamp}.json', 'w') as f:\n        json.dump(report, f, indent=2, default=str)\n    \n    print(\"\\nTPOT iterative pipeline completed successfully!\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}