{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":87793,"databundleVersionId":12276181,"sourceType":"competition"},{"sourceId":11962855,"sourceType":"datasetVersion","datasetId":7522315},{"sourceId":260872499,"sourceType":"kernelVersion"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-25T01:26:34.816222Z","iopub.execute_input":"2025-09-25T01:26:34.816916Z","iopub.status.idle":"2025-09-25T01:26:53.596122Z","shell.execute_reply.started":"2025-09-25T01:26:34.816887Z","shell.execute_reply":"2025-09-25T01:26:53.595412Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd \nfile_path = \"/kaggle/input/stanford-3d-rna-folding\"\ndirectory = \"kaggle/input/stanford-3d-rna-folding/MSA.fasta\",\n\"/kaggle/input/stanford-3d-rna-folding/PDB_RNA.cif\",\"/kaggle/input/stanford-3d-rna-folding/sample_submission.csv\",\"/kaggle/input/stanford-3d-rna-folding/test_sequences.csv\",\"/kaggle/input/stanford-3d-rna-folding/train_labels.csv\",\"/kaggle/input/stanford-3d-rna-folding/train_sequences.csv\",\"/kaggle/input/stanford-3d-rna-folding/train_sequences_v2.csv\",\"/kaggle/input/stanford-3d-rna-folding/validation_labels.csv\",\"/kaggle/input/stanford-3d-rna-folding/validation_sequences.csv\"\noutput_path = \"/kaggle/working/directory\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T01:26:53.597363Z","iopub.execute_input":"2025-09-25T01:26:53.597879Z","iopub.status.idle":"2025-09-25T01:26:53.602584Z","shell.execute_reply.started":"2025-09-25T01:26:53.597853Z","shell.execute_reply":"2025-09-25T01:26:53.601819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"MSA_data = \"/kaggle/input/stanford-3d-rna-folding/MSA.fasta\"\nPDB_RNA_data = \"/kaggle/input/stanford-3d-rna-folding/PDB_RNA.cif\"\nsample_submission_data = \"/kaggle/input/stanford-3d-rna-folding/sample_submission.csv\"\ntest_data = \"/kaggle/input/stanford-3d-rna-folding/test_sequences.csv\"\ntrain_labels_data = \"/kaggle/input/stanford-3d-rna-folding/train_labels.csv\"\ntrain_sequences_data = \"/kaggle/input/stanford-3d-rna-folding/train_sequences.csv\"\ntrain_sequences_v2_data = \"/kaggle/input/stanford-3d-rna-folding/train_sequences_v2.csv\"\nvalidation_labels_data = \"/kaggle/input/stanford-3d-rna-folding/validation_labels.csv\"\nvalidation_sequences_data = \"/kaggle/input/stanford-3d-rna-folding/validation_sequences.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T01:26:53.60337Z","iopub.execute_input":"2025-09-25T01:26:53.603715Z","iopub.status.idle":"2025-09-25T01:26:53.619071Z","shell.execute_reply.started":"2025-09-25T01:26:53.603687Z","shell.execute_reply":"2025-09-25T01:26:53.618582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  get APIs\nimport os\n\nimport pandas as pd\n\nimport numpy as np\n\nimport matplotlib.pyplot as plt\n\nimport seaborn as sns\n\nfrom pathlib import Path\n\nimport logging\n\nfrom typing import List, Dict, Optional, Tuple\n\nimport json\n\nfrom datetime import datetime\n\nimport subprocess\n\nimport sys\n\n\n# Import RNAdvisor\n!pip install rnadvisor \n\ntry:\n\n    from rnadvisor.rnadvisor_cli import RNAdvisorCLI\n\nexcept ImportError:\n\n    print(\"rnadvisor not installed. Run: pip install rnadvisor\")\n\n    sys.exit(1)\n\n\nclass RNAdvisorIntegrator:\n     A comprehensive wrapper class for RNAdvisor integration\n\n    \"\"\"\n\n    \n\n    def __init__(self, base_dir: str = \"rna_analysis\", log_level: str = \"INFO\"):\n\n        \"\"\"\n\n        Initialize the RNAdvisor integrator\n\n        \n\n        Args:\n\n            base_dir: Base directory for analysis\n\n            log_level: Logging level\n\n        \"\"\"\n\n        self.base_dir = Path(base_dir)\n\n        self.base_dir.mkdir(exist_ok=True)\n\n        \n\n        # Setup logging\n\n        logging.basicConfig(\n\n            level=getattr(logging, log_level),\n\n            format='%(asctime)s - %(levelname)s - %(message)s',\n\n            handlers=[\n\n                logging.FileHandler(self.base_dir / 'rnadvisor_analysis.log'),\n\n                logging.StreamHandler()\n\n            ]\n\n        )\n\n        self.logger = logging.getLogger(__name__)\n\n        \n\n        # Define available scores\n\n        self.all_scores = [\n\n            \"rmsd\", \"inf\", \"mcq\", \"lddt\", \"tm-score\", \"gdt-ts\", \n\n            \"ares\", \"pamnet\", \"p-value\", \"di\", \"cad-score\", \n\n            \"3drnascore\", \"lociparse\", \"tb-mcq\", \"cgrnasp\", \n\n            \"dfire\", \"rasp\", \"rs-rnasp\", \"clash\", \"rna-briq\"\n\n        ]\n\n        \n\n        self.fast_scores = [\"rmsd\", \"inf\", \"mcq\", \"lddt\", \"tm-score\", \"gdt-ts\"]\n\n        self.comprehensive_scores = [\n\n            \"rmsd\", \"inf\", \"mcq\", \"lddt\", \"tm-score\", \"gdt-ts\", \n\n            \"ares\", \"pamnet\", \"p-value\", \"di\", \"cad-score\"\n\n        ]\n\n\n    def single_structure_analysis(self, \n\n                                 pred_path: str, \n\n                                 native_path: str,\n\n                                 scores: List[str] = None,\n\n                                 output_name: str = \"single_analysis\") -> Tuple[pd.DataFrame, pd.DataFrame]:\n\n        \"\"\"\n\n        Analyze a single predicted structure against native\n\n        \n\n        Args:\n\n            pred_path: Path to predicted structure PDB file\n\n            native_path: Path to native structure PDB file\n\n            scores: List of scores to calculate\n\n            output_name: Name for output files\n\n            \n\n        Returns:\n\n            Tuple of (results_df, timing_df)\n\n        \"\"\"\n\n        if scores is None:\n\n            scores = self.fast_scores\n\n            \n\n        self.logger.info(f\"Starting single structure analysis: {pred_path}\")\n\n        \n\n        # Create output directory\n\n        output_dir = self.base_dir / output_name\n\n        output_dir.mkdir(exist_ok=True)\n\n        \n\n        # Setup RNAdvisor\n\n        rnadvisor_cli = RNAdvisorCLI(\n\n            pred_dir=pred_path,\n\n            native_path=native_path,\n\n            out_path=str(output_dir / \"results.csv\"),\n\n            scores=scores\n\n        )\n\n        \n\n        try:\n\n            # Run analysis\n\n            df_results, df_time = rnadvisor_cli.predict()\n\n            \n\n            # Save results\n\n            df_results.to_csv(output_dir / \"results.csv\", index=False)\n\n            df_time.to_csv(output_dir / \"timing.csv\", index=False)\n\n            \n\n            self.logger.info(f\"Analysis completed. Results saved to {output_dir}\")\n\n            \n\n            # Generate summary report\n\n            self._generate_single_report(df_results, df_time, output_dir)\n\n            \n\n            return df_results, df_time\n\n            \n\n        except Exception as e:\n\n            self.logger.error(f\"Analysis failed: {str(e)}\")\n\n            raise\n\n\n    def batch_analysis(self, \n\n                      pred_dir: str, \n\n                      native_path: str,\n\n                      scores: List[str] = None,\n\n                      output_name: str = \"batch_analysis\") -> Tuple[pd.DataFrame, pd.DataFrame]:\n\n        \"\"\"\n\n        Analyze multiple predicted structures against a native structure\n\n        \n\n        Args:\n\n            pred_dir: Directory containing predicted structure PDB files\n\n            native_path: Path to native structure PDB file\n\n            scores: List of scores to calculate\n\n            output_name: Name for output files\n\n            \n\n        Returns:\n\n            Tuple of (results_df, timing_df)\n\n        \"\"\"\n\n        if scores is None:\n\n            scores = self.comprehensive_scores\n\n            \n\n        self.logger.info(f\"Starting batch analysis: {pred_dir}\")\n\n        \n\n        # Create output directory\n\n        output_dir = self.base_dir / output_name\n\n        output_dir.mkdir(exist_ok=True)\n\n        \n\n        # Setup RNAdvisor\n\n        rnadvisor_cli = RNAdvisorCLI(\n\n            pred_dir=pred_dir,\n\n            native_path=native_path,\n\n            out_path=str(output_dir / \"batch_results.csv\"),\n\n            scores=scores\n\n        )\n\n        \n\n        try:\n\n            # Run analysis\n\n            df_results, df_time = rnadvisor_cli.predict()\n\n            \n\n            # Save results\n\n            df_results.to_csv(output_dir / \"batch_results.csv\", index=False)\n\n            df_time.to_csv(output_dir / \"batch_timing.csv\", index=False)\n\n            \n\n            self.logger.info(f\"Batch analysis completed. Results saved to {output_dir}\")\n\n            \n\n            # Generate comprehensive report\n\n            self._generate_batch_report(df_results, df_time, output_dir)\n\n            \n\n            return df_results, df_time\n\n            \n\n        except Exception as e:\n\n            self.logger.error(f\"Batch analysis failed: {str(e)}\")\n\n            raise\n\n\n    def comparative_analysis(self, \n\n                           experiments: Dict[str, Dict],\n\n                           output_name: str = \"comparative_analysis\") -> Dict[str, Tuple[pd.DataFrame, pd.DataFrame]]:\n\n        \"\"\"\n\n        Compare multiple experiments/methods\n\n        \n\n        Args:\n\n            experiments: Dict with experiment names as keys and config dicts as values\n\n                        Each config should have 'pred_dir', 'native_path', 'scores'\n\n            output_name: Name for output directory\n\n            \n\n        Returns:\n\n            Dict of experiment results\n\n        \"\"\"\n\n        self.logger.info(f\"Starting comparative analysis with {len(experiments)} experiments\")\n\n        \n\n        # Create output directory\n\n        output_dir = self.base_dir / output_name\n\n        output_dir.mkdir(exist_ok=True)\n\n        \n\n        all_results = {}\n\n        \n\n        for exp_name, config in experiments.items():\n\n            self.logger.info(f\"Running experiment: {exp_name}\")\n\n            \n\n            exp_output_dir = output_dir / exp_name\n\n            exp_output_dir.mkdir(exist_ok=True)\n\n            \n\n            # Setup RNAdvisor for this experiment\n\n            rnadvisor_cli = RNAdvisorCLI(\n\n                pred_dir=config['pred_dir'],\n\n                native_path=config['native_path'],\n\n                out_path=str(exp_output_dir / \"results.csv\"),\n\n                scores=config.get('scores', self.comprehensive_scores)\n\n            )\n\n            \n\n            try:\n\n                # Run analysis\n\n                df_results, df_time = rnadvisor_cli.predict()\n\n                \n\n                # Add experiment identifier\n\n                df_results['experiment'] = exp_name\n\n                df_time['experiment'] = exp_name\n\n                \n\n                # Save results\n\n                df_results.to_csv(exp_output_dir / \"results.csv\", index=False)\n\n                df_time.to_csv(exp_output_dir / \"timing.csv\", index=False)\n\n                \n\n                all_results[exp_name] = (df_results, df_time)\n\n                \n\n                self.logger.info(f\"Experiment {exp_name} completed\")\n\n                \n\n            except Exception as e:\n\n                self.logger.error(f\"Experiment {exp_name} failed: {str(e)}\")\n\n                continue\n\n        \n\n        # Generate comparative report\n\n        self._generate_comparative_report(all_results, output_dir)\n\n        \n\n        return all_results\n\n\n    def pipeline_integration(self, \n\n                           config_file: str,\n\n                           output_name: str = \"pipeline_analysis\") -> Dict:\n\n        \"\"\"\n\n        Run analysis based on configuration file for pipeline integration\n\n        \n\n        Args:\n\n            config_file: Path to JSON configuration file\n\n            output_name: Name for output directory\n\n            \n\n        Returns:\n\n            Dict with analysis results and metadata\n\n        \"\"\"\n\n        self.logger.info(f\"Starting pipeline integration from config: {config_file}\")\n\n        \n\n        # Load configuration\n\n        with open(config_file, 'r') as f:\n\n            config = json.load(f)\n\n        \n\n        # Create output directory\n\n        output_dir = self.base_dir / output_name\n\n        output_dir.mkdir(exist_ok=True)\n\n        \n\n        results = {\n\n            'config': config,\n\n            'timestamp': datetime.now().isoformat(),\n\n            'results': {},\n\n            'summary': {}\n\n        }\n\n        \n\n        # Run analysis based on config type\n\n        analysis_type = config.get('analysis_type', 'single')\n\n        \n\n        if analysis_type == 'single':\n\n            df_results, df_time = self.single_structure_analysis(\n\n                pred_path=config['pred_path'],\n\n                native_path=config['native_path'],\n\n                scores=config.get('scores'),\n\n                output_name=output_name\n\n            )\n\n            results['results']['single'] = {\n\n                'results': df_results.to_dict('records'),\n\n                'timing': df_time.to_dict('records')\n\n            }\n\n            \n\n        elif analysis_type == 'batch':\n\n            df_results, df_time = self.batch_analysis(\n\n                pred_dir=config['pred_dir'],\n\n                native_path=config['native_path'],\n\n                scores=config.get('scores'),\n\n                output_name=output_name\n\n            )\n\n            results['results']['batch'] = {\n\n                'results': df_results.to_dict('records'),\n\n                'timing': df_time.to_dict('records')\n\n            }\n\n            \n\n        elif analysis_type == 'comparative':\n\n            all_results = self.comparative_analysis(\n\n                experiments=config['experiments'],\n\n                output_name=output_name\n\n            )\n\n            results['results']['comparative'] = {}\n\n            for exp_name, (df_results, df_time) in all_results.items():\n\n                results['results']['comparative'][exp_name] = {\n\n                    'results': df_results.to_dict('records'),\n\n                    'timing': df_time.to_dict('records')\n\n                }\n\n        \n\n        # Save pipeline results\n\n        with open(output_dir / 'pipeline_results.json', 'w') as f:\n\n            json.dump(results, f, indent=2, default=str)\n\n        \n\n        return results\n\n\n    def _generate_single_report(self, df_results: pd.DataFrame, df_time: pd.DataFrame, output_dir: Path):\n\n        \"\"\"Generate report for single structure analysis\"\"\"\n\n        \n\n        # Create visualization\n\n        plt.figure(figsize=(12, 8))\n\n        \n\n        # Plot scores\n\n        numeric_cols = df_results.select_dtypes(include=[np.number]).columns\n\n        if len(numeric_cols) > 1:\n\n            plt.subplot(2, 2, 1)\n\n            df_results[numeric_cols].plot(kind='bar')\n\n            plt.title('Quality Scores')\n\n            plt.xticks(rotation=45)\n\n            \n\n            plt.subplot(2, 2, 2)\n\n            df_time.plot(x='score', y='time', kind='bar')\n\n            plt.title('Computation Time by Score')\n\n            plt.xticks(rotation=45)\n\n        \n\n        plt.tight_layout()\n\n        plt.savefig(output_dir / 'analysis_report.png', dpi=300, bbox_inches='tight')\n\n        plt.close()\n\n        \n\n        # Generate summary statistics\n\n        summary = {\n\n            'analysis_type': 'single_structure',\n\n            'timestamp': datetime.now().isoformat(),\n\n            'scores_calculated': len(df_results.columns) - 1,  # Exclude filename column\n\n            'total_time': df_time['time'].sum() if 'time' in df_time.columns else 0,\n\n            'best_scores': {}\n\n        }\n\n        \n\n        # Find best scores (lower is better for most metrics)\n\n        for col in numeric_cols:\n\n            if col not in ['filename']:\n\n                summary['best_scores'][col] = float(df_results[col].iloc[0]) if len(df_results) > 0 else None\n\n        \n\n        with open(output_dir / 'summary.json', 'w') as f:\n\n            json.dump(summary, f, indent=2)\n\n\n    def _generate_batch_report(self, df_results: pd.DataFrame, df_time: pd.DataFrame, output_dir: Path):\n\n        \"\"\"Generate report for batch analysis\"\"\"\n\n        \n\n        # Create visualizations\n\n        fig, axes = plt.subplots(2, 2, figsize=(15, 12))\n\n        \n\n        # Plot 1: Score distribution\n\n        numeric_cols = df_results.select_dtypes(include=[np.number]).columns\n\n        if len(numeric_cols) > 0:\n\n            df_results[numeric_cols].hist(bins=20, ax=axes[0,0])\n\n            axes[0,0].set_title('Score Distributions')\n\n        \n\n        # Plot 2: Best structures by different metrics\n\n        if 'rmsd' in df_results.columns:\n\n            best_rmsd = df_results.nsmallest(5, 'rmsd')\n\n            axes[0,1].bar(range(len(best_rmsd)), best_rmsd['rmsd'])\n\n            axes[0,1].set_title('Top 5 Structures by RMSD')\n\n            axes[0,1].set_xlabel('Structure Rank')\n\n            axes[0,1].set_ylabel('RMSD')\n\n        \n\n        # Plot 3: Correlation heatmap\n\n        if len(numeric_cols) > 1:\n\n            correlation_matrix = df_results[numeric_cols].corr()\n\n            sns.heatmap(correlation_matrix, annot=True, cmap='coolwarm', ax=axes[1,0])\n\n            axes[1,0].set_title('Score Correlations')\n\n        \n\n        # Plot 4: Timing analysis\n\n        if 'time' in df_time.columns:\n\n            df_time.plot(x='score', y='time', kind='bar', ax=axes[1,1])\n\n            axes[1,1].set_title('Computation Time by Score')\n\n            axes[1,1].tick_params(axis='x', rotation=45)\n\n        \n\n        plt.tight_layout()\n\n        plt.savefig(output_dir / 'batch_analysis_report.png', dpi=300, bbox_inches='tight')\n\n        plt.close()\n\n        \n\n        # Generate comprehensive summary\n\n        summary = {\n\n            'analysis_type': 'batch',\n\n            'timestamp': datetime.now().isoformat(),\n\n            'total_structures': len(df_results),\n\n            'scores_calculated': len(numeric_cols),\n\n            'total_computation_time': df_time['time'].sum() if 'time' in df_time.columns else 0,\n\n            'statistics': {}\n\n        }\n\n        \n\n        # Calculate statistics for each score\n\n        for col in numeric_cols:\n\n            if col not in ['filename']:\n\n                summary['statistics'][col] = {\n\n                    'mean': float(df_results[col].mean()),\n\n                    'std': float(df_results[col].std()),\n\n                    'min': float(df_results[col].min()),\n\n                    'max': float(df_results[col].max()),\n\n                    'median': float(df_results[col].median())\n\n                }\n\n        \n\n        with open(output_dir / 'batch_summary.json', 'w') as f:\n\n            json.dump(summary, f, indent=2)\n\n\n    def _generate_comparative_report(self, all_results: Dict, output_dir: Path):\n\n        \"\"\"Generate report for comparative analysis\"\"\"\n\n        \n\n        # Combine results for comparison\n\n        combined_results = []\n\n        combined_timing = []\n\n        \n\n        for exp_name, (df_results, df_time) in all_results.items():\n\n            combined_results.append(df_results)\n\n            combined_timing.append(df_time)\n\n        \n\n        if combined_results:\n\n            combined_df = pd.concat(combined_results, ignore_index=True)\n\n            combined_timing_df = pd.concat(combined_timing, ignore_index=True)\n\n            \n\n            # Create comparison visualizations\n\n            fig, axes = plt.subplots(2, 2, figsize=(15, 12))\n\n            \n\n            # Plot 1: RMSD comparison\n\n            if 'rmsd' in combined_df.columns:\n\n                combined_df.boxplot(column='rmsd', by='experiment', ax=axes[0,0])\n\n                axes[0,0].set_title('RMSD Comparison by Experiment')\n\n                axes[0,0].set_xlabel('Experiment')\n\n                axes[0,0].set_ylabel('RMSD')\n\n            \n\n            # Plot 2: Score comparison heatmap\n\n            numeric_cols = combined_df.select_dtypes(include=[np.number]).columns\n\n            if len(numeric_cols) > 1:\n\n                exp_means = combined_df.groupby('experiment')[numeric_cols].mean()\n\n                sns.heatmap(exp_means.T, annot=True, cmap='RdYlBu_r', ax=axes[0,1])\n\n                axes[0,1].set_title('Mean Scores by Experiment')\n\n            \n\n            # Plot 3: Timing comparison\n\n            if 'time' in combined_timing_df.columns:\n\n                timing_summary = combined_timing_df.groupby(['experiment', 'score'])['time'].sum().unstack()\n\n                timing_summary.plot(kind='bar', ax=axes[1,0])\n\n                axes[1,0].set_title('Total Computation Time by Experiment')\n\n                axes[1,0].tick_params(axis='x', rotation=45)\n\n            \n\n            # Plot 4: Structure count\n\n            structure_counts = combined_df['experiment'].value_counts()\n\n            structure_counts.plot(kind='bar', ax=axes[1,1])\n\n            axes[1,1].set_title('Number of Structures by Experiment')\n\n            axes[1,1].tick_params(axis='x', rotation=45)\n\n            \n\n            plt.tight_layout()\n\n            plt.savefig(output_dir / 'comparative_analysis_report.png', dpi=300, bbox_inches='tight')\n\n            plt.close()\n\n            \n\n            # Save combined results\n\n            combined_df.to_csv(output_dir / 'combined_results.csv', index=False)\n\n            combined_timing_df.to_csv(output_dir / 'combined_timing.csv', index=False)\n\n\n\ndef main():\n\n    \"\"\"\n\n    Example usage of RNAdvisor Integration\n\n    \"\"\"\n\n    \n\n    # Initialize integrator\n\n    integrator = RNAdvisorIntegrator(base_dir=\"rna_analysis_output\")\n\n    \n\n    # Example 1: Single structure analysis\n\n    print(\"=== Example 1: Single Structure Analysis ===\")\n\n    try:\n\n        results, timing = integrator.single_structure_analysis(\n\n            pred_path=\"data/example/PREDS/prediction1.pdb\",\n\n            native_path=\"data/example/NATIVE/R1107.pdb\",\n\n            scores=[\"rmsd\", \"inf\", \"mcq\", \"lddt\"],\n\n            output_name=\"example_single\"\n\n        )\n\n        print(f\"Single analysis completed. RMSD: {results['rmsd'].iloc[0] if 'rmsd' in results.columns else 'N/A'}\")\n\n    except Exception as e:\n\n        print(f\"Single analysis failed: {e}\")\n\n    \n\n    # Example 2: Batch analysis\n\n    print(\"\\n=== Example 2: Batch Analysis ===\")\n\n    try:\n\n        results, timing = integrator.batch_analysis(\n\n            pred_dir=\"data/example/PREDS/\",\n\n            native_path=\"data/example/NATIVE/R1107.pdb\",\n\n            scores=[\"rmsd\", \"inf\", \"mcq\", \"lddt\", \"tm-score\"],\n\n            output_name=\"example_batch\"\n\n        )\n\n        print(f\"Batch analysis completed. Analyzed {len(results)} structures\")\n\n    except Exception as e:\n\n        print(f\"Batch analysis failed: {e}\")\n\n    \n\n    # Example 3: Comparative analysis\n\n    print(\"\\n=== Example 3: Comparative Analysis ===\")\n\n    experiments = {\n\n        \"method_a\": {\n\n            \"pred_dir\": \"data/method_a/predictions/\",\n\n            \"native_path\": \"data/native/structure.pdb\",\n\n            \"scores\": [\"rmsd\", \"inf\", \"mcq\"]\n\n        },\n\n        \"method_b\": {\n\n            \"pred_dir\": \"data/method_b/predictions/\",\n\n            \"native_path\": \"data/native/structure.pdb\", \n\n            \"scores\": [\"rmsd\", \"inf\", \"mcq\"]\n\n        }\n\n    }\n\n    \n\n    try:\n\n        comp_results = integrator.comparative_analysis(\n\n            experiments=experiments,\n\n            output_name=\"example_comparative\"\n\n        )\n\n        print(f\"Comparative analysis completed for {len(comp_results)} experiments\")\n\n    except Exception as e:\n\n        print(f\"Comparative analysis failed: {e}\")\n\n    \n\n    # Example 4: Pipeline integration with config file\n\n    print(\"\\n=== Example 4: Pipeline Integration ===\")\n\n    \n\n    # Create example config file\n\n    config = {\n\n        \"analysis_type\": \"batch\",\n\n        \"pred_dir\": \"data/example/PREDS/\",\n\n        \"native_path\": \"data/example/NATIVE/R1107.pdb\",\n\n        \"scores\": [\"rmsd\", \"inf\", \"mcq\", \"lddt\", \"tm-score\", \"gdt-ts\"],\n\n        \"output_options\": {\n\n            \"generate_plots\": True,\n\n            \"save_timing\": True,\n\n            \"z_score\": True\n\n        }\n\n    }\n\n    \n\n    config_path = \"pipeline_config.json\"\n\n    with open(config_path, 'w') as f:\n\n        json.dump(config, f, indent=2)\n\n    \n\n    try:\n\n        pipeline_results = integrator.pipeline_integration(\n\n            config_file=config_path,\n\n            output_name=\"example_pipeline\"\n\n        )\n\n        print(f\"Pipeline integration completed. Config: {config_path}\")\n\n    except Exception as e:\n\n        print(f\"Pipeline integration failed: {e}\")\n\n\n\nif __name__ == \"__main__\":\n\n    main()\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T01:26:53.620336Z","iopub.execute_input":"2025-09-25T01:26:53.620565Z","iopub.status.idle":"2025-09-25T01:26:53.64942Z","shell.execute_reply.started":"2025-09-25T01:26:53.62054Z","shell.execute_reply":"2025-09-25T01:26:53.648615Z"}},"outputs":[],"execution_count":null}]}