""" Demo script for Bio-ACDC framework. Runs a minimal coevolution experiment with small models on CPU. """ import logging import os import sys # Setup logging logging.basicConfig( level=logging.INFO, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s', handlers=[logging.StreamHandler(sys.stdout)] ) from bio_acdc import BioACDC, BioACDCConfig from bio_acdc.tasks import BioTaskPool from bio_acdc.mergers import LinearMerge from bio_acdc.mutators import GaussianNoiseMutator from bio_acdc.evaluator import BioEvaluator from bio_acdc.archive import BioArchive def demo_task_pool(): """Demonstrate task pool functionality.""" print("\n" + "="*60) print("DEMO 1: Task Pool") print("="*60) pool = BioTaskPool(seed=42) print(f"\nInitial pool has {len(pool.tasks)} tasks") # Sample tasks tasks = pool.get_tasks(n=5) for i, task in enumerate(tasks): print(f"\nTask {i+1}: {task.task_id}") print(f" Type: {task.task_type}, Family: {task.task_family}") print(f" Prompt: {task.prompt[:80]}...") print(f" Difficulty: {task.difficulty:.2f}") print(f" Expected: {task.expected_answer}") # Generate new tasks archive = BioArchive(archive_size=5) new_tasks = pool.generate_new_tasks(archive, num_tasks=3) print(f"\nGenerated {len(new_tasks)} new tasks") print(f"Pool size now: {len(pool.tasks)}") return pool def demo_archive(): """Demonstrate archive functionality.""" print("\n" + "="*60) print("DEMO 2: Dominated Novelty Search Archive") print("="*60) archive = BioArchive(archive_size=5, k_neighbors=2) # Simulate adding solutions from bio_acdc.archive import BioSolution solutions = [ BioSolution("model_A", 0.7, {"task_1": 0.8, "task_2": 0.6, "task_3": 0.9}, generation=1), BioSolution("model_B", 0.6, {"task_1": 0.5, "task_2": 0.7, "task_3": 0.5}, generation=1), BioSolution("model_C", 0.5, {"task_1": 0.9, "task_2": 0.4, "task_3": 0.3}, generation=1), ] for sol in solutions: archive.add_solution(sol) print(f"\nAdded {len(solutions)} solutions to archive") print(f"Coverage: {archive.get_coverage():.2%}") best = archive.get_best() print(f"Best solution: {best.model_path} (fitness={best.fitness:.3f})") # Add new solutions new_solutions = [ BioSolution("model_D", 0.65, {"task_1": 0.6, "task_2": 0.8, "task_3": 0.4}, generation=2), BioSolution("model_E", 0.55, {"task_1": 0.3, "task_2": 0.9, "task_3": 0.2}, generation=2), ] archive.update(new_solutions) print(f"\nAfter update: {len(archive.solutions)} solutions") best = archive.get_best() print(f"New best: {best.model_path} (fitness={best.fitness:.3f})") print(f"Coverage: {archive.get_coverage():.2%}") def demo_merging(): """Demonstrate model merging with toy models.""" print("\n" + "="*60) print("DEMO 3: Model Merging (Toy Example)") print("="*60) import torch import json from safetensors.torch import save_file # Create two toy models model_dir = "/tmp/toy_models" os.makedirs(model_dir, exist_ok=True) # Create simple state dicts model_a_weights = { "layer1.weight": torch.randn(10, 10) * 0.1, "layer1.bias": torch.randn(10) * 0.1, } model_b_weights = { "layer1.weight": torch.randn(10, 10) * 0.1, "layer1.bias": torch.randn(10) * 0.1, } model_a_path = os.path.join(model_dir, "model_a") model_b_path = os.path.join(model_dir, "model_b") os.makedirs(model_a_path, exist_ok=True) os.makedirs(model_b_path, exist_ok=True) save_file(model_a_weights, os.path.join(model_a_path, "model.safetensors")) save_file(model_b_weights, os.path.join(model_b_path, "model.safetensors")) # Write config config = {"num_layers": 1, "hidden_size": 10} with open(os.path.join(model_a_path, "config.json"), "w") as f: json.dump(config, f) with open(os.path.join(model_b_path, "config.json"), "w") as f: json.dump(config, f) print(f"\nCreated toy models at {model_dir}") # Linear merge from bio_acdc.mergers import LinearMerge merger = LinearMerge(weights=[0.6, 0.4]) merged_path = merger.merge( parent_paths=[model_a_path, model_b_path], save_path=os.path.join(model_dir, "merged"), ) print(f"Merged model saved to: {merged_path}") # Verify merge from safetensors.torch import load_file merged = load_file(os.path.join(merged_path, "model.safetensors")) print(f"Merged weights shape: {merged['layer1.weight'].shape}") # Mutation from bio_acdc.mutators import GaussianNoiseMutator mutator = GaussianNoiseMutator(std=0.01) mutated_path = mutator.mutate(merged_path, os.path.join(model_dir, "mutated")) print(f"Mutated model saved to: {mutated_path}") def demo_bio_acdc_mock(): """Demonstrate full Bio-ACDC loop with mock components.""" print("\n" + "="*60) print("DEMO 4: Full Bio-ACDC Loop (Mock)") print("="*60) from bio_acdc.core import BioACDCConfig from bio_acdc.archive import BioSolution, BioArchive # Mock components that don't need real models class MockTaskPool: def __init__(self): self.tasks = [] self.rng = __import__('numpy').random.RandomState(42) def get_tasks(self, n=10): return [] def generate_new_tasks(self, archive, num_tasks=5, difficulty_weights=None): return [] class MockEvaluator: def evaluate_model(self, model_path, tasks): # Return random scores import random return {f"task_{i}": random.random() for i in range(10)} class MockMerger: def merge(self, parent_paths, save_path, base_model_path=None): os.makedirs(save_path, exist_ok=True) return save_path class MockMutator: def mutate(self, model_path, save_path): return save_path config = BioACDCConfig( archive_size=10, num_generations=3, offspring_per_gen=2, seed_model_paths=["/tmp/model_a", "/tmp/model_b"], output_dir="/tmp/bio_acdc_mock", ) task_pool = MockTaskPool() evaluator = MockEvaluator() merger = MockMerger() mutator = MockMutator() print("\nRunning mock Bio-ACDC for 3 generations...") # Simulate archive = BioArchive(archive_size=10) for gen in range(1, 4): print(f"\n--- Generation {gen} ---") # Mock offspring new_sols = [] for i in range(2): sol = BioSolution( model_path=f"/tmp/gen_{gen}_ind_{i}", fitness=0.5 + 0.1 * (gen + i), skill_vector={f"task_{j}": 0.5 + 0.1 * j for j in range(10)}, generation=gen, ) new_sols.append(sol) print(f" Created offspring {i}: fitness={sol.fitness:.3f}") # Update archive if gen == 1: for sol in new_sols: archive.add_solution(sol) else: archive.update(new_sols) best = archive.get_best() print(f" Archive size: {len(archive.solutions)}") if best: print(f" Best: {best.model_path} (fitness={best.fitness:.3f})") print("\nMock evolution complete!") def main(): """Run all demos.""" print("\n" + "="*60) print("BIO-ACDC FRAMEWORK DEMO") print("Biological Sequence Model Coevolution") print("="*60) demo_task_pool() demo_archive() demo_merging() demo_bio_acdc_mock() print("\n" + "="*60) print("ALL DEMOS COMPLETE!") print("="*60) print("\nBio-ACDC is ready for use with real biological LMs.") print("\nNext steps:") print("1. Install requirements: pip install -r requirements.txt") print("2. Download seed models (e.g., facebook/esm2_t33_650M_UR50D)") print("3. Run full coevolution: python -c 'from demo_bio_acdc import run_full; run_full()'") print("\nSee README.md for detailed usage instructions.") if __name__ == "__main__": main()