bio-acdc / demo.py
AliSaadatV's picture
Upload demo.py
be57fe4 verified
Raw History Blame Contribute Delete
8.35 kB
"""
Demo script for Bio-ACDC framework.
Runs a minimal coevolution experiment with small models on CPU.
"""
import logging
import os
import sys
# Setup logging
logging.basicConfig(
level=logging.INFO,
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s',
handlers=[logging.StreamHandler(sys.stdout)]
)
from bio_acdc import BioACDC, BioACDCConfig
from bio_acdc.tasks import BioTaskPool
from bio_acdc.mergers import LinearMerge
from bio_acdc.mutators import GaussianNoiseMutator
from bio_acdc.evaluator import BioEvaluator
from bio_acdc.archive import BioArchive
def demo_task_pool():
"""Demonstrate task pool functionality."""
print("\n" + "="*60)
print("DEMO 1: Task Pool")
print("="*60)
pool = BioTaskPool(seed=42)
print(f"\nInitial pool has {len(pool.tasks)} tasks")
# Sample tasks
tasks = pool.get_tasks(n=5)
for i, task in enumerate(tasks):
print(f"\nTask {i+1}: {task.task_id}")
print(f" Type: {task.task_type}, Family: {task.task_family}")
print(f" Prompt: {task.prompt[:80]}...")
print(f" Difficulty: {task.difficulty:.2f}")
print(f" Expected: {task.expected_answer}")
# Generate new tasks
archive = BioArchive(archive_size=5)
new_tasks = pool.generate_new_tasks(archive, num_tasks=3)
print(f"\nGenerated {len(new_tasks)} new tasks")
print(f"Pool size now: {len(pool.tasks)}")
return pool
def demo_archive():
"""Demonstrate archive functionality."""
print("\n" + "="*60)
print("DEMO 2: Dominated Novelty Search Archive")
print("="*60)
archive = BioArchive(archive_size=5, k_neighbors=2)
# Simulate adding solutions
from bio_acdc.archive import BioSolution
solutions = [
BioSolution("model_A", 0.7, {"task_1": 0.8, "task_2": 0.6, "task_3": 0.9}, generation=1),
BioSolution("model_B", 0.6, {"task_1": 0.5, "task_2": 0.7, "task_3": 0.5}, generation=1),
BioSolution("model_C", 0.5, {"task_1": 0.9, "task_2": 0.4, "task_3": 0.3}, generation=1),
]
for sol in solutions:
archive.add_solution(sol)
print(f"\nAdded {len(solutions)} solutions to archive")
print(f"Coverage: {archive.get_coverage():.2%}")
best = archive.get_best()
print(f"Best solution: {best.model_path} (fitness={best.fitness:.3f})")
# Add new solutions
new_solutions = [
BioSolution("model_D", 0.65, {"task_1": 0.6, "task_2": 0.8, "task_3": 0.4}, generation=2),
BioSolution("model_E", 0.55, {"task_1": 0.3, "task_2": 0.9, "task_3": 0.2}, generation=2),
]
archive.update(new_solutions)
print(f"\nAfter update: {len(archive.solutions)} solutions")
best = archive.get_best()
print(f"New best: {best.model_path} (fitness={best.fitness:.3f})")
print(f"Coverage: {archive.get_coverage():.2%}")
def demo_merging():
"""Demonstrate model merging with toy models."""
print("\n" + "="*60)
print("DEMO 3: Model Merging (Toy Example)")
print("="*60)
import torch
import json
from safetensors.torch import save_file
# Create two toy models
model_dir = "/tmp/toy_models"
os.makedirs(model_dir, exist_ok=True)
# Create simple state dicts
model_a_weights = {
"layer1.weight": torch.randn(10, 10) * 0.1,
"layer1.bias": torch.randn(10) * 0.1,
}
model_b_weights = {
"layer1.weight": torch.randn(10, 10) * 0.1,
"layer1.bias": torch.randn(10) * 0.1,
}
model_a_path = os.path.join(model_dir, "model_a")
model_b_path = os.path.join(model_dir, "model_b")
os.makedirs(model_a_path, exist_ok=True)
os.makedirs(model_b_path, exist_ok=True)
save_file(model_a_weights, os.path.join(model_a_path, "model.safetensors"))
save_file(model_b_weights, os.path.join(model_b_path, "model.safetensors"))
# Write config
config = {"num_layers": 1, "hidden_size": 10}
with open(os.path.join(model_a_path, "config.json"), "w") as f:
json.dump(config, f)
with open(os.path.join(model_b_path, "config.json"), "w") as f:
json.dump(config, f)
print(f"\nCreated toy models at {model_dir}")
# Linear merge
from bio_acdc.mergers import LinearMerge
merger = LinearMerge(weights=[0.6, 0.4])
merged_path = merger.merge(
parent_paths=[model_a_path, model_b_path],
save_path=os.path.join(model_dir, "merged"),
)
print(f"Merged model saved to: {merged_path}")
# Verify merge
from safetensors.torch import load_file
merged = load_file(os.path.join(merged_path, "model.safetensors"))
print(f"Merged weights shape: {merged['layer1.weight'].shape}")
# Mutation
from bio_acdc.mutators import GaussianNoiseMutator
mutator = GaussianNoiseMutator(std=0.01)
mutated_path = mutator.mutate(merged_path, os.path.join(model_dir, "mutated"))
print(f"Mutated model saved to: {mutated_path}")
def demo_bio_acdc_mock():
"""Demonstrate full Bio-ACDC loop with mock components."""
print("\n" + "="*60)
print("DEMO 4: Full Bio-ACDC Loop (Mock)")
print("="*60)
from bio_acdc.core import BioACDCConfig
from bio_acdc.archive import BioSolution, BioArchive
# Mock components that don't need real models
class MockTaskPool:
def __init__(self):
self.tasks = []
self.rng = __import__('numpy').random.RandomState(42)
def get_tasks(self, n=10):
return []
def generate_new_tasks(self, archive, num_tasks=5, difficulty_weights=None):
return []
class MockEvaluator:
def evaluate_model(self, model_path, tasks):
# Return random scores
import random
return {f"task_{i}": random.random() for i in range(10)}
class MockMerger:
def merge(self, parent_paths, save_path, base_model_path=None):
os.makedirs(save_path, exist_ok=True)
return save_path
class MockMutator:
def mutate(self, model_path, save_path):
return save_path
config = BioACDCConfig(
archive_size=10,
num_generations=3,
offspring_per_gen=2,
seed_model_paths=["/tmp/model_a", "/tmp/model_b"],
output_dir="/tmp/bio_acdc_mock",
)
task_pool = MockTaskPool()
evaluator = MockEvaluator()
merger = MockMerger()
mutator = MockMutator()
print("\nRunning mock Bio-ACDC for 3 generations...")
# Simulate
archive = BioArchive(archive_size=10)
for gen in range(1, 4):
print(f"\n--- Generation {gen} ---")
# Mock offspring
new_sols = []
for i in range(2):
sol = BioSolution(
model_path=f"/tmp/gen_{gen}_ind_{i}",
fitness=0.5 + 0.1 * (gen + i),
skill_vector={f"task_{j}": 0.5 + 0.1 * j for j in range(10)},
generation=gen,
)
new_sols.append(sol)
print(f" Created offspring {i}: fitness={sol.fitness:.3f}")
# Update archive
if gen == 1:
for sol in new_sols:
archive.add_solution(sol)
else:
archive.update(new_sols)
best = archive.get_best()
print(f" Archive size: {len(archive.solutions)}")
if best:
print(f" Best: {best.model_path} (fitness={best.fitness:.3f})")
print("\nMock evolution complete!")
def main():
"""Run all demos."""
print("\n" + "="*60)
print("BIO-ACDC FRAMEWORK DEMO")
print("Biological Sequence Model Coevolution")
print("="*60)
demo_task_pool()
demo_archive()
demo_merging()
demo_bio_acdc_mock()
print("\n" + "="*60)
print("ALL DEMOS COMPLETE!")
print("="*60)
print("\nBio-ACDC is ready for use with real biological LMs.")
print("\nNext steps:")
print("1. Install requirements: pip install -r requirements.txt")
print("2. Download seed models (e.g., facebook/esm2_t33_650M_UR50D)")
print("3. Run full coevolution: python -c 'from demo_bio_acdc import run_full; run_full()'")
print("\nSee README.md for detailed usage instructions.")
if __name__ == "__main__":
main()