Spaces:
Running
Running
Download solar_eval/cli/commands/runs.py from dev-strender/proofread-demo: direct link, hf CLI and curl.
- Browser
- Download file 41.7 kB
-
https://huggingface.co/spaces/dev-strender/proofread-demo/resolve/483134ace86f21c444532ef540c77505aadb29b9/solar_eval/cli/commands/runs.py
- Command line
-
hf download hf://spaces/dev-strender/proofread-demo@483134ace86f21c444532ef540c77505aadb29b9/solar_eval/cli/commands/runs.py
-
curl -L -o runs.py https://huggingface.co/spaces/dev-strender/proofread-demo/resolve/483134ace86f21c444532ef540c77505aadb29b9/solar_eval/cli/commands/runs.py
41.7 kB
| """Run management commands โ local execution.""" | |
| import asyncio | |
| import json | |
| import subprocess | |
| import time | |
| from datetime import datetime | |
| from pathlib import Path | |
| from typing import Any | |
| import click | |
| from rich.progress import Progress, SpinnerColumn, BarColumn, TextColumn, TimeElapsedColumn | |
| from solar_eval.cli.commands.projects import _resolve_project_id | |
| from solar_eval.cli.formatters import format_runs_table, format_results_table, format_scores | |
| #: run ์ด ๋๋ ๋ค ์๋ ์ ์ฌ์ ํ์ฉํ๋ ์๊ฐ. ์ ์ฒด ์์นด์ด๋ธ๊ฐ ์๋๋ผ ํ๋ก์ ํธ ํ๋๋ง | |
| #: ์ ์ฌํ๋ฏ๋ก ๋ณดํต ์ ์ด๋ฉด ๋๋๋ค โ ๋์ด๊ฐ๋ฉด ๋ญ๊ฐ ์๋ชป๋ ๊ฒ์ด๋ผ ๊ธฐ๋ค๋ฆฌ์ง ์๋๋ค. | |
| INGEST_TIMEOUT_SECONDS = 180 | |
| def runs_group() -> None: | |
| """Manage evaluation runs.""" | |
| def list_runs(ctx: click.Context, project: str, task: str | None) -> None: | |
| """List runs for a project.""" | |
| cfg = ctx.obj["config"] | |
| if not cfg.is_remote: | |
| # List local run directories | |
| artifacts = cfg.artifacts_dir(project) | |
| if not artifacts.exists(): | |
| click.secho("No local runs found.", fg="yellow") | |
| return | |
| runs = [] | |
| for d in sorted(artifacts.iterdir()): | |
| if d.is_dir() and (d / "run.json").exists(): | |
| run_data = json.loads((d / "run.json").read_text()) | |
| if task and run_data.get("task") != task: | |
| continue | |
| runs.append( | |
| { | |
| "id": d.name, | |
| "task": run_data.get("task", "-"), | |
| "model": run_data.get("model", "-"), | |
| "prompt": run_data.get("prompt", run_data.get("prompt_version", "-")), | |
| "pipeline": run_data.get("pipeline", "-"), | |
| "status": run_data.get("status", "unknown"), | |
| "total_samples": run_data.get("total_samples", 0), | |
| "completed_samples": run_data.get("completed_samples", 0), | |
| "created_at": run_data.get("started_at", "-"), | |
| } | |
| ) | |
| if not runs: | |
| click.secho("No local runs found.", fg="yellow") | |
| return | |
| click.secho(format_runs_table(runs), fg="blue") | |
| return | |
| from solar_eval.cli.client import EvalClient | |
| client = EvalClient(cfg.remote_url, cfg.timeout) | |
| project_id = _resolve_project_id(client, project) | |
| params = {"task": task} if task else {} | |
| runs = client.get(f"/api/projects/{project_id}/runs", params=params) | |
| if not runs: | |
| click.secho("No runs found.", fg="yellow") | |
| return | |
| click.secho(format_runs_table(runs), fg="blue") | |
| def start_run( | |
| ctx: click.Context, | |
| project: str, | |
| task: str, | |
| prompt_name: str, | |
| model: str, | |
| pipeline_name: str | None, | |
| reasoning_effort: str | None, | |
| temperature: float, | |
| max_tokens: int, | |
| max_workers: int, | |
| limit: int | None, | |
| note: str | None, | |
| set_options: tuple[str, ...], | |
| no_ingest: bool, | |
| ) -> None: | |
| """Start an evaluation run. | |
| Example: | |
| solar-eval runs start chosun-proofreading -t ci-v1-all \\ | |
| -p dev_260329_pro3_v1 -m solar-pro3 -P 251231_pro3 \\ | |
| --note "v24 SC+judge ๊ฐ ci-v1 ์์๋ prod ๋ฅผ ๋๋์ง" | |
| solar-eval runs start chosun-proofreading -t paragraph-small \\ | |
| -p prompt_dev_v23 -m solar-pro3 -P pipeline_dev_v24 \\ | |
| --set steps.basic_correction.repeat_temperatures='[0.0, 0.5]' \\ | |
| --note "SC ์จ๋ ํญ ์ค์" | |
| """ | |
| cfg = ctx.obj["config"] | |
| _start_local( | |
| cfg, | |
| project=project, | |
| task=task, | |
| prompt_name=prompt_name, | |
| model=model, | |
| pipeline_name=pipeline_name, | |
| reasoning_effort=reasoning_effort, | |
| temperature=temperature, | |
| max_tokens=max_tokens, | |
| max_workers=max_workers, | |
| limit=limit, | |
| note=note, | |
| set_options=set_options, | |
| no_ingest=no_ingest, | |
| ) | |
| def _resolve_prompt(prompts_dir: Path, prompt_name: str) -> Path: | |
| """Resolve prompt from prompts/ directory (directory or YAML file). | |
| Resolution order: | |
| 1. prompts/{name}/ (directory TXT format) | |
| 2. prompts/{name}.yaml (flat YAML layout) | |
| 3. prompts/_library/{name}.yaml (legacy layout) | |
| """ | |
| # Directory TXT format (new) | |
| dir_path = prompts_dir / prompt_name | |
| if dir_path.is_dir(): | |
| return dir_path | |
| # Flat YAML layout | |
| yaml_path = prompts_dir / f"{prompt_name}.yaml" | |
| if yaml_path.exists(): | |
| return yaml_path | |
| # Legacy _library layout | |
| legacy_path = prompts_dir / "_library" / f"{prompt_name}.yaml" | |
| if legacy_path.exists(): | |
| return legacy_path | |
| raise click.ClickException( | |
| f"Prompt '{prompt_name}' not found. Searched:\n" | |
| f" {dir_path}/ (directory)\n" | |
| f" {yaml_path} (YAML)\n" | |
| f" {legacy_path} (legacy)" | |
| ) | |
| def _resolve_pipeline_file(project_dir: Path, pipeline_name: str) -> Path: | |
| """Resolve pipeline YAML from pipelines/ directory.""" | |
| path = project_dir / "pipelines" / f"{pipeline_name}.yaml" | |
| if not path.exists(): | |
| raise click.ClickException(f"Pipeline '{pipeline_name}' not found: {path}") | |
| return path | |
| def _compose_pipeline_config(project_dir: Path, pipeline_name: str, overrides: dict) -> dict: | |
| """ํ์ดํ๋ผ์ธ์ `extends`/override ๊น์ง ํด์ํด ๋๋ ค์ค๋ค. | |
| ์กฐ๋ฆฝ ์คํจ(์๋ base, ์๋ ์คํ ์ ๊ฐ๋ฆฌํค๋ override ๋ฑ)๋ run ์ ์์ํ๊ธฐ ์ ์ | |
| ์ฃฝ์ธ๋ค โ ํจ๊ณผ ์๋ override ๋ก ๋์๊ฐ run ์ ๊ฒฐ๋ก ์ ์กฐ์ฉํ ์ค์ผ์ํจ๋ค. | |
| """ | |
| from solar_eval.core.pipeline_compose import PipelineCompositionError, compose_pipeline | |
| from solar_eval.core.project_loader import load_pipeline_file | |
| pipelines_dir = _resolve_pipeline_file(project_dir, pipeline_name).parent | |
| try: | |
| composed = load_pipeline_file(pipelines_dir, pipeline_name) | |
| if overrides: | |
| composed = compose_pipeline(composed, overrides=overrides) | |
| return composed | |
| except (PipelineCompositionError, FileNotFoundError) as e: | |
| raise click.ClickException(str(e)) from None | |
| def _build_run_name(task: str, pipeline_name: str, prompt_name: str) -> str: | |
| """Build a readable run name: {task}__{pipeline}__{prompt}__{YYMMDD}_{HHMMSS}""" | |
| ts = datetime.now().strftime("%y%m%d_%H%M%S") | |
| return f"{task}__{pipeline_name}__{prompt_name}__{ts}" | |
| _RUN_DIR_MAX_ATTEMPTS = 20 | |
| _RUN_DIR_RETRY_DELAY_SECONDS = 0.05 | |
| def _make_run_dir( | |
| artifacts_dir: Path, task: str, pipeline_name: str, prompt_name: str | |
| ) -> tuple[str, Path]: | |
| """run_name ์ ์ง๊ณ ๊ทธ ๋๋ ํฐ๋ฆฌ๋ฅผ ์์์ ์ผ๋ก ๋ง๋ ๋ค. | |
| ๊ฐ์ ์ด์ ๊ฐ์ (task, pipeline, prompt) ์กฐํฉ์ผ๋ก ๋ run ์ด ์์๋๋ฉด | |
| `_build_run_name` ์ด ๊ฐ์ ์ด๋ฆ์ ๋ธ๋ค. `mkdir(exist_ok=False)` ๋ฅผ ์ฐ๋ฉด ๋ ํ๋ก์ธ์ค๊ฐ | |
| ๋์์ ๊ฐ์ ์ด๋ฆ์ผ๋ก mkdir ํด๋ ์ ํํ ํ๋๋ง ์ฑ๊ณตํ๊ณ , ์ง ์ชฝ์ ์ ํ์์คํฌํ๋ก | |
| ์ฌ์๋ํ๋ค โ ํ์ผ ํฌ๋งท ์์ฒด๋ ์ ๋ฐ๊พธ๋ฏ๋ก evalhub ์ run_id ํ์ฑ๊ณผ ํธํ๋๋ค. | |
| """ | |
| for _ in range(_RUN_DIR_MAX_ATTEMPTS): | |
| run_name = _build_run_name(task, pipeline_name, prompt_name) | |
| run_dir = artifacts_dir / run_name | |
| try: | |
| run_dir.mkdir(parents=True, exist_ok=False) | |
| return run_name, run_dir | |
| except FileExistsError: | |
| time.sleep(_RUN_DIR_RETRY_DELAY_SECONDS) | |
| raise click.ClickException(f"Could not allocate a unique run directory under {artifacts_dir}") | |
| def _config_source(cfg, project: str) -> str: | |
| """run ์ config ์ถ์ฒ โ ๋ ํฌ ๊ด๋ฆฌ ํ๋ก์ ํธ๋ฉด ๋ ํฌ ์๋ ๊ฒฝ๋ก, ์๋๋ฉด ๋ฐ์ดํฐ ํ ์ ๋ ๊ฒฝ๋ก.""" | |
| config_dir = cfg.config_dir(project) | |
| if cfg.is_repo_managed(project) and cfg.repo_root is not None: | |
| return str(config_dir.relative_to(cfg.repo_root)) | |
| return str(config_dir) | |
| def _git_commit(repo_root: Path | None) -> str | None: | |
| """์ฌํ์ฉ ๋์ฅ: ๋ ํฌ HEAD ์ปค๋ฐ. ์์ ํธ๋ฆฌ๊ฐ ๋๋ฌ์ฐ๋ฉด '-dirty' ๋ฅผ ๋ถ์ธ๋ค.""" | |
| if repo_root is None: | |
| return None | |
| try: | |
| head = subprocess.run( | |
| ["git", "-C", str(repo_root), "rev-parse", "--short", "HEAD"], | |
| capture_output=True, | |
| text=True, | |
| check=True, | |
| ).stdout.strip() | |
| status = subprocess.run( | |
| ["git", "-C", str(repo_root), "status", "--porcelain"], | |
| capture_output=True, | |
| text=True, | |
| check=True, | |
| ).stdout.strip() | |
| return f"{head}-dirty" if status else head | |
| except (OSError, subprocess.CalledProcessError): | |
| return None | |
| def _start_local( | |
| cfg, | |
| project: str, | |
| task: str, | |
| prompt_name: str, | |
| model: str, | |
| pipeline_name: str | None, | |
| reasoning_effort: str | None, | |
| temperature: float, | |
| max_tokens: int, | |
| max_workers: int, | |
| limit: int | None, | |
| note: str | None = None, | |
| set_options: tuple[str, ...] = (), | |
| no_ingest: bool = False, | |
| ) -> None: | |
| """Start run locally with the new prompt/model/pipeline interface.""" | |
| from solar_eval.core.dataset_loader import DatasetLoader | |
| from solar_eval.core.fingerprint import pipeline_fingerprint, prompt_fingerprint | |
| from solar_eval.core.pipeline_compose import PipelineCompositionError, parse_set_option | |
| from solar_eval.core.project_loader import load_all_project_configs | |
| from solar_eval.core.runner import BatchRunner | |
| from solar_eval.models.prompt_version import ( | |
| detect_prompt_format, | |
| load_prompt_messages, | |
| load_step_prompts, | |
| ) | |
| from solar_eval.providers import UpstageProvider | |
| from solar_eval.providers.openai_provider import OpenAIProvider | |
| from solar_eval.stores import JsonlStore | |
| # Load project config | |
| configs = load_all_project_configs(cfg.projects_dir, cfg.config_dirs) | |
| project_config = next((c for c in configs if c["name"] == project), None) | |
| if not project_config: | |
| raise click.ClickException(f"Project not found in {cfg.projects_dir}: {project}") | |
| task_config = next((t for t in project_config.get("tasks", []) if t["name"] == task), None) | |
| if not task_config: | |
| raise click.ClickException(f"Task '{task}' not found in project '{project}'") | |
| # Resolve prompt file | |
| prompts_dir = cfg.prompts_dir(project) | |
| prompt_path = _resolve_prompt(prompts_dir, prompt_name) | |
| fmt = detect_prompt_format(prompt_path) | |
| if fmt == "multi_step": | |
| step_prompts = load_step_prompts(prompt_path) | |
| prompt = { | |
| "step_prompts": step_prompts, | |
| "model": model, | |
| "temperature": temperature, | |
| "max_tokens": max_tokens, | |
| } | |
| else: | |
| messages = load_prompt_messages(prompt_path) | |
| prompt = { | |
| "messages": messages, | |
| "model": model, | |
| "temperature": temperature, | |
| "max_tokens": max_tokens, | |
| } | |
| if reasoning_effort: | |
| prompt["reasoning_effort"] = reasoning_effort | |
| # Resolve pipeline โ config ๋ ๋ ํฌ ๊ด๋ฆฌ ํ๋ก์ ํธ๋ฉด git ์ชฝ์ด ์ ๋ณธ. | |
| # `extends` ์์๊ณผ `--set` override ๋ ์ฌ๊ธฐ์ ์ ๋ถ ํด์๋ผ, ์คํ ์์ง์๋ ์์ฑ๋ | |
| # steps ๋ชฉ๋ก๋ง ๋์ด๊ฐ๋ค. | |
| try: | |
| cli_overrides = parse_set_option(set_options) | |
| except PipelineCompositionError as e: | |
| raise click.ClickException(str(e)) from None | |
| effective_pipeline = pipeline_name | |
| if pipeline_name: | |
| pipeline_config = _compose_pipeline_config( | |
| cfg.config_dir(project), pipeline_name, cli_overrides | |
| ) | |
| task_config = {**task_config, "pipeline_config": pipeline_config} | |
| else: | |
| # Use default from project.yaml (already resolved by project_loader) | |
| pc = task_config.get("pipeline_config", {}) | |
| effective_pipeline = pc.get("name", "default") if isinstance(pc, dict) else str(pc) | |
| if cli_overrides: | |
| raise click.ClickException("--set requires an explicit --pipeline/-P") | |
| pipeline_config = pc if isinstance(pc, dict) else {} | |
| # Build run name and directory -- ์ด๋ฆ ์ถฉ๋์ ์ฌ๊ธฐ์ ์์์ ์ผ๋ก ๊ฑธ๋ฌ์ง๋ค(F10). | |
| run_name, run_dir = _make_run_dir( | |
| cfg.artifacts_dir(project), task, effective_pipeline, prompt_name | |
| ) | |
| store = JsonlStore(run_dir) | |
| # Save run metadata | |
| run_meta = { | |
| "task": task, | |
| "model": model, | |
| "prompt": prompt_name, | |
| "pipeline": effective_pipeline, | |
| "temperature": temperature, | |
| "max_tokens": max_tokens, | |
| "reasoning_effort": reasoning_effort, | |
| # ์ ๋๋ ธ๋์ง โ ์์ ์๋ ๋ณ๋ ๋ฉ๋ ธํธ(experiments/*.yaml)๊ฐ ๋ด์์ง๋ง run ๊ณผ ๋ถ๋ฆฌ๋ผ ์์ด | |
| # ์ฐธ์กฐ๊ฐ ๋์ด์ง๊ณ ์ฉ์๋ค. run ๊ณผ ๊ฐ์ ํ์ผ์ ๋๋ฉด ๋์ด์ง ์๊ฐ ์๋ค. | |
| "note": note, | |
| # provenance: ์ด run ์ config(ํ๋กฌํํธยทํ์ดํ๋ผ์ธ)๊ฐ ์ด๋์ ์๋์ง (W&B/DVC ์ ๋์ฅ) | |
| "config_source": _config_source(cfg, project), | |
| "git_commit": _git_commit(cfg.repo_root) if cfg.is_repo_managed(project) else None, | |
| # ๋ด์ฉ ์ง๋ฌธ โ ์ด๋ฆ์ด ์๋๋ผ ๋ด์ฉ์ผ๋ก ์ค์ ์ ์๋ณํ๋ค. ์ด๋ฆ๋ง ๋ค๋ฅธ ์ฌ๋ณธ, | |
| # `extends` ๋ก ์กฐ๋ฆฝํ ๊ฒ, `--set` ์ผ๋ก ์ฆ์์์ ๋ง๋ ๊ฒ์ด ๋ชจ๋ ๊ฐ์ ๊ฐ์ด๋ฉด | |
| # ๊ฐ์ ์ง๋ฌธ์ ๊ฐ๋๋ค (core/fingerprint.py). | |
| "prompt_sha": prompt_fingerprint(prompt_path), | |
| "pipeline_sha": pipeline_fingerprint(pipeline_config), | |
| # ํ์ผ ์์ด ๋๋ฆฐ ์คํ์ด๋ฉด ๋ฌด์์ ๋ฐ๊ฟจ๋์ง๊ฐ ์ฌ๊ธฐ ๋จ๋๋ค. | |
| "pipeline_overrides": cli_overrides or None, | |
| } | |
| # run_dir ์ _make_run_dir ์ด ์ด๋ฏธ ์์์ ์ผ๋ก ๋ง๋ค์๋ค -- ์ฌ๊ธฐ์ ๋ค์ mkdir ํ์ง ์๋๋ค. | |
| (run_dir / "run.json").write_text(json.dumps(run_meta, ensure_ascii=False, indent=2)) | |
| # Log run info | |
| click.secho(f"Prompt: {prompt_name} [{fmt}]", fg="blue") | |
| click.secho(f"Model: {model}", fg="blue") | |
| click.secho(f"Pipeline: {effective_pipeline}", fg="blue") | |
| for path, value in cli_overrides.items(): | |
| click.secho(f" override: {path} = {value!r}", fg="magenta") | |
| click.secho( | |
| f"Fingerprint: prompt {run_meta['prompt_sha']} / pipeline {run_meta['pipeline_sha']}", | |
| fg="blue", | |
| ) | |
| if reasoning_effort: | |
| click.secho(f"Reasoning effort: {reasoning_effort}", fg="blue") | |
| if limit: | |
| click.secho(f"Limit: {limit} samples", fg="yellow") | |
| click.secho(f"Starting local run: {project}/{run_name}", fg="blue") | |
| click.secho(f"Output: {run_dir}", fg="blue") | |
| # Create runner | |
| provider = UpstageProvider() | |
| judge_provider = OpenAIProvider() | |
| dataset_loader = DatasetLoader(base_dir=cfg.projects_dir) | |
| runner = BatchRunner( | |
| store=store, | |
| inference_provider=provider, | |
| judge_provider=judge_provider, | |
| dataset_loader=dataset_loader, | |
| ) | |
| # Run with progress display | |
| _run_with_progress(runner, run_name, project_config, task_config, prompt, max_workers, limit) | |
| _warn_if_incomplete(run_dir) | |
| click.echo(f"\nResults saved to: {run_dir}/") | |
| if not no_ingest: | |
| _ingest_into_evalhub(cfg, project, run_dir) | |
| def _ingest_into_evalhub(cfg, project: str, run_dir: Path) -> None: | |
| """๋ฐฉ๊ธ ๋ง๋ run ์ evalhub ๋ก ์ ์ฌํ๋ค. | |
| **์คํจํด๋ run ์ ์ฑ๊ณต์ด๋ค** โ ๊ฒฐ๊ณผ๋ ์ด๋ฏธ ๋์คํฌ์ ์๊ณ , ์ ์ฌ๋ ๋์ค์ ๋ค์ | |
| ๋๋ฆฌ๋ฉด ๋๋ ๋ฉฑ๋ฑ ์์ ์ด๋ค. ๊ทธ๋์ ์ด๋ค ์คํจ๋ exit code ๋ฅผ ๋ฐ๊พธ์ง ์๊ณ | |
| "๋ฌด์์ด ์ ๋๊ณ ์ด๋ป๊ฒ ๋์ด๋ฆฌ๋์ง"๋ง ์๋ฆฐ๋ค. ์ ์ฌ๋ฅผ ๋ชป ํ๋ค๋ ์ฌ์ค ์์ฒด๋ฅผ | |
| ์กฐ์ฉํ ๋๊ธฐ์ง๋ ์๋๋ค โ ๊ทธ๋ฌ๋ฉด evalhub ํ๋ฉด์ด ๋ก์ ์ฑ๋ก ๋จ๋๋ค. | |
| ๋ ํฌ ๋ฐ standalone ์ค์น์๋ evalhub ์์ฒด๊ฐ ์์ผ๋ฏ๋ก ์กฐ์ฉํ ๊ฑด๋๋ด๋ค. | |
| """ | |
| if cfg.repo_root is None: | |
| return | |
| click.secho("\nevalhub ์ ์ฌ ์ค...", fg="blue") | |
| try: | |
| result = subprocess.run( | |
| ["pnpm", "--filter", "@poc/eval-store", "ingest", "--only", project], | |
| cwd=str(cfg.repo_root), | |
| capture_output=True, | |
| text=True, | |
| timeout=INGEST_TIMEOUT_SECONDS, | |
| ) | |
| except FileNotFoundError: | |
| _warn_ingest_skipped(project, run_dir, "pnpm ์ ์ฐพ์ ์ ์์ต๋๋ค") | |
| return | |
| except subprocess.TimeoutExpired: | |
| _warn_ingest_skipped(project, run_dir, f"{INGEST_TIMEOUT_SECONDS}์ด ์์ ๋๋์ง ์์์ต๋๋ค") | |
| return | |
| except OSError as e: | |
| _warn_ingest_skipped(project, run_dir, str(e)) | |
| return | |
| if result.returncode == 0: | |
| click.secho(f" evalhub ์ ์ฌ ์๋ฃ โ {project}", fg="green") | |
| return | |
| _warn_ingest_skipped(project, run_dir, _ingest_failure_reason(result)) | |
| def _ingest_failure_reason(result: subprocess.CompletedProcess[str]) -> str: | |
| """ingest ์คํจ ์ถ๋ ฅ์์ ์ฌ๋์ด ์ฝ์ ํ ์ค์ ๋ฝ๋๋ค. | |
| stderr ์ ์ฒด๋ node ์คํํธ๋ ์ด์ค๋ผ ๊ทธ๋๋ก ๋ณด์ฌ์ฃผ๋ฉด ์ ์ ์์ธ์ด ๋ฌปํ๋ค. | |
| ๊ฐ์ฅ ํํ ์์ธ(๋ฐฑ์๋ ๋ฏธ๊ธฐ๋)์ ์ฐ๊ฒฐ ๊ฑฐ๋ถ๋ก ๋ํ๋๋ฏ๋ก ๋จผ์ ์ง๋๋ค. | |
| """ | |
| output = f"{result.stderr}\n{result.stdout}" | |
| if "ECONNREFUSED" in output or "connect ECONNREFUSED" in output: | |
| return "Postgres ์ ์ฐ๊ฒฐํ ์ ์์ต๋๋ค (๋ฐฑ์๋๊ฐ ๊บผ์ ธ ์๋ ๊ฒ ๊ฐ์ต๋๋ค)" | |
| for line in result.stderr.splitlines(): | |
| stripped = line.strip() | |
| if stripped and not stripped.startswith("at ") and "node_modules" not in stripped: | |
| return stripped[:160] | |
| return f"ingest ๊ฐ exit {result.returncode} ๋ก ๋๋ฌ์ต๋๋ค" | |
| def _warn_ingest_skipped(project: str, run_dir: Path, reason: str) -> None: | |
| """์ ์ฌ ์คํจ๋ฅผ ์๋ฆฌ๊ณ ๋์ด๋ฆฌ๋ ๋ช ๋ น์ ๊ทธ๋๋ก ์ค์ด ์ค๋ค (exit code ๋ 0 ์ ์ง).""" | |
| click.secho(f"\n โ evalhub ์ ์ฌ๋ฅผ ๊ฑด๋๋ฐ์์ต๋๋ค โ {reason}", fg="yellow", bold=True) | |
| click.secho(f" run ๊ฒฐ๊ณผ๋ ๊ทธ๋๋ก ์์ต๋๋ค: {run_dir}", fg="yellow") | |
| click.secho(" ๋ฐฑ์๋๋ฅผ ๋์ด ๋ค ์๋๋ฅผ ์คํํ๋ฉด ๋ฐ์๋ฉ๋๋ค:", fg="yellow") | |
| click.secho(" make evalhub-up", fg="yellow") | |
| click.secho(f" make evalhub-ingest ONLY={project}", fg="yellow") | |
| def _warn_if_incomplete(run_dir: Path) -> None: | |
| """์ํ์ด ์ ์ค๋ ์ฑ ๋๋ฌ์ผ๋ฉด ์ ์ ์์ ํฌ๊ฒ ์๋ฆฐ๋ค. | |
| ์ ์ค์ ์กฐ์ฉํ๋ค: ์คํจํ ์ํ์ `results.jsonl` ์ ํ ์์ฒด๊ฐ ๋จ์ง ์๊ณ (error ํ๋๋ | |
| ์๋ค) ์ ์๋ ์ด์๋จ์ ์ํ๋ง์ผ๋ก ์ง๊ณ๋๋ค. ์ค์ ๋ก 429 ์ฌ์๋ ์์ง์ผ๋ก 159๊ฑด ์ค | |
| 17๊ฑด์ด ๋น ์ง run ์ด "์๋ฃ"๋ก ๋๋๋ฉฐ ์ ์๋ฅผ ๋๊ณ , `run.json` ์ | |
| completed/total ์ ์ง์ ๋์กฐํ๊ธฐ ์ ๊น์ง ์๋ฌด๋ ๋ชฐ๋๋ค. | |
| ๋ถ๋ถ ๊ฒฐ๊ณผ๋ผ๋ ์ง์ฐ์ง๋ ์๋๋ค โ ์ฌํ ๋น์ฉ์ด ํฌ๊ณ , ์๋ฃ์จ์ ์๊ณ ๋ณด๋ฉด ์ธ๋ชจ๊ฐ ์๋ค. | |
| """ | |
| try: | |
| meta = json.loads((run_dir / "run.json").read_text()) | |
| except (OSError, json.JSONDecodeError): | |
| return | |
| total, done = meta.get("total_samples"), meta.get("completed_samples") | |
| if not isinstance(total, int) or not isinstance(done, int) or done >= total: | |
| return | |
| click.secho( | |
| f"\n โ ์ํ {total - done}๊ฑด์ด ์ ์ค๋์ต๋๋ค ({done}/{total} ์๋ฃ). " | |
| f"์ ์๋ ์๋ฃ๋ ์ํ๋ง์ผ๋ก ๊ณ์ฐ๋ ๊ฐ์ด๋ผ ๋ค๋ฅธ run ๊ณผ ์ง์ ๋น๊ตํ ์ ์์ต๋๋ค.", | |
| fg="yellow", | |
| bold=True, | |
| ) | |
| click.secho( | |
| " ํํ ์์ธ์ API 429 ์ ๋๋ค โ `-w/--max-workers` ๋ฅผ ๋ฎ์ถฐ ๋ค์ ๋๋ฆฌ์ธ์.", | |
| fg="yellow", | |
| ) | |
| def _run_with_progress( | |
| runner, | |
| run_id, | |
| project_config, | |
| task_config, | |
| prompt, | |
| max_workers, | |
| limit, | |
| completed_results: list[dict[str, Any]] | None = None, | |
| ): | |
| """Execute a run with Rich progress bar display. | |
| `completed_results` ๋ resume ์ด ์ด๋ฏธ ๋๋ ์ํ์ ๋ค์ ๋๋ฆฌ์ง ์๊ฒ ํ๋ ๋ค๋ฆฌ๋ค -- | |
| ์ฌ๊ธฐ์ ์ ๋๊ธฐ๋ฉด runner ๊ฐ ํ๋ผ๋ฏธํฐ๋ฅผ ๊ฐ๊ณ ์์ด๋ ์์ฉ์ด ์๋ค(F2). | |
| `execute_run` ์ ์น๋ช ์ ์คํจ์์ raise ํ๋ฏ๋ก ์ด ํจ์๋ raise ํ๋ค -- | |
| ์ผ๋ถ๋ฌ ๊ฐ์ธ์ง ์๋๋ค. ์์ชฝ `on_progress` ์ "error" ์ด๋ฒคํธ๊ฐ ์ด๋ฏธ ์ฌ๋์ด ์ฝ์ | |
| ๋ฉ์์ง๋ฅผ ์ฐ์๊ณ , ์ฌ๊ธฐ์ ๋ ๊ฐ์ธ๋ฉด ๊ฐ์ ๋ด์ฉ์ด ๋ ๋ฒ ๋์จ๋ค. | |
| """ | |
| import logging as _logging | |
| from rich.console import Console | |
| from rich.live import Live | |
| from rich.text import Text | |
| console = Console(stderr=True) | |
| use_rich = True | |
| live_ctx: Live | None = None | |
| progress_bar = Progress( | |
| SpinnerColumn(), | |
| TextColumn("[progress.description]{task.description}"), | |
| BarColumn(), | |
| TextColumn("{task.completed}/{task.total}"), | |
| TextColumn("[progress.percentage]{task.percentage:>3.0f}%"), | |
| TimeElapsedColumn(), | |
| ) | |
| try: | |
| live_ctx = Live(progress_bar, console=console, refresh_per_second=8) | |
| live_ctx.start() | |
| except Exception: | |
| use_rich = False | |
| # Redirect pipeline/runner logs above the progress bar | |
| class _LiveLogHandler(_logging.Handler): | |
| def emit(self, record): | |
| if live_ctx: | |
| live_ctx.console.print(Text(record.getMessage(), style="dim")) | |
| _log_names = ( | |
| "solar_eval.pipelines.pipeline", | |
| "solar_eval.pipelines.steps.llm", | |
| "solar_eval.core.runner", | |
| "solar_eval.providers.upstage", | |
| ) | |
| if use_rich: | |
| _live_handler = _LiveLogHandler() | |
| for name in _log_names: | |
| lg = _logging.getLogger(name) | |
| lg.addHandler(_live_handler) | |
| lg.propagate = False | |
| _logging.getLogger("httpx").setLevel(_logging.WARNING) | |
| def _cleanup_live(): | |
| if use_rich: | |
| live_ctx.stop() | |
| for name in _log_names: | |
| _logging.getLogger(name).removeHandler(_live_handler) | |
| _logging.getLogger(name).propagate = True | |
| inference_task_id = None | |
| eval_task_id = None | |
| async def on_progress(event): | |
| nonlocal inference_task_id, eval_task_id | |
| etype = event.get("type") | |
| if etype == "dataset_loaded": | |
| total = event["total"] | |
| if use_rich: | |
| inference_task_id = progress_bar.add_task("Inference", total=total) | |
| else: | |
| click.secho(f" Dataset loaded: {total} samples", fg="blue") | |
| elif etype == "inference_progress": | |
| if use_rich and inference_task_id is not None: | |
| progress_bar.update(inference_task_id, completed=event["completed"]) | |
| elif not use_rich: | |
| c, t = event["completed"], event["total"] | |
| if c % 10 == 0 or c == t: | |
| click.echo(f" Inference: {c}/{t}") | |
| elif etype == "inference_complete": | |
| if use_rich and inference_task_id is not None: | |
| progress_bar.update(inference_task_id, completed=event["total"]) | |
| else: | |
| click.secho(f" Inference complete: {event['total']} samples", fg="green") | |
| elif etype == "evaluation_start": | |
| total = event.get("total", 0) | |
| if use_rich: | |
| eval_task_id = progress_bar.add_task("Evaluation", total=total or 1) | |
| else: | |
| click.secho(" Starting evaluation...", fg="blue") | |
| elif etype == "evaluation_progress": | |
| if use_rich and eval_task_id is not None: | |
| progress_bar.update(eval_task_id, completed=event["completed"]) | |
| elif etype == "done": | |
| if use_rich and eval_task_id is not None: | |
| progress_bar.update(eval_task_id, completed=progress_bar.tasks[eval_task_id].total) | |
| _cleanup_live() | |
| click.echo() | |
| click.secho("Run completed!", fg="green", bold=True) | |
| overall = event.get("overall_score") | |
| if overall is not None: | |
| click.echo(f" Overall Score: {overall:.4f}") | |
| scores = event.get("scores", {}) | |
| if scores: | |
| click.echo(f" Scores: {format_scores(scores)}") | |
| elif etype == "error": | |
| _cleanup_live() | |
| click.echo() | |
| click.secho(f" Run failed: {event.get('error')}", fg="red", bold=True) | |
| asyncio.run( | |
| runner.execute_run( | |
| run_id=run_id, | |
| project_config=project_config, | |
| task_config=task_config, | |
| prompt=prompt, | |
| on_progress=on_progress, | |
| max_workers=max_workers, | |
| limit=limit, | |
| completed_results=completed_results, | |
| ) | |
| ) | |
| def batch_runs(ctx, project, task, prompts, model, pipeline_name, reasoning_effort, max_workers): | |
| """Run evaluation for multiple prompts sequentially.""" | |
| cfg = ctx.obj["config"] | |
| if cfg.is_remote: | |
| raise click.ClickException("Batch is local-mode only") | |
| prompt_list = [p.strip() for p in prompts.split(",")] | |
| click.secho(f"Batch run: {project}/{task} โ prompts: {prompt_list}", fg="blue", bold=True) | |
| for pname in prompt_list: | |
| click.secho(f"\n{'=' * 50}", fg="blue") | |
| click.secho(f"Running prompt: {pname}...", fg="blue", bold=True) | |
| click.secho(f"{'=' * 50}", fg="blue") | |
| try: | |
| _start_local( | |
| cfg, | |
| project=project, | |
| task=task, | |
| prompt_name=pname, | |
| model=model, | |
| pipeline_name=pipeline_name, | |
| reasoning_effort=reasoning_effort, | |
| temperature=0.0, | |
| max_tokens=4000, | |
| max_workers=max_workers, | |
| limit=None, | |
| ) | |
| except Exception as e: | |
| click.secho(f" {pname} failed: {e}", fg="red") | |
| click.secho(f"\n{'=' * 50}", fg="blue") | |
| click.secho("Batch complete!", fg="green", bold=True) | |
| def resume_run(ctx: click.Context, project: str, run_id: str, max_workers: int) -> None: | |
| """Resume an interrupted evaluation run.""" | |
| cfg = ctx.obj["config"] | |
| if cfg.is_remote: | |
| raise click.ClickException("Resume is local-mode only") | |
| from solar_eval.core.dataset_loader import DatasetLoader | |
| from solar_eval.core.project_loader import load_all_project_configs | |
| from solar_eval.core.runner import BatchRunner | |
| from solar_eval.providers import UpstageProvider | |
| from solar_eval.providers.openai_provider import OpenAIProvider | |
| from solar_eval.stores import JsonlStore | |
| # Find the run directory | |
| run_dir = cfg.artifacts_dir(project) / run_id | |
| if not run_dir.exists(): | |
| raise click.ClickException(f"Run directory not found: {run_dir}") | |
| store = JsonlStore(run_dir) | |
| run_meta = store.load_run_meta() | |
| if not run_meta: | |
| raise click.ClickException(f"No run.json found in {run_dir}") | |
| status = run_meta.get("status", "unknown") | |
| if status == "completed": | |
| raise click.ClickException(f"Run {run_id} is already completed. Nothing to resume.") | |
| task = run_meta.get("task") | |
| if not task: | |
| raise click.ClickException("Cannot determine task from run.json") | |
| # Load existing results | |
| existing_results = store.load_existing_results() | |
| completed_indices = {r["sample_idx"] for r in existing_results} | |
| click.secho(f"Resuming run: {run_id}", fg="blue", bold=True) | |
| click.secho(f" Task: {task}, Already completed: {len(completed_indices)} samples", fg="blue") | |
| # Load project/task config | |
| configs = load_all_project_configs(cfg.projects_dir, cfg.config_dirs) | |
| project_config = next((c for c in configs if c["name"] == project), None) | |
| if not project_config: | |
| raise click.ClickException(f"Project not found: {project}") | |
| task_config = next((t for t in project_config.get("tasks", []) if t["name"] == task), None) | |
| if not task_config: | |
| raise click.ClickException(f"Task '{task}' not found in project '{project}'") | |
| # Resolve prompt from run.json | |
| prompt_name = run_meta.get("prompt") | |
| pipeline_name = run_meta.get("pipeline") | |
| model = run_meta.get("model", "solar-pro2") | |
| if not prompt_name: | |
| raise click.ClickException( | |
| "Cannot determine prompt from run.json. Only runs created with the new CLI can be resumed." | |
| ) | |
| from solar_eval.models.prompt_version import ( | |
| detect_prompt_format, | |
| load_prompt_messages, | |
| load_step_prompts, | |
| ) | |
| prompt_path = _resolve_prompt(cfg.prompts_dir(project), prompt_name) | |
| fmt = detect_prompt_format(prompt_path) | |
| if fmt == "multi_step": | |
| prompt = { | |
| "step_prompts": load_step_prompts(prompt_path), | |
| "model": model, | |
| "temperature": run_meta.get("temperature", 0.0), | |
| "max_tokens": run_meta.get("max_tokens", 4000), | |
| } | |
| else: | |
| prompt = { | |
| "messages": load_prompt_messages(prompt_path), | |
| "model": model, | |
| "temperature": run_meta.get("temperature", 0.0), | |
| "max_tokens": run_meta.get("max_tokens", 4000), | |
| } | |
| if run_meta.get("reasoning_effort"): | |
| prompt["reasoning_effort"] = run_meta["reasoning_effort"] | |
| # Override pipeline if specified โ ์๋ run ์ด `--set` ์ผ๋ก ๋ฐ๊ฟ ๋๋ฆฐ ๊ฒ์ด๋ฉด | |
| # ๊ทธ override ๊น์ง ๋ณต์ํด์ผ ๊ฐ์ ์ค์ ์ผ๋ก ์ด์ด์ง๋ค. | |
| if pipeline_name and pipeline_name != "default": | |
| task_config = { | |
| **task_config, | |
| "pipeline_config": _compose_pipeline_config( | |
| cfg.config_dir(project), | |
| pipeline_name, | |
| run_meta.get("pipeline_overrides") or {}, | |
| ), | |
| } | |
| click.secho(f" Model: {model}", fg="blue") | |
| # ์ด์ ํ๊ฐ๋ฅผ ๋ฏธ๋ฆฌ ์ง์ฐ์ง ์๋๋ค(F3) -- ์ฌํ๊ฐ๊ฐ ๋์ค์ ์ฃฝ์ผ๋ฉด ๋ณต๊ตฌํ ์ ์๊ฐ | |
| # ์์ด์ง๋ค. create_evaluation() ์ด ์์์ ๊ต์ฒด๋ผ ์ง์ฐ์ง ์์๋ ์์ ํ๋ค. | |
| # Setup providers and runner | |
| provider = UpstageProvider() | |
| judge_provider = OpenAIProvider() | |
| dataset_loader = DatasetLoader(base_dir=cfg.projects_dir) | |
| runner = BatchRunner( | |
| store=store, | |
| inference_provider=provider, | |
| judge_provider=judge_provider, | |
| dataset_loader=dataset_loader, | |
| ) | |
| _run_with_progress( | |
| runner, | |
| run_id, | |
| project_config, | |
| task_config, | |
| prompt, | |
| max_workers, | |
| limit=None, | |
| # ์ด๋ฏธ ๋๋ ์ํ์ ๊ฑด๋๋ด๋ค -- ์ ๋๊ธฐ๋ฉด resume ์ด ์ ์ํ์ ๋ค์ ์ ๋ฃ ํธ์ถํ๊ณ | |
| # results.jsonl ์ ๊ฐ์ sample_idx ๋ฅผ ์ค๋ณต append ํ๋ค(F2). | |
| completed_results=existing_results, | |
| ) | |
| click.echo(f"\nResults saved to: {run_dir}/") | |
| def eval_run(ctx: click.Context, project: str, run_id: str) -> None: | |
| """Run evaluation only on existing inference results (no new inference).""" | |
| cfg = ctx.obj["config"] | |
| if cfg.is_remote: | |
| raise click.ClickException("Eval is local-mode only") | |
| from solar_eval.core.project_loader import load_all_project_configs | |
| from solar_eval.core.runner import build_eval_sample_from_result | |
| from solar_eval.evaluators.registry import create_evaluator | |
| from solar_eval.stores import JsonlStore | |
| # Find run directory | |
| run_dir = cfg.artifacts_dir(project) / run_id | |
| if not run_dir.exists(): | |
| raise click.ClickException(f"Run directory not found: {run_dir}") | |
| store = JsonlStore(run_dir) | |
| run_meta = store.load_run_meta() | |
| if not run_meta: | |
| raise click.ClickException(f"No run.json found in {run_dir}") | |
| task = run_meta.get("task") | |
| if not task: | |
| raise click.ClickException("Cannot determine task from run.json") | |
| # Load results | |
| results = store.load_existing_results() | |
| if not results: | |
| raise click.ClickException(f"No inference results found in {run_dir}/results.jsonl") | |
| click.secho(f"Evaluating run: {run_id}", fg="blue", bold=True) | |
| click.secho(f" Task: {task}, Samples: {len(results)}", fg="blue") | |
| # Load task config for evaluator settings | |
| configs = load_all_project_configs(cfg.projects_dir, cfg.config_dirs) | |
| project_config = next((c for c in configs if c["name"] == project), None) | |
| if not project_config: | |
| raise click.ClickException(f"Project not found: {project}") | |
| task_config = next((t for t in project_config.get("tasks", []) if t["name"] == task), None) | |
| if not task_config: | |
| raise click.ClickException(f"Task '{task}' not found in project '{project}'") | |
| # ์ด์ ํ๊ฐ๋ฅผ ๋ฏธ๋ฆฌ ์ง์ฐ์ง ์๋๋ค(F3) -- ์๋ ์ฌํ๊ฐ๊ฐ ์คํจํด๋ ์ ์ ์๋ ๋จ๋๋ค. | |
| # Run evaluation | |
| evaluator = create_evaluator(task_config.get("evaluator", {"type": "llm_judge"})) | |
| # judge provider ๋ฅผ ์ ๋๊ธฐ๋ฉด llm_judge ๊ณ์ด์ ์ ์ํ์์ ์คํจํ๋ค -- ๊ทธ๋ฆฌ๊ณ | |
| # ๊ทธ ์คํจ๊ฐ ์์ ์ 0์ placeholder ๋ก ๋ฎ์ฌ ์กฐ์ฉํ completed ๋ก ๋๋ฌ๋ค(F8). | |
| from solar_eval.providers.openai_provider import OpenAIProvider | |
| judge_provider = OpenAIProvider() | |
| from datetime import timezone | |
| async def run_eval(): | |
| # ์ฑ์ ์คํจ๋ ์ ์๊ฐ ์๋๋ผ ์คํจ๋ก ๋จ๊ธด๋ค -- aggregate() ์๋ ์ฑ๊ณต๋ถ๋ง ๋๊ธด๋ค. | |
| eval_results: list[dict[str, Any]] = [] | |
| eval_failures: list[dict[str, Any]] = [] | |
| for result in results: | |
| # ๋ฆฌ์คํธ ์์น๊ฐ ์๋๋ผ ์๋ณธ sample_idx ๊ฐ ์ ๋ณธ์ด๋ค(F6). | |
| sample_idx = result["sample_idx"] | |
| try: | |
| eval_sample = build_eval_sample_from_result(result, reference=result.get("golden")) | |
| evaluator.validate_required_fields(eval_sample) | |
| eval_result = await evaluator.evaluate(sample=eval_sample, provider=judge_provider) | |
| eval_results.append({**eval_result, "sample_idx": sample_idx}) | |
| except Exception as e: | |
| click.secho(f" Sample {sample_idx} eval failed: {e}", fg="yellow") | |
| eval_failures.append({"sample_idx": sample_idx, "error": str(e)}) | |
| # Aggregate and save | |
| aggregated = evaluator.aggregate(eval_results) | |
| # ์ฑ๊ณตํ ์ฑ์ ์ด ํ๋๋ ์์ผ๋ฉด ์ ์ ์๋ฆฌ๋ฅผ ๋น์ด๋ค -- 0.0 ์ "0์ "์ด๋ผ๋ ์ธก์ ๊ฐ์ด๋ผ | |
| # "์ธก์ ์ด ์์๋ค"์ ๊ตฌ๋ถ๋์ง ์๋๋ค(runner.py ์ ๊ฐ์ ๊ท์น). | |
| overall_score = aggregated.get("overall_score", 0.0) if eval_results else None | |
| eval_id = await store.create_evaluation( | |
| { | |
| "run_id": run_id, | |
| "scores": aggregated.get("scores", {}), | |
| "overall_score": overall_score, | |
| "eval_model": "local", | |
| "eval_success_count": len(eval_results), | |
| "eval_failed_count": len(eval_failures), | |
| "failed_sample_indices": [f["sample_idx"] for f in eval_failures], | |
| } | |
| ) | |
| # Save per-sample eval details -- ์ฑ๊ณต/์คํจ ๋ ๋ค ํ ํ์ฉ ๋จ๊ธด๋ค. | |
| eval_detail_docs = [] | |
| for er in eval_results: | |
| eval_detail_docs.append( | |
| { | |
| "evaluation_id": eval_id, | |
| "sample_idx": er["sample_idx"], | |
| "category_scores": er.get("category_scores", {}), | |
| "error_counts": { | |
| k: v.get("error_count", 0) | |
| for k, v in er.get("details", {}).items() | |
| if isinstance(v, dict) | |
| }, | |
| "severity": er.get("severity", ""), | |
| "score": er.get("score", 0.0), | |
| } | |
| ) | |
| for f in eval_failures: | |
| eval_detail_docs.append( | |
| { | |
| "evaluation_id": eval_id, | |
| "sample_idx": f["sample_idx"], | |
| "category_scores": {}, | |
| "error_counts": {}, | |
| "severity": None, | |
| # 0.0 ์ด ์๋๋ผ None -- ์ฑ์ ์คํจ๋ฅผ ์ต์ ์ ์์ ๊ตฌ๋ถํ๋ค. | |
| "score": None, | |
| "error": f["error"], | |
| } | |
| ) | |
| await store.insert_eval_details(eval_detail_docs) | |
| # Update run status | |
| await store.update_run( | |
| run_id, | |
| { | |
| "status": "completed", | |
| "completed_at": datetime.now(timezone.utc), | |
| }, | |
| ) | |
| return {**aggregated, "overall_score": overall_score, "failed": len(eval_failures)} | |
| aggregated = asyncio.run(run_eval()) | |
| click.echo() | |
| click.secho("Evaluation completed!", fg="green", bold=True) | |
| overall = aggregated.get("overall_score") | |
| if overall is None: | |
| # ์ฑ์ ์ด ํ ๊ฑด๋ ์ฑ๊ณตํ์ง ๋ชปํ๋ค -- ์ซ์๋ฅผ ์ฐ์ผ๋ฉด "0์ "์ผ๋ก ์ฝํ๋ค. | |
| click.secho(" Overall Score: ์ธก์ ์์ (์ ์ํ ์ฑ์ ์คํจ)", fg="red", bold=True) | |
| else: | |
| click.echo(f" Overall Score: {overall:.4f}") | |
| failed = aggregated.get("failed", 0) | |
| if failed: | |
| click.secho(f" ์ฑ์ ์คํจ: {failed}๊ฑด (score: null ๋ก ๊ธฐ๋ก๋จ)", fg="yellow") | |
| scores = aggregated.get("scores", {}) | |
| if scores: | |
| click.echo(f" Scores: {format_scores(scores)}") | |
| click.echo(f"\nResults saved to: {run_dir}/") | |
| def show_results(ctx: click.Context, run_ref: str, limit: int | None) -> None: | |
| """Show per-sample results for a run.""" | |
| cfg = ctx.obj["config"] | |
| if cfg.is_remote: | |
| # run_ref format: "project/run_id" | |
| from solar_eval.cli.client import EvalClient | |
| client = EvalClient(cfg.remote_url, cfg.timeout) | |
| parts = run_ref.split("/") | |
| if len(parts) != 2: | |
| raise click.ClickException("Remote mode: use 'project-name/run-id' format") | |
| project_id = _resolve_project_id(client, parts[0]) | |
| results = client.get(f"/api/projects/{project_id}/runs/{parts[1]}/results") | |
| else: | |
| # run_ref format: "project/run_name" | |
| parts = run_ref.split("/") | |
| if len(parts) != 2: | |
| raise click.ClickException("Local mode: use 'project-name/run-name' format") | |
| results_file = cfg.artifacts_dir(parts[0]) / parts[1] / "results.jsonl" | |
| if not results_file.exists(): | |
| raise click.ClickException(f"Results not found: {results_file}") | |
| results = [ | |
| json.loads(line) for line in results_file.read_text().strip().split("\n") if line | |
| ] | |
| if not results: | |
| click.secho("No results found.", fg="yellow") | |
| return | |
| if limit: | |
| results = results[:limit] | |
| click.secho(format_results_table(results), fg="blue") | |
| click.echo(f"\nShowing {len(results)} result(s)") | |