File size: 6,003 Bytes
9c84f9d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
"""๋ฐ์ดํ„ฐ์…‹ ํŒŒ์ผ์—์„œ ํƒœ์Šคํฌ๋ฅผ ๋ฐœ๊ฒฌํ•œ๋‹ค โ€” ๋งค๋‹ˆํŽ˜์ŠคํŠธ์— ์†์œผ๋กœ ์ ์ง€ ์•Š๋Š”๋‹ค.

์™œ: ํƒœ์Šคํฌ๋Š” ์‚ฌ์‹ค์ƒ "๋ฐ์ดํ„ฐ์…‹ ํŒŒ์ผ + ์ฑ„์  ์„ค์ •"์ธ๋ฐ, ํŒŒ์ผ์ด ์žˆ๋‹ค๋Š” ์‚ฌ์‹ค์„
`project.yaml` ์— ํ•œ ์ค„์”ฉ ์˜ฎ๊ฒจ ์ ์–ด ์™”๋‹ค(chosun-proofreading ์€ 59์ค„). ์†์œผ๋กœ
์˜ฎ๊ฒจ ์ ๋Š” ๋ชฉ๋ก์€ ๋ฐ˜๋“œ์‹œ ์›๋ณธ๊ณผ ๊ฐˆ๋ผ์ง„๋‹ค โ€” ์‹ค์ œ๋กœ ๋ฐ์ดํ„ฐ์…‹์ด ์‚ฌ๋ผ์ง„ ๋’ค์—๋„
๋งค๋‹ˆํŽ˜์ŠคํŠธ์—๋งŒ ๋‚จ์€ ์œ ๋ น ํƒœ์Šคํฌ ๋‘ ๊ฐœ(`critical-issue`ยท`edge-case`)๊ฐ€ ์ƒ๊ฒผ๊ณ ,
`solar-eval tasks list` ์—๋Š” ๋ณด์ด์ง€๋งŒ ๋Œ๋ฆฌ๋ฉด ์ฃฝ๋Š” ์ƒํƒœ๋กœ ๋ฐฉ์น˜๋๋‹ค.

๋ฐœ๊ฒฌ ๊ทœ์น™์€ ํŒŒ์ผ ์ด๋ฆ„์—์„œ ํƒœ์Šคํฌ ์ด๋ฆ„์„ ๋งŒ๋“ ๋‹ค:

    discover:
      - glob: critical_issue/*_v2.jsonl   # datasets/ ๊ธฐ์ค€ ์ƒ๋Œ€ glob
        name: ci-v2-{stem}                # {stem} = ํŒŒ์ผ stem, '_' โ†’ '-'
        strip_suffix: _v2                 # stem ์—์„œ ๋จผ์ € ๋–ผ๋Š” ๊ผฌ๋ฆฌํ‘œ

**๋ช…์‹œ ํƒœ์Šคํฌ๊ฐ€ ์ด๊ธด๋‹ค.** ์ด๋ฏธ `tasks:` ์— ์žˆ๋Š” dataset_path ๋Š” ๋ฐœ๊ฒฌ ๋Œ€์ƒ์—์„œ
๋น ์ง€๋ฏ€๋กœ, ๊ทœ์น™์— ์•ˆ ๋งž๋Š” ์ด๋ฆ„(`paragraph`, `yul-ryul`)์€ ์˜ˆ์ „ ๊ทธ๋Œ€๋กœ ์ ์–ด ๋‘๋ฉด
๋œ๋‹ค. ๊ณผ๊ฑฐ run ์ด ๊ทธ ์ด๋ฆ„์œผ๋กœ ๊ธฐ๋ก๋ผ ์žˆ์–ด ์ด๋ฆ„์„ ๋ฐ”๊พธ๋ฉด evalhub ์—์„œ ๊ณ„๋ณด๊ฐ€
๋Š์–ด์ง€๊ธฐ ๋•Œ๋ฌธ์—, ์ด ์šฐ์„ ์ˆœ์œ„๋Š” ํƒ€ํ˜‘ ๋Œ€์ƒ์ด ์•„๋‹ˆ๋‹ค.
"""

import logging
from pathlib import Path
from typing import Any

logger = logging.getLogger(__name__)

#: `name` ํ…œํ”Œ๋ฆฟ์—์„œ ํŒŒ์ผ stem ์œผ๋กœ ์น˜ํ™˜๋˜๋Š” ์ž๋ฆฌํ‘œ์‹œ์ž.
STEM_PLACEHOLDER = "{stem}"


class TaskDiscoveryError(ValueError):
    """๋ฐœ๊ฒฌ ๊ทœ์น™์ด ์ž˜๋ชป๋์„ ๋•Œ โ€” ์กฐ์šฉํžˆ ๋นˆ ๋ชฉ๋ก์„ ๋งŒ๋“ค์ง€ ์•Š๋Š”๋‹ค."""


def discover_tasks(config: dict[str, Any], data_root: Path | str | None) -> list[dict[str, Any]]:
    """`discover:` ๊ทœ์น™์œผ๋กœ ํƒœ์Šคํฌ ๋ชฉ๋ก์„ ๋งŒ๋“ ๋‹ค.

    Args:
        config: project.yaml dict. `discover`, `dataset.repo`, `tasks` ๋ฅผ ์ฝ๋Š”๋‹ค.
        data_root: ๋ฐ์ดํ„ฐ ๋ฃจํŠธ(projects_dir). `dataset.repo` ๊ฐ€ ์ด ์•„๋ž˜์— ์žˆ๋‹ค.

    Returns:
        ์ƒˆ๋กœ ๋ฐœ๊ฒฌ๋œ ํƒœ์Šคํฌ dict ๋ชฉ๋ก (๋ช…์‹œ ํƒœ์Šคํฌ์™€ ์ค‘๋ณต๋˜์ง€ ์•Š๋Š” ๊ฒƒ๋งŒ).
        `discover` ๊ฐ€ ์—†์œผ๋ฉด ๋นˆ ๋ชฉ๋ก.

    Raises:
        TaskDiscoveryError: ๊ทœ์น™์— `glob`/`name` ์ด ์—†๊ฑฐ๋‚˜ ์ด๋ฆ„์ด ์ถฉ๋Œํ•  ๋•Œ.
    """
    rules = config.get("discover")
    if not rules:
        return []
    if not isinstance(rules, list):
        raise TaskDiscoveryError("`discover` must be a list of rules")

    dataset_root = _dataset_root(config, data_root)
    if dataset_root is None or not dataset_root.is_dir():
        logger.warning(
            "Task discovery skipped for %s: dataset root not found (%s)",
            config.get("name"),
            dataset_root,
        )
        return []

    claimed_paths = {
        t.get("dataset_path") for t in config.get("tasks", []) or [] if isinstance(t, dict)
    }
    claimed_names = {t.get("name") for t in config.get("tasks", []) or [] if isinstance(t, dict)}

    discovered: list[dict[str, Any]] = []
    seen_names: dict[str, str] = {}
    for rule in rules:
        discovered.extend(_apply_rule(rule, dataset_root, claimed_paths, claimed_names, seen_names))
    return sorted(discovered, key=lambda t: t["name"])


def _apply_rule(
    rule: Any,
    dataset_root: Path,
    claimed_paths: set[Any],
    claimed_names: set[Any],
    seen_names: dict[str, str],
) -> list[dict[str, Any]]:
    """๊ทœ์น™ ํ•˜๋‚˜๋ฅผ ์ ์šฉํ•ด ํƒœ์Šคํฌ๋ฅผ ๋งŒ๋“ ๋‹ค."""
    if not isinstance(rule, dict):
        raise TaskDiscoveryError(f"discover rule must be a mapping, got {type(rule).__name__}")
    pattern = rule.get("glob")
    name_template = rule.get("name")
    if not pattern or not name_template:
        raise TaskDiscoveryError(f"discover rule needs `glob` and `name`: {rule!r}")
    if STEM_PLACEHOLDER not in name_template:
        raise TaskDiscoveryError(
            f"discover `name` must contain {STEM_PLACEHOLDER}: {name_template!r}"
        )

    strip_suffix = rule.get("strip_suffix") or ""
    extra = {k: v for k, v in rule.items() if k not in ("glob", "name", "strip_suffix")}

    tasks: list[dict[str, Any]] = []
    matches = sorted(dataset_root.glob(pattern))
    if not matches:
        logger.warning("discover rule matched no files: %s (under %s)", pattern, dataset_root)

    for file in matches:
        if not file.is_file():
            continue
        dataset_path = file.relative_to(dataset_root).as_posix()
        if dataset_path in claimed_paths:
            continue  # ๋ช…์‹œ ํƒœ์Šคํฌ๊ฐ€ ์ด๋ฏธ ๊ฐ€๋ฆฌํ‚ค๋Š” ํŒŒ์ผ โ€” ์ด๋ฆ„ ๊ณ„๋ณด๋ฅผ ์ง€ํ‚จ๋‹ค
        task_name = _task_name(file.stem, strip_suffix, name_template)
        if task_name in claimed_names:
            continue
        previous = seen_names.get(task_name)
        if previous is not None and previous != dataset_path:
            raise TaskDiscoveryError(
                f"discover produced duplicate task name {task_name!r} "
                f"for {previous} and {dataset_path}"
            )
        seen_names[task_name] = dataset_path
        tasks.append({"name": task_name, "dataset_path": dataset_path, **extra})
    return tasks


def _task_name(stem: str, strip_suffix: str, template: str) -> str:
    """ํŒŒ์ผ stem โ†’ ํƒœ์Šคํฌ ์ด๋ฆ„. ๊ผฌ๋ฆฌํ‘œ๋ฅผ ๋–ผ๊ณ  `_` ๋ฅผ `-` ๋กœ ๋ฐ”๊พผ๋‹ค."""
    if strip_suffix and stem.endswith(strip_suffix):
        stem = stem[: -len(strip_suffix)]
    return template.replace(STEM_PLACEHOLDER, stem.replace("_", "-"))


def _dataset_root(config: dict[str, Any], data_root: Path | str | None) -> Path | None:
    """`dataset.repo` ๋ฅผ ๋ฐ์ดํ„ฐ ๋ฃจํŠธ ๊ธฐ์ค€์œผ๋กœ ํ‘ผ๋‹ค. ์›๊ฒฉ(HF) ๋ฐ์ดํ„ฐ์…‹์€ ์Šค์บ”ํ•˜์ง€ ์•Š๋Š”๋‹ค."""
    dataset = config.get("dataset") or {}
    if dataset.get("source") not in (None, "local"):
        return None
    repo = dataset.get("repo")
    if not repo:
        return None
    repo_path = Path(repo)
    if repo_path.is_absolute():
        return repo_path
    if data_root is None:
        return None
    return Path(data_root) / repo_path