"""Blind annotation sheets for v0.2+. export Write a shuffled CSV without expected labels, categories or groups: dataset/v/annotations/.csv compare Compare a filled sheet with the draft labels: agreement, Cohen's kappa, disagreements and naturalness flags. In the sheet, `etiket` takes the option number or the option key. Use `?` when more than one option fits or none does. Any text in `dogal_degil` flags the Turkish as unnatural; `not` is free text. """ import argparse import csv import random import sys from collections import Counter from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from benchmark.dataset import VERSIONS, load_cases, load_tasks, version_paths # noqa: E402 COLUMNS = ["no", "id", "soru", "secenekler", "metin", "etiket", "dogal_degil", "not"] UNSURE = "?" def sheet_dir(version): return version_paths(version, "public")[0].parent / "annotations" def options_text(criteria): return "\n".join(f"{i}. {key}: {desc}" for i, (key, desc) in enumerate(criteria.items(), 1)) def parse_label(raw, criteria): raw = raw.strip() if not raw: return None if raw == UNSURE: return UNSURE keys = list(criteria) if raw.isdigit() and 1 <= int(raw) <= len(keys): return keys[int(raw) - 1] if raw in criteria: return raw raise ValueError(f"unknown label {raw!r}; use 1-{len(keys)}, a key or {UNSURE}") def read_sheet(path, cases, tasks, ignore_unknown=False): """Return {case_id: (label, unnatural, note)} for rows with a label or a flag.""" by_id = {c["id"]: c for c in cases} out, errors = {}, [] with path.open(encoding="utf-8-sig", newline="") as f: for row in csv.DictReader(f): cid = row["id"].strip() if cid not in by_id: if not ignore_unknown: errors.append(f"row {row['no']}: unknown id {cid!r}") continue try: label = parse_label(row["etiket"], tasks[by_id[cid]["task_id"]]["criteria"]) except ValueError as e: errors.append(f"row {row['no']} ({cid}): {e}") continue unnatural = bool(row.get("dogal_degil", "").strip()) note = row.get("not", "").strip() if label or unnatural or note: out[cid] = (label, unnatural, note) return out, errors def cohen_kappa(pairs): n = len(pairs) if not n: return None observed = sum(a == b for a, b in pairs) / n ca, cb = Counter(a for a, _ in pairs), Counter(b for _, b in pairs) expected = sum(ca[k] * cb[k] for k in ca) / (n * n) return 1.0 if expected == 1 else (observed - expected) / (1 - expected) def export(args, cases, tasks): path = sheet_dir(args.version) / f"{args.annotator}.csv" if path.exists() and not args.force: sys.exit(f"{path} already exists; use --force to overwrite (filled labels will be lost)") rows = list(cases) random.Random(args.seed).shuffle(rows) path.parent.mkdir(parents=True, exist_ok=True) with path.open("w", encoding="utf-8-sig", newline="") as f: writer = csv.writer(f) writer.writerow(COLUMNS) for no, c in enumerate(rows, 1): task = tasks[c["task_id"]] writer.writerow([no, c["id"], task["instructions"], options_text(task["criteria"]), c["state"], "", "", ""]) print(f"Wrote {len(rows)} rows to {path}") def compare(args, cases, tasks): path = Path(args.file) if args.file else sheet_dir(args.version) / f"{args.annotator}.csv" sheet, errors = read_sheet(path, cases, tasks) for e in errors: print("ERROR:", e) by_id = {c["id"]: c for c in cases} labelled = {cid: v for cid, v in sheet.items() if v[0]} print(f"Labelled {len(labelled)}/{len(cases)} cases in {path.name}") scored = [(by_id[cid], v) for cid, v in labelled.items() if by_id[cid]["scored"]] decided = [(c, v) for c, v in scored if v[0] != UNSURE] pairs = [(c["expected"], v[0]) for c, v in decided] if pairs: agree = sum(a == b for a, b in pairs) print(f"Scored: agreement {agree}/{len(pairs)} ({agree / len(pairs):.0%}), " f"Cohen's kappa {cohen_kappa(pairs):.2f}, marked '{UNSURE}': {len(scored) - len(decided)}") ambiguous = [(by_id[cid], v) for cid, v in labelled.items() if not by_id[cid]["scored"]] if ambiguous: hits = sum(v[0] == UNSURE or v[0] in c["valid_answers"] for c, v in ambiguous) print(f"Ambiguous: {hits}/{len(ambiguous)} marked '{UNSURE}' or within valid answers") disputed = [(c, v) for c, v in scored if v[0] != c["expected"]] disputed += [(c, v) for c, v in ambiguous if v[0] != UNSURE and v[0] not in c["valid_answers"]] if disputed: print("\nDisagreements:") for c, (label, _, note) in sorted(disputed, key=lambda x: x[0]["id"]): gold = c["expected"] or "/".join(c["valid_answers"]) print(f" {c['id']} [{c['task_id']}] draft={gold} annotator={label}") print(f" {c['state']}") if note: print(f" not: {note}") flagged = [(by_id[cid], v) for cid, v in sheet.items() if v[1]] if flagged: print("\nFlagged as unnatural:") for c, (_, _, note) in sorted(flagged, key=lambda x: x[0]["id"]): print(f" {c['id']}: {c['state']}" + (f"\n not: {note}" if note else "")) noted = [(by_id[cid], v) for cid, v in sheet.items() if v[2] and not v[1] and (by_id[cid], v) not in disputed] if noted: print("\nOther notes:") for c, (_, _, note) in sorted(noted, key=lambda x: x[0]["id"]): print(f" {c['id']}: {note}") if errors: sys.exit(1) def main(): parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) sub = parser.add_subparsers(dest="command", required=True) for name in ("export", "compare"): p = sub.add_parser(name) p.add_argument("--version", default="0.2", choices=sorted(VERSIONS)) p.add_argument("--annotator", required=name == "export", help="sheet name, e.g. your first name") if name == "export": p.add_argument("--seed", type=int, default=13) p.add_argument("--force", action="store_true") else: p.add_argument("--file", help="filled CSV (default: annotations/.csv)") args = parser.parse_args() if args.command == "compare" and not (args.file or args.annotator): parser.error("compare needs --annotator or --file") public_path, tasks_path = version_paths(args.version, "public") cases, tasks = load_cases(public_path), load_tasks(tasks_path) (export if args.command == "export" else compare)(args, cases, tasks) if __name__ == "__main__": main()