#!/usr/bin/env python3 """Validate rebuilt atomic-model predictions and generate leaderboard data.""" from __future__ import annotations import argparse import json from collections import Counter, defaultdict from pathlib import Path import pyarrow.parquet as pq MAIN_TASKS = ("sentence_1", "sentence_5", "sentence_10", "sentence_50") SOURCE_TASKS = ( "source_held_out_1", "source_held_out_5", "source_held_out_10", "source_held_out_50", ) VERSE_TASKS = ("verse_1", "verse_5", "verse_10", "verse_50") def read_json(path: Path) -> dict: return json.loads(path.read_text()) def read_predictions(path: Path) -> list[dict[str, str]]: rows = [json.loads(line) for line in path.read_text().splitlines() if line.strip()] if not rows: raise ValueError(f"empty predictions: {path}") return rows def macro_f1(gold: list[str], predicted: list[str]) -> float: scores = [] for label in sorted(set(gold)): tp = sum(g == label and p == label for g, p in zip(gold, predicted)) fp = sum(g != label and p == label for g, p in zip(gold, predicted)) fn = sum(g == label and p != label for g, p in zip(gold, predicted)) denominator = 2 * tp + fp + fn scores.append(2 * tp / denominator if denominator else 0.0) return sum(scores) / len(scores) def official_test(dataset_root: Path, task: str) -> dict[str, str]: files = sorted((dataset_root / task).glob("test-*.parquet")) if not files: raise FileNotFoundError(f"missing official test parquet for {task}") rows = pq.read_table(files, columns=["id", "author"]).to_pylist() return {str(row["id"]): str(row["author"]) for row in rows} def diagnostics( predictions_path: Path, dataset_root: Path, task: str, expected_macro_f1: float, ) -> tuple[float, dict[str, float], dict[str, dict[str, int]]]: predictions = read_predictions(predictions_path) official = official_test(dataset_root, task) ids = [str(row["id"]) for row in predictions] if len(ids) != len(set(ids)): raise ValueError(f"duplicate prediction IDs for {task}: {predictions_path}") if set(ids) != set(official): missing = len(set(official) - set(ids)) extra = len(set(ids) - set(official)) raise ValueError(f"prediction ID mismatch for {task}: missing={missing}, extra={extra}") for row in predictions: if str(row["gold"]) != official[str(row["id"])]: raise ValueError(f"gold-label mismatch for {task}, row {row['id']}") gold = [str(row["gold"]) for row in predictions] predicted = [str(row["prediction"]) for row in predictions] score = macro_f1(gold, predicted) if abs(score - expected_macro_f1) > 1e-10: raise ValueError( f"macro-F1 mismatch for {task}: recomputed={score}, saved={expected_macro_f1}" ) correct: Counter[str] = Counter() support: Counter[str] = Counter(gold) confusion: defaultdict[str, Counter[str]] = defaultdict(Counter) for actual, prediction in zip(gold, predicted): correct[actual] += actual == prediction confusion[actual][prediction] += 1 accuracy = {label: 100 * correct[label] / support[label] for label in sorted(support)} sparse_confusion = { label: dict(sorted(confusion[label].items())) for label in sorted(confusion) } return 100 * score, accuracy, sparse_confusion def merge_author_accuracy( target: dict[str, dict[str, float]], task: str, values: dict[str, float] ) -> None: for author, value in values.items(): target.setdefault(author, {})[task] = value def ts(value: object) -> str: return json.dumps(value, ensure_ascii=False, indent=2) def feature_count(metrics: dict) -> int: return int(metrics["feature_metadata"]["train_shape"][1]) def build_row( *, model: str, url: str, dimensionality: str, classifier: str, status: str, task_specs: list[tuple[str, Path, Path, float]], dataset_roots: dict[str, Path], ) -> dict: scores: dict[str, float] = {} author_accuracy: dict[str, dict[str, float]] = {} confusions: dict[str, dict[str, dict[str, int]]] = {} for task, predictions_path, _metrics_path, saved_macro_f1 in task_specs: dataset_root = dataset_roots["verse" if task.startswith("verse_") else "sphragis"] score, accuracy, confusion = diagnostics( predictions_path, dataset_root, task, saved_macro_f1 ) scores[task] = score merge_author_accuracy(author_accuracy, task, accuracy) confusions[task] = confusion return { "model": model, "url": url, "dimensionality": dimensionality, "classifier": classifier, "status": status, "scores": scores, "authorAccuracy": author_accuracy, "confusions": confusions, } def finetune_specs(root: Path, family: str, tasks: tuple[str, ...], encoder: str = "greberta"): family_root = root / f"{encoder}-finetune" / family metrics_path = family_root / "metrics.json" metrics = read_json(metrics_path) return [ (task, family_root / task / "predictions.jsonl", metrics_path, float(metrics["test"][task]["macro_f1"])) for task in tasks ] def siamese_specs(root: Path, family: str, tasks: tuple[str, ...]): """Specs for the siamese contender, whose finalizer wrote a separate tree. The training run never reads test; ``siamese-test`` is written by ``sphragis_models.siamese_finalize``, which reloads the validation-selected checkpoint and rule and evaluates test once. The flag is checked here so a run that never reached the finalizer cannot reach the leaderboard. """ family_root = root / "siamese-test" / family / "supcon_ce" metrics_path = family_root / "metrics.json" metrics = read_json(metrics_path) if metrics.get("test_loaded") is not True: raise ValueError(f"siamese test not finalized: {metrics_path}") if metrics.get("selection_split") != "validation": raise ValueError(f"siamese selection was not made on validation: {metrics_path}") return [ (task, family_root / task / "predictions.jsonl", metrics_path, float(metrics["test"][task]["macro_f1"])) for task in tasks ], str(metrics["selected_rule"]), int(metrics["selected_epoch"]) def mosteller_wallace_specs(root: Path, family: str, model: str, tasks: tuple[str, ...]): """Specs for one Mosteller-Wallace count model, plus its selected settings. The grid job never reads test; ``mosteller-wallace-test`` is written by ``sphragis_models.mosteller_wallace_finalize``, which refits the selected settings on train plus validation and evaluates test once. """ family_root = root / "mosteller-wallace-test" / family / model metrics_path = family_root / "metrics.json" metrics = read_json(metrics_path) if metrics.get("test_loaded") is not True: raise ValueError(f"Mosteller-Wallace test not finalized: {metrics_path}") if metrics.get("selection_split") != "validation": raise ValueError(f"Mosteller-Wallace selection was not made on validation: {metrics_path}") specs = [ (task, family_root / task / "predictions.jsonl", metrics_path, float(metrics["test"][task]["macro_f1"])) for task in tasks ] chosen = [metrics["selected"][task] for task in tasks] return specs, chosen def ordinary_specs(root: Path, base: str, family: str, tasks: tuple[str, ...]): specs = [] for task in tasks: task_root = root / base / family / "models" / task metrics_path = task_root / "metrics.json" metrics = read_json(metrics_path) specs.append((task, task_root / "predictions.jsonl", metrics_path, float(metrics["test_metrics"]["macro_f1"]))) return specs def burrows_specs(root: Path, family: str, tasks: tuple[str, ...]): specs = [] for task in tasks: task_root = root / "burrows-logreg" / family / task metrics_path = task_root / "metrics.json" metrics = read_json(metrics_path) specs.append((task, task_root / "predictions.jsonl", metrics_path, float(metrics["test_metrics"]["macro_f1"]))) return specs def syntax_specs(root: Path, family: str, tasks: tuple[str, ...]): specs, dimensions, counts = [], [], [] for task in tasks: selection_path = root / "syntax-logreg" / family / "selected" / task / "selection.json" selection = read_json(selection_path) if selection.get("test_loaded") is not True: raise ValueError(f"syntax test not finalized: {selection_path}") dimension = int(selection["selected"]["max_dimensions"]) task_root = root / "syntax-logreg" / family / f"d{dimension}-all-train" / "models" / task metrics_path = task_root / "metrics.json" metrics = read_json(metrics_path) dimensions.append(dimension) counts.append(feature_count(metrics)) specs.append((task, task_root / "predictions.jsonl", metrics_path, float(metrics["test_metrics"]["macro_f1"]))) return specs, dimensions, counts def pair_specs(root: Path, pair: str, family: str, tasks: tuple[str, ...]): """Specs for a double-combination model, one selected scaling per task. ``sphragis_models.pair_select`` chooses each task's block scaling on validation and only then reads test, so the flag is checked here: a task whose scaling was never finalized cannot reach the leaderboard. """ specs, scalings, counts = [], [], [] for task in tasks: selection_path = root / "pair-logreg" / pair / family / "selected" / task / "selection.json" selection = read_json(selection_path) if selection.get("test_loaded") is not True: raise ValueError(f"pair test not finalized: {selection_path}") metrics_path = Path(selection["metrics_path"]) metrics = read_json(metrics_path) if metrics["feature_metadata"]["representation"] != "arm_pair": raise ValueError(f"not a pair model: {metrics_path}") scalings.append(str(selection["selected"]["scaling"])) counts.append(feature_count(metrics)) specs.append((task, metrics_path.parent / "predictions.jsonl", metrics_path, float(metrics["test_metrics"]["macro_f1"]))) return specs, scalings, counts def method_specs(root: Path, family: str, tasks: tuple[str, ...]) -> dict[str, tuple]: """Specs for every combination whose method was settled on validation. ``scripts/run_method_search.py`` scores concatenation and five posterior combiners on validation, keeps the best, and reads test once for it. Each cell records the winning method and where that method's predictions live: a pooling win writes its own, a concatenation win points back into the pair grid. Returns {combination: (specs, methods)} keyed by the arms joined with a hyphen, in the search's sorted order. """ out: dict[str, tuple] = {} for combo_dir in sorted((root / "combination-method").glob("*")): family_dir = combo_dir / family if not family_dir.is_dir(): continue specs, methods = [], [] for task in tasks: selection_path = family_dir / task / "selection.json" if not selection_path.is_file(): continue selection = read_json(selection_path) if selection.get("test_loaded") is not True or not selection.get("predictions_path"): raise ValueError(f"combination not finalized: {selection_path}") methods.append(str(selection["selected"]["method"])) predictions = Path(selection["predictions_path"]) if not predictions.is_absolute(): # The search was run from the models repository root with a # relative output root; resolve against that root. predictions = root.parent.parent / predictions specs.append((task, predictions, selection_path, float(selection["test_metrics"]["macro_f1"]))) if specs: out[combo_dir.name] = (specs, methods) return out def alm_specs(root: Path, family: str, tasks: tuple[str, ...]): """Specs for the remade authorial ensemble, or None until it exists. ``scripts/alm_attribution_rows.py`` writes one predictions/metrics pair per task from the remade models' scores. The directory is absent until the remake has landed, and a leaderboard build before then simply carries no ensemble row rather than a stale one. """ family_root = root.parent / "alm-remake-20260902" / "leaderboard" / family if not family_root.is_dir(): return None specs, models = [], set() for task in tasks: metrics_path = family_root / task / "metrics.json" if not metrics_path.is_file(): return None metrics = read_json(metrics_path) models.update(metrics["models"]) specs.append((task, family_root / task / "predictions.jsonl", metrics_path, float(metrics["test_metrics"]["macro_f1"]))) return specs, len(models) def dimensions_for_specs(specs: list[tuple[str, Path, Path, float]]) -> list[int]: return [feature_count(read_json(metrics_path)) for _, _, metrics_path, _ in specs] def make_rows(results_root: Path, sphragis_data: Path, metre_data: Path): roots = {"sphragis": sphragis_data, "verse": metre_data} main_and_source = MAIN_TASKS + SOURCE_TASKS ft_sphragis = finetune_specs(results_root, "sphragis-main", MAIN_TASKS) ft_sphragis += finetune_specs(results_root, "sphragis-source-held-out", SOURCE_TASKS) ft_metre = finetune_specs(results_root, "sphragis-metre", VERSE_TASKS) kaino_ft_sphragis = finetune_specs( results_root, "sphragis-main", MAIN_TASKS, encoder="kainobert" ) + finetune_specs( results_root, "sphragis-source-held-out", SOURCE_TASKS, encoder="kainobert" ) kaino_ft_metre = finetune_specs( results_root, "sphragis-metre", VERSE_TASKS, encoder="kainobert" ) roberta_ft_sphragis = finetune_specs( results_root, "sphragis-main", MAIN_TASKS, encoder="kainoberta" ) + finetune_specs( results_root, "sphragis-source-held-out", SOURCE_TASKS, encoder="kainoberta" ) roberta_ft_metre = finetune_specs( results_root, "sphragis-metre", VERSE_TASKS, encoder="kainoberta" ) greberta_sphragis = ordinary_specs( results_root, "greberta-logreg", "sphragis-main", MAIN_TASKS ) + ordinary_specs( results_root, "greberta-logreg", "sphragis-source-held-out", SOURCE_TASKS ) greberta_metre = ordinary_specs( results_root, "greberta-logreg", "sphragis-metre", VERSE_TASKS ) kainobert_sphragis = ordinary_specs( results_root, "kainobert-logreg", "sphragis-main", MAIN_TASKS ) + ordinary_specs( results_root, "kainobert-logreg", "sphragis-source-held-out", SOURCE_TASKS ) kainobert_metre = ordinary_specs( results_root, "kainobert-logreg", "sphragis-metre", VERSE_TASKS ) tfidf_sphragis = ordinary_specs( results_root, "tfidf-logreg", "sphragis-main", MAIN_TASKS ) + ordinary_specs( results_root, "tfidf-logreg", "sphragis-source-held-out", SOURCE_TASKS ) tfidf_metre = ordinary_specs( results_root, "tfidf-logreg", "sphragis-metre", VERSE_TASKS ) burrows_sphragis = burrows_specs( results_root, "sphragis-main", MAIN_TASKS ) + burrows_specs(results_root, "sphragis-source-held-out", SOURCE_TASKS) burrows_metre = burrows_specs(results_root, "sphragis-metre", VERSE_TASKS) syntax_sphragis, syntax_sphragis_d, syntax_sphragis_counts = syntax_specs( results_root, "sphragis-main", MAIN_TASKS ) source_specs, source_d, source_counts = syntax_specs( results_root, "sphragis-source-held-out", SOURCE_TASKS ) syntax_sphragis += source_specs syntax_sphragis_d += source_d syntax_sphragis_counts += source_counts syntax_metre, syntax_metre_d, syntax_metre_counts = syntax_specs( results_root, "sphragis-metre", VERSE_TASKS ) metre_specs = ordinary_specs(results_root, "metre-logreg", "", VERSE_TASKS) pairs = ( "greberta-alm", "greberta-tfidf", "greberta-char", "greberta-burrows", "char-alm", "tfidf-alm", "burrows-alm", "char-tfidf", "burrows-char", "burrows-tfidf", "greberta-syntax", "burrows-syntax", "syntax-tfidf", "syntax-char", "syntax-alm", # Cube cells, admitted only where all three constituent pairs were # positive in the square grid. "greberta-char-tfidf", "greberta-tfidf-alm", "char-tfidf-alm", "greberta-burrows-alm", "greberta-burrows-char", "burrows-char-alm", "greberta-char-alm", # The same cells with KainoBERT in place of GreBERTa. KainoBERT was # pretrained on a corpus that provably excludes the test split, so # each twin says what the combination is worth without the exposure # GreBERTa's pretraining corpus cannot rule out. "kainobert-alm", "kainobert-tfidf", "kainobert-char", "kainobert-burrows", "kainobert-syntax", "kainobert-char-tfidf", "kainobert-tfidf-alm", "kainobert-burrows-alm", "kainobert-burrows-char", "kainobert-char-alm", ) alm_main = alm_specs(results_root, "sphragis-main", MAIN_TASKS) alm_source = alm_specs(results_root, "sphragis-source-held-out", SOURCE_TASKS) alm_metre = alm_specs(results_root, "sphragis-metre", VERSE_TASKS) alm_status = ( "One OLMo-1B model further pretrained per author on the current release with " "validation-loss early stopping; attribution is the lowest mean per-token surprisal; " "no base-model or epoch search; test evaluated once" ) alm_rows = [] if alm_main and alm_source: alm_rows.append(build_row( model="OLMo-1B authorial language models", url="https://github.com/Urdatorn/sphragis_models", dimensionality=f"1.18B parameters per author; {alm_main[1]} and {alm_source[1]} models", classifier="Lowest mean per-token surprisal", status=alm_status, task_specs=alm_main[0] + alm_source[0], dataset_roots=roots)) alm_metre_rows = [] if alm_metre: alm_metre_rows.append(build_row( model="OLMo-1B authorial language models", url="https://github.com/Urdatorn/sphragis_models", dimensionality=f"1.18B parameters per author; {alm_metre[1]} models", classifier="Lowest mean per-token surprisal", status=alm_status, task_specs=alm_metre[0], dataset_roots=roots)) method_main = method_specs(results_root, "sphragis-main", MAIN_TASKS) method_source = method_specs(results_root, "sphragis-source-held-out", SOURCE_TASKS) method_sphragis: dict[str, tuple] = {} for combo in sorted(set(method_main) | set(method_source)): a = method_main.get(combo, ([], [])); b = method_source.get(combo, ([], [])) method_sphragis[combo] = (a[0] + b[0], a[1] + b[1]) method_metre = method_specs(results_root, "sphragis-metre", VERSE_TASKS) settled = {frozenset(c.split("-")) for c in list(method_sphragis) + list(method_metre)} # A concatenation-only row stays only where the method search never ran, # which today means the KainoBERT twins. pairs = tuple(p for p in pairs if frozenset(p.split("-")) not in settled) pair_data: dict[str, tuple] = {} for pair in pairs: main = pair_specs(results_root, pair, "sphragis-main", MAIN_TASKS) source = pair_specs(results_root, pair, "sphragis-source-held-out", SOURCE_TASKS) pair_data[pair] = tuple(a + b for a, b in zip(main, source)) metre_pairs = ( "greberta-tfidf", "greberta-char", "greberta-burrows", "greberta-metre", "char-tfidf", "burrows-char", "burrows-tfidf", "burrows-metre", "char-metre", "tfidf-metre", "greberta-alm", "char-alm", "tfidf-alm", "burrows-alm", "metre-alm", "greberta-syntax", "syntax-char", "syntax-metre", "syntax-alm", "burrows-greberta-metre", "burrows-metre-tfidf", "greberta-metre-tfidf", "alm-greberta-tfidf", "char-greberta-tfidf", "burrows-char-metre", "alm-char-tfidf", "burrows-greberta-tfidf", "alm-char-greberta", "char-metre-tfidf", "burrows-char-greberta", "char-greberta-metre", "burrows-char-tfidf", ) metre_pair_data: dict[str, tuple] = {} metre_pairs = tuple(p for p in metre_pairs if frozenset(p.split("-")) not in settled) for pair in metre_pairs: try: metre_pair_data[pair] = pair_specs(results_root, pair, "sphragis-metre", VERSE_TASKS) except (FileNotFoundError, KeyError, ValueError): continue mw_specs: dict[tuple[str, str], tuple] = {} for model in ("dirichlet_multinomial", "multinomial_nb"): mw_specs[(model, "sphragis")] = tuple( a + b for a, b in zip( mosteller_wallace_specs(results_root, "sphragis-main", model, MAIN_TASKS), mosteller_wallace_specs( results_root, "sphragis-source-held-out", model, SOURCE_TASKS ), ) ) mw_specs[(model, "verse")] = mosteller_wallace_specs( results_root, "sphragis-metre", model, VERSE_TASKS ) ARM_NAMES = { "greberta": "GreBERTa", "kainobert": "KainoBERT", "tfidf": "lemma TF-IDF", "char": "Char n-gram", "syntax": "syntax rates", "burrows": "Burrows z-scores", "alm": "ALM", "metre": "metrical line features", "greberta_e2e": "GreBERTa (end-to-end)", "siamese": "Siamese GreBERTa", } METHOD_NAMES = { "concatenation": "concatenation", "vote_hard": "hard vote", "vote_soft": "soft vote", "logit_mean": "geometric mean", "rank_mean": "mean rank", "vote_weighted": "validation-weighted vote", } def method_classifier(methods: list[str]) -> str: counts = Counter(METHOD_NAMES[m] for m in methods) return "Validation-selected per task: " + ", ".join( f"{name} \u00d7{n}" if n > 1 else name for name, n in counts.most_common()) def method_status(combo: str) -> str: arms = combo.split("-") note = "" if "alm" in arms: note = "; the authorial language models are being remade and this row will be refreshed" return ( f"{len(arms)} atomic models combined; the combination method is a hyperparameter " "chosen per task on validation macro-F1 among concatenation of prepared feature " "blocks and five training-free posterior combiners (hard vote, soft vote, geometric " "mean, mean rank, validation-weighted vote); test evaluated once for the chosen " "method" + note ) def pair_model_name(pair: str) -> str: return " + ".join(ARM_NAMES[arm] for arm in pair.split("-")) def pair_dimensionality(pair: str, counts: list[int]) -> str: members = [ARM_NAMES[arm] for arm in pair.split("-")] joined = ", ".join(members[:-1]) + " and " + members[-1] return f"{joined} columns stacked: " + slash(counts) def pair_status(scalings: list[str]) -> str: return ( "Two prepared single-channel feature blocks concatenated, each arm at the " "variant its own standalone model selected on validation; block scaling (" + "/".join(scalings) + ") selected per task on validation macro-F1, where balanced divides each " "block by its mean training row norm and native keeps the standalone scales; " + common_status ) def mw_dimensionality(chosen: list[dict]) -> str: return "Validation-selected n=" + "/".join(str(entry["n"]) for entry in chosen) def mw_status(chosen: list[dict], bursty: bool) -> str: alphas = "/".join(str(entry["alpha"]) for entry in chosen) base = ( "Character n-gram counts with a uniform author prior fixed a priori; " f"validation-selected n and Dirichlet smoothing alpha={alphas}; " "vocabulary from the atomic training split only; train+validation refit; " "test evaluated once" ) if not bursty: return base scales = "/".join( "raw posterior" if entry["scale"] is None else str(entry["scale"]) for entry in chosen ) return ( base + f"; validation-selected Dirichlet concentration={scales}, where the raw " "posterior is the non-bursty limit that reproduces naive Bayes" ) siamese_main, siamese_main_rule, siamese_main_epoch = siamese_specs( results_root, "sphragis-main", MAIN_TASKS ) siamese_source, siamese_source_rule, siamese_source_epoch = siamese_specs( results_root, "sphragis-source-held-out", SOURCE_TASKS ) siamese_sphragis = siamese_main + siamese_source siamese_metre, siamese_metre_rule, siamese_metre_epoch = siamese_specs( results_root, "sphragis-metre", VERSE_TASKS ) def siamese_status(rules: list[str], epochs: list[int]) -> str: return ( "Shared GreBERTa encoder trained with supervised contrastive plus cross-entropy " "on author-balanced batches; epoch and decision rule (" + "/".join(rules) + " at epoch " + "/".join(map(str, epochs)) + ") both selected on the full validation split; larger-task scores are means " "over exact constituents; test evaluated once" ) def slash(values: list[int]) -> str: return " / ".join(f"{value:,}" for value in values) def selected_n(specs): return [int(read_json(path)["selected"]["n"]) for _, _, path, _ in specs] common_status = ( "Five-point learning-rate search on validation macro-F1; validation-loss " "early stopping; train+validation refit; test evaluated once" ) sphragis_rows = [ build_row( model="GreBERTa (end-to-end)", url="https://huggingface.co/bowphs/GreBerta", dimensionality="~126M trainable parameters per benchmark track", classifier="Linear softmax head", status=("Fine-tuned on the atomic task with validation-loss early stopping; " "larger-task logits are means over exact constituents; test evaluated once"), task_specs=ft_sphragis, dataset_roots=roots, ), build_row( model="KainoBERT (end-to-end)", url="https://huggingface.co/Urdatorn/KainoBERT-sphragis", dimensionality="~136M trainable parameters per benchmark track", classifier="Linear softmax head", status=("Fine-tuned on the atomic task with validation-loss early stopping; " "larger-task logits are means over exact constituents; test evaluated once"), task_specs=kaino_ft_sphragis, dataset_roots=roots, ), build_row( model="KainoBERTa (end-to-end)", url="https://huggingface.co/Urdatorn/KainoBERTa-sphragis", dimensionality="~112M trainable parameters per benchmark track", classifier="Linear softmax head", status=("Fine-tuned on the atomic task with validation-loss early stopping; larger-task logits are means over exact constituents; test evaluated once. Architecture control for KainoBERT: same pretraining corpus, blocks, tokenizer, masking and schedule, stopped by the same validation plateau rule"), task_specs=roberta_ft_sphragis, dataset_roots=roots, ), build_row( model="GreBERTa", url="https://huggingface.co/bowphs/GreBerta", dimensionality="768 frozen features", classifier="PyTorch multinomial logistic regression", status=common_status, task_specs=greberta_sphragis, dataset_roots=roots, ), build_row( model="KainoBERT", url="https://huggingface.co/Urdatorn/KainoBERT-sphragis", dimensionality="768 frozen features", classifier="PyTorch multinomial logistic regression", status=common_status, task_specs=kainobert_sphragis, dataset_roots=roots, ), build_row( model="Lemma TF-IDF", url="https://github.com/Urdatorn/sphragis_models", dimensionality=slash(dimensions_for_specs(tfidf_sphragis)), classifier="PyTorch multinomial logistic regression", status=common_status, task_specs=tfidf_sphragis, dataset_roots=roots, ), build_row( model="Burrows lemma z-scores", url="https://github.com/Urdatorn/sphragis_models", dimensionality="Validation-selected MFW n=" + "/".join(map(str, selected_n(burrows_sphragis))), classifier="PyTorch multinomial logistic regression", status=common_status, task_specs=burrows_sphragis, dataset_roots=roots, ), build_row( model="Syntax feature rates", url="https://github.com/Urdatorn/sphragis_models", dimensionality=("Validation-selected d=" + "/".join(map(str, syntax_sphragis_d)) + "; features=" + slash(syntax_sphragis_counts)), classifier="PyTorch multinomial logistic regression", status=("Combination dimension 1-4 selected independently per task on validation " "macro-F1; " + common_status), task_specs=syntax_sphragis, dataset_roots=roots, ), build_row( model="Siamese GreBERTa (SupCon + CE)", url="https://github.com/Urdatorn/sphragis_models", dimensionality="~127M trainable parameters; 256-d L2-normalized embedding", classifier="Nearest author centroid and linear softmax head", status=siamese_status( [siamese_main_rule, siamese_source_rule], [siamese_main_epoch, siamese_source_epoch], ), task_specs=siamese_sphragis, dataset_roots=roots, ), build_row( model="Char n-gram Dirichlet-multinomial", url="https://github.com/Urdatorn/sphragis_models", dimensionality=mw_dimensionality(mw_specs[("dirichlet_multinomial", "sphragis")][1]), classifier="Dirichlet compound multinomial posterior predictive", status=mw_status(mw_specs[("dirichlet_multinomial", "sphragis")][1], bursty=True), task_specs=list(mw_specs[("dirichlet_multinomial", "sphragis")][0]), dataset_roots=roots, ), build_row( model="Char n-gram naive Bayes", url="https://github.com/Urdatorn/sphragis_models", dimensionality=mw_dimensionality(mw_specs[("multinomial_nb", "sphragis")][1]), classifier="Multinomial naive Bayes", status=mw_status(mw_specs[("multinomial_nb", "sphragis")][1], bursty=False), task_specs=list(mw_specs[("multinomial_nb", "sphragis")][0]), dataset_roots=roots, ), *( build_row( model=pair_model_name(pair), url="https://github.com/Urdatorn/sphragis_models", dimensionality=pair_dimensionality(pair, pair_data[pair][2]), classifier="PyTorch multinomial logistic regression", status=pair_status(pair_data[pair][1]), task_specs=list(pair_data[pair][0]), dataset_roots=roots, ) for pair in pairs ), *alm_rows, *( build_row( model=pair_model_name(combo), url="https://github.com/Urdatorn/sphragis_models", dimensionality=f"{len(combo.split('-'))} arms", classifier=method_classifier(method_sphragis[combo][1]), status=method_status(combo), task_specs=list(method_sphragis[combo][0]), dataset_roots=roots, ) for combo in method_sphragis ), ] metre_rows = [ build_row( model="GreBERTa (end-to-end)", url="https://huggingface.co/bowphs/GreBerta", dimensionality="~126M trainable parameters", classifier="Linear softmax head", status=("Fine-tuned on verse_1 with validation-loss early stopping; larger-task " "logits are means over exact constituents; test evaluated once"), task_specs=ft_metre, dataset_roots=roots, ), build_row( model="KainoBERT (end-to-end)", url="https://huggingface.co/Urdatorn/KainoBERT-sphragis", dimensionality="~136M trainable parameters", classifier="Linear softmax head", status=("Fine-tuned on verse_1 with validation-loss early stopping; larger-task " "logits are means over exact constituents; test evaluated once"), task_specs=kaino_ft_metre, dataset_roots=roots, ), build_row( model="KainoBERTa (end-to-end)", url="https://huggingface.co/Urdatorn/KainoBERTa-sphragis", dimensionality="~112M trainable parameters", classifier="Linear softmax head", status=("Fine-tuned on verse_1 with validation-loss early stopping; larger-task logits are means over exact constituents; test evaluated once. Architecture control for KainoBERT: same pretraining corpus, blocks, tokenizer, masking and schedule, stopped by the same validation plateau rule"), task_specs=roberta_ft_metre, dataset_roots=roots, ), build_row( model="GreBERTa", url="https://huggingface.co/bowphs/GreBerta", dimensionality="768 frozen features", classifier="PyTorch multinomial logistic regression", status=common_status, task_specs=greberta_metre, dataset_roots=roots, ), build_row( model="KainoBERT", url="https://huggingface.co/Urdatorn/KainoBERT-sphragis", dimensionality="768 frozen features", classifier="PyTorch multinomial logistic regression", status=common_status, task_specs=kainobert_metre, dataset_roots=roots, ), build_row( model="Lemma TF-IDF", url="https://github.com/Urdatorn/sphragis_models", dimensionality=slash(dimensions_for_specs(tfidf_metre)), classifier="PyTorch multinomial logistic regression", status=common_status, task_specs=tfidf_metre, dataset_roots=roots, ), build_row( model="Burrows lemma z-scores", url="https://github.com/Urdatorn/sphragis_models", dimensionality="Validation-selected MFW n=" + "/".join(map(str, selected_n(burrows_metre))), classifier="PyTorch multinomial logistic regression", status=common_status, task_specs=burrows_metre, dataset_roots=roots, ), build_row( model="Syntax feature rates", url="https://github.com/Urdatorn/sphragis_models", dimensionality=("Validation-selected d=" + "/".join(map(str, syntax_metre_d)) + "; features=" + slash(syntax_metre_counts)), classifier="PyTorch multinomial logistic regression", status=("Combination dimension 1-4 selected independently per task on validation " "macro-F1; " + common_status), task_specs=syntax_metre, dataset_roots=roots, ), build_row( model="Metrical line features", url="https://github.com/Urdatorn/sphragis_models", dimensionality=slash(dimensions_for_specs(metre_specs)), classifier="PyTorch multinomial logistic regression", status=("All metrical features observed in atomic training lines; " + common_status), task_specs=metre_specs, dataset_roots=roots, ), build_row( model="Siamese GreBERTa (SupCon + CE)", url="https://github.com/Urdatorn/sphragis_models", dimensionality="~127M trainable parameters; 256-d L2-normalized embedding", classifier="Nearest author centroid and linear softmax head", status=siamese_status([siamese_metre_rule], [siamese_metre_epoch]), task_specs=siamese_metre, dataset_roots=roots, ), build_row( model="Char n-gram Dirichlet-multinomial", url="https://github.com/Urdatorn/sphragis_models", dimensionality=mw_dimensionality(mw_specs[("dirichlet_multinomial", "verse")][1]), classifier="Dirichlet compound multinomial posterior predictive", status=mw_status(mw_specs[("dirichlet_multinomial", "verse")][1], bursty=True), task_specs=list(mw_specs[("dirichlet_multinomial", "verse")][0]), dataset_roots=roots, ), build_row( model="Char n-gram naive Bayes", url="https://github.com/Urdatorn/sphragis_models", dimensionality=mw_dimensionality(mw_specs[("multinomial_nb", "verse")][1]), classifier="Multinomial naive Bayes", status=mw_status(mw_specs[("multinomial_nb", "verse")][1], bursty=False), task_specs=list(mw_specs[("multinomial_nb", "verse")][0]), dataset_roots=roots, ), *( build_row( model=pair_model_name(pair), url="https://github.com/Urdatorn/sphragis_models", dimensionality=pair_dimensionality(pair, metre_pair_data[pair][2]), classifier="PyTorch multinomial logistic regression", status=pair_status(metre_pair_data[pair][1]), task_specs=list(metre_pair_data[pair][0]), dataset_roots=roots, ) for pair in metre_pairs if pair in metre_pair_data ), *alm_metre_rows, *( build_row( model=pair_model_name(combo), url="https://github.com/Urdatorn/sphragis_models", dimensionality=f"{len(combo.split('-'))} arms", classifier=method_classifier(method_metre[combo][1]), status=method_status(combo), task_specs=list(method_metre[combo][0]), dataset_roots=roots, ) for combo in method_metre ), ] return sphragis_rows, metre_rows def main() -> None: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--results-root", type=Path, required=True) parser.add_argument("--sphragis-data", type=Path, required=True) parser.add_argument("--metre-data", type=Path, required=True) parser.add_argument("--output", type=Path, required=True) args = parser.parse_args() sphragis_rows, metre_rows = make_rows( args.results_root, args.sphragis_data, args.metre_data ) content = ( "// Generated by scripts/publish_atomic_results.py; do not edit manually.\n" "import type { ResultRow } from './data';\n\n" f"export const sphragisAtomicResults: ResultRow[] = {ts(sphragis_rows)};\n\n" f"export const metreAtomicResults: ResultRow[] = {ts(metre_rows)};\n" ) args.output.write_text(content) print( json.dumps({ "output": str(args.output), "sphragis_rows": len(sphragis_rows), "metre_rows": len(metre_rows), "validated_tasks": sum(len(row["scores"]) for row in sphragis_rows + metre_rows), }) ) if __name__ == "__main__": main()