Files
nvidia__skills/.github/scripts/tests/test_aggregate_benchmarks.py
Moshe Abramovitch 13083bcbbc fix(benchmarks): retire the profile field, accept v3 threshold nulls
SkillEvaluator 0.2.1 redesigned BENCHMARK.md and stopped emitting two
fields the aggregator read. The metadata job has failed on every sync
since TAO's 37 re-signed skills landed in #482, because the null-loss
guard added in #468 correctly refused to write the result.

The two fields need different answers:

profile — retired. v1/v2 emitted '- NVSkills-Eval profile: external'.
NVSkills-Eval is an internal name and does not belong in a published
report, so v3 dropping it was right. Carrying a permanently-null column
named after an internal tool is worse than removing it.

pass_threshold_pct — kept, and left null on v3. It is a real provenance
value that v3 turned into template prose: the same '50%' sentence
appears on skills with 4, 5 and 18 tasks, so scraping it would stamp one
number on every skill regardless of what it was evaluated against. The
existing NO_FABRICATION note already says so. Asking SkillEvaluator to
emit it as a real field again is tracked separately.

benchmarks.json regenerated once with --allow-null-regressions to land
both changes. A strict --check passes against the result, so the guard
stays armed for the next format change.

Signed-off-by: Moshe Abramovitch <moshea@nvidia.com>
2026-08-25 13:31:01 -05:00

137 lines
5.6 KiB
Python

#!/usr/bin/env python3
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
"""Tests for aggregate_benchmarks.py report parsing.
Fixtures are verbatim BENCHMARK.md files from the catalog, one per report
layout in circulation:
v2 — SkillEvaluator 0.9.x (skills/cuopt-developer, evaluated 2026-06)
v3 — SkillEvaluator 1.3.x (skills/nemotron-speech, evaluated 2026-08)
v3 dropped two fields the parser read (`NVSkills-Eval profile` and
`Pass threshold`) and added three it does not (`Evaluator version`,
`Dataset digest`, `Validation status`).
`profile` is now retired: NVSkills-Eval is an internal name and does not
belong in a published report, so the parser no longer looks for it.
`pass_threshold_pct` is kept and left None on v3 — it is a real provenance
value that v3 turned into template prose. These tests pin both decisions.
"""
import sys
import unittest
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
import aggregate_benchmarks as agg # noqa: E402
FIXTURES = Path(__file__).parent / "fixtures"
V2 = FIXTURES / "v2_cuopt_developer.md"
V3 = FIXTURES / "v3_nemotron_speech.md"
class TestV3ProvenanceFields(unittest.TestCase):
"""v3 carries per-run provenance the parser currently discards."""
def test_captures_evaluator_version(self):
entry = agg.parse_benchmark(V3)
self.assertEqual(entry["evaluator_version"], "1.3.2")
def test_captures_dataset_digest(self):
entry = agg.parse_benchmark(V3)
self.assertEqual(
entry["dataset_digest"],
"sha256:7da18a129d0ad5efdc392d764333668f68f9acf996152684bc2464427afdb20f",
)
def test_captures_validation_status(self):
entry = agg.parse_benchmark(V3)
self.assertEqual(entry["validation_status"], "passed")
def test_v2_reports_lack_the_new_fields_and_stay_none(self):
"""Older reports must not error, just leave the new fields empty."""
entry = agg.parse_benchmark(V2)
self.assertIsNone(entry["evaluator_version"])
self.assertIsNone(entry["dataset_digest"])
self.assertIsNone(entry["validation_status"])
class TestNoFabricatedProvenance(unittest.TestCase):
"""The v3 glossary mentions '50%' as static template prose.
It is byte-identical across skills with 4, 5 and 18 tasks, so it is not a
per-skill measurement. Parsing it would stamp 50.0 on every skill
regardless of what it was actually evaluated against — a fabricated
provenance claim in the file whose job is recording provenance. None is
the honest value.
"""
def test_pass_threshold_is_none_for_v3_not_scraped_from_glossary(self):
entry = agg.parse_benchmark(V3)
self.assertIsNone(entry["pass_threshold_pct"])
def test_profile_is_not_emitted_at_all(self):
"""Retired field: absent from every entry, both layouts."""
self.assertNotIn("profile", agg.parse_benchmark(V3))
self.assertNotIn("profile", agg.parse_benchmark(V2))
class TestExistingBehaviourStillWorks(unittest.TestCase):
"""Regression guard: the v2 path and shared fields must not change."""
def test_v2_still_parses_legacy_fields(self):
entry = agg.parse_benchmark(V2)
self.assertEqual(entry["pass_threshold_pct"], 50.0)
def test_v3_shared_fields_parse(self):
entry = agg.parse_benchmark(V3)
self.assertEqual(entry["skill"], "nemotron-speech")
self.assertEqual(entry["evaluation_date"], "2026-08-20")
self.assertEqual(entry["environment"], "local")
self.assertEqual(entry["tasks"], 18)
self.assertEqual(entry["attempts_per_task"], 1)
self.assertEqual(entry["verdict"], "PASS")
class TestNullRateRegressionGuard(unittest.TestCase):
"""A field emptying across a regeneration must fail loudly.
This is the generic guard for the whole class of silent degradation:
the regeneration succeeds, the schema stays valid, --check passes, and a
column quietly goes null. Both the pass_threshold drift and the
2026-08-03 disappearance of cuopt-multi-objective-exploration are this
shape.
"""
def test_flags_a_field_that_lost_values(self):
old = {"skills": [{"skill": "a", "environment": "k8s-sandbox"},
{"skill": "b", "environment": "k8s-sandbox"}]}
new = {"skills": [{"skill": "a", "environment": None},
{"skill": "b", "environment": None}]}
regressions = agg.null_rate_regressions(old, new)
self.assertIn("environment", regressions)
self.assertEqual(regressions["environment"], (0, 2))
def test_silent_on_unchanged_null_rates(self):
old = {"skills": [{"skill": "a", "environment": None}]}
new = {"skills": [{"skill": "a", "environment": None}]}
self.assertEqual(agg.null_rate_regressions(old, new), {})
def test_silent_when_a_field_gains_values(self):
old = {"skills": [{"skill": "a", "environment": None}]}
new = {"skills": [{"skill": "a", "environment": "k8s-sandbox"}]}
self.assertEqual(agg.null_rate_regressions(old, new), {})
def test_ignores_skills_absent_from_the_old_file(self):
"""A newly added skill with empty fields is not a regression."""
old = {"skills": [{"skill": "a", "environment": "k8s-sandbox"}]}
new = {"skills": [{"skill": "a", "environment": "k8s-sandbox"},
{"skill": "b", "environment": None}]}
self.assertEqual(agg.null_rate_regressions(old, new), {})
if __name__ == "__main__":
unittest.main(verbosity=2)