mirror of
https://github.com/NVIDIA/skills.git
synced 2026-09-14 15:41:50 +08:00
13083bcbbc
SkillEvaluator 0.2.1 redesigned BENCHMARK.md and stopped emitting two fields the aggregator read. The metadata job has failed on every sync since TAO's 37 re-signed skills landed in #482, because the null-loss guard added in #468 correctly refused to write the result. The two fields need different answers: profile — retired. v1/v2 emitted '- NVSkills-Eval profile: external'. NVSkills-Eval is an internal name and does not belong in a published report, so v3 dropping it was right. Carrying a permanently-null column named after an internal tool is worse than removing it. pass_threshold_pct — kept, and left null on v3. It is a real provenance value that v3 turned into template prose: the same '50%' sentence appears on skills with 4, 5 and 18 tasks, so scraping it would stamp one number on every skill regardless of what it was evaluated against. The existing NO_FABRICATION note already says so. Asking SkillEvaluator to emit it as a real field again is tracked separately. benchmarks.json regenerated once with --allow-null-regressions to land both changes. A strict --check passes against the result, so the guard stays armed for the next format change. Signed-off-by: Moshe Abramovitch <moshea@nvidia.com>
137 lines
5.6 KiB
Python
137 lines
5.6 KiB
Python
#!/usr/bin/env python3
|
|
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
|
# SPDX-License-Identifier: Apache-2.0
|
|
"""Tests for aggregate_benchmarks.py report parsing.
|
|
|
|
Fixtures are verbatim BENCHMARK.md files from the catalog, one per report
|
|
layout in circulation:
|
|
|
|
v2 — SkillEvaluator 0.9.x (skills/cuopt-developer, evaluated 2026-06)
|
|
v3 — SkillEvaluator 1.3.x (skills/nemotron-speech, evaluated 2026-08)
|
|
|
|
v3 dropped two fields the parser read (`NVSkills-Eval profile` and
|
|
`Pass threshold`) and added three it does not (`Evaluator version`,
|
|
`Dataset digest`, `Validation status`).
|
|
|
|
`profile` is now retired: NVSkills-Eval is an internal name and does not
|
|
belong in a published report, so the parser no longer looks for it.
|
|
`pass_threshold_pct` is kept and left None on v3 — it is a real provenance
|
|
value that v3 turned into template prose. These tests pin both decisions.
|
|
"""
|
|
|
|
import sys
|
|
import unittest
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
|
|
import aggregate_benchmarks as agg # noqa: E402
|
|
|
|
FIXTURES = Path(__file__).parent / "fixtures"
|
|
V2 = FIXTURES / "v2_cuopt_developer.md"
|
|
V3 = FIXTURES / "v3_nemotron_speech.md"
|
|
|
|
|
|
class TestV3ProvenanceFields(unittest.TestCase):
|
|
"""v3 carries per-run provenance the parser currently discards."""
|
|
|
|
def test_captures_evaluator_version(self):
|
|
entry = agg.parse_benchmark(V3)
|
|
self.assertEqual(entry["evaluator_version"], "1.3.2")
|
|
|
|
def test_captures_dataset_digest(self):
|
|
entry = agg.parse_benchmark(V3)
|
|
self.assertEqual(
|
|
entry["dataset_digest"],
|
|
"sha256:7da18a129d0ad5efdc392d764333668f68f9acf996152684bc2464427afdb20f",
|
|
)
|
|
|
|
def test_captures_validation_status(self):
|
|
entry = agg.parse_benchmark(V3)
|
|
self.assertEqual(entry["validation_status"], "passed")
|
|
|
|
def test_v2_reports_lack_the_new_fields_and_stay_none(self):
|
|
"""Older reports must not error, just leave the new fields empty."""
|
|
entry = agg.parse_benchmark(V2)
|
|
self.assertIsNone(entry["evaluator_version"])
|
|
self.assertIsNone(entry["dataset_digest"])
|
|
self.assertIsNone(entry["validation_status"])
|
|
|
|
|
|
class TestNoFabricatedProvenance(unittest.TestCase):
|
|
"""The v3 glossary mentions '50%' as static template prose.
|
|
|
|
It is byte-identical across skills with 4, 5 and 18 tasks, so it is not a
|
|
per-skill measurement. Parsing it would stamp 50.0 on every skill
|
|
regardless of what it was actually evaluated against — a fabricated
|
|
provenance claim in the file whose job is recording provenance. None is
|
|
the honest value.
|
|
"""
|
|
|
|
def test_pass_threshold_is_none_for_v3_not_scraped_from_glossary(self):
|
|
entry = agg.parse_benchmark(V3)
|
|
self.assertIsNone(entry["pass_threshold_pct"])
|
|
|
|
def test_profile_is_not_emitted_at_all(self):
|
|
"""Retired field: absent from every entry, both layouts."""
|
|
self.assertNotIn("profile", agg.parse_benchmark(V3))
|
|
self.assertNotIn("profile", agg.parse_benchmark(V2))
|
|
|
|
|
|
class TestExistingBehaviourStillWorks(unittest.TestCase):
|
|
"""Regression guard: the v2 path and shared fields must not change."""
|
|
|
|
def test_v2_still_parses_legacy_fields(self):
|
|
entry = agg.parse_benchmark(V2)
|
|
self.assertEqual(entry["pass_threshold_pct"], 50.0)
|
|
|
|
def test_v3_shared_fields_parse(self):
|
|
entry = agg.parse_benchmark(V3)
|
|
self.assertEqual(entry["skill"], "nemotron-speech")
|
|
self.assertEqual(entry["evaluation_date"], "2026-08-20")
|
|
self.assertEqual(entry["environment"], "local")
|
|
self.assertEqual(entry["tasks"], 18)
|
|
self.assertEqual(entry["attempts_per_task"], 1)
|
|
self.assertEqual(entry["verdict"], "PASS")
|
|
|
|
|
|
class TestNullRateRegressionGuard(unittest.TestCase):
|
|
"""A field emptying across a regeneration must fail loudly.
|
|
|
|
This is the generic guard for the whole class of silent degradation:
|
|
the regeneration succeeds, the schema stays valid, --check passes, and a
|
|
column quietly goes null. Both the pass_threshold drift and the
|
|
2026-08-03 disappearance of cuopt-multi-objective-exploration are this
|
|
shape.
|
|
"""
|
|
|
|
def test_flags_a_field_that_lost_values(self):
|
|
old = {"skills": [{"skill": "a", "environment": "k8s-sandbox"},
|
|
{"skill": "b", "environment": "k8s-sandbox"}]}
|
|
new = {"skills": [{"skill": "a", "environment": None},
|
|
{"skill": "b", "environment": None}]}
|
|
regressions = agg.null_rate_regressions(old, new)
|
|
self.assertIn("environment", regressions)
|
|
self.assertEqual(regressions["environment"], (0, 2))
|
|
|
|
def test_silent_on_unchanged_null_rates(self):
|
|
old = {"skills": [{"skill": "a", "environment": None}]}
|
|
new = {"skills": [{"skill": "a", "environment": None}]}
|
|
self.assertEqual(agg.null_rate_regressions(old, new), {})
|
|
|
|
def test_silent_when_a_field_gains_values(self):
|
|
old = {"skills": [{"skill": "a", "environment": None}]}
|
|
new = {"skills": [{"skill": "a", "environment": "k8s-sandbox"}]}
|
|
self.assertEqual(agg.null_rate_regressions(old, new), {})
|
|
|
|
def test_ignores_skills_absent_from_the_old_file(self):
|
|
"""A newly added skill with empty fields is not a regression."""
|
|
old = {"skills": [{"skill": "a", "environment": "k8s-sandbox"}]}
|
|
new = {"skills": [{"skill": "a", "environment": "k8s-sandbox"},
|
|
{"skill": "b", "environment": None}]}
|
|
self.assertEqual(agg.null_rate_regressions(old, new), {})
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main(verbosity=2)
|