Files
Dan Guido ce9ae2e2dc Add post-patch-validation plugin (#302)
* Add post-patch-validation plugin

* Use public contact email for post-patch-validation

* Fix post-patch validation CLI and artifact edge cases
2026-09-14 01:49:16 -04:00

378 lines
15 KiB
JavaScript

export const meta = {
name: 'validate-patch',
description:
'Build and execute an evidence plan for an existing security patch, then return the deterministic S1-S5 verdict plus independent coverage concerns',
whenToUse:
'Use after a security patch exists. Pass finding, baseRef, and patchRef or patchFile. The workflow runs local project code and cannot ask for missing input after launch.',
phases: [
{ title: 'Inventory', detail: 'Pin the finding, baseline, patch, diff hash, and changed files' },
{ title: 'Coverage', detail: 'Map variants, behavior, adjacent security, and test infrastructure' },
{ title: 'Plan', detail: 'Create executable checks and pass the machine plan validator' },
{ title: 'Execute', detail: 'Run checks in isolated worktrees and assign the S-score in Python' },
{ title: 'Review', detail: 'Flag omitted paths or evidence that did not exercise real code' },
],
}
const INVENTORY_SCHEMA = {
type: 'object',
additionalProperties: false,
required: [
'planPath',
'baseCommit',
'patchSha256',
'changedFiles',
'evidenceLevel',
'submodules',
'findingSummary',
],
properties: {
planPath: { type: 'string' },
baseCommit: { type: 'string' },
patchedCommit: { type: ['string', 'null'] },
patchSha256: { type: 'string' },
changedFiles: { type: 'array', items: { type: 'string' } },
evidenceLevel: { enum: ['source', 'build', 'runtime'] },
submodules: { type: 'array', items: { type: 'string' } },
findingSummary: { type: 'string' },
},
}
const PROPOSAL_SCHEMA = {
type: 'object',
additionalProperties: false,
required: ['lens', 'observations', 'proposals'],
properties: {
lens: { type: 'string' },
observations: { type: 'array', items: { type: 'string' } },
proposals: {
type: 'array',
items: {
type: 'object',
additionalProperties: false,
required: ['id', 'kind', 'rationale', 'covers', 'strategy'],
properties: {
id: { type: 'string' },
kind: {
enum: ['control', 'exploit', 'variant', 'behavior', 'regression', 'security', 'suite'],
},
rationale: { type: 'string' },
covers: { type: 'array', items: { type: 'string' } },
strategy: { type: 'string' },
},
},
},
},
}
const PLAN_SCHEMA = {
type: 'object',
additionalProperties: false,
required: [
'complete',
'planPath',
'checkCount',
'coverage',
'evidenceLevel',
'validationOutput',
],
properties: {
complete: { type: 'boolean' },
planPath: { type: 'string' },
checkCount: { type: 'integer' },
coverage: {
type: 'array',
items: { enum: ['control', 'exploit', 'variant', 'behavior', 'regression', 'security', 'suite'] },
},
evidenceLevel: { enum: ['source', 'build', 'runtime'] },
validationOutput: { type: 'string' },
blocker: { type: 'string' },
allowEnv: { type: 'array', items: { type: 'string' } },
},
}
const EXECUTION_SCHEMA = {
type: 'object',
additionalProperties: false,
required: ['resultPath', 'reportPath', 'verdict', 'label', 'evidenceLevel', 'reasons'],
properties: {
resultPath: { type: 'string' },
reportPath: { type: 'string' },
verdict: { enum: ['S1', 'S2', 'S3', 'S4', 'S5', 'INCONCLUSIVE'] },
label: { type: 'string' },
evidenceLevel: { enum: ['source', 'build', 'runtime'] },
reasons: { type: 'array', items: { type: 'string' } },
},
}
const REVIEW_SCHEMA = {
type: 'object',
additionalProperties: false,
required: ['lens', 'approved', 'concerns', 'evidenceRead'],
properties: {
lens: { type: 'string' },
approved: { type: 'boolean' },
concerns: { type: 'array', items: { type: 'string' } },
// A reviewer that read nothing cannot approve anything. Without minItems an agent could
// return approved:true with empty arrays and advance an S1 having inspected no evidence.
evidenceRead: { type: 'array', items: { type: 'string' }, minItems: 1 },
},
}
function normalizeArgs(raw) {
const trim = value => (typeof value === 'string' ? value.trim() : '')
const source = typeof raw === 'string' ? { finding: raw } : raw && typeof raw === 'object' ? raw : {}
const patchFile = trim(source.patchFile)
const explicitPatchRef = trim(source.patchRef)
return {
finding: trim(source.finding),
baseRef: trim(source.baseRef),
patchRef: explicitPatchRef || (patchFile ? '' : 'HEAD'),
patchFile,
workdir: trim(source.workdir) || 'post-patch-validation',
}
}
function argProblems(value) {
const problems = []
if (!value.finding) problems.push('finding')
if (!value.baseRef) problems.push('baseRef')
if (!value.patchRef && !value.patchFile) problems.push('patchRef or patchFile')
if (value.patchRef && value.patchFile) problems.push('only one of patchRef or patchFile')
if (
value.workdir.startsWith('/') ||
value.workdir.split('/').some(part => part === '..' || part === '')
) {
problems.push('workdir (must be a non-empty relative path without ..)')
}
return problems
}
function finalStatus(verdict, reviews) {
if (verdict !== 'S1') return 'REJECTED'
if (!Array.isArray(reviews) || reviews.length !== 2) return 'REVIEW_REQUIRED'
if (reviews.some(review => !review || !review.approved)) return 'REVIEW_REQUIRED'
// Approval only counts when the reviewer names what it read. An empty evidenceRead is a
// reviewer that rubber-stamped, which is indistinguishable from no review at all.
if (reviews.some(review => !Array.isArray(review.evidenceRead) || !review.evidenceRead.length)) {
return 'REVIEW_REQUIRED'
}
return 'READY_FOR_HUMAN_REVIEW'
}
const input = normalizeArgs(args)
const problems = argProblems(input)
if (problems.length) {
log(`BLOCKED: missing or unsafe workflow input: ${problems.join(', ')}`)
return {
status: 'BLOCKED',
deterministicVerdict: 'INCONCLUSIVE',
humanReviewRequired: true,
reason: `Supply structured args before launch: ${problems.join(', ')}`,
}
}
phase('Inventory')
const patchOption = input.patchFile
? `--patch-file ${JSON.stringify(input.patchFile)}`
: `--patched-ref ${JSON.stringify(input.patchRef)}`
const inventory = await agent(
`Load the post-patch-validation skill and perform only its scaffold step in the current Git
repository. This is a local validation; do not use the network and do not modify tracked project
files. Read the finding if it is a path, reduce it to one root-cause-and-impact sentence, and run
the bundled Python script with these inputs:
finding: ${input.finding}
base ref: ${input.baseRef}
patch option: ${patchOption}
scaffold flag: --evidence-level source
plan path: ${input.workdir}/plan.json
Use the finding's stable ID when one exists; otherwise use "patch-finding". Run one command at a
time with no pipes or chained shell operators. Return the values printed by scaffold.`,
{ schema: INVENTORY_SCHEMA, label: 'inventory', phase: 'Inventory' },
)
if (!inventory) {
return {
status: 'BLOCKED',
deterministicVerdict: 'INCONCLUSIVE',
humanReviewRequired: true,
reason: 'inventory agent returned no pinned plan',
}
}
phase('Coverage')
const LENSES = [
{
key: 'root-cause',
brief:
'Trace the vulnerable invariant through every sibling caller and alternate entry/output path. Propose the original exploit assertion plus at least one meaningfully distinct root-cause variant. Include error and teardown paths when ownership or lifetime is involved.',
},
{
key: 'behavior',
brief:
'Identify stable benign behavior that must remain byte-identical, plus targeted non-security regressions at the boundaries changed by the patch. Avoid outputs containing time, randomness, addresses, or unordered collections.',
},
{
key: 'adjacent-security',
brief:
'Look specifically for a new vulnerability introduced by the patch: ownership, cleanup, authorization, bounds, concurrency, error handling, and state-transition regressions. Propose checks that should pass on both base and patch.',
},
{
key: 'harness',
brief:
'Find the smallest deterministic project-native build/test command, sanitizer, or bounded fixed-seed fuzz corpus. Also propose a benign harness control that proves both worktrees can execute the relevant component.',
},
]
const proposals = await parallel(
LENSES.map(lens => () =>
agent(
`Read the finding, the pinned plan at ${inventory.planPath}, the changed files, and both Git
revisions without checking either revision out over the user's working tree. Work read-only.
Your coverage lens is ${lens.key}: ${lens.brief}
Return concrete check proposals for the post-patch-validation plan. Each proposal needs a stable
ID, one supported kind, a falsifiable rationale, the exact code path/invariant it covers, and an
implementation strategy. Do not claim a path is covered merely because the diff mentions it.
Do not write tests yet; the Plan phase owns all artifacts.`,
{ schema: PROPOSAL_SCHEMA, label: `coverage:${lens.key}`, phase: 'Coverage' },
),
),
)
if (proposals.length !== LENSES.length || proposals.some(value => !value)) {
return {
status: 'BLOCKED',
deterministicVerdict: 'INCONCLUSIVE',
humanReviewRequired: true,
reason: 'one or more fixed coverage lenses returned no result',
}
}
phase('Plan')
const planned = await agent(
`Load the post-patch-validation skill. Turn the fixed-lens proposals below into a complete,
executable validation plan at ${inventory.planPath}. You may write helper test artifacts only
under ${input.workdir}/checks; do not edit tracked project files or the patch.
PROPOSALS
${JSON.stringify(proposals, null, 2)}
Run the bundled script's print-schema command before editing the plan. Use argv arrays and the
{checkout}, {plan_dir}, {scratch}, and {side} placeholders; never use shell command strings, pipes,
redirections, or chained commands, and write only under {scratch}. The {side} placeholder and
PPV_SIDE are unavailable to exploit and variant checks. Every helper must invoke real project code.
Sort checks by ID and supply every required kind. Exploit and variant safety
assertions must fail on the vulnerable base and pass on the patch, and each must print PPV_REACHED
flushed immediately before its assertion or the runner will return INCONCLUSIVE. Security checks
must pass on both revisions. Behavior output must be stable enough for exact comparison.
Replace the scaffolded source evidence_level with the highest level the completed checks honestly
support: keep source when only source or patch invariants run, use build when target code is
compiled or analyzed without executing the reported behavior, and use runtime only when the
exploit and variant checks execute the affected component. Return that value as evidenceLevel.
Exploit and variant assertions must
test only the security invariant; put liveness, exact error type, timing, and compatibility in
behavior or regression checks. Preserve the scaffolded submodules list unless another pinned
submodule is demonstrably required.
Checks run under a fixed minimal environment. If the project's toolchain needs host variables
beyond PATH and HOME, list their names in allowEnv rather than hardcoding host paths into the
plan. Never list a credential; the values are recorded in result.json.
Run validate-plan when done. If the evidence cannot honestly cover every required kind, do not
invent a check: return complete=false with the blocker and preserve the incomplete plan.`,
{ schema: PLAN_SCHEMA, label: 'plan', phase: 'Plan' },
)
if (!planned || !planned.complete) {
return {
status: 'BLOCKED',
deterministicVerdict: 'INCONCLUSIVE',
humanReviewRequired: true,
reason: planned ? planned.blocker || planned.validationOutput : 'plan agent returned nothing',
planPath: planned ? planned.planPath : inventory.planPath,
}
}
phase('Execute')
const allowEnvFlags = (Array.isArray(planned.allowEnv) ? planned.allowEnv : [])
.filter(name => /^[A-Za-z_][A-Za-z0-9_]*$/.test(name))
.map(name => `--allow-env ${name}`)
.join(' ')
const execution = await agent(
`Load the post-patch-validation skill and execute its Python runner exactly once:
plan: ${planned.planPath}
output: ${input.workdir}/results
extra flags: ${allowEnvFlags || '(none)'}
The runner intentionally exits nonzero for S2-S5 and INCONCLUSIVE. A nonzero Bash result is not a
reason to rerun it. Read result.json after the command, return its exact verdict, label, evidence
level, reasons, and artifact paths, and do not edit the plan, patch, checks, or result.`,
{ schema: EXECUTION_SCHEMA, label: 'execute', phase: 'Execute' },
)
if (!execution) {
return {
status: 'BLOCKED',
deterministicVerdict: 'INCONCLUSIVE',
humanReviewRequired: true,
reason: 'execution agent returned no machine result',
planPath: planned.planPath,
}
}
phase('Review')
const REVIEW_LENSES = [
{
key: 'surface-completeness',
brief:
'Try to find a root-cause sibling, alternate entry/output path, error path, boundary, or teardown path omitted from the declared covers fields. Read the relevant source rather than trusting the plan summary.',
},
{
key: 'evidence-integrity',
brief:
'Read result.json, patch.diff, plan.snapshot.json, the raw logs, and every content-addressed helper artifact referenced by each run\'s argv_files (including scripts passed to interpreters, not only argv[0]). Verify the recorded hashes, and reject a relevant helper whose argv_files record has no artifact. Reject mocks, reimplemented vulnerable logic, a test that never invokes real project code, empty output, a harness that prints the marker without exercising the vulnerable path, or commands whose observations do not support their declared kind.',
},
]
const reviews = await parallel(
REVIEW_LENSES.map(lens => () =>
agent(
`Review the completed post-patch evidence read-only under ${input.workdir}/results.
Your lens is ${lens.key}: ${lens.brief}
The Python verdict is ${execution.verdict}. You cannot change or vote on that S-score. Set
approved=false when the supplied evidence is incomplete or invalid under your lens, list concrete
concerns with file/function/check IDs, and list every artifact you actually read. Missing evidence
is a concern, not consent. Do not run the validation again and do not edit artifacts.`,
{ schema: REVIEW_SCHEMA, label: `review:${lens.key}`, phase: 'Review' },
),
),
)
const status = finalStatus(execution.verdict, reviews)
return {
status,
deterministicVerdict: execution.verdict,
verdictLabel: execution.label,
evidenceLevel: execution.evidenceLevel,
reasons: execution.reasons,
humanReviewRequired: true,
planPath: planned.planPath,
resultPath: execution.resultPath,
reportPath: execution.reportPath,
reviews: reviews.filter(Boolean),
}