// `adjudicate` — the AI-adjudication workflow for the criteria the static engine cannot // decide. Where the engine leaves a judgment/needs-rendering success criterion as a // `manual` residual risk, this turns it into a WORKLIST: one entry per manual criterion, // pre-loaded with the concrete evidence the engine already captured (every image's alt, // every link's text + context, literal colour pairs, control labels…). The AI agent reads // the evidence and records a verdict — C / NC / NA / manual — with a justification (for C // and NA), a groundable finding (for NC), or a reason (for a still-`manual` residual that // truly needs a rendered DOM via `scan`, or is genuinely undecidable). `applyAdjudication` // folds the verdicts back into the audit, FAIL-CLOSED: no null verdict, no unjustified // C/NA, no ungroundable NC, no reasonless manual, full coverage of the residual set. The // decisions are the AGENT's, statically, gated — not a deferral to a human. import { mkdirSync, statSync, writeFileSync } from "node:fs"; import { isAbsolute, join, relative, resolve } from "node:path"; import type { AuditResult, Automatability, CriterionCitation, Finding, Lang, PackCriterionAdjudication, ResidualRisk, Severity, Status } from "./types.js"; import { SCHEMA_VERSION } from "./types.js"; import { discover } from "./discover.js"; import { readText } from "./util.js"; import { parseSource } from "./parse/source.js"; import { attachSignals, PAGES_DIR, snapshotPageId } from "./snapshot.js"; import { loadConfig } from "./config.js"; import { type Harvested, harvestSubjects, isSnapshotFile, PACK_SUBJECTS, pageOfDoc, SC_SUBJECTS } from "./adjudicate-subjects.js"; import type { Doc } from "./parse/html.js"; import { ADJUDICATION, adjudicationForWcagRefs } from "./adjudication-data.js"; import { scTitle, getSC, hasSC, techniquesFor, allSC, guidelineTitle, understanding } from "./wcag.js"; import { groundFinding, type GroundingSummary } from "./grounding.js"; import { type StandardId, CORE, isCore, loadPack, hasId, getCriterion, derivePackResults, isProvisionalJudgmentInapplicable, criterionUrl, glossaryAnchorsOf, localize, resolveGlossary, siblingCriteria, presupposedCriterion, type StandardPack, type PackCriterion, } from "./standards/index.js"; import { guidanceForCriterion } from "./guidance/index.js"; import { guidanceExampleBlock } from "./prd.js"; import { INAPPLICABLE_STATUS } from "./types.js"; import { derivePages, pageGridModel } from "./pages.js"; /** Cap on CONTENT CLASSES shown per criterion — not on anchors. * * The distinction is the whole point. A cap on anchors makes the evidence a sample, and a * `C` over a sample is a conformity claim about elements nobody looked at. A cap on classes * bounds the reading while keeping the population complete, because the population collapses: * 887 links over 38 captured pages are 97 distinct (text, href) pairs. When even the class * count exceeds this, `evidenceComplete` goes false and the fold refuses a `C` outright — * an incomplete reading may still find a real failure, it may never clear one. */ export const ADJUDICATE_MAX_EVIDENCE_CLASSES = 1200; // The number is set by what a real application actually contains, measured, and then given // room to grow. On a 338-file product with 38 captured pages the largest per-criterion // populations run to 592 classes (what survives when CSS is off), 531 (reading order) and 487 // (structure) — and those ARE the populations, not samples of them. // // The headroom is the point. A cap sitting just above today's largest criterion is a gate // that flips the day someone adds a heading: the criterion becomes unclearable for reasons of // VOLUME rather than of uncertainty, which is the wrong reason to refuse a verdict and an // especially confusing one, since nothing about the code got worse. Refusing to conclude is // honest only when something really was not looked at. /** Sibling anchors RENDERED per class. The data keeps them all — the citation gate reads that * list to decide whether an anchor belongs to the criterion, so a bound here would make the * gate refuse real occurrences for being ninth. Measured on a real run: 31 of 37 refusals * were citations of elements the criterion genuinely carried. Only the reading is bounded. */ const ALSO_AT_SHOWN = 8; /** Lines a citation may sit from the anchor it was given and still count as citing it. */ const CITE_DRIFT_DEFAULT = 10; /** How much there actually was to look at, so a reader can tell a complete reading from a * glance at the first few. */ export interface EvidencePopulation { /** Distinct content classes — what the agent is shown, one representative each. */ classes: number; /** Raw anchors behind them. `occurrences` per class sums to this. */ occurrences: number; files: number; pages: number; } export interface Evidence { file: string; line: number; selector: string; snippet: string; note?: string; // extra context the harvester surfaced (e.g. a link's nearest heading) /** Anchors sharing this content class — the same header link on 38 pages is one decision, * not 38. Absent when the class has a single anchor. */ occurrences?: number; /** Every other anchor of this class, as `file:line`. Complete on purpose: the citation gate * reads it to decide membership, so a bound here would refuse real occurrences. The * worklist renders only the first few (ALSO_AT_SHOWN). */ alsoAt?: string[]; /** Page ids this class appears on, when it was harvested from rendered captures. */ pages?: string[]; } export type CriterionVerdict = "C" | "NC" | "NA" | "manual" | null; /** An agent-declared non-conformity — same shape a `Finding` needs to render + re-gate. */ export interface AgentFinding { file: string; line: number; selector?: string; message: string; snippet?: string; severity?: Severity; // The precise normative test the agent cites for an NC verdict — a WCAG SC id for the // core (e.g. "1.1.1"), or a pack criterion / test id when adjudicating a pack standard // (e.g. "1.1" or "1.1.1"). REQUIRED for any NC finding: `applyAdjudication` fail-closes // when it is absent or does not resolve against the active standard. A recommendation // (advisory) needs none — a good practice has no normative test by definition. normativeRef?: string; } export interface AdjudicationItem { criteriaId: string; automatability: Automatability; /** Feedback from an adversarial review that reopened this criterion. It is context for the * next adjudicator, never a verdict and never copied into the final decision implicitly. */ previousReview?: string; /** A deterministic rendered test exists, but the criterion still needs judgment for a C. * This keeps the missing-render warning independent from the final verdict owner. */ needsRenderedEvidence?: boolean; title?: string; /** Exact numbered tests still open on this criterion. Pack worklists persist this so the * AI cannot silently skip a judgment test or re-review a test already decided NC. */ testIds?: string[]; evidence: Evidence[]; /** Static/rendered findings whose rule contract is explicitly non-conclusive. These are * leads for the adjudicator, never findings that can be folded without a fresh verdict. */ signals?: Array<{ ruleId: string; tests: string[]; file: string; line: number; selector?: string; message: string; snippet?: string; }>; /** What the harvest actually found, before collapsing to classes. */ population?: EvidencePopulation; /** False when even the CLASS count exceeded the cap, so some distinct thing was never * shown. A `C` is refused on such an item: a reading that skipped part of the population * can report a failure it saw, but it cannot clear what it never looked at. */ evidenceComplete?: boolean; evidenceTruncated?: { shown: number; total: number }; /** The markup the harvest actually found, as flat tokens (see `Harvested.markup`). Read by * the brief to say which of the criterion's numbered tests the evidence TOUCHES — never to * say which it does not. Derived, so `hydrateAdjudication` restores it with the evidence. */ markup?: string[]; verdict: CriterionVerdict; // the agent fills this justification: string; // REQUIRED for C and NA reason: string | null; // REQUIRED for a still-`manual` verdict ("needs-rendered-dom" | "undecidable") findings: AgentFinding[]; // REQUIRED (≥1, groundable, each with a normativeRef) for NC // What the agent CLEARED, for a C (or ruled out of scope, for an NA). The mirror image of // `findings`, and required for the same reason: without a slot to cite in, a conforming // verdict was pure prose, and the gate could only ever check that the prose was non-empty // — so a model answering "C" to everything with the justification "x" passed, and the // report published 91 conformant criteria nobody had assessed. // // Each citation must ground (the file/line/snippet must resolve, exactly like an NC // finding) AND match one of this item's own `evidence` anchors by file+line — that pairing // is what turns "cite something real" into "cite the evidence you were shown". When the // harvester found no evidence at all there is nothing to clear, and the honest verdicts // are `manual` or `NA`, never `C`. // Objects are what the contract asks for; a bare `"file:line"` string is accepted too, // because that is the shape the worklist's own `alsoAt` uses and a real run reached for it. // See `readCitation` — the gate is identical either way, a string just carries no snippet. citations?: (Evidence | string)[]; // Non-normative good practices the agent noted on this criterion — folded back into the // audit as ADVISORY findings (grounded exactly like an NC finding, but never affecting // status: they cannot flip the criterion to NC nor enter conformancePct). Optional. recommendations?: AgentFinding[]; decidedBy: "agent"; } export interface AdjudicationFile { tool: "ultra11y"; kind: "adjudication"; schemaVersion: number; standard: StandardId; auditDate: string; // The contract, carried BY the worklist. Whoever fills this file — an agent in a coding // harness, an orchestrator's tool node, a script — reads the file, not ultra11y's source, // so the file has to say what a verdict may be and what each one additionally requires. // Advisory to the reader, never read back by `applyAdjudication` (which validates against // its own constants), so a hand-edited or stale header can never widen what is accepted. contract?: AdjudicationContract; items: AdjudicationItem[]; } /** What the filler of an adjudication file is allowed to write, stated in the file. */ export interface AdjudicationContract { verdicts: readonly string[]; manualReasons: readonly string[]; requires: Record; } /** The contract as written into every worklist. Derived from the same constants the gate * validates against, so the two cannot drift. */ export function adjudicationContract(): AdjudicationContract { return { verdicts: [...VERDICTS], manualReasons: [...MANUAL_REASON_VALUES], requires: { C: "a non-empty justification AND citations[] naming the harvested evidence it cleared (each anchor resolvable and drawn from this criterion's own evidence, and about the same kind of element the harvest recorded there — copy the evidence's own `snippet` rather than retyping the element); a criterion with no harvested evidence cannot be C at all", NA: "a non-empty justification, and citations[] whenever evidence was presented — a criterion whose subject exists NOWHERE in the audited scope is NA, never NC", NC: "at least one groundable finding, each naming the `file` it was observed in (an NC with no location is refused as flatly as an uncited C — and a non-conformity that rests on an ABSENCE is still observed on an element of a page, so cite that element) and each citing a normativeRef that resolves against the active standard", manual: `a reason ∈ {${MANUAL_REASON_VALUES.join(", ")}}`, }, }; } /** Resolve the audit's scope inputs back to parsed docs (harvesting reads the same files * the audit did — run `verify --manual` from the audit's cwd). Best-effort: unreadable / * vanished files are skipped, exactly like the audit's own read loop. */ function docsForAudit(audit: AuditResult, cwd?: string): Doc[] { const inputs = audit.scope.inputs.filter((i) => i !== "-" && i !== ""); if (!inputs.length) return []; const { files } = discover(inputs, {}); const docs: Doc[] = []; for (const f of files) { try { // `resolve`, not `join`: a cwd resolves RELATIVE paths and must leave an absolute one // alone. `join("/repo", "/tmp/x/page.html")` produced `/repo/tmp/x/page.html`, which is // unreadable — and the catch below swallows it, so every criterion silently arrived with // ZERO evidence. The gate then refused each C verdict for "no evidence harvested", // blaming the adjudicator for a path bug. Grounding has always used `resolve` for the // same reason; these two must agree, since one harvests the anchors the other checks. const doc = parseSource(readText(cwd ? resolve(cwd, f) : f), f); // A page snapshot is a DIRECTORY of signals (computed styles, boxes, stylesheets, the // screenshot), not a lone .html — and `parseSource` only reads the DOM. Without this // call the harvest sees every captured page as inert markup, so a criterion decided on // what the browser MEASURED (an image of text, a target's size, a sticky header over a // focused element) arrived with nothing to rule on. `runAudit` has always attached them // (src/audit.ts); the two must agree, since one harvests the anchors the other checks. attachSignals(doc); docs.push(doc); } catch { /* unreadable — skip, mirrors runAudit */ } } return docs; } /** The numbers that decide how much of a criterion is shown and how strictly a citation is * read. Every one of them was a constant compiled into the engine, and every one is a * judgement about a CODEBASE rather than a fact about accessibility — so the audited * repository owns them, through `.ultra11yrc.json`. Absent ⇒ these defaults, which is what * every repository that never opens the question keeps. */ export interface AdjudicationLimits { maxClasses: number; showAlsoAt: number; citationDrift: number; } export const ADJUDICATION_DEFAULTS: AdjudicationLimits = { maxClasses: ADJUDICATE_MAX_EVIDENCE_CLASSES, showAlsoAt: ALSO_AT_SHOWN, citationDrift: CITE_DRIFT_DEFAULT, }; /** Read them from the audited repository, falling back to the defaults per field — a config * that sets one number must not silently reset the others. */ export function adjudicationLimits(cwd?: string): AdjudicationLimits { const cfg = loadConfig(cwd ?? ".")?.adjudication; if (!cfg) return ADJUDICATION_DEFAULTS; const positive = (v: unknown, fallback: number): number => (typeof v === "number" && Number.isFinite(v) && v > 0 ? Math.floor(v) : fallback); return { maxClasses: positive(cfg.maxClasses, ADJUDICATION_DEFAULTS.maxClasses), showAlsoAt: positive(cfg.showAlsoAt, ADJUDICATION_DEFAULTS.showAlsoAt), // 0 is meaningful here (exact-anchor matching), so it is allowed through. citationDrift: typeof cfg.citationDrift === "number" && Number.isFinite(cfg.citationDrift) && cfg.citationDrift >= 0 ? Math.floor(cfg.citationDrift) : ADJUDICATION_DEFAULTS.citationDrift, }; } /** Collapse harvested anchors to one representative per CONTENT CLASS, and record how big * the population really was. * * The old shape was `harvested.slice(0, 30)` — a SAMPLE. A `C` over a sample is a claim * about a population nobody looked at, and the numbers were not close: measured on a real * audit, RGAA 11.1 was ruled on 30 of 2652 anchors, none of them from a rendered page. But * the population is only large when counted as occurrences. 887 links across 38 captured * pages are 97 distinct (text, href) pairs; 47 images are 8 distinct (alt, src) pairs. One * representative per class, with its occurrence count, is therefore the WHOLE population, * said once per distinct thing — which is what makes an honest `C` reachable at all. */ function collapse( harvested: Harvested[], limits: AdjudicationLimits, ): { evidence: Evidence[]; population: EvidencePopulation; complete: boolean; markup: string[] } { const byClass = new Map(); for (const item of harvested) { const g = byClass.get(item.cls); if (g) g.push(item); else byClass.set(item.cls, [item]); } const groups = [...byClass.values()]; const evidence = groups.slice(0, limits.maxClasses).map((group) => { // Prefer a RENDERED anchor as the representative: it proves what the browser actually // produced, and it is intrinsically page-scoped. The source anchor that produced it is // not lost — it travels in `alsoAt`, which is what someone editing the fix needs. const rep = group.find((x) => isSnapshotFile(x.ev.file)) ?? group[0]!; const others = group.filter((x) => x !== rep); const pages = [...new Set(group.map((x) => pageOfDoc(x.ev.file)).filter((x): x is string => x !== undefined))]; return { ...rep.ev, ...(group.length > 1 ? { occurrences: group.length } : {}), ...(others.length ? { alsoAt: others.map((x) => `${x.ev.file}:${x.ev.line}`) } : {}), ...(pages.length ? { pages } : {}), }; }); return { evidence, population: { classes: groups.length, occurrences: harvested.length, files: new Set(harvested.map((x) => x.ev.file)).size, pages: new Set(harvested.map((x) => pageOfDoc(x.ev.file)).filter((x) => x !== undefined)).size, }, complete: groups.length <= limits.maxClasses, // Over the WHOLE harvest, not the shown representatives: the cap drops classes, and a // mechanism that exists in the audited scope must not stop being reported because its // class fell past the limit. This only ever LIGHTS a test up, so erring wide is the safe // direction — see `testMarkupTokens`. markup: [...new Set(harvested.flatMap((x) => x.markup ?? []))].sort(), }; } /** A pack criterion's automatability: the WORST among the success criteria it maps to. One * needing a rendered DOM for any of them needs one, full stop. A criterion whose SCs are all * outside the core (RGAA 8.1 → the removed 4.1.1) is still the agent's to decide from source. * * Exported because the pack audit DOCUMENT (src/standards/document.ts) has to label its * residual risks the same way this worklist labels its items — two answers for one criterion * would have `audit --standard rgaa` and `verify --manual --standard rgaa` disagree about * whether a browser is needed. */ export function packAutomatability(scs: readonly string[], criterion?: PackCriterion): Automatability { if (criterion?.automation) { const tiers = Object.values(criterion.automation.tests); if (criterion.automation.completeBySilence === true) return tiers.includes("rendered") ? "needs-rendering" : "static"; // A deterministic failure detector is not a positive conformance proof. Unless the // criterion explicitly closes by measured silence, its residual C verdict is judgment. return "judgment"; } const autos = scs.map((sc) => getSC(sc)?.automatability).filter((a): a is Automatability => !!a); return autos.includes("needs-rendering") ? "needs-rendering" : "judgment"; } function blankItem( criteriaId: string, automatability: Automatability, title: string | undefined, harvested: Harvested[], limits: AdjudicationLimits, signals: AdjudicationItem["signals"] = [], testIds: string[] = [], ): AdjudicationItem { const { evidence, population, complete, markup } = collapse(harvested, limits); return { criteriaId, automatability, ...(title ? { title } : {}), ...(testIds.length ? { testIds } : {}), evidence, ...(signals.length ? { signals } : {}), ...(markup.length ? { markup } : {}), population, evidenceComplete: complete, ...(complete ? {} : { evidenceTruncated: { shown: evidence.length, total: population.classes } }), verdict: null, justification: "", reason: null, findings: [], recommendations: [], decidedBy: "agent" as const, }; } /** The subjects that decide a success criterion. Empty ⇒ the criterion has none declared, * which `tests/harvest-coverage.test.ts` refuses for any criterion the engine hands over. */ const subjectsForSc = (sc: string): string[] => SC_SUBJECTS[sc] ?? []; /** The subjects that decide a PACK criterion: its own when it declares them, else the union * of its mapped success criteria's. The override is what stops RGAA 11.1 (are the fields * labelled?) from being handed the page's heading outline because 1.3.1 happens to come * first in its `wcag` list. */ function subjectsForPackCriterion(standard: StandardId, id: string, scs: string[]): string[] { const own = PACK_SUBJECTS[standard]?.[id]; if (own) return own; const out: string[] = []; for (const sc of scs) for (const subject of subjectsForSc(sc)) if (!out.includes(subject)) out.push(subject); return out; } /** Whether a pack result still needs an agent ruling. * * Manual rows are the ordinary residual set. A judgment criterion provisionally closed as * inapplicable is included as well: absence of a harvested subject is useful deterministic * evidence, but a FULL audit still asks the agent to confirm that absence and the criterion's * particular cases. A prior agent decision is never reopened. */ function packResultNeedsAdjudication(pc: ReturnType[number], criterion?: PackCriterion): boolean { if (pc.status === "manual") return true; return isProvisionalJudgmentInapplicable(pc, criterion); } /** Candidate rules are observations the pack deliberately leaves to judgment. When a * criterion has no subject harvester of its own (reflow is document-wide), those findings are * also the only concrete anchors the model can cite. Prefer persisted snapshot anchors over * transient URLs and keep the signal's wording as the note the adjudicator reads. */ function candidateFindingEvidence(findings: readonly Finding[], limit: number): Evidence[] { const snapshots = findings.filter((finding) => isSnapshotFile(finding.file)); const local = findings.filter((finding) => !/^https?:\/\//i.test(finding.file)); const pool = snapshots.length ? snapshots : local; const seen = new Set(); const evidence: Evidence[] = []; for (const finding of pool) { const key = `${finding.file}:${finding.line}:${finding.selectorHint}`; if (seen.has(key)) continue; seen.add(key); evidence.push({ file: finding.file, line: finding.line, selector: finding.selectorHint === "document" ? "" : finding.selectorHint, snippet: finding.snippet, note: finding.message, ...(finding.page ? { pages: [finding.page] } : {}), }); if (evidence.length >= limit) break; } return evidence; } /** Build the adjudication worklist. * * For the WCAG core: one item per residual-risk (manual) success criterion. * * For a COUNTRY STANDARD: one item per PACK criterion that derives `manual`, plus every * judgment criterion provisionally closed for absence. When the audit carries pages, a * criterion that is definitively NC for the RUN is also included if some page cells are still * open: the deterministic finding settles the run, not the unaffected pages, and the strict * page gate must not be structurally impossible to satisfy. * Keying by the pack's own criteria is not cosmetic: it is what lets an item carry the * criterion's numbered tests, and therefore what lets `normativeRefResolves` check a citation * against THIS criterion's tests instead of accepting any id of the right shape. */ export function buildAdjudicationWorklist(audit: AuditResult, opts: { cwd?: string; standard?: StandardId } = {}): AdjudicationItem[] { const docs = docsForAudit(audit, opts.cwd); const standard = opts.standard; const limits = adjudicationLimits(opts.cwd); if (standard !== undefined && !isCore(standard)) { const pack = loadPack(standard); const pages = derivePages(audit, audit.scope.pages ?? []); const grid = pageGridModel(audit, pages, standard, "en"); const openOnPage = new Set([...grid.status.entries()].filter(([, byPage]) => [...byPage.values()].some((status) => status === "manual")).map(([id]) => id)); const alreadyAdjudicated = new Set( audit.packAdjudication?.standard === standard ? audit.packAdjudication.criteria.filter((criterion) => criterion.status !== "manual").map((criterion) => criterion.id) : [], ); return derivePackResults(audit, standard) .filter( (pc) => packResultNeedsAdjudication(pc, getCriterion(pack, pc.id)) || (pc.status === "NC" && openOnPage.has(pc.id) && !alreadyAdjudicated.has(pc.id)), ) .map((pc) => { const crit = getCriterion(pack, pc.id); const scs = crit?.wcag ?? pc.scs; const item = blankItem( pc.id, packAutomatability(scs, crit), // THE STANDARD'S OWN LOCALE, not a literal "fr". Identical output for RGAA, which // publishes in French and only in French — but the PACK is what says so, and a // standard publishing in another language must not be titled through a locale this // call happened to name. `localize` rather than `titlePlain` because a pack's // locales are not the UI frame's `Lang`: casting one to the other to satisfy a // signature would be asserting something about a country standard that is not true. crit ? localize(pack, crit.titlePlain, pack.defaultLocale) : undefined, harvestSubjects(subjectsForPackCriterion(standard, pc.id, scs), docs), limits, (pc.candidateFindings ?? []).map((finding) => ({ ruleId: finding.ruleId, tests: crit?.automation?.rules.find((rule) => rule.id === finding.ruleId)?.tests ?? [], file: finding.file, line: finding.line, message: finding.message, ...(finding.snippet ? { snippet: finding.snippet } : {}), })), Object.keys(crit?.tests ?? {}).map((test) => `${pc.id}.${test}`), ); const candidates = item.evidence.length ? [] : candidateFindingEvidence(pc.candidateFindings ?? [], limits.maxClasses); return { ...item, ...(pc.status === "manual" && pc.justification?.match(/Contre-expertise\s*:/i) ? { previousReview: pc.justification } : {}), ...(candidates.length ? { evidence: candidates, population: { classes: candidates.length, occurrences: candidates.length, files: new Set(candidates.map((evidence) => evidence.file)).size, pages: new Set(candidates.flatMap((evidence) => evidence.pages ?? [])).size, }, } : {}), ...(Object.values(crit?.automation?.tests ?? {}).includes("rendered") ? { needsRenderedEvidence: true } : {}), }; }); } return audit.residualRisks.map((r: ResidualRisk) => blankItem(r.criteriaId, r.automatability, scTitle(r.criteriaId) ?? undefined, harvestSubjects(subjectsForSc(r.criteriaId), docs), limits), ); } /** THE CRITERIA THIS RUN COULD NOT POSSIBLY HAVE DECIDED — because nothing was rendered. * * `applyAdjudication` already refuses a `needs-rendered-dom` verdict whose own evidence sits * in a page capture: that is a deferral to a tier which has already run. This is the mirror, * and it was the more expensive failure of the two. Measured on the 2026-08-20 RGAA cascade — * three passes, 311 turns, $24.90 — seven criteria came back `needs-rendered-dom` and every * one of them was RIGHT: the workflow audited sources only, and no page was ever snapshotted. * Nothing in the run said so, so the bill bought the news that a step nobody had run had not * run. * * `scope.pagesAudited` is the evidence, and `undefined` is read as UNKNOWN rather than as * none: an audit written before that field existed knows nothing either way, and a warning * that fires on "unknown" is a warning people learn to scroll past. So the answer is empty * unless we can say positively that no page's real DOM was read. * * Advisory by construction — it returns a list, it does not refuse anything. A source-only * audit is a legitimate thing to want, and a worklist is not where a project's scope gets * decided; `check --require-rendered` is the opt-in that fails. */ export function unrenderedResidual(audit: AuditResult, items: AdjudicationItem[]): string[] { const audited = audit.scope.pagesAudited; const readSomePage = audited === undefined ? (audit.scope.pages ?? []).length > 0 : audited.length > 0; if (readSomePage) return []; return items.filter((it) => it.automatability === "needs-rendering" || it.needsRenderedEvidence === true).map((it) => it.criteriaId); } /** Read a citation whichever way it was written. * * The contract asks for `{file, line, …}`, and a real run wrote `"src/Foo.tsx:15"` instead — * 63 verdicts refused for a shape, not for a claim. It is an unambiguous form and the * worklist itself is full of it (`alsoAt` is exactly `file:line`), so an adjudicator * reaching for it is being consistent, not careless. * * Nothing about the gate moves: membership and grounding run on the result either way. A * string simply carries no snippet, so grounding falls back to its selector/line checks — * weaker evidence, and the reason the object form stays what the contract asks for. */ export function readCitation(c: Evidence | string): Evidence | null { if (typeof c !== "string") return c; const at = c.lastIndexOf(":"); if (at <= 0) return null; const line = Number.parseInt(c.slice(at + 1), 10); if (!Number.isFinite(line)) return null; return { file: c.slice(0, at), line, selector: "", snippet: "" }; } /** The files this audit actually read. Computed from the audit's own scope inputs, so it is * the same set the harvest walked — `discover` only globs, it does not parse, so this costs * little and is memoised per fold. */ function auditFiles(audit: AuditResult, cwd?: string): Set { const inputs = audit.scope.inputs.filter((i) => i !== "-" && i !== ""); if (!inputs.length) return new Set(); try { void cwd; // `discover` returns the same repo-relative paths the harvest anchors use return new Set(discover(inputs, {}).files); } catch { return new Set(); } } /** Snapshot paths identify committed evidence, not the runner checkout that happened to hold * it. Keep ordinary source paths strict while comparing any snapshot through its published * `.ultra11y/pages//…` suffix. */ function canonicalEvidenceFile(file: string): string { const posix = file.replace(/\\/g, "/"); const marker = `${PAGES_DIR}/`; const at = posix.lastIndexOf(marker); return at >= 0 ? posix.slice(at) : posix; } /** A directory input scopes its supporting files too. The parser may select only markup, but * an adjudicator is explicitly told to open linked CSS/JS to answer judgment criteria. Such a * citation remains bounded to the audited directory and is still content-grounded. */ function withinAuditInput(file: string, audit: AuditResult, cwd?: string): boolean { const target = resolve(cwd ?? process.cwd(), file); for (const input of audit.scope.inputs) { if (input === "-" || input === "") continue; const root = resolve(cwd ?? process.cwd(), input); try { if (statSync(root).isDirectory()) { const rel = relative(root, target); if (rel === "" || (rel !== ".." && !rel.startsWith("../") && !isAbsolute(rel))) return true; } else if (root === target) return true; } catch { // A glob or vanished input is covered by the discovered-file set when possible. } } return false; } /** Does this citation name evidence the criterion actually carries? * * Same file, and within the drift `groundFinding` already tolerates. Exact line equality was * too literal to be useful: the harvest anchors a at line 19, the agent cites the *
at line 20 — which is the very thing RGAA 5.4 asks about, inside the element it * was shown — and the gate called it fabricated. Measured on a real run, that class of * refusal cost more criteria than every other cause combined. * * What the check still catches is what it was written for: a citation pointing at a file the * criterion was never given. And it is only half the gate — `groundFinding` still has to find * the cited content in the real source, so a plausible-looking file:line in the right * neighbourhood proves nothing on its own. */ /** Why a citation was held to the literal check, and what to cite instead. * * Appended to a `cited snippet not found` refusal when the citation missed THIS criterion's * harvest. Without it the message names a symptom — a transcription that did not match — and * hides the cause, which is that the citation left the evidence the criterion was shown. The * two are fixed differently: one asks the adjudicator to copy more carefully, the other asks * it to cite something else entirely. */ export function offHarvestHint(criteriaId: string, evidence: Evidence[]): string { const anchors = [...new Set(evidence.map((e) => e.selector?.trim() || `${e.file}:${e.line}`).filter(Boolean))]; const shown = anchors.slice(0, 4).join(", "); const rest = anchors.length > 4 ? `, +${anchors.length - 4}` : ""; return anchors.length ? ` — this anchor is not among the evidence harvested for ${criteriaId}, so the snippet was verified literally rather than vouched for by the harvest. Cite one of: ${shown}${rest}.` : ` — ${criteriaId} was harvested no evidence, so nothing can vouch for a transcription here. If the criterion has no subject in scope, the verdict is NA with a justification and no citation to ground.`; } /** The tag a citation or an anchor is about, from its snippet first and its selector second. * Lowercased — HTML tag names are case-insensitive. Undefined when neither says. */ function tagOf(x: { snippet?: string; selector?: string }): string | undefined { const fromSnippet = /<\s*([a-zA-Z][\w-]*)/.exec(x.snippet ?? ""); if (fromSnippet) return fromSnippet[1]!.toLowerCase(); const fromSelector = /^\s*([a-zA-Z][\w-]*)/.exec(x.selector ?? ""); return fromSelector ? fromSelector[1]!.toLowerCase() : undefined; } /** The distinctive words of a markup fragment: attribute VALUES and text, minus the noise. * * Values and text, never attribute names — every `` has an `href`, so names carry no * identity. Short tokens go too: `""`, `en`, `0` are shared by half a document. */ function contentTokens(markup: string): Set { const out = new Set(); const withoutTags = markup.replace(/<[^>]*>/g, " "); for (const m of markup.matchAll(/=\s*"([^"]*)"|=\s*'([^']*)'/g)) { for (const w of (m[1] ?? m[2] ?? "").split(/[\s/_-]+/)) if (w.length >= 3) out.add(w.toLowerCase()); } for (const w of withoutTags.split(/[\s/_-]+/)) if (w.length >= 3) out.add(w.toLowerCase()); return out; } /** Is the citation RECOGNISABLY the element the harvest recorded at that anchor? * * This is what survives when byte-exact snippet matching is given up, and it has to do two * jobs at once. An adjudicator that retypes an element — attributes reordered, `class` left * off — named the right thing and must not be refused. An adjudicator that writes down a * DIFFERENT element at the same anchor (another link, another href, another label) has not * read what it claims to have read, and must be. * * So: same tag, and at least one distinctive word in common — an attribute value or a word of * the text. `` against * `` shares `assets` and * `help.svg`; `Nowhere` against `Read more` * shares nothing at all. * * Silence is not contradiction: a citation with no snippet, or an anchor whose markup carries * no distinctive word (`
`), is judged on the tag alone. Membership and the anchor's own * grounding still stand behind every one of them. */ function recognisablySame(cite: { snippet?: string; selector?: string }, anchor: { snippet?: string; selector?: string }): boolean { const ta = tagOf(cite); const tb = tagOf(anchor); if (ta !== undefined && tb !== undefined && ta !== tb) return false; if (!cite.snippet || !anchor.snippet) return true; const want = contentTokens(anchor.snippet); if (!want.size) return true; const got = contentTokens(cite.snippet); if (!got.size) return true; for (const w of got) if (want.has(w)) return true; return false; } /** How much of the anchor's distinctive content the citation repeats. 0 = nothing in common. */ function overlap(cite: { snippet?: string }, anchor: { snippet?: string }): number { if (!cite.snippet || !anchor.snippet) return 0; const want = contentTokens(anchor.snippet); if (!want.size) return 0; let n = 0; for (const w of contentTokens(cite.snippet)) if (want.has(w)) n++; return n; } /** The harvested anchor a citation resolves to, or undefined. * * Two shapes, because a class carries two kinds of anchor. The REPRESENTATIVE has a snippet * the engine read out of the file itself; a SIBLING in `alsoAt` is a bare `file:line` — the * same content class at another occurrence, with no snippet recorded. Which one matched * decides what can be grounded, so the caller is told. * * CONTENT BREAKS THE TIE, because the line often cannot. A page snapshot is * `documentElement.outerHTML`: one line, the whole document. Every anchor harvested from it * therefore sits at line 2, so "the anchor at the cited line" names dozens of elements at * once and taking the first is a coin toss — measured on a real run, that is how a citation * of an `` came to be checked against an `` and refused for describing the wrong * element. So among the anchors the line admits, the one the citation actually describes * wins; ties and empty overlaps keep document order, which is what a single-anchor file * always did. */ function anchorFor( evidence: Evidence[], c: { file: string; line: number; selector?: string; snippet?: string }, drift: number, ): { at: Evidence; representative: boolean } | undefined { const citedFile = canonicalEvidenceFile(c.file); const reps = evidence.filter((e) => canonicalEvidenceFile(e.file) === citedFile && Math.abs(e.line - c.line) <= drift); const best = (cands: Evidence[]): Evidence | undefined => { let top: Evidence | undefined; let score = -1; for (const e of cands) { const n = overlap(c, e); if (n > score) { score = n; top = e; } } return top; }; const rep = best(reps); if (rep) return { at: rep, representative: true }; // Browser probes are first attached to a compact snapshot anchor (often line 1), while the // persisted DOM may later be pretty-printed and cited at its real element line. On the same // canonical snapshot file, an exact selector or recognisable content can bridge that // serializer drift; ordinary source files keep the configured line bound. if (snapshotPageId(c.file) !== undefined) { const anywhere = evidence.filter((e) => { if (canonicalEvidenceFile(e.file) !== citedFile) return false; const selectorMatch = !!c.selector && !!e.selector && c.selector === e.selector; return selectorMatch || (overlap(c, e) > 0 && recognisablySame(c, e)); }); const moved = best(anywhere); if (moved) return { at: moved, representative: true }; } const siblings = evidence.filter((e) => cites0(e.alsoAt ?? [], c, drift)); const sib = best(siblings); return sib ? { at: sib, representative: false } : undefined; } function cites0(anchors: string[], c: { file: string; line: number }, drift: number): boolean { for (const a of anchors) { const at = a.lastIndexOf(":"); if (at < 0) continue; if (canonicalEvidenceFile(a.slice(0, at)) !== canonicalEvidenceFile(c.file)) continue; const line = Number.parseInt(a.slice(at + 1), 10); if (Number.isFinite(line) && Math.abs(line - c.line) <= drift) return true; } return false; } export interface ApplyAdjudicationResult { ok: boolean; audit: AuditResult; issues: string[]; applied: number; stillManual: number; /** Criteria whose verdict was REFUSED by the gate and therefore not applied. They stay * « to assess », each carrying the refusal as its residual reason. Always 0 in strict mode, * where a single refusal discards the whole fold. */ rejected: number; /** The ids behind `rejected`, in file order — so a caller can name them without re-parsing * `issues` (one criterion can raise several). */ rejectedCriteria: string[]; grounding: GroundingSummary; } const NC_SEVERITY_DEFAULT: Severity = "majeur"; const MANUAL_REASONS = new Set(["needs-rendered-dom", "undecidable"]); /** The verdicts an adjudication may carry, in the exact spelling the file must use. Exported * because it is a CONTRACT, not an implementation detail: the worklist declares it in its own * header and the rejection message names it, so whoever fills the file never has to read this * source to learn what they are allowed to write. */ export const VERDICTS = ["C", "NC", "NA", "manual"] as const; /** The reasons a still-`manual` verdict may cite — same contract, same reason to export it. */ export const MANUAL_REASON_VALUES = ["needs-rendered-dom", "undecidable"] as const; /** Canonicalise a verdict written in any case ("na", "Nc", "MANUAL" → "NA", "NC", "manual"). * * Three of the four verdicts are upper-case and the fourth is not, which is exactly the kind * of detail an agent filling the worklist gets wrong — and it used to cost the entire run, * because `applyAdjudication` fail-closes on an unknown verdict. Case carries no meaning * here: "na" can only ever mean NA. What stays rejected is a verdict outside the vocabulary, * which is a real disagreement about the contract rather than a spelling accident. Returns * undefined when there is no match. */ export function normalizeVerdict(v: unknown): Exclude | undefined { if (typeof v !== "string") return undefined; const k = v.trim().toLowerCase(); return VERDICTS.find((x) => x.toLowerCase() === k); } /** The same tolerance for a `manual` verdict's reason. */ function normalizeManualReason(r: unknown): string | undefined { if (typeof r !== "string") return undefined; const k = r.trim().toLowerCase(); return MANUAL_REASON_VALUES.find((x) => x === k); } /** Rewrite an adjudication's verdicts and manual reasons into their canonical spelling, in * place. Anything unrecognised is left untouched, so the per-item validation below still * reports it as the contract violation it is — this normalises spelling, it never invents a * decision the file did not carry. */ export function canonicalizeAdjudication(adj: AdjudicationFile): AdjudicationFile { for (const it of adj.items) { if (it.verdict !== null) { const v = normalizeVerdict(it.verdict); if (v !== undefined) it.verdict = v; } if (it.reason !== null && it.reason !== undefined) { const r = normalizeManualReason(it.reason); if (r !== undefined) it.reason = r; } } return adj; } /** Does an NC finding's `normativeRef` resolve against the ACTIVE standard? * * Core: a real success-criterion id (reuses `hasSC`). * * Pack: the criterion the ref names must be `itemCriterionId` ITSELF — either cited bare * ("11.2") or as one of its own tests ("11.2.1"). * * That last constraint is load-bearing, not pedantry. A pack test id has the same `N.N.N` * shape as a WCAG success criterion, so a laxer check accepted the WCAG id an agent would * naturally reach for and silently read it as an unrelated pack test: citing "1.4.3" * (Contrast Minimum) resolved as RGAA test 1.4.3, which is about CAPTCHA images. Binding the * citation to the item's own criterion removes the collision entirely. * * Fail-closed: an absent/blank/unresolvable ref fails the whole adjudication (mirrors * verify.ts). */ function normativeRefResolves(ref: string | undefined, standard: StandardId, itemCriterionId?: string): boolean { const r = (ref ?? "").trim(); if (!r) return false; if (isCore(standard)) return hasSC(r); const pack = loadPack(standard); // Which criterion does the ref name? Either the criterion itself, or ".". let critId = hasId(pack, r) ? r : undefined; let testKey: string | undefined; if (critId === undefined) { const dot = r.lastIndexOf("."); if (dot <= 0) return false; const head = r.slice(0, dot); if (!hasId(pack, head)) return false; critId = head; testKey = r.slice(dot + 1); } const crit = getCriterion(pack, critId); if (!crit) return false; if (testKey !== undefined && !(crit.tests && Object.hasOwn(crit.tests, testKey))) return false; // The citation must belong to the criterion being adjudicated. return itemCriterionId === undefined || critId === itemCriterionId; } /** Fold an adjudication file back into the audit. Returns a NEW AuditResult with the decided * statuses, agent findings, recomputed conformancePct, a shrunk residual set, and the * `adjudicated` marker. * * FAIL-CLOSED PER VERDICT, which is not the same thing as fail-closed per FILE. Every check * below is unchanged and no refused verdict is ever applied — but a refusal now costs its own * criterion and nothing else. The all-or-nothing fold was measured to be worse than useless: * a CI run that filled 95 of 96 verdicts correctly had all 96 discarded, so the audit paid a * full adjudication and published « to assess » across the board. Since the point of the gate * is to keep an unproven verdict out of the report — not to punish the ones that proved * themselves — a refused criterion simply stays « to assess », carrying the refusal as its * residual reason so the next run knows what to fix. * * `strict: true` restores the old file-level behaviour for a caller that genuinely wants * all-or-nothing (a conformance deliverable signed off in one pass). */ export function applyAdjudication( audit: AuditResult, adj: AdjudicationFile, opts: { cwd?: string; strict?: boolean; /** Criteria the caller KNOWS this adjudication does not cover, each with the reason to * record. A ledger replay supplies them (« stale », « never adjudicated ») so an absence * it fully expected is not reported as a coverage violation — while the criterion still * stays to assess, carrying that reason instead of a blank cell. */ residualReasons?: Record; } = {}, ): ApplyAdjudicationResult { const issues: string[] = []; const expected = opts.residualReasons ?? {}; // Per-criterion attribution. The gate used to collect one flat list, which is why it could // only ever reject the whole file: an issue did not know which verdict it condemned. const itemIssues = new Map(); // Criteria that will NOT fold and must keep `manual` with a reason — the refused ones plus // the ones the caller declared uncovered. const notFolded = new Set(); const blame = (criteriaId: string, issue: string) => { issues.push(issue); notFolded.add(criteriaId); const list = itemIssues.get(criteriaId); if (list) list.push(issue); else itemIssues.set(criteriaId, [issue]); }; /** An open criterion this adjudication deliberately does not carry. Not a gate violation. */ const uncovered = (criteriaId: string) => notFolded.add(criteriaId); // Spelling first, decisions second: "na" is NA, and rejecting the run over the case of a // verdict taught the caller nothing about accessibility. Everything below stays fail-closed. canonicalizeAdjudication(adj); const itemCounts = new Map(); for (const item of adj.items) itemCounts.set(item.criteriaId, (itemCounts.get(item.criteriaId) ?? 0) + 1); for (const [criteriaId, count] of itemCounts) { if (count > 1) blame(criteriaId, `criterion ${criteriaId}: duplicate criterion id appears ${count} times in the adjudication — no implicit winner is accepted`); } const byId = new Map(adj.items.map((it) => [it.criteriaId, it])); // Coverage must use the exact same definition of "open" as the producer. Re-deriving only // the run-level manual rows here made a strict page worklist impossible to fold: criteria // reopened to close manual page cells were accepted by the model, then rejected as surplus. const packMode = !isCore(adj.standard); const open = new Set(); if (packMode) { for (const item of buildAdjudicationWorklist(audit, { standard: adj.standard, cwd: opts.cwd })) { open.add(item.criteriaId); if (byId.has(item.criteriaId)) continue; if (Object.hasOwn(expected, item.criteriaId)) uncovered(item.criteriaId); else blame(item.criteriaId, `criterion ${item.criteriaId}: missing from the adjudication (coverage gap)`); } } else { for (const r of audit.residualRisks) { open.add(r.criteriaId); if (byId.has(r.criteriaId)) continue; if (Object.hasOwn(expected, r.criteriaId)) uncovered(r.criteriaId); else blame(r.criteriaId, `criterion ${r.criteriaId}: missing from the adjudication (coverage gap)`); } } // …and the other direction. The coverage check above only proves every open criterion was // ruled on; it says nothing about a SURPLUS item. That mattered: the fold resolved a // criterion by id against the whole audit, so an extra `{criteriaId: "3.1.1", verdict: // "C"}` overwrote a non-conformity the deterministic engine had decided — no finding, no // citation, no normativeRef. Adjudication may only ever decide what the engine left open. for (const it of adj.items) { if (!open.has(it.criteriaId)) { blame(it.criteriaId, `criterion ${it.criteriaId}: not open for adjudication — the engine already decided it, or it is not part of ${adj.standard}`); } } // Per-item fail-closed validation. Grounding inputs are collected PER ITEM (not into one // flat list) for the same reason as `blame`: a failed citation has to condemn its own // criterion and no other. type Ground = { file: string; line: number; selector?: string; snippet?: string }; const groundInputs = new Map(); // Memoised: only a citation that missed its criterion's own anchors ever asks for it. let scopeCache: Set | undefined; const { citationDrift } = adjudicationLimits(opts.cwd); const scopeFiles = (): Set => (scopeCache ??= auditFiles(audit, opts.cwd)); const inAuditScope = (file: string): boolean => scopeFiles().has(file) || scopeFiles().has(canonicalEvidenceFile(file)) || withinAuditInput(file, audit, opts.cwd); // Which pack criteria the ENGINE already ruled non-conformant, indexed by the exact anchor // it ruled them on. Memoised: only an NC on a criterion that HAS a mechanical neighbour ever // asks for it, which is a minority of a worklist. `agent:` findings are excluded on purpose // — this asks what the deterministic engine established, not what a previous pass claimed. let engineNcCache: Map> | undefined; const engineNcAt = (key: string): ReadonlySet => { if (!engineNcCache) { engineNcCache = new Map(); if (!isCore(adj.standard)) { for (const pc of derivePackResults(audit, adj.standard)) { if (pc.status !== "NC") continue; for (const f of pc.findings) { if (f.advisory || f.ruleId.startsWith("agent:")) continue; const k = anchorKey(f.file, f.line, f.selectorHint); (engineNcCache.get(k) ?? engineNcCache.set(k, new Set()).get(k)!).add(pc.id); } } } } return engineNcCache.get(key) ?? EMPTY_IDS; }; const toGround = ( criteriaId: string, g: { file: string; line: number; selector?: string; snippet?: string }, fallback?: { file: string; line: number; selector?: string; snippet?: string }, offHarvest?: string, ) => { const entry = { g, ...(fallback ? { fallback } : {}), ...(offHarvest ? { offHarvest } : {}) }; const list = groundInputs.get(criteriaId); if (list) list.push(entry); else groundInputs.set(criteriaId, [entry]); }; for (const it of adj.items) { const v = it.verdict; if (v === null) { blame(it.criteriaId, `criterion ${it.criteriaId}: unadjudicated (verdict is null)`); } else if (v === "C" || v === "NA") { if (!it.justification?.trim()) blame(it.criteriaId, `criterion ${it.criteriaId}: a ${v} verdict requires a justification`); // A clearing verdict is gated exactly like an accusing one. Before this, the only // check was "the justification is a non-empty string", so `"x"` cleared a criterion — // and a model answering C to everything published a conformance nobody had assessed. const cites = (it.citations ?? []).map(readCitation).filter((c): c is Evidence => c !== null); // An INCOMPLETE reading may report a failure it saw; it may never clear what it never // looked at. Before class-based harvesting the evidence was a 30-anchor sample of a // population that ran to thousands, so a `C` was routinely a conformity claim over // elements nobody had been shown. Clearing is the direction that needs the whole set. if (v === "C" && it.evidenceComplete === false) { blame( it.criteriaId, `criterion ${it.criteriaId}: a C verdict needs the COMPLETE evidence set, and this criterion's harvest was capped at ${it.evidence.length} of ${it.population?.classes ?? "?"} content classes — record "manual", or "NC" if what you did see fails`, ); } if (it.evidence.length === 0) { // Nothing was harvested for this criterion, so there is nothing the agent could have // read to clear it. `NA` is still legitimate (the honest "no element in scope is // concerned"); `C` is not — it must stay `manual` and be ruled on against the // criterion's own tests, from a rendered capture or by hand. if (v === "C") { blame( it.criteriaId, `criterion ${it.criteriaId}: a C verdict needs evidence to cite, and none was harvested for this criterion — record "manual" (reason "undecidable"), or "NA" if nothing in scope is concerned`, ); } } else if (cites.length === 0 && v === "C") { // Only a `C` must cite. A `C` says "everything here conforms", so it has to name what // it cleared; an `NA` says "none of this is what the criterion is about", and there is // nothing to clear — asking it to cite the very elements it just ruled out of scope is // a contradiction, and one the engine does not hold ITSELF to: an engine-proved NA // (src/audit.ts subjectMatterReason) carries a justification naming what was searched // for, and no citations at all. Measured on a real run, six criteria were refused for // this alone, each of them an honest NA. // // The justification is what keeps an NA falsifiable, and it is still required above. blame( it.criteriaId, `criterion ${it.criteriaId}: a C verdict must cite at least one of the ${it.evidence.length} evidence item(s) it was shown (citations: [{file, line, …}])`, ); } else if (cites.length > 0) { // Each citation must name evidence THIS item carried. Grounding alone would only // prove the anchor exists somewhere in the tree; the pairing proves the agent ruled // on what it was actually given. // The anchor set is every occurrence of a class this criterion carries — the // representative AND its siblings — not the representatives alone. // // The check proves "you ruled on what you were given". Matching only representatives // proved something narrower and wrong: a class holds one anchor per rendered page, an // agent with file access naturally cites the occurrence it opened, and the gate then // called a real, grounded `file:line` fabricated. Measured on a real run, that is // exactly how RGAA 5.8, 9.2 and 11.1 were refused. Membership was the defect; // grounding — which proves the citation resolves against real source — is unchanged // and still runs on every one of them. const anchors = it.evidence.flatMap((e) => [`${e.file}:${e.line}`, ...(e.alsoAt ?? [])]); for (const c of cites) { // Two ways to belong, and the second one is what a good adjudicator does. // // The harvest anchors a rendered page (`.ultra11y/pages//dom.html`) because that // is what proves the browser's output — but the place a reader goes to FIX it is the // component that produced it, and that is what the agent naturally cites. Measured // on a real run: 15 of the 16 citations still refused after the drift tolerance were // in files this audit had read, pointing at the source behind the evidence. Refusing // them called correct work fabricated. // // So a citation also belongs when it lands in a file the audit read. That is a real // bound — a file outside the audited scope is still refused — and it is not the last // one: `groundFinding` below still has to find the cited content at that file:line in // the real source, which is what "not fabricated" actually means. if (!cites0(anchors, c, citationDrift) && !inAuditScope(c.file)) { blame(it.criteriaId, `criterion ${it.criteriaId}: citation ${c.file}:${c.line} is not among this criterion's harvested evidence (fabricated?)`); } // GROUND THE ANCHOR, NOT THE TRANSCRIPTION. // // Once a citation lands on evidence this criterion was shown, the harvested anchor is // the ground truth: the engine read it out of the file, so it is there by // construction. Re-checking the agent's RETYPING of it against the file tests // spelling, not evidence — and it fails constantly, because an adjudicator writes the // element out from the brief rather than copying the `snippet` field byte for byte. // // Measured on a real run: 81 criteria adjudicated, 78 of 82 citations landing exactly // on a harvested anchor, and 54 verdicts refused — every one of them for // `cited snippet not found`. A rendered snapshot serializes the document on ONE line, // so its anchors all sit at line 2 and the grounding window is the whole page: what // failed was never the location, only the transcription of it. // // A sibling from `alsoAt` carries no snippet (it is a bare `file:line`), so there is // nothing authoritative to ground — the selector probe runs against the real file // instead, which is the same check a citation with no snippet has always had. // // Anything OUTSIDE the harvest keeps the strict check below, unchanged. That is where // a fabricated location would hide, and this must not become a way to launder one. // THE CITATION FIRST, THE ANCHOR ONLY AS A FALLBACK — and the order is the whole // point. // // Whatever the agent wrote is checked against the real file, exactly as before. That // is the strongest proof there is, and it covers the case the drift window exists // for: RGAA 5.2 asks about a table's summary, the harvest anchors the ``, and // the honest citation is the `
` a line below. Grounding the anchor INSTEAD // refused precisely that — measured on a real run, 5.2, 5.4, 5.5, 11.6, 11.7 and // 11.9 all died on it, each a correct citation of a neighbour. // // Only when the citation does not ground — which is what a RETYPING looks like, and // on a one-line snapshot that is most of them — does the harvested anchor stand in. // It is authoritative there: the engine read it out of the file itself. And only // then does recognisability have anything to say, because only then is the anchor // being used to vouch for something the file did not confirm on its own. const anchor = anchorFor(it.evidence, c, citationDrift); const cite = { file: c.file, line: c.line, selector: c.selector, snippet: c.snippet }; if (anchor && recognisablySame(c, anchor.at)) { toGround( it.criteriaId, cite, anchor.representative ? { file: anchor.at.file, line: anchor.at.line, selector: anchor.at.selector ?? c.selector, snippet: anchor.at.snippet } : { file: c.file, line: c.line, selector: anchor.at.selector ?? c.selector }, ); } else { // WHY THE STRICT PATH APPLIED, said at the point it applies. // // The citation belongs (it is in a file this audit read) but it is not on an anchor // THIS criterion was shown, so the paragraph above cannot let the harvest vouch for // the transcription — and the refusal that follows says only `cited snippet not // found`, which reads as « you mistyped » when the actionable fact is « you cited // outside this criterion's evidence, so your retyping was verified literally ». // // Measured: that message sent a reader diagnosing RGAA 12.5 to the wrong conclusion // — a fold bug — with this source open. The risk is not the lost criterion, it is // the « fix » that would follow: relaxing the strict path is exactly the hole the // two bounds above exist to close. So the cause travels with the refusal. // // It bites hardest where the criterion's subject is an ABSENCE. 12.1 and 12.5 are // the repeat offenders the adjudicator prompt already names: asked to prove there is // no search engine, a model cites the search form it is arguing about — which is, by // definition, not among the anchors harvested for a criterion about navigation. // ONLY when the citation is genuinely off-harvest. Landing on an anchor and then // claiming a different KIND of element (an cited as a