diff --git a/spikes/editor-toolkit/Cargo.lock b/spikes/editor-toolkit/Cargo.lock index 9421b1b..79707e9 100644 --- a/spikes/editor-toolkit/Cargo.lock +++ b/spikes/editor-toolkit/Cargo.lock @@ -3807,6 +3807,27 @@ dependencies = [ "serde_json", ] +[[package]] +name = "round2-a11y-oracle" +version = "0.1.0" +dependencies = [ + "round2-textkit", + "serde", + "serde_json", + "unicode-normalization", + "unicode-segmentation", +] + +[[package]] +name = "round2-candidatekit" +version = "0.1.0" +dependencies = [ + "round2-diff", + "round2-textkit", + "serde", + "serde_json", +] + [[package]] name = "round2-diff" version = "0.1.0" diff --git a/spikes/editor-toolkit/Cargo.toml b/spikes/editor-toolkit/Cargo.toml index a5a539f..a1d4ed2 100644 --- a/spikes/editor-toolkit/Cargo.toml +++ b/spikes/editor-toolkit/Cargo.toml @@ -12,6 +12,8 @@ members = [ "round2-textkit", "round2-svgref", "round2-reference", + "round2-candidatekit", + "round2-a11y-oracle", ] # a11y-verifier is a standalone Python script (a11y-verifier/verify.py), not diff --git a/spikes/editor-toolkit/a11y-verifier/.gitignore b/spikes/editor-toolkit/a11y-verifier/.gitignore new file mode 100644 index 0000000..c18dd8d --- /dev/null +++ b/spikes/editor-toolkit/a11y-verifier/.gitignore @@ -0,0 +1 @@ +__pycache__/ diff --git a/spikes/editor-toolkit/a11y-verifier/test_verify.py b/spikes/editor-toolkit/a11y-verifier/test_verify.py new file mode 100644 index 0000000..cf6d607 --- /dev/null +++ b/spikes/editor-toolkit/a11y-verifier/test_verify.py @@ -0,0 +1,1230 @@ +"""Plain `unittest` coverage for `verify.classify` and +`verify.validate_expectations_file` — check 5's whole scoring logic and its +fail-closed artifact validation, isolated from any live AT-SPI bus. + +`verify.py`'s `gi.repository.Atspi` import happens inside `main()`, not at +module import time, specifically so this file can `import verify` and drive +`classify()` directly with synthetic `ObservedNode` trees — no bus, no +desktop, no display required. Run with: + + python3 -m unittest + +from this directory (`a11y-verifier/`). + +Every test constructs the *wrong* input for the property it names and +asserts the *specific* verdict/outcome that input must produce — never a +bare "FAIL" or "not PASS", because a classifier branch that silently +degrades to the wrong FAIL reason would pass a weaker test and is exactly +the kind of regression this file exists to catch. +""" +import unittest + +from verify import ( + EXPECTED_FIXTURE_IDS, + PLATFORM, + VISUAL_ORDER_TRAP, + ObservedNode, + Verdict, + _is_source_bearing_fragment, + classify, + validate_expectations_file, +) + + +def node(role, name, children=None): + """A synthetic `ObservedNode`, the same shape `walk_for_check5` builds + from a live AT-SPI tree.""" + return ObservedNode(role=role, name=name, children=list(children) if children else []) + + +def expectation( + expected_name, + accepted_roles=("label", "static", "text", "paragraph"), + prohibited_roles=("image", "canvas", "filler", "panel", "unknown"), + alternative_forms=None, + visual_order_name=None, + source_atoms=None, +): + """A synthetic fixture entry with the same shape + `round2-a11y-oracle/a11y_expectations.json` emits.""" + return { + "fixture_id": "F-TEST", + "expected_name": expected_name, + "expected_name_hex": expected_name.encode("utf-8").hex(), + "expected_name_byte_len": len(expected_name.encode("utf-8")), + "accepted_roles": list(accepted_roles), + "prohibited_roles": list(prohibited_roles), + "alternative_forms": alternative_forms or {}, + "visual_order_name": visual_order_name, + # D1: per-segment source atoms (`round2-a11y-oracle`'s + # `source_atoms`). Defaults to `None` (classify's own `.get(...) or + # []` treats that as no atoms), since most tests don't need one. + "source_atoms": source_atoms, + } + + +class ByteExactPass(unittest.TestCase): + def test_single_node_with_accepted_role_and_exact_name_passes(self): + exp = expectation("Allegro affettuoso — al fine") + verdict = classify(exp, [node("text", "Allegro affettuoso — al fine")]) + self.assertEqual(verdict.verdict, "PASS") + self.assertIsNone(verdict.prohibited_outcome) + self.assertEqual(verdict.observed_role, "text") + self.assertEqual(verdict.observed_name, exp["expected_name"]) + + def test_a_one_byte_difference_does_not_pass(self): + """Mutation guard: if byte comparison were replaced by e.g. a + case-insensitive or trimmed comparison, this would wrongly PASS.""" + exp = expectation("Allegro") + verdict = classify(exp, [node("text", "allegro")]) + self.assertNotEqual(verdict.verdict, "PASS") + + +class CompositionConcatenationPass(unittest.TestCase): + def test_two_segment_names_concatenated_in_tree_order_pass(self): + exp = expectation("Coro אבג") + # No single node carries the whole name — only the concatenation of + # two text descendants of a common parent, in tree/logical order, + # does. + run = node("frame", "", children=[node("text", "Coro "), node("text", "אבג")]) + verdict = classify(exp, [run]) + self.assertEqual(verdict.verdict, "PASS") + self.assertIsNone(verdict.prohibited_outcome) + self.assertEqual(verdict.observed_name, exp["expected_name"]) + + def test_concatenation_in_the_wrong_order_does_not_pass(self): + """Mutation guard: if concatenation order were unspecified (e.g. a + set instead of an ordered list), swapping the two nodes would still + wrongly PASS.""" + exp = expectation("Coro אבג") + run = node("frame", "", children=[node("text", "אבג"), node("text", "Coro ")]) + verdict = classify(exp, [run]) + self.assertNotEqual(verdict.verdict, "PASS") + + +class ProhibitedOutcomes(unittest.TestCase): + """One test per `round2_textkit::a11y::PROHIBITED_OUTCOMES` name.""" + + def test_absent_from_tree_when_no_candidate_node_is_found(self): + exp = expectation("Allegro") + verdict = classify(exp, []) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "absent-from-tree") + + def test_name_empty_is_distinct_from_absent_from_tree(self): + """§8.3: "name-empty ... absence wearing a role." A node is present + (unlike the absent-from-tree case above) but its name is the empty + string — these must classify differently.""" + exp = expectation("Allegro") + verdict = classify(exp, [node("label", "")]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "name-empty") + + def test_name_normalized_matches_the_precommitted_nfc_form(self): + # F-E's case: source is NFD ("Cafe" + combining acute, "Café"), + # a wrong tree exposes the NFC "Café" ("Café", a *different* + # string byte-for-byte even though the two render identically) + # instead. Written with explicit \N escapes rather than the literal + # glyph so the two forms cannot be silently typed as the same string + # by accident — that mistake produced a false PASS here once already. + nfd = "Cafe\N{COMBINING ACUTE ACCENT}" + nfc = "Caf\N{LATIN SMALL LETTER E WITH ACUTE}" + self.assertNotEqual(nfd, nfc, "anchor: the two forms must be different strings") + exp = expectation( + nfd, + alternative_forms={"name-normalized": [nfc]}, + ) + verdict = classify(exp, [node("text", nfc)]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "name-normalized") + self.assertEqual(verdict.observed_name, nfc) + + def test_name_is_shaped_glyphs_matches_the_precommitted_ligature_form(self): + # F-A's case: the `ff` ligature collapses, so a wrong tree drops a + # letter relative to the source string. + exp = expectation( + "Allegro affettuoso — al fine", + alternative_forms={"name-is-shaped-glyphs": ["Allegro afettuoso — al fne"]}, + ) + verdict = classify(exp, [node("text", "Allegro afettuoso — al fne")]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "name-is-shaped-glyphs") + + def test_name_drops_unresolved_codepoints_matches_the_precommitted_form(self): + # F-C's case: U+0627 is covered by no declared face and is dropped. + exp = expectation( + "Coro ا", + alternative_forms={"name-drops-unresolved-codepoints": ["Coro "]}, + ) + verdict = classify(exp, [node("static", "Coro ")]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "name-drops-unresolved-codepoints") + + def test_a_correct_name_does_not_spuriously_match_an_alternative_form(self): + """Mutation guard: if alternative-form matching ran before the exact + match, a correct name would risk matching an alternative_forms entry + by accident and wrongly FAIL.""" + exp = expectation( + "Coro ا", + alternative_forms={"name-drops-unresolved-codepoints": ["Coro "]}, + ) + verdict = classify(exp, [node("static", "Coro ا")]) + self.assertEqual(verdict.verdict, "PASS") + + +class VisualOrderDiagnosis(unittest.TestCase): + def test_visual_order_concatenation_is_named_specifically(self): + # F-D's designed trap: a tree walking visual runs left to right + # reverses the embedded RTL segment's codepoint order. + exp = expectation( + "Allegro אבג con brio", + visual_order_name="Allegro גבא con brio", + ) + run = node( + "frame", + "", + children=[node("text", "Allegro "), node("text", "גבא"), node("text", " con brio")], + ) + verdict = classify(exp, [run]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, VISUAL_ORDER_TRAP) + self.assertEqual(verdict.observed_name, exp["visual_order_name"]) + + def test_visual_order_diagnosis_is_not_reported_as_a_generic_mismatch(self): + """Mutation guard: if the visual-order check were deleted, this + would still FAIL but with `prohibited_outcome=None` (the generic + fallback) instead of the specific diagnosis — asserting the exact + name, not just FAIL, is what catches that.""" + exp = expectation( + "Allegro אבג con brio", + visual_order_name="Allegro גבא con brio", + ) + run = node( + "frame", + "", + children=[node("text", "Allegro "), node("text", "גבא"), node("text", " con brio")], + ) + verdict = classify(exp, [run]) + self.assertIsNotNone(verdict.prohibited_outcome) + self.assertNotEqual(verdict.prohibited_outcome, "absent-from-tree") + self.assertNotEqual(verdict.prohibited_outcome, "name-empty") + + +class ProhibitedRole(unittest.TestCase): + def test_a_prohibited_role_fails_even_with_the_exact_name(self): + exp = expectation("Allegro") + verdict = classify(exp, [node("canvas", "Allegro")]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.observed_role, "canvas") + # This is a role-vocabulary failure (§8.2), not one of the five + # name-transformation PROHIBITED_OUTCOMES (§8.3) — it must not be + # reported as one. + self.assertIsNone(verdict.prohibited_outcome) + self.assertEqual( + verdict.reason, + "a node's name matches expected_name byte-for-byte, but its role 'canvas' is in the " + "at-spi2 prohibited set (no accepted-role node also carries it)", + ) + + def test_an_accepted_role_with_the_exact_name_is_not_penalized(self): + """Mutation guard: confirms the previous test is actually exercising + the role check, not some other reason that name would fail.""" + exp = expectation("Allegro") + verdict = classify(exp, [node("label", "Allegro")]) + self.assertEqual(verdict.verdict, "PASS") + + def test_an_unlisted_role_carrying_the_exact_name_is_not_absent_from_tree(self): + """A role outside `accepted_roles | prohibited_roles` (e.g. `push + button`) is never a text-*candidate* for the composition/single-node + role checks — `_flatten_candidates`/`_subtree_contributors` exclude + it. But a node under that role can still carry the run's exact + text, and §8.3's absent-from-tree ("the default outcome for a + toolkit that draws to a canvas and stops") does not describe that: + the run *is* in the tree. This must FAIL naming the actual observed + role, not report absent-from-tree — conflating "wrong role" with + "nothing there at all" would hide evidence a real candidate + produced.""" + exp = expectation("Allegro") + verdict = classify(exp, [node("push button", "Allegro")]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertNotEqual(verdict.prohibited_outcome, "absent-from-tree") + self.assertEqual(verdict.observed_role, "push button") + self.assertEqual(verdict.observed_name, "Allegro") + + def test_a_truly_empty_tree_is_still_absent_from_tree(self): + """The companion case to the one above, pinned side by side so a + regression that merges the two back together (e.g. by making the + new unlisted-role scan fire unconditionally) is caught: with + nothing in the tree at all, the outcome must still be + `absent-from-tree`.""" + exp = expectation("Allegro") + verdict = classify(exp, []) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "absent-from-tree") + + def test_an_unlisted_role_carrying_an_alternative_form_is_also_not_absent_from_tree(self): + """The whole-tree scan must check precommitted alternative forms + too, not only `expected_name` — a candidate that shaped the name + wrong *and* exposed it under an unlisted role has still put the + (wrong) text in the tree, which is a different, more specific, + finding than "nothing is there.""" + exp = expectation( + "Coro ا", + alternative_forms={"name-drops-unresolved-codepoints": ["Coro "]}, + ) + verdict = classify(exp, [node("push button", "Coro ")]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertNotEqual(verdict.prohibited_outcome, "absent-from-tree") + self.assertEqual(verdict.observed_role, "push button") + self.assertEqual(verdict.observed_name, "Coro ") + + def test_an_unlisted_role_carrying_the_visual_order_form_is_also_not_absent_from_tree(self): + exp = expectation( + "Allegro אבג con brio", + visual_order_name="Allegro גבא con brio", + ) + verdict = classify(exp, [node("push button", "Allegro גבא con brio")]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertNotEqual(verdict.prohibited_outcome, "absent-from-tree") + self.assertEqual(verdict.observed_role, "push button") + + def test_an_unlisted_role_carrying_an_unrelated_name_is_still_absent_from_tree(self): + """Mutation guard: the whole-tree scan must only match a name + against `expected_name`/alternative forms/`visual_order_name` — a + node with an unlisted role and completely unrelated text must not + rescue the verdict away from absent-from-tree either.""" + exp = expectation("Allegro") + verdict = classify(exp, [node("push button", "something unrelated")]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "absent-from-tree") + + +class MultipleFormsPerOutcome(unittest.TestCase): + """O2: one outcome can carry more than one precommitted rendering — + `round2-a11y-oracle` gives `name-is-shaped-glyphs` both a + cluster-collapse form and a ligature presentation-form substitution for + F-A. Either one observed must classify the same outcome.""" + + def _f_a_like_expectation(self): + return expectation( + "Allegro affettuoso — al fine", + alternative_forms={ + "name-is-shaped-glyphs": [ + "Allegro afettuoso — al fne", + "Allegro a\N{LATIN SMALL LIGATURE FF}ettuoso — al \N{LATIN SMALL LIGATURE FI}ne", + ] + }, + ) + + def test_the_cluster_collapse_form_classifies(self): + verdict = classify( + self._f_a_like_expectation(), [node("text", "Allegro afettuoso — al fne")] + ) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "name-is-shaped-glyphs") + + def test_the_presentation_form_classifies_the_same_outcome(self): + presentation = ( + "Allegro a\N{LATIN SMALL LIGATURE FF}ettuoso — al \N{LATIN SMALL LIGATURE FI}ne" + ) + verdict = classify(self._f_a_like_expectation(), [node("text", presentation)]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "name-is-shaped-glyphs") + + def test_a_third_string_matches_neither_form(self): + """Mutation guard: confirms the two tests above are matching a + specific listed string, not merely "anything different from + expected_name" — if list matching degenerated to that, this would + wrongly report name-is-shaped-glyphs too.""" + verdict = classify( + self._f_a_like_expectation(), + [node("text", "something else entirely")], + ) + self.assertNotEqual(verdict.prohibited_outcome, "name-is-shaped-glyphs") + + +class SubtreeScopedComposition(unittest.TestCase): + """B1: composition is scored against one run subtree's own descendants, + never the whole application flattened into one list. These are the two + cases the coordinator reproduced against the pre-B1 flat classifier: + + - `[("canvas", "Coro "), ("text", "אבג")]` wrongly PASSed. + - `[("text", "Coro "), ("text", "אבג"), ("label", "MyApp Window")]` + wrongly FAILed, even though `label` is an accepted at-spi2 role that + every real application's window frame carries, elsewhere in the tree. + """ + + def _f_b_like_expectation(self): + return expectation("Coro אבג") + + def test_a_prohibited_role_sibling_in_the_same_run_subtree_fails_with_the_role_named(self): + """The false-PASS case (B1's first reproduction), expressed as a + real tree: `canvas` and `text` are siblings under one run subtree — + together they still spell out expected_name byte-for-byte, but a + `canvas` contributed to it, which must FAIL, naming `canvas`, not + PASS.""" + run = node("frame", "", children=[node("canvas", "Coro "), node("text", "אבג")]) + verdict = classify(self._f_b_like_expectation(), [run]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.observed_role, "canvas") + self.assertEqual(verdict.observed_name, "Coro אבג") + self.assertIn("canvas", verdict.reason) + self.assertIn("otherwise-correct composition", verdict.reason) + # Not one of the five PROHIBITED_OUTCOMES names — this is a §8.2 + # role-vocabulary failure inside a composition, not a §8.3 name + # transformation. + self.assertIsNone(verdict.prohibited_outcome) + + def test_an_unrelated_accepted_role_node_elsewhere_does_not_poison_a_correct_composition(self): + """The false-FAIL case (B1's second reproduction), expressed as a + real tree: the run's own two text nodes are correctly grouped under + their own subtree; an unrelated `label` (an *accepted* at-spi2 role + — every real window frame carries one) sits elsewhere in the same + application. The label must not be able to corrupt the run's own, + otherwise-correct, composition into a FAIL.""" + application = node( + "frame", + "", + children=[ + node("group", "", children=[node("text", "Coro "), node("text", "אבג")]), + node("label", "MyApp Window"), + ], + ) + verdict = classify(self._f_b_like_expectation(), [application]) + self.assertEqual(verdict.verdict, "PASS") + self.assertIsNone(verdict.prohibited_outcome) + self.assertEqual(verdict.observed_name, "Coro אבג") + + def test_the_legitimate_accepted_role_split_still_passes_when_it_is_the_whole_tree(self): + """Sanity companion to the two reproductions above: a run correctly + split across two accepted-role text nodes, with nothing else in the + tree at all, must still PASS — B1's fix must not have become so + conservative that it stopped recognizing the ordinary case.""" + run = node("paragraph", "", children=[node("text", "Coro "), node("text", "אבג")]) + verdict = classify(self._f_b_like_expectation(), [run]) + self.assertEqual(verdict.verdict, "PASS") + self.assertIsNone(verdict.prohibited_outcome) + + def test_a_prohibited_contributor_is_named_even_when_a_correct_subtree_exists_elsewhere(self): + """The composition scan must not let a PASS found in one subtree + erase evidence of a bad contributor found in *another* — but it must + still prefer reporting the PASS overall, since a candidate that gets + it right anywhere in a legitimate run subtree has satisfied §8.1. + This test pins the reverse: when NO subtree passes cleanly, the + reported reason must name the actual bad contributor, not a generic + mismatch.""" + run = node( + "frame", + "", + children=[node("canvas", "Coro "), node("text", "אבג")], + ) + verdict = classify(self._f_b_like_expectation(), [run]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.observed_role, "canvas") + self.assertNotIn("no node or composition matched", verdict.reason) + + def test_a_composition_that_silently_dropped_a_prohibited_contributor_would_be_wrong(self): + """Guards the specific failure mode B1 warns against: a subtree must + not PASS by concatenating only its *accepted*-role contributors and + ignoring a prohibited one. Here, dropping the `canvas` node would + leave just `"אבג"`, which does not equal expected_name either — so + this also confirms the concatenation includes every contributor's + name, not a filtered subset, before the role check ever runs.""" + run = node("frame", "", children=[node("canvas", "Coro "), node("text", "אבג")]) + verdict = classify(self._f_b_like_expectation(), [run]) + # If contributors had been filtered to accepted-only before + # concatenating, the concat would be "אבג" (not expected_name), and + # this subtree would be silently skipped rather than FAILed with a + # named reason — falling through to a *weaker* diagnosis than the + # sharp one B1 requires. + self.assertEqual(verdict.verdict, "FAIL") + self.assertIn("canvas", verdict.reason) + + +class ContainerNamingDoesNotChangeBlame(unittest.TestCase): + """An empty-named structural wrapper (a `frame` around the real + contributors, exposing no name of its own) must never be blamed for a + bad composition. It contributes zero bytes to the concatenation, so it + cannot be what made the composition wrong; blaming it hides the actual + offender — here, a *prohibited*-role `canvas` that carried half the + run, which is the real §8.2 violation the report exists to name. + + Same tree content in all three shapes below; only the container's own + name (or its absence) differs. All three must name `canvas`.""" + + def _f_b_like_expectation(self): + return expectation("Coro אבג") + + def test_canvas_and_text_inside_an_unnamed_frame_blames_canvas(self): + tree = node( + "application", + "p", + children=[node("frame", "", children=[node("canvas", "Coro "), node("text", "אבג")])], + ) + verdict = classify(self._f_b_like_expectation(), [tree]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.observed_role, "canvas") + + def test_canvas_and_text_inside_a_named_frame_blames_canvas(self): + tree = node( + "application", + "p", + children=[ + node("frame", "MyApp", children=[node("canvas", "Coro "), node("text", "אבג")]) + ], + ) + verdict = classify(self._f_b_like_expectation(), [tree]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.observed_role, "canvas") + + def test_canvas_and_text_as_direct_siblings_blames_canvas(self): + tree = node( + "application", + "p", + children=[node("canvas", "Coro "), node("text", "אבג")], + ) + verdict = classify(self._f_b_like_expectation(), [tree]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.observed_role, "canvas") + + def test_an_unlisted_but_not_prohibited_contributor_is_still_named_when_no_prohibited_one_exists( + self, + ): + """The weaker fallback branch stays reachable: with no + prohibited-role contributor at all, an unlisted-role one is still + named (not silently dropped just because it is the weaker case). + Nested under an unnamed wrapper, so this test also isolates the + empty-name exclusion on its own: with no prohibited contributor + present, the "prefer prohibited" rule cannot be what saves this + case from blaming the wrapper — only excluding the empty-named + `frame` from the contributor set can. A mutation that deleted the + empty-name exclusion (but kept the prohibited-preference) would + wrongly blame `frame` here, even though the same mutation happens + to survive the two `blames_canvas` tests above (where a prohibited + `canvas` is also present and the preference rule alone rescues + them).""" + tree = node( + "application", + "p", + children=[ + node( + "frame", "", children=[node("push button", "Coro "), node("text", "אבג")] + ) + ], + ) + verdict = classify(self._f_b_like_expectation(), [tree]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.observed_role, "push button") + self.assertIn("neither accepted nor prohibited", verdict.reason) + + def test_a_prohibited_contributor_is_preferred_over_an_unlisted_one_listed_first(self): + """Isolates the "prefer prohibited" rule specifically, with no + empty-named node anywhere in the tree: an unlisted-role + (`push button`) contributor is listed *before* a prohibited-role + (`canvas`) one, both non-empty-named. Naming "the first non-accepted + contributor" (no preference) would wrongly blame `push button`; only + the explicit prohibited-preference blames `canvas`.""" + tree = node( + "frame", + "", + children=[node("push button", "Coro "), node("canvas", "אבג")], + ) + verdict = classify(self._f_b_like_expectation(), [tree]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.observed_role, "canvas") + self.assertIn("prohibited", verdict.reason) + + def test_a_correct_split_run_nested_under_an_unnamed_container_still_passes(self): + """Confirms excluding empty-named nodes does not disturb a + legitimate PASS: the concatenation is unchanged whether the + empty-named wrapper is included or excluded (it contributes zero + bytes either way), so a correct split run nested under one must + still pass.""" + tree = node( + "application", + "p", + children=[node("frame", "", children=[node("text", "Coro "), node("text", "אבג")])], + ) + verdict = classify(self._f_b_like_expectation(), [tree]) + self.assertEqual(verdict.verdict, "PASS") + + +class ExactNamePrecedence(unittest.TestCase): + """C1: exact-name matching must evaluate every node before deciding, + never return on the first match — an accepted-role node carrying + expected_name wins regardless of where in the tree it sits, even when a + prohibited-role node carrying the *same* exact name is listed first.""" + + @staticmethod + def _forest(canvas_first): + canvas = node("canvas", "Coro אבג") + text = node("text", "Coro אבג") + return [canvas, text] if canvas_first else [text, canvas] + + def test_prohibited_role_node_listed_first_still_passes(self): + exp = expectation("Coro אבג") + verdict = classify(exp, self._forest(canvas_first=True)) + self.assertEqual(verdict.verdict, "PASS") + self.assertEqual(verdict.observed_role, "text") + self.assertIsNone(verdict.prohibited_outcome) + + def test_accepted_role_node_listed_first_still_passes(self): + exp = expectation("Coro אבג") + verdict = classify(exp, self._forest(canvas_first=False)) + self.assertEqual(verdict.verdict, "PASS") + self.assertEqual(verdict.observed_role, "text") + self.assertIsNone(verdict.prohibited_outcome) + + def test_both_orderings_of_the_forest_produce_the_same_verdict(self): + """The direct C1 reproduction: the coordinator measured opposite + verdicts for the two orderings of this exact forest. Pinned here as + one assertion comparing both `classify` calls, not two independently + hand-written expectations that could each be individually wrong in + the same direction.""" + exp = expectation("Coro אבג") + v_canvas_first = classify(exp, self._forest(canvas_first=True)) + v_text_first = classify(exp, self._forest(canvas_first=False)) + self.assertEqual(v_canvas_first.verdict, v_text_first.verdict) + self.assertEqual(v_canvas_first.observed_role, v_text_first.observed_role) + self.assertEqual(v_canvas_first.prohibited_outcome, v_text_first.prohibited_outcome) + self.assertEqual(v_canvas_first.verdict, "PASS") + + def test_without_any_accepted_role_match_the_prohibited_one_still_fails(self): + """Mutation guard: confirms the PASSes above happen *because* an + accepted-role match exists, not because exact-name matching became + unconditional PASS — with only the prohibited-role node present, + this must still FAIL.""" + exp = expectation("Coro אבג") + verdict = classify(exp, [node("canvas", "Coro אבג")]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.observed_role, "canvas") + + +class SubtreeScopedDiagnoses(unittest.TestCase): + """C2: the alternative-form and visual-order diagnoses are scored per + subtree, exactly like composition (B1) — an unrelated accepted-role node + elsewhere in the application (e.g. a window `label`) must not poison a + legitimate subtree's diagnosis into a generic mismatch.""" + + def test_visual_order_trap_is_found_despite_an_unrelated_window_label(self): + """The direct C2 reproduction: F-D's designed visual-order trap, + with an unrelated `label` elsewhere in the application.""" + exp = expectation( + "Allegro אבג con brio", + visual_order_name="Allegro גבא con brio", + ) + application = node( + "application", + "p", + children=[ + node("label", "MyApp Window"), + node( + "frame", + "", + children=[ + node("text", "Allegro "), + node("text", "גבא"), + node("text", " con brio"), + ], + ), + ], + ) + verdict = classify(exp, [application]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, VISUAL_ORDER_TRAP) + self.assertNotIn("no node or composition matched", verdict.reason) + + def test_alternative_form_composition_is_found_despite_an_unrelated_window_label(self): + """The same poisoning bug, for an alternative-form composition (not + visual-order) — the ligature-collapse form split across two text + nodes, so this exercises the *subtree-concatenation* alt-form path + specifically, not the already-order-independent single-node one.""" + exp = expectation( + "Allegro affettuoso — al fine", + alternative_forms={"name-is-shaped-glyphs": ["Allegro afettuoso — al fne"]}, + ) + application = node( + "application", + "p", + children=[ + node("label", "MyApp Window"), + node( + "frame", + "", + children=[ + node("text", "Allegro afettuoso — al "), + node("text", "fne"), + ], + ), + ], + ) + verdict = classify(exp, [application]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "name-is-shaped-glyphs") + self.assertNotIn("no node or composition matched", verdict.reason) + + def test_without_the_unrelated_label_the_same_composition_still_matches(self): + """Mutation guard: confirms the two tests above are really about the + label not poisoning the result, not about some other property of + the tree shape — remove the label and the same diagnosis must still + fire.""" + exp = expectation( + "Allegro אבג con brio", + visual_order_name="Allegro גבא con brio", + ) + frame = node( + "frame", + "", + children=[node("text", "Allegro "), node("text", "גבא"), node("text", " con brio")], + ) + verdict = classify(exp, [frame]) + self.assertEqual(verdict.prohibited_outcome, VISUAL_ORDER_TRAP) + + +class UnlistedRoleComposition(unittest.TestCase): + """C3: text composed across two or more unlisted-role descendants must + not be misreported as absent-from-tree — the same per-subtree + composition scoring applied to roles in neither `accepted_roles` nor + `prohibited_roles`.""" + + def test_two_unlisted_role_nodes_composing_the_exact_name_is_not_absent(self): + """The direct C3 reproduction: two `push button` nodes whose + concatenation is the run's exact text.""" + exp = expectation("Coro אבג") + application = node( + "application", + "p", + children=[node("push button", "Coro "), node("push button", "אבג")], + ) + verdict = classify(exp, [application]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertNotEqual(verdict.prohibited_outcome, "absent-from-tree") + self.assertEqual(verdict.observed_name, "Coro אבג") + + def test_unlisted_role_composition_with_unrelated_text_is_still_absent(self): + """The composition-scan analogue of the single-node absent-from-tree + guard: two unlisted-role nodes whose concatenation is *not* the + run's text, an alternative form, or visual_order_name, must still + classify as genuine absence — the scan must not over-fire just + because *some* unlisted-role composition exists.""" + exp = expectation("Coro אבג") + application = node( + "application", + "p", + children=[node("push button", "something"), node("push button", "unrelated")], + ) + verdict = classify(exp, [application]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "absent-from-tree") + + +class AbsentFromTreeVsMisorderedFragments(unittest.TestCase): + """User ruling, following C3: a permutation probe found that reversing + the same two contributors gave `absent-from-tree` under unlisted roles + but a plain generic mismatch under accepted roles — two different named + outcomes for the same underlying defect (contributors in the wrong + order). Three cases, side by side, each asserting its own distinct + outcome, so a future change cannot silently re-merge them. + + **Behaviour change, reported explicitly (not adjusted quietly), from the + reorder that moved source-bearing detection to run before absence/ + name-empty (the very next ruling in this same sequence):** case 1 below + used to assert "unchanged, a generic composition FAIL" — that was true + only because the fragment scan, at the time, ran solely inside the + `flat_candidates`-empty branch and so never even looked at accepted-role + contributors. Once source-bearing detection (fragments included) was + unified to run over *every* role unconditionally, the identical + reversed-`text` case is now *also* caught by the fragment scan, with a + more specific message than the old generic mismatch — which is the + intended, uniform consequence of "misordered source fragments -> + composition/role failure, never absence" applying without a role + exception. Nothing about *this* file's tests silently changed; the + updated assertion below is that report. + + 1. reversed **accepted**-role contributors — a role/composition FAIL + (fragment-scan diagnosis, naming the fragments), never + `absent-from-tree`; + 2. reversed **unlisted**-role contributors — the original fix: a + role/composition FAIL, never `absent-from-tree`; + 3. the true-absence control — nothing resembling the run's text + anywhere — still reaches `absent-from-tree`, proving the fix + narrowed the bug without making the outcome unreachable. + """ + + def test_reversed_accepted_role_contributors_are_a_composition_failure_not_absence(self): + """Behaviour change (see class docstring): this used to assert a + generic mismatch (`prohibited_outcome=None`, "no node or + composition matched..."). It now asserts the fragment-scan + diagnosis — still `prohibited_outcome=None` (a §8.2 role/composition + failure, not a §8.3 PROHIBITED_OUTCOMES name), but a more specific + reason, because the fragment scan is no longer gated to unlisted + roles only.""" + exp = expectation("Coro אבג") + run = node("frame", "", children=[node("text", "אבג"), node("text", "Coro ")]) + verdict = classify(exp, [run]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertIsNone(verdict.prohibited_outcome) + self.assertNotEqual(verdict.prohibited_outcome, "absent-from-tree") + self.assertIn("role/composition failure", verdict.reason) + + def test_reversed_unlisted_role_contributors_are_a_role_failure_not_absence(self): + """Item 2, the fix itself — the coordinator's exact reproduction: + the identical shape, under unlisted roles, must NOT be + `absent-from-tree`. The text is genuinely present; only the order + (and the role) is wrong.""" + exp = expectation("Coro אבג") + application = node( + "application", + "p", + children=[node("push button", "אבג"), node("push button", "Coro ")], + ) + verdict = classify(exp, [application]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertNotEqual(verdict.prohibited_outcome, "absent-from-tree") + # Not a §8.3 PROHIBITED_OUTCOMES name either — this is the §8.2 + # role/composition failure category, the same as C3's other cases. + self.assertIsNone(verdict.prohibited_outcome) + + def test_true_absence_is_still_reachable(self): + """Item 1, the control: with nothing resembling the run's text + anywhere, `absent-from-tree` must still fire — proving the fix + narrowed the bug rather than making the outcome unreachable.""" + exp = expectation("Coro אבג") + verdict = classify(exp, []) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "absent-from-tree") + + +class SourceBearingFragmentGuards(unittest.TestCase): + """Prove `_is_source_bearing_fragment` (C3's fix) is not loose enough to + make `absent-from-tree` unreachable in practice — the user's own stated + concern: "a rule loose enough that an application's ordinary window + label counts as a fragment would make absent-from-tree unreachable in + practice, which is a worse failure than the one being fixed." Each + guard below is the specific scenario that concern describes.""" + + def test_a_single_shared_character_is_not_a_fragment(self): + """A one-character coincidence — "o" appears in both "Coro" and + almost any ordinary English text — must not count; this is exactly + the case the length-2 floor exists to exclude.""" + self.assertFalse(_is_source_bearing_fragment("o", {"Coro אבג"})) + + def test_an_ordinary_window_label_is_not_a_fragment(self): + """The user's own example, direct: a whole, realistic window title + is longer than (and unrelated to) the run's short text, so it can + never be a literal substring of it.""" + self.assertFalse(_is_source_bearing_fragment("MyApp Window", {"Coro אבג"})) + + def test_a_whitespace_only_name_is_not_a_fragment(self): + self.assertFalse(_is_source_bearing_fragment(" ", {"Coro אבג"})) + + def test_an_empty_name_is_not_a_fragment(self): + self.assertFalse(_is_source_bearing_fragment("", {"Coro אבג"})) + + def test_a_two_character_real_fragment_does_count(self): + """Anchors the floor at exactly two characters, not three or more — + confirms the guards above are testing the length-1 boundary + specifically, not merely "short strings never match".""" + self.assertTrue(_is_source_bearing_fragment("בג", {"Coro אבג"})) + + def test_an_ordinary_window_label_does_not_rescue_a_tree_from_absence_end_to_end(self): + """The end-to-end version of the guard above: a real, unrelated, + realistic window label under an unlisted role, with nothing else in + the tree, must still classify as absent-from-tree.""" + exp = expectation("Coro אבג") + application = node( + "application", + "p", + # An unlisted role, not "label" (which is an accepted at-spi2 + # role and would exit the C3 branch this test is about via the + # ordinary accepted-role path instead). + children=[node("push button", "MyApp Window")], + ) + verdict = classify(exp, [application]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "absent-from-tree") + + +class FCTwoSegmentComposition(unittest.TestCase): + """D1/D2 regression group, required together — all four reproduce + against F-C's real shape (`"Coro "` + `"ا"`, with `"Coro "` also being + F-C's own precommitted `name-drops-unresolved-codepoints` form) and must + all hold simultaneously: + + unlisted "ا" for F-C -> NOT absent-from-tree + unrelated single ASCII "o" -> still NOT source-bearing + F-C accepted split ["Coro ", "ا"] -> PASS + F-C lone accepted node named exactly "Coro " -> still name-drops-unresolved-codepoints + + The first two prove D1 (an unresolved segment can be a single character, + which the length-2 substring rule alone cannot catch, but the + coincidence guard must still hold); the last two prove D2 (a legitimate + two-node split now PASSes despite the first segment alone matching a + precommitted alternative form, and the single-node case — which has no + composition to find — still names that outcome exactly as before). + """ + + def _f_c_like_expectation(self): + return expectation( + "Coro ا", + alternative_forms={"name-drops-unresolved-codepoints": ["Coro "]}, + source_atoms=["Coro ", "ا"], + ) + + def test_unlisted_role_carrying_f_cs_unresolved_segment_is_not_absent_from_tree(self): + """D1, the reported finding: F-C's unresolved segment `ا` is a + single character — below `_is_source_bearing_fragment`'s length-2 + floor — but it is still a precommitted `source_atoms` entry, so a + node carrying it under an unlisted role must be a role/composition + failure, never absence. §8.3: "the accessibility tree carries the + text, not the ink".""" + exp = self._f_c_like_expectation() + verdict = classify(exp, [node("push button", "ا")]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertNotEqual(verdict.prohibited_outcome, "absent-from-tree") + self.assertEqual(verdict.observed_role, "push button") + + def test_an_unrelated_single_ascii_character_is_still_not_source_bearing(self): + """Mutation guard, D1: confirms the atom-matching path is additive + and precommitted, not a blanket "any single character counts" rule + — an unrelated stray `"o"` (present in `"Coro"` only by coincidence, + and not a `source_atoms` entry) must still not rescue the tree from + absence, exactly as `_is_source_bearing_fragment`'s own coincidence + guard already requires on its own.""" + exp = self._f_c_like_expectation() + verdict = classify(exp, [node("push button", "o")]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "absent-from-tree") + + def test_the_legitimate_two_node_split_passes(self): + """D2, the fix itself — the coordinator's exact reproduction: two + ACCEPTED-role text nodes, one per direction run (exactly what §8.1 + permits: "a tree that exposes one text node per direction run is + not wrong"), must PASS even though the first segment alone happens + to equal F-C's own precommitted `name-drops-unresolved-codepoints` + form.""" + exp = self._f_c_like_expectation() + run = node("frame", "", children=[node("text", "Coro "), node("text", "ا")]) + verdict = classify(exp, [run]) + self.assertEqual(verdict.verdict, "PASS") + self.assertIsNone(verdict.prohibited_outcome) + + def test_a_lone_accepted_node_named_exactly_coro_is_still_the_named_outcome(self): + """The required guard proving D2's fix did not simply disable the + alternative-form diagnosis: with no second node, there is no + composition to find, so a single `text:"Coro "` node must still be + classified by name, exactly as before D2.""" + exp = self._f_c_like_expectation() + verdict = classify(exp, [node("text", "Coro ")]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "name-drops-unresolved-codepoints") + + +class AbsentFromTreeVsNameEmpty(unittest.TestCase): + """User ruling: `absent-from-tree` must be gated on *no source-bearing + text existing anywhere*, not on *no accepted-or-prohibited-role + candidate existing* — the latter gate is nearly always false for a real + application (a window `label` alone defeats it), which made + `absent-from-tree` — §8.3's own words, "the one this check will most + likely actually catch" — effectively unreachable in practice. + + The pinned precedence: + + 1. Source-bearing text or a precommitted form present anywhere -> + classify its name/role/composition outcome. This runs first, over + the whole forest, any role — before any absence/empty-name + determination. + 2. Otherwise, `name-empty` only when both hold: at least one + **accepted**-role candidate node exists, and every + accepted-or-prohibited-role candidate's name is empty. + 3. Every other no-source-bearing case -> `absent-from-tree`. + + The intended taxonomy, and the required regression lock: four cases, + side by side, so a future change cannot re-merge them. + + unrelated UI text only -> absent-from-tree + empty prohibited canvas only -> absent-from-tree + empty accepted text/label node -> name-empty + misordered source fragments -> composition/role failure, never absence + + The distinction being preserved: `name-empty` means an attempted + static-text exposure without a name; `absent-from-tree` covers + drawing-only or unrelated trees. A prohibited-role empty node is not + "wearing a role" in §8.3's sense — it is the draw-and-stop case. + """ + + def test_unrelated_ui_text_only_is_absent_from_tree(self): + """The coordinator's exact reproduction: ordinary application + chrome — a button, a window label — none of it related to the run. + Every real application has role-listed nodes like this, which is + exactly why the old "no candidate exists anywhere" gate made + `absent-from-tree` nearly unreachable.""" + exp = expectation("Coro אבג") + application = node( + "application", + "p", + children=[node("push button", "Save"), node("label", "MyApp Window")], + ) + verdict = classify(exp, [application]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "absent-from-tree") + + def test_empty_prohibited_canvas_only_is_absent_from_tree(self): + """The case that matters most (§8.3's own words): a toolkit that + drew to a canvas and stopped. A `canvas` node exposing no name is + not "wearing a role" in §8.3's name-empty sense — with no + accepted-role candidate anywhere in the tree, this is the + draw-and-stop case, `absent-from-tree`, not `name-empty`.""" + exp = expectation("Coro אבג") + application = node("frame", "MyApp", children=[node("canvas", "")]) + verdict = classify(exp, [application]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "absent-from-tree") + + def test_empty_accepted_text_node_is_name_empty(self): + """The companion case: an *accepted*-role node attempting to expose + static text, but with no name — this is the genuine name-empty + case, "absence wearing a role".""" + exp = expectation("Coro אבג") + application = node("frame", "MyApp", children=[node("label", "")]) + verdict = classify(exp, [application]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertEqual(verdict.prohibited_outcome, "name-empty") + + def test_misordered_source_fragments_are_a_composition_failure_not_absence_or_empty(self): + """The fourth leg: fragments of the run's actual text, present but + in the wrong order, must never be classified as absence or + name-empty — a composition/role failure, per item 1's precedence + over items 2 and 3.""" + exp = expectation("Coro אבג") + application = node( + "application", + "p", + children=[node("push button", "אבג"), node("push button", "Coro ")], + ) + verdict = classify(exp, [application]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertNotEqual(verdict.prohibited_outcome, "absent-from-tree") + self.assertNotEqual(verdict.prohibited_outcome, "name-empty") + + def test_an_accepted_node_with_real_text_alongside_an_empty_one_is_not_name_empty(self): + """Mutation guard: `name-empty` requires *every* candidate to be + empty, not merely *an* accepted-role candidate existing — an + accepted node with real (if unrelated) text sitting alongside an + empty one must not trigger name-empty, since not every + text-candidate node is actually empty.""" + exp = expectation("Coro אבג") + application = node( + "frame", + "MyApp", + children=[node("label", "Random"), node("label", "")], + ) + verdict = classify(exp, [application]) + self.assertEqual(verdict.verdict, "FAIL") + self.assertNotEqual(verdict.prohibited_outcome, "name-empty") + + +class ExpectationsFileValidation(unittest.TestCase): + """O1/B2: the loader must fail closed on a malformed or stale oracle + before any live AT-SPI readback — never a FAIL, always a usage error + that the caller (`run_check5`) turns into exit 2.""" + + VALID_DIGEST = "deadbeef" * 8 # a plausible-looking 64-hex-char sha256 + + @staticmethod + def _fixture(fixture_id, name, alternative_forms=None, source_atoms=None): + return { + "fixture_id": fixture_id, + "expected_name": name, + "expected_name_hex": name.encode("utf-8").hex(), + "expected_name_byte_len": len(name.encode("utf-8")), + "accepted_roles": ["label", "static", "text", "paragraph"], + "prohibited_roles": ["image", "canvas", "filler", "panel", "unknown"], + # Defaults to one atom equal to the whole name (trivially + # satisfies the join-equals-name invariant) — tests of other + # fields don't need more than one segment. + "source_atoms": source_atoms if source_atoms is not None else [name], + "alternative_forms": alternative_forms or {}, + } + + def _valid_file(self): + """A fully self-consistent, five-fixture file — every B2/O1/D1 check + passes against this by construction. Each test below mutates + exactly one thing away from it, so a raised error is attributable to + the one defect under test rather than an incidental other one.""" + return { + "contract": "spec/CONTRACT_EDITOR_T4_SPIKE.md pin 13", + "recipe": "spikes/editor-toolkit/ROUND2_TEXT_RECIPE.md §8", + "platform": PLATFORM, + "source_fixtures_digest": self.VALID_DIGEST, + "fixtures": [ + self._fixture("F-A", "Allegro affettuoso — al fine"), + self._fixture("F-B", "Coro אבג", source_atoms=["Coro ", "אבג"]), + self._fixture("F-C", "Coro ا", source_atoms=["Coro ", "ا"]), + self._fixture( + "F-D", "Allegro אבג con brio", source_atoms=["Allegro ", "אבג", " con brio"] + ), + self._fixture("F-E", "Café"), + ], + } + + def _validate(self, file): + validate_expectations_file( + file, expected_platform=PLATFORM, expected_source_digest=self.VALID_DIGEST + ) + + # ---- baseline ---- + + def test_a_fully_valid_file_is_accepted(self): + self._validate(self._valid_file()) # must not raise + + # ---- platform ---- + + def test_a_wrong_platform_is_refused(self): + bad = self._valid_file() + bad["platform"] = "aria" + with self.assertRaises(ValueError) as ctx: + self._validate(bad) + msg = str(ctx.exception) + self.assertIn("aria", msg) + self.assertIn(PLATFORM, msg) + + # ---- source_fixtures_digest (B2) ---- + + def test_a_stale_source_digest_is_refused(self): + bad = self._valid_file() + bad["source_fixtures_digest"] = "stale" + "0" * 60 + with self.assertRaises(ValueError) as ctx: + self._validate(bad) + msg = str(ctx.exception) + self.assertIn("stale", msg) + self.assertIn(self.VALID_DIGEST, msg) + + # ---- fixture id completeness/uniqueness ---- + + def test_a_duplicate_fixture_id_is_refused(self): + bad = self._valid_file() + bad["fixtures"].append(self._fixture("F-A", "duplicate")) + with self.assertRaises(ValueError) as ctx: + self._validate(bad) + self.assertIn("duplicate", str(ctx.exception).lower()) + self.assertIn("F-A", str(ctx.exception)) + + def test_a_missing_fixture_is_refused(self): + bad = self._valid_file() + bad["fixtures"] = [f for f in bad["fixtures"] if f["fixture_id"] != "F-E"] + with self.assertRaises(ValueError) as ctx: + self._validate(bad) + self.assertIn("F-E", str(ctx.exception)) + + def test_an_extra_fixture_id_is_refused(self): + bad = self._valid_file() + bad["fixtures"].append(self._fixture("F-Z", "unexpected")) + with self.assertRaises(ValueError) as ctx: + self._validate(bad) + self.assertIn("F-Z", str(ctx.exception)) + + def test_the_expected_fixture_id_set_is_exactly_f_a_through_f_e(self): + """Anchors `EXPECTED_FIXTURE_IDS` itself, independent of + `validate_expectations_file` — if this constant silently gained or + lost an id, the two tests above could pass against the wrong set.""" + self.assertEqual(EXPECTED_FIXTURE_IDS, frozenset({"F-A", "F-B", "F-C", "F-D", "F-E"})) + + # ---- expected_name / expected_name_hex / expected_name_byte_len self-consistency ---- + + def test_a_wrong_hex_is_refused(self): + bad = self._valid_file() + bad["fixtures"][0]["expected_name_hex"] = "ff" * 10 + with self.assertRaises(ValueError) as ctx: + self._validate(bad) + msg = str(ctx.exception) + self.assertIn("F-A", msg) + self.assertIn("hex", msg) + + def test_an_uppercase_hex_is_refused(self): + """§8.1 specifically requires *lowercase* hex — an otherwise-correct + but uppercase rendering must still be refused, not accepted as + "close enough".""" + bad = self._valid_file() + bad["fixtures"][0]["expected_name_hex"] = bad["fixtures"][0]["expected_name_hex"].upper() + with self.assertRaises(ValueError) as ctx: + self._validate(bad) + self.assertIn("hex", str(ctx.exception)) + + def test_a_wrong_byte_length_is_refused(self): + bad = self._valid_file() + bad["fixtures"][0]["expected_name_byte_len"] += 1 + with self.assertRaises(ValueError) as ctx: + self._validate(bad) + msg = str(ctx.exception) + self.assertIn("F-A", msg) + self.assertIn("byte_len", msg) + + def test_a_correct_hex_and_length_pair_is_not_refused(self): + """Mutation guard: confirms the two tests above are checking the + actual computed hex/length, not merely "is a string of digits" or + some other weaker property.""" + self._validate(self._valid_file()) # must not raise + + # ---- O1: cross-outcome collision (unchanged, folded into this entry point) ---- + + def test_a_colliding_file_is_refused_naming_the_fixture_and_both_outcomes(self): + bad = self._valid_file() + bad["fixtures"][2]["alternative_forms"] = { # F-C + "name-drops-unresolved-codepoints": ["Coro "], + "name-is-shaped-glyphs": ["Coro "], + } + with self.assertRaises(ValueError) as ctx: + self._validate(bad) + msg = str(ctx.exception) + self.assertIn("F-C", msg) + self.assertIn("name-drops-unresolved-codepoints", msg) + self.assertIn("name-is-shaped-glyphs", msg) + + def test_a_collision_between_list_entries_is_also_refused(self): + """The collision can be between any entry in one outcome's list and + any entry in another's, not just single-value outcomes.""" + bad = self._valid_file() + bad["fixtures"][0]["alternative_forms"] = { # F-A + "name-normalized": ["alpha", "shared"], + "name-is-shaped-glyphs": ["beta", "shared"], + } + with self.assertRaises(ValueError) as ctx: + self._validate(bad) + self.assertIn("shared", str(ctx.exception)) + + def test_a_repeated_value_within_the_same_outcome_is_not_a_collision(self): + """Mutation guard: if the collision check fired on *any* repeated + value rather than specifically a *cross-outcome* one, this would + wrongly raise — two entries in one outcome's own list happening to + repeat is not the ambiguity O1 refuses.""" + fine = self._valid_file() + fine["fixtures"][0]["alternative_forms"] = {"name-normalized": ["same", "same"]} + self._validate(fine) # must not raise + + def test_the_real_committed_file_is_valid(self): + """Grounds the synthetic tests above in the actual generated + artifact — the file `run_check5` will really load. Uses the file's + own `platform`/`source_fixtures_digest` as the "expected" values + (this is the one place a real digest isn't known statically), so + this test is really only exercising the fixture-id/name-consistency/ + collision checks against real data, not the digest-mismatch check.""" + import json + import os + + path = os.path.join( + os.path.dirname(__file__), "..", "round2-a11y-oracle", "a11y_expectations.json" + ) + if not os.path.exists(path): + self.skipTest(f"{path} absent — run the round2-a11y-oracle generator first") + with open(path, "r", encoding="utf-8") as f: + real_file = json.load(f) + validate_expectations_file( + real_file, + expected_platform=real_file.get("platform"), + expected_source_digest=real_file.get("source_fixtures_digest"), + ) # must not raise + + +if __name__ == "__main__": + unittest.main() diff --git a/spikes/editor-toolkit/a11y-verifier/verify.py b/spikes/editor-toolkit/a11y-verifier/verify.py index 7addf6f..861dc38 100644 --- a/spikes/editor-toolkit/a11y-verifier/verify.py +++ b/spikes/editor-toolkit/a11y-verifier/verify.py @@ -1,42 +1,53 @@ #!/usr/bin/env python3 -"""Round 0 accessibility readback verifier. +"""AT-SPI2 accessibility verifier — Round 0 readback mode, and Round 2 check 5. -An AT-SPI client, independent of any candidate's own process, that walks -the live platform accessibility tree (via the AT-SPI2 registry over D-Bus) -looking for an accessible node with a given role and name. This is a real -client query of the tree, per CONTRACT_EDITOR_T4_SPIKE.md Round 0: "Setting -the node in your own process and printing your own struct is NOT a -readback." +An AT-SPI client, independent of any candidate's own process, that walks the +live platform accessibility tree (via the AT-SPI2 registry over D-Bus). Two +modes, selected by which flags are given: -Uses gi.repository.Atspi, the official GObject-introspection binding for -AT-SPI2 (the same library backing Orca and Accerciser). This is used in -place of the `atspi` Rust crate as an "equivalent AT-SPI client" (the -contract's own wording) — chosen because its API is stable, documented, and -already verified reachable on this machine, rather than reverse-engineering -an unfamiliar async zbus proxy API under this round's timebox. That -substitution is a named deviation, reported as such. +Round 0 mode — unchanged, byte-for-byte, from the version that produced +`round0-evidence/c1-egui-readback.txt` and `c2-vello-readback.txt`: -Usage: verify.py --role "push button" --name "EpiphanyProbeButton" [--app-name SUBSTR] [--max-depth N] [--timeout SECONDS] -Exit code 0 and prints "READBACK: PASS" with the path from desktop root to -the matched node, if found within the timeout. Exit code 1 and prints -"READBACK: FAIL" with a dump of what *was* found, if the bus is reachable -but no match appears before the timeout. Exit code 2 and prints -"READBACK: NOT RUN" if the AT-SPI bus itself cannot be reached at all. +Looks for one exact (role, name) match anywhere under the desktop (optionally +restricted to apps whose name contains --app-name). Exit 0 "READBACK: PASS", +exit 1 "READBACK: FAIL", exit 2 "READBACK: NOT RUN" (bus unreachable). + +Round 2 check 5 mode — `spikes/editor-toolkit/ROUND2_TEXT_RECIPE.md` §8, an +accessibility oracle packet 2B-A precommits (`round2-a11y-oracle`): + + verify.py --expectations round2-a11y-oracle/a11y_expectations.json --fixture F-A \ + --expect-source-digest \ + --app-name SUBSTR [--timeout N] [--json PATH] + +Scores one fixture's check 5 against the live tree under the candidate's +application (matched by --app-name, required in this mode). Exit 0 "CHECK5: +PASS", exit 1 "CHECK5: FAIL" (naming exactly one of +`round2_textkit::a11y::PROHIBITED_OUTCOMES`, or a role/composition-specific +diagnosis, when applicable), exit 2 "CHECK5: NOT RUN" — reserved *only* for +the AT-SPI bus itself being unreachable. A candidate that simply never built +an accessibility tree is `absent-from-tree`, which is a FAIL, not NOT RUN. + +Uses gi.repository.Atspi, the official GObject-introspection binding for +AT-SPI2 (the same library backing Orca and Accerciser). Used in place of the +`atspi` Rust crate as an "equivalent AT-SPI client" (the contract's own +wording) — chosen because its API is stable, documented, and already +verified reachable on this machine, rather than reverse-engineering an +unfamiliar async zbus proxy API under this round's timebox. That substitution +is a named deviation, reported as such. """ import argparse +import json import sys import time +from dataclasses import dataclass, field +from typing import Dict, List, Optional, Tuple -try: - import gi - - gi.require_version("Atspi", "2.0") - from gi.repository import Atspi -except Exception as exc: # pragma: no cover - environment probe - print(f"READBACK: NOT RUN — could not import gi.repository.Atspi: {exc}") - sys.exit(2) +# --------------------------------------------------------------------------- +# Round 0 mode — unmodified from the version that produced the committed +# round0-evidence transcripts. Do not change this function's behaviour. +# --------------------------------------------------------------------------- def walk(node, role, name, app_name_substr, max_depth, path, found, all_seen): @@ -77,16 +88,7 @@ def walk(node, role, name, app_name_substr, max_depth, path, found, all_seen): ) -def main(): - ap = argparse.ArgumentParser() - ap.add_argument("--role", required=True) - ap.add_argument("--name", required=True) - ap.add_argument("--app-name", default=None, help="only descend into apps whose name contains this substring") - ap.add_argument("--max-depth", type=int, default=12) - ap.add_argument("--timeout", type=float, default=20.0) - ap.add_argument("--poll-interval", type=float, default=0.5) - args = ap.parse_args() - +def run_round0(args, Atspi): try: Atspi.init() except Exception as exc: @@ -162,5 +164,939 @@ def main(): sys.exit(1) +# --------------------------------------------------------------------------- +# Round 2 check 5 mode. +# --------------------------------------------------------------------------- + +# A verifier-specific diagnostic name for the §8.1 composition trap — the +# concatenation matches `visual_order_name`, not `expected_name`. This is +# deliberately *not* one of `round2_textkit::a11y::PROHIBITED_OUTCOMES`: it is +# a structural composition failure (assembled in the wrong order), not one of +# the five name-transformation outcomes §8.3 pins. Naming it distinctly is +# the whole point of the requirement: "the report must name it rather than +# emit a generic mismatch." +VISUAL_ORDER_TRAP = "composed-in-visual-order" + +# The platform row this verifier scores against — this machine's live AT +# client is AT-SPI2 (recipe §8.2, round0-evidence's precedent), matching +# `round2_a11y_oracle::PLATFORM`. Used by `validate_expectations_file` (B2) +# to refuse an expectations file generated for a different platform, rather +# than silently scoring against the wrong role vocabulary. +PLATFORM = "at-spi2" + +# The exact five fixture ids the recipe names (ROUND2_TEXT_RECIPE.md §2), +# restated here — not read back out of the file being validated — the same +# discipline `round2_textkit::output::FixtureFile::validate`'s +# `EXPECTED_FIXTURES` uses, so a file missing one or carrying an extra id is +# caught against a literal, not against its own other contents. +EXPECTED_FIXTURE_IDS = frozenset({"F-A", "F-B", "F-C", "F-D", "F-E"}) + + +def validate_expectations_file( + expectations_file: dict, *, expected_platform: str, expected_source_digest: str +) -> None: + """B2/O1: fail closed on a malformed or stale oracle **before** any live + AT-SPI readback. Check 5 is disqualifying, so a defect in the oracle + artifact itself must never be silently absorbed into a candidate's + verdict — every check below raises `ValueError` (which the caller turns + into a usage error, exit 2, never a FAIL: a malformed oracle is not a + candidate defect) naming exactly what disagreed. + + - `platform` must equal `expected_platform` — scoring F-A's at-spi2 role + vocabulary against a file generated for a different platform would + silently check the wrong roles. + - `source_fixtures_digest` must equal `expected_source_digest` — the + caller passes `round2_textkit::output::expected_artifact_digest()` + (via `--expect-source-digest`), so an oracle generated against a + *different* `fixtures.json` (stale, or regenerated on a machine with + different fonts — recipe §1) cannot score a candidate under the + pretense of being current. + - The fixture id set is exactly `EXPECTED_FIXTURE_IDS`: no duplicates, none + missing, none extra. + - Per fixture, `expected_name` / `expected_name_hex` / `expected_name_byte_len` + are mutually consistent — recipe §8.1 carries the name three ways + specifically so a divergence between them is detectable; this is what + detects it. (lowercase hex, per §8.1's own "lowercase hex" wording.) + - D1: `source_atoms` is a list of strings whose concatenation, in order, + equals `expected_name` — the same partition property + `round2_a11y_oracle::source_atoms`'s own doc comment claims and tests + on the generation side; this is the verifier-side half of that same + check, so a hand-edited or differently-generated file cannot silently + carry atoms that no longer add up to the name they are supposed to be + components of. + - O1: no two different outcome names in one fixture's `alternative_forms` + produce the same string (unchanged from the earlier fix, folded into + this same fail-closed entry point). + + Does not touch AT-SPI or any live state, so it is testable without a bus, + the same as `classify`. + """ + actual_platform = expectations_file.get("platform") + if actual_platform != expected_platform: + raise ValueError( + f"platform is {actual_platform!r}, expected {expected_platform!r} — this oracle was " + "not generated for the platform being scored" + ) + + actual_digest = expectations_file.get("source_fixtures_digest") + if actual_digest != expected_source_digest: + raise ValueError( + f"source_fixtures_digest is {actual_digest!r}, expected {expected_source_digest!r} " + "(round2_textkit::output::expected_artifact_digest()) — this oracle may have been " + "generated against a different fixtures.json and must not score a candidate" + ) + + fixtures = expectations_file.get("fixtures", []) + ids = [fx.get("fixture_id") for fx in fixtures] + if len(ids) != len(set(ids)): + duplicates = sorted({i for i in ids if ids.count(i) > 1}) + raise ValueError(f"duplicate fixture_id(s) in expectations file: {duplicates}") + id_set = set(ids) + missing = sorted(EXPECTED_FIXTURE_IDS - id_set) + extra = sorted(id_set - EXPECTED_FIXTURE_IDS) + if missing or extra: + raise ValueError( + f"fixture id set is {sorted(id_set)}, expected exactly {sorted(EXPECTED_FIXTURE_IDS)} " + f"(missing: {missing}, extra: {extra})" + ) + + for fx in fixtures: + fixture_id = fx.get("fixture_id", "") + + name = fx.get("expected_name") + name_hex = fx.get("expected_name_hex") + name_byte_len = fx.get("expected_name_byte_len") + if not isinstance(name, str): + raise ValueError(f"{fixture_id!r}: expected_name is not a string: {name!r}") + actual_name_bytes = name.encode("utf-8") + actual_hex = actual_name_bytes.hex() # Python's .hex() is always lowercase + if name_hex != actual_hex: + raise ValueError( + f"{fixture_id!r}: expected_name_hex is {name_hex!r}, but the lowercase hex of " + f"expected_name's UTF-8 bytes is {actual_hex!r} — the name and its hex have " + "diverged" + ) + if name_byte_len != len(actual_name_bytes): + raise ValueError( + f"{fixture_id!r}: expected_name_byte_len is {name_byte_len!r}, but " + f"expected_name's UTF-8 byte length is {len(actual_name_bytes)}" + ) + + atoms = fx.get("source_atoms") + if not isinstance(atoms, list) or not all(isinstance(a, str) for a in atoms): + raise ValueError(f"{fixture_id!r}: source_atoms is not a list of strings: {atoms!r}") + joined_atoms = "".join(atoms) + if joined_atoms != name: + raise ValueError( + f"{fixture_id!r}: source_atoms {atoms!r} concatenate to {joined_atoms!r}, which " + f"does not equal expected_name {name!r} — the atoms no longer partition the name " + "they are supposed to be components of" + ) + + forms_by_outcome: Dict[str, List[str]] = fx.get("alternative_forms", {}) or {} + owner_of: Dict[str, str] = {} + for outcome, forms in forms_by_outcome.items(): + for form in forms: + existing = owner_of.get(form) + if existing is not None and existing != outcome: + raise ValueError( + f"{fixture_id!r}: alternative forms {existing!r} and {outcome!r} both " + f"produce {form!r} — an oracle that returns two different " + "classifications for the same observed string must fail closed, not " + "let iteration order pick one" + ) + owner_of[form] = outcome + + +@dataclass +class Verdict: + """One check-5 scoring outcome. `verdict` is always exactly one of + "PASS" / "FAIL" (`prohibited_outcome` distinguishes NOT RUN, which is + handled by the caller before a `Verdict` is ever constructed — NOT RUN is + reserved for the AT-SPI bus itself being unreachable, never for a + classification the tree walk produced).""" + + verdict: str + reason: str + observed_role: Optional[str] = None + observed_name: Optional[str] = None + prohibited_outcome: Optional[str] = None + + +@dataclass +class ObservedNode: + """One node of a live AT-SPI subtree, as walked by `walk_for_check5` — + role, name, and children, preserving the structure `classify` needs to + score §8.1 composition per-subtree (B1). Deliberately holds nothing else + (no live AT-SPI object reference): once built, this is inert data, which + is what lets `classify` stay pure and bus-free.""" + + role: str + name: str + children: List["ObservedNode"] = field(default_factory=list) + + +def _iter_nodes(node: ObservedNode): + """Every node in `node`'s subtree, `node` itself included, pre-order.""" + yield node + for child in node.children: + yield from _iter_nodes(child) + + +def _iter_forest(roots: List[ObservedNode]): + """Every node in every tree in `roots`, pre-order, roots first.""" + for root in roots: + yield from _iter_nodes(root) + + +def _flatten_candidates( + roots: List[ObservedNode], accepted: set, prohibited: set +) -> List[Tuple[str, str]]: + """Every `(role, name)` pair, for every node anywhere in the forest whose + role is a text-candidate (`accepted | prohibited`), in tree order. + + Used for exactly one thing now: `classify`'s final `name-empty` vs. + `absent-from-tree` decision (user ruling), reached only after every + source-bearing scan (which considers *every* role, not just + `accepted | prohibited`) has found nothing. `name-empty` is specifically + about accepted/prohibited-role candidates existing with no name, so it + is the one remaining check that legitimately wants this narrower, + role-filtered list rather than the whole forest. + """ + return [ + (n.role, n.name) for n in _iter_forest(roots) if n.role in accepted or n.role in prohibited + ] + + +def _all_descendants(root: ObservedNode) -> List[Tuple[str, str]]: + """Every `(role, name)` pair for **every** descendant of `root` **with a + non-empty name**, regardless of role — `root` itself excluded, since a + node's own name matching `expected_name` (or an alternative/visual-order + form) is the separate single-node case (§8.1's first alternative; this + is its second), in tree order. + + Deliberately **not** filtered by role before the caller concatenates: a + non-accepted-role contributor's name is still part of what the + subtree's composition actually says, and dropping it before summing + would let a subtree "pass" by silently ignoring a contributor it + doesn't like — precisely the wrong fix for B1. The caller concatenates + first, checks role-acceptability only once the concatenation is already + confirmed to equal `expected_name` (or a precommitted alternative/ + visual-order form). + + **Empty-named nodes are excluded entirely, not merely ignored when + picking whom to blame.** An empty name contributes zero bytes to the + concatenation — including or excluding it never changes `subtree_concat` + — so the only thing including it can do is let a purely structural + wrapper (a `frame` or `panel` around the real contributors, exposing no + name of its own) be *named* as the offending contributor merely because + it happens to sort first in tree order, hiding the actual, non-empty, + possibly prohibited-role contributor that is the real §8.2 violation. + Excluding it here, at the source, fixes this the same way regardless of + which subtree in `classify`'s scan happens to be tried (and matched) + first — relying on the wrapper's own name to corrupt a *different* + subtree's concatenation would only fix the cases where that subtree + happened to be visited later. + + Used by `classify`'s composition scan for **every** subtree, regardless + of which roles (if any) appear elsewhere in the tree — an earlier + version of this function only admitted unlisted-role contributors when + *no* accepted-or-prohibited-role node existed anywhere in the tree, + which is exactly the gating the "absent-from-tree vs. name-empty" fix + removed: a real application's window `label` must not prevent the run's + actual text, exposed under an unlisted role elsewhere in the same tree, + from being found. + """ + out: List[Tuple[str, str]] = [] + for child in root.children: + for n in _iter_nodes(child): + if n.name != "": + out.append((n.role, n.name)) + return out + + +def _is_source_bearing_fragment(name: str, targets) -> bool: + """**One of two additive paths** (D1) `classify`'s fragment scan uses to + decide "the run's text is present, even if not composed correctly" + (user ruling, following C3) — this is the general, heuristic, + coincidence-guarded substring rule; `source_atoms` exact matching (see + `classify`'s fragment scan) is the other, precommitted, no-length-floor + path. To distinguish real (if misordered or incomplete) evidence of the + run from an unrelated node's text that happens to share a coincidental + substring, `name` counts as a source-bearing fragment of one of + `targets` (`expected_name`, or a precommitted alternative/visual-order + form) only if it is: + + - non-empty and not whitespace-only (`name.strip()` is non-empty) — a + bare space is not evidence of anything, even though a space is + technically a substring of e.g. `"Coro "`; + - **at least two characters** after stripping — a single character is + not distinguishable from coincidence: almost any two unrelated + strings of ordinary language share *some* one character (a window + title and `"Coro אבג"` both very plausibly contain the letter `"o"`); + - a literal substring of at least one target, compared **as given** — + never normalized, and the containment test itself uses the raw + (unstripped) `name`, so incidental surrounding whitespace in `name` + that isn't present in the target correctly fails to match; only the + length/whitespace *gate* above is computed on the stripped form. + + **Stated limit, not hidden — this rule deliberately under-detects, and + is deliberately never loosened to cover it.** A genuine run fragment + shorter than two characters — F-C's unresolved segment `ا` is exactly + this case, a single character — is never caught by *this* function, on + purpose: loosening the floor to catch it would risk exactly what + `SourceBearingFragmentGuards`' guard tests exist to catch — an + application's ordinary window title coincidentally sharing a short + substring with `expected_name` and permanently disabling + `absent-from-tree` for that fixture, "which is a worse failure than the + one being fixed" (the ruling's own words). F-C's single-character + segment is instead caught by the *other* path — an exact match against + a precommitted `source_atoms` entry, which needs no length floor at all + because it is a comparison against precommitted data, not a heuristic + guess from length alone. The two paths are independent; this function's + own contract does not change. + """ + stripped = name.strip() + if len(stripped) < 2: + return False + return any(name in target for target in targets) + + +def classify(expectation: dict, roots: List[ObservedNode]) -> Verdict: + """The whole of check 5's scoring logic, and nothing else. + + `expectation` is one fixture's entry from `a11y_expectations.json` + (`round2-a11y-oracle`) — a plain dict with `expected_name`, + `accepted_roles`, `prohibited_roles`, `alternative_forms` (an outcome + name mapped to a **list** of precommitted forms — O2: one outcome can + have more than one plausible rendering, e.g. `name-is-shaped-glyphs` + carries both a cluster-collapse form and a ligature presentation-form + substitution for F-A; matched if the observed name equals *any* entry), + and (optionally) `visual_order_name`. The caller must have already run + this file through `validate_expectations_file` (O1/B2) — `classify` + itself does not re-check the oracle's own integrity, since a malformed + oracle is exactly what validation exists to refuse before this function + ever runs. + + `roots` is the forest of `ObservedNode` trees the live tree walk found + under the candidate's application (usually one tree, the matched app's + own node) — this function does not touch AT-SPI, D-Bus, or any live + state, which is what makes it testable without a bus. + + **Shape (user ruling): source-bearing detection runs first, in full, + across every role, before any absence or empty-name determination — + never the other way around.** An earlier version of this function only + looked for the run's text under unlisted roles when *no* + accepted-or-prohibited-role node existed anywhere in the tree at all. + That gate was wrong: a real application always has *some* accepted-role + node (a window title `label`, at minimum), so the run's actual text, + exposed under an unlisted role *alongside* that unrelated label, was + never even looked for — the tree fell straight into the ordinary + (non-source-bearing) scoring path and reported whatever that path says + for "some accepted-role text exists, none of it matches," which used to + be a generic mismatch and is now (see below) `absent-from-tree`. So + every check in this section runs over the **whole forest, every role, + unconditionally** — never against just the first match, and never + gated on whether some *other*, unrelated node happens to carry an + accepted or prohibited role. + + **PRECEDENCE (pinned, C1) — exact-name matches.** §8.1's rule — "the + run's own accessible name ... must equal the source string" — is + evaluated over **every** node in the forest, regardless of role, not the + first one found. If *any* node with an **accepted** role carries + `expected_name` byte-for-byte, the verdict is PASS, regardless of where + in the tree that node sits or whether some *other* node (prohibited- or + unlisted-role) also happens to carry it. Failing that, a **prohibited**- + role match is named preferentially over an **unlisted**-role one (more + specific, per §8.2's own vocabulary); failing that, an unlisted-role + match is named. This is a pinned rule, not an implementation shortcut: a + tree that lists a `canvas` node before the real `text` node is exactly + the same candidate as one that lists them in the other order, and must + score the same way. Do not "simplify" this back to returning on the + first exact-name match — that reintroduces order-dependence on a + disqualifying check. + + **B1/C2: composition and its alternative-form/visual-order diagnoses are + all scored per subtree, never against a whole-application + concatenation, and admit every role as a contributor.** §8.1's second + alternative — "the names of its text descendants concatenated in + logical order" — is a statement about *one run's* subtree, and nothing + in §8.1 restricts which roles may compose it (an unlisted role composing + correctly is still wrong — see the fragment/role check below — but that + is a role failure to report, not a reason to exclude the node from the + concatenation in the first place). This function tries every node in the + forest as a candidate "this is the run" subtree root in turn, and for + each one: + + - if that subtree's own descendants (any role) concatenate + byte-exactly to `expected_name` **and** every one of those + descendants has an accepted role, PASS; + - if they concatenate to `expected_name` but include a non-accepted-role + contributor, that is a FAIL naming that contributor specifically — an + otherwise-correct composition failed by one contributor's role, + **never** silently dropped from consideration or averaged away by + unrelated nodes elsewhere in the tree (B1's original bug: a stray + `canvas` node absorbed into a whole-application PASS; C2's bug on the + diagnosis side: a stray `label` node corrupting an F-D-style + visual-order composition into a generic mismatch instead of naming + `composed-in-visual-order`); + - if instead they concatenate to one of a `PROHIBITED_OUTCOMES` + alternative form, or to `visual_order_name`, that subtree's diagnosis + is recorded (not returned immediately — a PASS found in a *different* + subtree still wins, since a candidate that got it right anywhere in a + legitimate run subtree has satisfied §8.1). + + A single node's own name is also checked against every alternative form + and `visual_order_name` (not only `expected_name`), regardless of role — + that has no subtree/aggregation ambiguity (one node's own name is + unambiguous regardless of tree position), so it stays a simple + whole-forest scan. + + **PRECEDENCE (pinned, D2) — a byte-exact PASS outranks every + alternative-form or visual-order diagnosis, per-node or per-subtree.** + Both PASS checks above (exact-name, and composition) already scan the + *entire* forest before either can return a PASS, so evaluating them + first and in full is what makes this safe: nothing is skipped to get to + the diagnosis checks below them. Concretely, the single-node and + subtree-level alternative-form/visual-order checks run **only after** + both PASS checks have been exhausted with nothing found — never + interleaved with them. This is why F-C's legitimate two-node split + (`text:"Coro "` + `text:"ا"`, both accepted — exactly the "one text node + per direction run" composition §8.1 permits) PASSes even though + `"Coro "` alone happens to equal F-C's own precommitted + `name-drops-unresolved-codepoints` form: the composition check finds the + byte-exact two-node PASS first. Do not "simplify" this by moving an + alternative-form check earlier for convenience — doing so previously + turned a legitimate F-C composition into a false FAIL naming a + `PROHIBITED_OUTCOMES` name that did not apply. + + **C3: fragments of the run's text present anywhere, under any role, even + out of the logical order §8.1 requires, are still evidence against + absence.** Failing an exact single-node or composition match above, any + node meeting the narrow `_is_source_bearing_fragment` definition (see + its own doc comment for the rule and its stated limits) is still + evidence the run's text is present, however it is arranged — this is + the case an exact-match/composition scan alone cannot see: text that is + genuinely present but misordered or incomplete. + + **`absent-from-tree` vs. `name-empty` (user ruling, pinned) — decided + only after every check above has found nothing.** The distinction being + preserved: `name-empty` means an attempted **static-text exposure** + without a name (§8.3: "absence wearing a role"); `absent-from-tree` + covers a **drawing-only** tree or an **unrelated** one. Concretely: + + - `name-empty` fires **only** when both hold: at least one + **accepted**-role candidate node exists somewhere in the tree, *and* + every accepted-or-prohibited-role candidate's name is empty. A lone + empty **prohibited**-role node (a canvas that drew nothing and + exposed nothing) is *not* "wearing a role" in §8.3's sense — it is + the draw-and-stop case §8.3 calls "the one this check will most + likely actually catch," and it is `absent-from-tree`. + - every other case that reaches this point — a genuinely empty tree, a + drawing-only tree, or a tree whose only text (under any role) bears no + relation to the run at all — is `absent-from-tree`. + + The required regression lock for this exact distinction lives in + `AbsentFromTreeVsNameEmpty` (`test_verify.py`): + + unrelated UI text only -> absent-from-tree + empty prohibited canvas only -> absent-from-tree + empty accepted text/label node -> name-empty + misordered source fragments -> composition/role failure, never absence + + **Contributor order stays semantically significant everywhere in this + function** — only **non-contributor** permutations (an unrelated + sibling moving around the tree) are required to be verdict-invariant. + This function never "fixes" composition into an order-insensitive + match; that would defeat the entire point of the F-D visual-order trap + (§8.1). + + Comparisons are always on the Python `str` (which is Unicode + codepoints), never bytes directly, but every string compared here is + already the exact source string on the Rust side (`str == str` is + codepoint-exact, which for valid UTF-8 is byte-exact) — the caller is + responsible for hex-encoding whatever `observed_name` this returns if a + byte-level report is needed (see `run_check5`). + """ + expected_name = expectation["expected_name"] + accepted = set(expectation["accepted_roles"]) + prohibited = set(expectation["prohibited_roles"]) + alt_forms: Dict[str, List[str]] = expectation.get("alternative_forms", {}) or {} + visual_order_name = expectation.get("visual_order_name") + # D1: precommitted per-segment source atoms (`round2-a11y-oracle`'s + # `source_atoms`), e.g. F-C's `["Coro ", "ا"]`. A node name exactly + # matching one is source-bearing regardless of length — this is what + # catches F-C's single-character unresolved segment `ا`, which the + # length-2 `_is_source_bearing_fragment` substring rule cannot (and must + # not be loosened to) catch on its own. + source_atoms = set(expectation.get("source_atoms", []) or []) + + interesting_names = {expected_name} + for forms in alt_forms.values(): + interesting_names.update(forms) + if visual_order_name is not None: + interesting_names.add(visual_order_name) + + # Every node in the forest, any role — the source-bearing scans below + # are unconditional on role, per the user ruling: gating them on whether + # some *other*, unrelated node happens to carry an accepted/prohibited + # role is exactly the bug being fixed. + all_nodes: List[Tuple[str, str]] = [(n.role, n.name) for n in _iter_forest(roots)] + + # 1. C1: exact-name matches, evaluated over the *entire* forest, every + # role, before deciding anything — never the first match found, and + # never gated on some other node's role. + exact_matches = [(role, name) for role, name in all_nodes if name == expected_name] + if exact_matches: + accepted_matches = [rn for rn in exact_matches if rn[0] in accepted] + if accepted_matches: + role, name = accepted_matches[0] + return Verdict( + "PASS", + f"a node with an accepted role ({role!r}) carries the accessible name " + "byte-for-byte", + observed_role=role, + observed_name=name, + ) + prohibited_matches = [rn for rn in exact_matches if rn[0] in prohibited] + if prohibited_matches: + role, name = prohibited_matches[0] + return Verdict( + "FAIL", + f"a node's name matches expected_name byte-for-byte, but its role {role!r} " + "is in the at-spi2 prohibited set (no accepted-role node also carries it)", + observed_role=role, + observed_name=name, + ) + # Every remaining match's role is in neither accepted nor prohibited. + role, name = exact_matches[0] + return Verdict( + "FAIL", + f"a node's name matches expected_name byte-for-byte, but its role {role!r} is " + "neither accepted nor prohibited for at-spi2 (no accepted- or prohibited-role node " + "also carries it)", + observed_role=role, + observed_name=name, + ) + + # 2. B1/C2: composition and its alternative-form/visual-order diagnoses, + # all scored per subtree in one pass, every role admitted as a + # contributor. Try every node in the forest as a candidate run-subtree + # root; every check below is decided by that node's own descendants + # alone, never by nodes outside it. + first_bad_composition: Optional[Verdict] = None + first_alt_form_fail: Optional[Verdict] = None + first_visual_order_fail: Optional[Verdict] = None + for candidate_root in _iter_forest(roots): + contributors = _all_descendants(candidate_root) + if not contributors: + continue + subtree_concat = "".join(name for _, name in contributors) + + if subtree_concat == expected_name: + bad = [(role, name) for role, name in contributors if role not in accepted] + if not bad: + return Verdict( + "PASS", + "the descendants of one run subtree concatenate to expected_name " + "byte-for-byte, and every contributor's role is accepted", + observed_name=subtree_concat, + ) + if first_bad_composition is None: + # Prefer naming a prohibited-role contributor over a merely + # unlisted one: prohibited is the specific, named §8.2 + # divergence, and the report exists to say that, not the + # weaker "nobody listed this role" case — pick the first + # prohibited-role entry if any exists, else fall back to the + # first non-accepted entry (necessarily unlisted-role, since + # `bad` excludes accepted roles by construction). + prohibited_bad = [rn for rn in bad if rn[0] in prohibited] + bad_role, _bad_name = prohibited_bad[0] if prohibited_bad else bad[0] + classification = ( + "prohibited" if bad_role in prohibited else "neither accepted nor prohibited" + ) + other_count = len(bad) - 1 + mention_others = ( + f" ({other_count} other non-accepted contributor(s) also present)" + if other_count > 0 + else "" + ) + first_bad_composition = Verdict( + "FAIL", + "a run subtree's descendants concatenate to expected_name byte-for-byte, " + f"but contributor role {bad_role!r} is {classification} for at-spi2{mention_others} " + "— an otherwise-correct composition, failed by this contributor's role", + observed_role=bad_role, + observed_name=subtree_concat, + ) + continue + + if first_alt_form_fail is None: + for outcome, forms in alt_forms.items(): + if subtree_concat in forms: + first_alt_form_fail = Verdict( + "FAIL", + "one run subtree's concatenated contributors match a precommitted " + f"{outcome!r} alternative form byte-for-byte", + observed_name=subtree_concat, + prohibited_outcome=outcome, + ) + break + + if ( + first_visual_order_fail is None + and visual_order_name is not None + and subtree_concat == visual_order_name + ): + first_visual_order_fail = Verdict( + "FAIL", + "one run subtree's concatenated contributors match visual_order_name, not " + "expected_name — the tree was assembled by walking the visual runs left to " + "right instead of logical order", + observed_name=subtree_concat, + prohibited_outcome=VISUAL_ORDER_TRAP, + ) + + if first_bad_composition is not None: + return first_bad_composition + + # D2 (user ruling): a byte-exact PASS — single-node (step 1, above) or + # subtree composition (step 2, above) — outranks every alternative-form + # or visual-order diagnosis, per-node or per-subtree. Both PASS checks + # already scan the *entire* forest before this point is ever reached, so + # by construction nothing above this line has skipped a legitimate PASS + # to get here. Only now, with every PASS opportunity exhausted, do the + # alternative-form/visual-order diagnoses get a turn — starting with a + # single node's own name (no subtree ambiguity: one node's own name is + # unambiguous regardless of position or role, so this stays a flat, + # whole-forest scan), then the subtree-level matches the composition + # loop above already recorded. + # + # This ordering is why F-C's legitimate two-node split + # (`text:"Coro "` + `text:"ا"`, both accepted) now PASSes even though + # `"Coro "` alone is also F-C's precommitted `name-drops-unresolved- + # codepoints` form: the composition loop above finds the byte-exact PASS + # across both nodes and returns before this per-node check ever runs. A + # single `text:"Coro "` node with **no** second node still reaches this + # check (no composition to find), so the outcome stays named exactly as + # before — see `FCTwoSegmentComposition`'s regression group + # (`test_verify.py`) for both halves of that guarantee. + for role, name in all_nodes: + for outcome, forms in alt_forms.items(): + if name in forms: + return Verdict( + "FAIL", + f"a node's name matches a precommitted {outcome!r} alternative form " + "byte-for-byte", + observed_role=role, + observed_name=name, + prohibited_outcome=outcome, + ) + if visual_order_name is not None and name == visual_order_name: + return Verdict( + "FAIL", + "a node's name matches visual_order_name, not expected_name — the tree was " + "assembled by walking the visual runs left to right instead of logical order", + observed_role=role, + observed_name=name, + prohibited_outcome=VISUAL_ORDER_TRAP, + ) + if first_alt_form_fail is not None: + return first_alt_form_fail + if first_visual_order_fail is not None: + return first_visual_order_fail + + # 3. C3/D1: fragments of the run's text present anywhere, any role, even + # when they do not compose to any target string in the required + # logical order — the case an exact-match/composition scan alone + # cannot see: text that is genuinely present but misordered or + # incomplete. Whole-forest, not subtree-scoped: the safety valve here + # is the narrow fragment definition itself + # (`_is_source_bearing_fragment`), not tree structure — the policy is + # "any fragment anywhere is evidence against absence," which a + # subtree restriction would contradict. + # + # D1: a node counts as source-bearing via **either** of two additive + # paths — `_is_source_bearing_fragment`'s length-2-or-more substring + # rule, **or** an exact match against a precommitted `source_atoms` + # entry, regardless of length. The atom path is what catches F-C's + # unresolved segment `ا`: a single character, which the substring + # rule's coincidence guard correctly refuses (an unrelated stray "o" + # must never rescue a tree from absence) but which is nonetheless a + # real, precommitted, exact source component §8.3 requires to appear + # in the name. The two paths are independent and neither replaces the + # other — F-A (a single-segment run) has no atom shorter than its + # whole `expected_name`, so it depends entirely on the substring path, + # same as before D1. + fragments = [ + (role, name) + for role, name in all_nodes + if _is_source_bearing_fragment(name, interesting_names) or name in source_atoms + ] + if fragments: + roles = sorted({role for role, _ in fragments}) + fragment_concat = "".join(name for _, name in fragments) + return Verdict( + "FAIL", + f"{len(fragments)} fragment(s) of the run's text are present under role(s) {roles}, " + "but do not compose to expected_name or a precommitted form in the required logical " + "order — a role/composition failure, not absent-from-tree", + observed_role=roles[0] if len(roles) == 1 else None, + observed_name=fragment_concat, + ) + + # 4. Nothing above found any source-bearing evidence anywhere, under any + # role, in any shape. Only one distinction remains (user ruling, + # pinned in the docstring above): `name-empty` requires an attempted + # *static-text* exposure — at least one accepted-role candidate node + # — with every accepted-or-prohibited-role candidate's name empty. + # Every other no-source-bearing case, including a lone empty + # prohibited-role node (draw-and-stop, §8.3's own headline case) and + # unrelated text under any role, is `absent-from-tree`. + flat_candidates = _flatten_candidates(roots, accepted, prohibited) + has_accepted_candidate = any(role in accepted for role, _ in flat_candidates) + if ( + flat_candidates + and has_accepted_candidate + and all(name == "" for _, name in flat_candidates) + ): + return Verdict( + "FAIL", + "an accepted-role candidate node is present, but every accepted- or prohibited-role " + "candidate's accessible name is empty — an attempted static-text exposure with no " + "name", + observed_name="", + prohibited_outcome="name-empty", + ) + + return Verdict( + "FAIL", + "no accessible-text-candidate node (accepted or prohibited role) found under the " + "candidate's application, on any single node, composed across any subtree, or as a " + "source-bearing fragment, under any role", + prohibited_outcome="absent-from-tree", + ) + + +def walk_for_check5(node, path, all_seen, max_depth) -> Optional[ObservedNode]: + """Recursively mirrors the live AT-SPI subtree under `node` into an + `ObservedNode` tree, and records every node's `role:name` into + `all_seen` for the human/JSON "full tree" report — the same diagnostic + output this produced before B1, alongside a tree instead of a flat list. + + Unlike the pre-B1 version, this does **not** decide which nodes are + text-candidates — that decision now happens in `classify`, scoped per + subtree (B1): filtering roles *while* flattening the walk into a list is + exactly what threw away the subtree structure composition scoring needs. + """ + if node is None: + return None + try: + name = node.get_name() + except Exception: + name = "" + try: + role = node.get_role_name() + except Exception: + role = "" + all_seen.append(" / ".join(path + [f"{role}:{name!r}"])) + observed = ObservedNode(role=role, name=name) + if max_depth <= 0: + return observed + try: + n = node.get_child_count() + except Exception: + return observed + for i in range(n): + try: + child = node.get_child_at_index(i) + except Exception: + continue + child_observed = walk_for_check5( + child, path + [f"{role}:{name!r}"], all_seen, max_depth - 1 + ) + if child_observed is not None: + observed.children.append(child_observed) + return observed + + +def hex_lower(s: Optional[str]) -> Optional[str]: + if s is None: + return None + return s.encode("utf-8").hex() + + +def run_check5(args, Atspi): + try: + with open(args.expectations, "r", encoding="utf-8") as f: + expectations_file = json.load(f) + except Exception as exc: + print(f"CHECK5: usage error — could not read/parse {args.expectations!r}: {exc}") + sys.exit(2) + + try: + validate_expectations_file( + expectations_file, + expected_platform=PLATFORM, + expected_source_digest=args.expect_source_digest, + ) + except ValueError as exc: + print(f"CHECK5: usage error — {args.expectations!r} failed validation: {exc}") + sys.exit(2) + + expectation = next( + (f for f in expectations_file.get("fixtures", []) if f.get("fixture_id") == args.fixture), + None, + ) + if expectation is None: + print( + f"CHECK5: usage error — {args.fixture!r} is not a fixture in {args.expectations!r} " + f"(has: {[f.get('fixture_id') for f in expectations_file.get('fixtures', [])]})" + ) + sys.exit(2) + + try: + Atspi.init() + except Exception as exc: + print(f"CHECK5: NOT RUN — Atspi.init() failed: {exc}") + sys.exit(2) + + deadline = time.monotonic() + args.timeout + verdict = None + all_seen: List[str] = [] + attempt = 0 + while time.monotonic() < deadline: + attempt += 1 + try: + desktop = Atspi.get_desktop(0) + except Exception as exc: + print(f"CHECK5: NOT RUN — Atspi.get_desktop(0) failed: {exc}") + sys.exit(2) + if desktop is None: + print("CHECK5: NOT RUN — Atspi.get_desktop(0) returned None (no AT-SPI registry?)") + sys.exit(2) + + roots: List[ObservedNode] = [] + all_seen = [] + try: + n_apps = desktop.get_child_count() + except Exception as exc: + print(f"CHECK5: NOT RUN — desktop.get_child_count() failed: {exc}") + sys.exit(2) + + for i in range(n_apps): + try: + app = desktop.get_child_at_index(i) + except Exception: + continue + if app is None: + continue + try: + app_name = app.get_name() + except Exception: + app_name = "" + if args.app_name not in app_name: + continue + app_observed = walk_for_check5(app, ["desktop"], all_seen, args.max_depth) + if app_observed is not None: + roots.append(app_observed) + + verdict = classify(expectation, roots) + if verdict.verdict == "PASS": + break + time.sleep(args.poll_interval) + + assert verdict is not None # the while loop above always runs at least once before a timeout + + print(f"CHECK5: {verdict.verdict}") + print(f"fixture: {args.fixture}") + print(f"attempt: {attempt}, timeout: {args.timeout}s") + print(f"reason: {verdict.reason}") + if verdict.observed_role is not None: + print(f"observed role: {verdict.observed_role}") + if verdict.observed_name is not None: + print(f"observed name: {verdict.observed_name!r}") + print(f"observed name (hex): {hex_lower(verdict.observed_name)}") + if verdict.prohibited_outcome is not None: + print(f"prohibited outcome: {verdict.prohibited_outcome}") + print("full tree (role:name) seen during the last walk:") + if not all_seen: + print(" ") + for line in all_seen: + print(" " + line) + + if args.json: + payload = { + "fixture_id": args.fixture, + "verdict": verdict.verdict, + "reason": verdict.reason, + "observed_role": verdict.observed_role, + "observed_name": verdict.observed_name, + "observed_name_hex": hex_lower(verdict.observed_name), + "prohibited_outcome": verdict.prohibited_outcome, + "walked_tree": all_seen, + } + with open(args.json, "w", encoding="utf-8") as f: + json.dump(payload, f, indent=2) + f.write("\n") + + sys.exit({"PASS": 0, "FAIL": 1}[verdict.verdict]) + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--role", default=None, help="Round 0 mode: exact role to match") + ap.add_argument("--name", default=None, help="Round 0 mode: exact name to match") + ap.add_argument( + "--expectations", + default=None, + help="Round 2 check 5 mode: path to round2-a11y-oracle's a11y_expectations.json", + ) + ap.add_argument("--fixture", default=None, help="Round 2 check 5 mode: fixture id (e.g. F-A)") + ap.add_argument( + "--expect-source-digest", + default=None, + help="Round 2 check 5 mode (required): round2_textkit::output::expected_artifact_digest() " + "— refuses the expectations file (usage error, exit 2) if its source_fixtures_digest " + "disagrees, so a stale oracle cannot score a candidate (B2)", + ) + ap.add_argument("--json", default=None, help="Round 2 check 5 mode: write the machine-readable verdict here") + ap.add_argument("--app-name", default=None, help="only descend into apps whose name contains this substring") + ap.add_argument("--max-depth", type=int, default=12) + ap.add_argument("--timeout", type=float, default=20.0) + ap.add_argument("--poll-interval", type=float, default=0.5) + args = ap.parse_args() + + round0_mode = args.role is not None and args.name is not None + check5_mode = args.expectations is not None and args.fixture is not None + + if round0_mode and check5_mode: + ap.error("--role/--name (Round 0 mode) and --expectations/--fixture (check 5 mode) are mutually exclusive") + if not round0_mode and not check5_mode: + ap.error("either --role and --name, or --expectations and --fixture, must be given") + if check5_mode and not args.app_name: + ap.error("--app-name is required in check 5 mode, to scope the walk to the candidate's application") + if check5_mode and not args.expect_source_digest: + ap.error( + "--expect-source-digest is required in check 5 mode (B2) — pass " + "round2_textkit::output::expected_artifact_digest()" + ) + + try: + import gi + + gi.require_version("Atspi", "2.0") + from gi.repository import Atspi + except Exception as exc: # pragma: no cover - environment probe + label = "READBACK" if round0_mode else "CHECK5" + print(f"{label}: NOT RUN — could not import gi.repository.Atspi: {exc}") + sys.exit(2) + + if round0_mode: + run_round0(args, Atspi) + else: + run_check5(args, Atspi) + + if __name__ == "__main__": main() diff --git a/spikes/editor-toolkit/round2-a11y-oracle/Cargo.toml b/spikes/editor-toolkit/round2-a11y-oracle/Cargo.toml new file mode 100644 index 0000000..97afa95 --- /dev/null +++ b/spikes/editor-toolkit/round2-a11y-oracle/Cargo.toml @@ -0,0 +1,40 @@ +[package] +name = "round2-a11y-oracle" +version = "0.1.0" +edition.workspace = true +publish.workspace = true + +# Packet 2B-A (ROUND2_TEXT_RECIPE.md §8, spec/CONTRACT_EDITOR_T4_SPIKE.md pin +# 13): the check-5 accessibility oracle's *comparison-data* half. This crate +# reads the already-committed, already-validated `round2-textkit/fixtures.json` +# (via `round2_textkit::output::load_fixtures`, which validates it against its +# embedded digest — see that crate) and writes +# `round2-a11y-oracle/a11y_expectations.json`: per fixture, the exact accepted/ +# prohibited at-spi2 roles and the exact byte strings each `PROHIBITED_OUTCOMES` +# classification would produce, so `a11y-verifier/verify.py`'s live-tree mode +# compares observed bytes against precommitted bytes rather than guessing what +# a wrong name "looks like". +# +# This crate must exist and be reviewed *before* either Round 2 candidate +# builds a tree (pin 13): if a candidate's own tree shaped what this oracle +# expects, the oracle would no longer be neutral evidence. +# +# Deliberately no rendering, windowing, or accessibility crate here — this +# crate never touches a live tree; that is `a11y-verifier/verify.py`'s job. + +[dependencies] +round2-textkit = { path = "../round2-textkit" } +serde = { version = "1", features = ["derive"] } +serde_json = "1" +# Pinned to the same version already resolved transitively (epiphany-core / +# epiphany-ops both already pull it), so adding this dependency changes +# nothing in Cargo.lock's resolution. +unicode-normalization = "=0.1.25" +# Same version round2-textkit shapes fixtures with, for the same reason it +# pins it: grapheme-boundary derivation here must agree with the grapheme +# boundaries `crate::hittest`'s caret stops were built from. +unicode-segmentation = "=1.13.3" + +[[bin]] +name = "generate_a11y_expectations" +path = "src/bin/generate_a11y_expectations.rs" diff --git a/spikes/editor-toolkit/round2-a11y-oracle/a11y_expectations.json b/spikes/editor-toolkit/round2-a11y-oracle/a11y_expectations.json new file mode 100644 index 0000000..b5f1ab5 --- /dev/null +++ b/spikes/editor-toolkit/round2-a11y-oracle/a11y_expectations.json @@ -0,0 +1,144 @@ +{ + "contract": "spec/CONTRACT_EDITOR_T4_SPIKE.md pin 13", + "recipe": "spikes/editor-toolkit/ROUND2_TEXT_RECIPE.md §8", + "platform": "at-spi2", + "source_fixtures_digest": "acc13c0d02624a0741cca5dffa7470a8971d3ecef5c6fb6f9e533ded684e7ed1", + "fixtures": [ + { + "fixture_id": "F-A", + "expected_name": "Allegro affettuoso — al fine", + "expected_name_hex": "416c6c6567726f20616666657474756f736f20e2809420616c2066696e65", + "expected_name_byte_len": 30, + "accepted_roles": [ + "label", + "static", + "text", + "paragraph" + ], + "prohibited_roles": [ + "image", + "canvas", + "filler", + "panel", + "unknown" + ], + "source_atoms": [ + "Allegro affettuoso — al fine" + ], + "alternative_forms": { + "name-is-shaped-glyphs": [ + "Allegro afettuoso — al fne", + "Allegro affettuoso — al fine" + ] + } + }, + { + "fixture_id": "F-B", + "expected_name": "Coro אבג", + "expected_name_hex": "436f726f20d790d791d792", + "expected_name_byte_len": 11, + "accepted_roles": [ + "label", + "static", + "text", + "paragraph" + ], + "prohibited_roles": [ + "image", + "canvas", + "filler", + "panel", + "unknown" + ], + "source_atoms": [ + "Coro ", + "אבג" + ], + "alternative_forms": {}, + "visual_order_name": "Coro גבא", + "visual_order_name_hex": "436f726f20d792d791d790" + }, + { + "fixture_id": "F-C", + "expected_name": "Coro ا", + "expected_name_hex": "436f726f20d8a7", + "expected_name_byte_len": 7, + "accepted_roles": [ + "label", + "static", + "text", + "paragraph" + ], + "prohibited_roles": [ + "image", + "canvas", + "filler", + "panel", + "unknown" + ], + "source_atoms": [ + "Coro ", + "ا" + ], + "alternative_forms": { + "name-drops-unresolved-codepoints": [ + "Coro " + ] + } + }, + { + "fixture_id": "F-D", + "expected_name": "Allegro אבג con brio", + "expected_name_hex": "416c6c6567726f20d790d791d79220636f6e206272696f", + "expected_name_byte_len": 23, + "accepted_roles": [ + "label", + "static", + "text", + "paragraph" + ], + "prohibited_roles": [ + "image", + "canvas", + "filler", + "panel", + "unknown" + ], + "source_atoms": [ + "Allegro ", + "אבג", + " con brio" + ], + "alternative_forms": {}, + "visual_order_name": "Allegro גבא con brio", + "visual_order_name_hex": "416c6c6567726f20d792d791d79020636f6e206272696f" + }, + { + "fixture_id": "F-E", + "expected_name": "Café — resumé", + "expected_name_hex": "43616665cc8120e2809420726573756d65cc81", + "expected_name_byte_len": 19, + "accepted_roles": [ + "label", + "static", + "text", + "paragraph" + ], + "prohibited_roles": [ + "image", + "canvas", + "filler", + "panel", + "unknown" + ], + "source_atoms": [ + "Café — resumé" + ], + "alternative_forms": { + "name-normalized": [ + "Café — resumé" + ] + } + } + ] +} diff --git a/spikes/editor-toolkit/round2-a11y-oracle/src/bin/generate_a11y_expectations.rs b/spikes/editor-toolkit/round2-a11y-oracle/src/bin/generate_a11y_expectations.rs new file mode 100644 index 0000000..a27a0b4 --- /dev/null +++ b/spikes/editor-toolkit/round2-a11y-oracle/src/bin/generate_a11y_expectations.rs @@ -0,0 +1,58 @@ +//! `generate_a11y_expectations` — Packet 2B-A's entry point. +//! +//! Loads `round2-textkit/fixtures.json` (validating it against its own +//! embedded digest via `round2_textkit::output::load_fixtures`), derives this +//! machine's check-5 comparison data for all five fixtures, and writes +//! `round2-a11y-oracle/a11y_expectations.json`. +//! +//! Exit behavior mirrors `round2-textkit`'s own `bin/generate`: a missing +//! `fixtures.json` (never generated, or generated on a machine without the +//! declared faces) is reported and this binary exits non-zero rather than +//! writing a partial or empty file — pin 13's ordering requires the oracle to +//! exist and be reviewed before a candidate consumes it, so silently writing +//! nothing would be worse than a loud failure. + +use std::path::PathBuf; + +fn main() { + let manifest_dir = PathBuf::from(env!("CARGO_MANIFEST_DIR")); + let textkit_dir = manifest_dir + .parent() + .expect("round2-a11y-oracle has a parent directory") + .join("round2-textkit"); + let fixtures_path = textkit_dir.join("fixtures.json"); + + let fixtures = round2_textkit::output::load_fixtures(&fixtures_path).unwrap_or_else(|e| { + panic!( + "{}: {e} — run `cargo run -p round2-textkit --bin generate` first", + fixtures_path.display() + ) + }); + + let expectations = round2_a11y_oracle::build_expectations_file(&fixtures); + + let out_path = manifest_dir.join("a11y_expectations.json"); + let json = serde_json::to_string_pretty(&expectations) + .expect("ExpectationsFile is always serializable"); + std::fs::write(&out_path, format!("{json}\n")) + .unwrap_or_else(|e| panic!("failed to write {}: {e}", out_path.display())); + + println!( + "wrote {} ({} fixtures, platform {})", + out_path.display(), + expectations.fixtures.len(), + expectations.platform + ); + for f in &expectations.fixtures { + println!( + " {}: {} alternative form(s), visual_order_name {}", + f.fixture_id, + f.alternative_forms.len(), + if f.visual_order_name.is_some() { + "present" + } else { + "omitted (identical to expected_name)" + } + ); + } +} diff --git a/spikes/editor-toolkit/round2-a11y-oracle/src/findings.rs b/spikes/editor-toolkit/round2-a11y-oracle/src/findings.rs new file mode 100644 index 0000000..3de019c --- /dev/null +++ b/spikes/editor-toolkit/round2-a11y-oracle/src/findings.rs @@ -0,0 +1,113 @@ +//! Findings routed back to `spikes/editor-toolkit/ROUND2_TEXT_RECIPE.md`, +//! discovered while building this crate. +//! +//! Recorded here — not only in a review conversation — so whoever next +//! revises the recipe finds it in the artifact rather than a chat transcript, +//! the same discipline `round2_textkit::findings` uses for the findings it +//! routes back to the W3 `.tex` amendment. These are findings *about the +//! recipe's own prose*, not about `epiphany-layout-ir`, so they are recorded +//! here rather than in `round2_textkit::findings`. +//! +//! This crate does not edit `ROUND2_TEXT_RECIPE.md` — that document belongs +//! to the coordinator's commit and a separate review thread. + +/// Recipe §8.1 claims: "a tree assembled by walking the visual runs left to +/// right produces a different string, and only there \[F-D\]." +/// +/// That is false under this crate's own generated data. F-B diverges the +/// same way: its logical name is `"Coro אבג"` and +/// `round2_a11y_oracle::visual_order_form` produces `"Coro גבא"` for it — a +/// real, non-empty `visual_order_name` entry in `a11y_expectations.json`, +/// exactly the same mechanism F-D exercises. +/// +/// The general shape, not just the one counterexample: under +/// [`crate::visual_order_form`]'s model (concatenate segments in stored +/// order, reversing an `Rtl` segment's own text by grapheme), **any +/// non-palindromic** RTL run of two or more graphemes diverges under a +/// visual-order walk, because reversing a grapheme sequence is a no-op +/// exactly when that sequence is a palindrome (a repeated single grapheme, +/// e.g. `"aa"`, is a palindrome and is therefore **not** a counterexample to +/// this narrower claim — it was a counterexample to the unqualified "any RTL +/// run of two or more graphemes" claim an earlier revision of this finding +/// made). F-D is not the *unique* case; it is the case where the RTL run is +/// *interior* to the string (`"Allegro "` ... `"אבג"` ... `" con brio"`) +/// rather than trailing (`"Coro "` ... `"אבג"`), which is why F-D's +/// divergence reads as obviously wrong to a human glancing at it and F-B's — +/// a suffix silently reversed — reads as more easily missed. That +/// readability difference is a real reason to prefer F-D as the check-5 +/// accessibility exemplar; it is not a reason to claim F-B does not exhibit +/// the same property. +/// +/// The recipe should either say "F-D and F-B" at §8.1, or drop the +/// uniqueness claim and state the actual distinguishing property: F-D is the +/// fixture where the RTL run is interior, not the fixture where the +/// divergence uniquely occurs. +/// +/// ## The same stale claim is also baked into a digest-bound artifact +/// +/// The recipe's prose is not the only place this claim lives. +/// `round2-textkit/src/a11y.rs`'s `note_for("F-D")` reads: "the concatenation +/// is logical-order, so a tree built by walking the visual runs left to +/// right fails here and only here" — the identical uniqueness claim, in +/// code. That note is compiled into every generated `fixtures.json` as +/// `fixtures[3].accessibility.note`, and `fixtures.json`'s own +/// `EXPECTED_ARTIFACT_DIGEST_HEX` (`round2_textkit::output`) binds the whole +/// serialized file, note text included, to `acc13c0d…` — a frozen, +/// user-reviewed artifact (Packet 2A). Editing the note's wording to correct +/// the claim would change that digest and break every consumer pinned to it, +/// which is a strictly larger and differently-scoped change than this +/// finding. +/// +/// **This half of the finding is tracked, not fixed**, and is recorded +/// explicitly so a later reader does not "helpfully" edit +/// `round2-textkit/src/a11y.rs`'s F-D note on the strength of this finding +/// alone and silently move `acc13c0d…` out from under Packet 2A. Fixing it +/// is a decision for whoever owns that digest and that packet's re-freeze, +/// not a drive-by edit from this crate. +pub const RECIPE_F1_VISUAL_ORDER_NOT_UNIQUE_TO_F_D: &str = "ROUND2_TEXT_RECIPE.md §8.1 claims \ + visual-order-walk assembly produces a different string \"and only there [F-D]\". It does \ + not: F-B's visual_order_name (\"Coro גבא\") also differs from its expected_name (\"Coro \ + אבג\"), and under this crate's visual_order_form, any non-palindromic RTL run of two or more \ + graphemes diverges the same way (a repeated-grapheme run like \"aa\" is a palindrome and does \ + not diverge, which is why the claim is qualified). F-D is not unique in exhibiting the \ + divergence; it is the fixture where the RTL run is interior to the string rather than \ + trailing, which is why the divergence is more obviously wrong to a reader. The recipe should \ + say \"F-D and F-B\" or state the interior-run property instead of a uniqueness claim. The \ + identical stale claim is also baked into round2-textkit/src/a11y.rs's note_for(\"F-D\") \ + (\"fails here and only here\"), which is compiled into fixtures.json and covered by its \ + frozen EXPECTED_ARTIFACT_DIGEST_HEX (acc13c0d...) — that half is TRACKED, NOT FIXED here, \ + because correcting it would move the digest and break Packet 2A; do not edit that note on \ + the strength of this finding alone."; + +#[cfg(test)] +mod tests { + use super::*; + + /// The finding must actually name both fixtures — a mutation that + /// silently dropped one of them from the constant would still compile + /// and would still "record a finding," just not the right one. + #[test] + fn the_finding_names_both_f_d_and_f_b() { + assert!(RECIPE_F1_VISUAL_ORDER_NOT_UNIQUE_TO_F_D.contains("F-D")); + assert!(RECIPE_F1_VISUAL_ORDER_NOT_UNIQUE_TO_F_D.contains("F-B")); + } + + /// The universal claim must be qualified — an unqualified "any RTL run + /// of two or more graphemes diverges" is false (a palindromic run does + /// not), which is exactly the over-claim B3 asked to be narrowed. + #[test] + fn the_finding_qualifies_the_claim_as_non_palindromic() { + assert!(RECIPE_F1_VISUAL_ORDER_NOT_UNIQUE_TO_F_D.contains("non-palindromic")); + } + + /// The digest-bound, tracked-not-fixed half of the finding must name the + /// actual frozen digest prefix and say explicitly that it is not fixed + /// here — a reader skimming only for "is this fixed" must not be able to + /// mistake "recorded" for "corrected." + #[test] + fn the_finding_names_the_frozen_digest_and_says_tracked_not_fixed() { + assert!(RECIPE_F1_VISUAL_ORDER_NOT_UNIQUE_TO_F_D.contains("acc13c0d")); + assert!(RECIPE_F1_VISUAL_ORDER_NOT_UNIQUE_TO_F_D.contains("TRACKED, NOT FIXED")); + assert!(RECIPE_F1_VISUAL_ORDER_NOT_UNIQUE_TO_F_D.contains("a11y.rs")); + } +} diff --git a/spikes/editor-toolkit/round2-a11y-oracle/src/lib.rs b/spikes/editor-toolkit/round2-a11y-oracle/src/lib.rs new file mode 100644 index 0000000..acd6e29 --- /dev/null +++ b/spikes/editor-toolkit/round2-a11y-oracle/src/lib.rs @@ -0,0 +1,971 @@ +//! Packet 2B-A: the check-5 accessibility oracle's comparison-data half. +//! +//! `spec/CONTRACT_EDITOR_T4_SPIKE.md` pin 13 requires this oracle to be +//! committed and reviewed **before any candidate builds a tree against it** — +//! if a candidate wrote the verifier, the oracle would be shaped by that +//! candidate's tree, which is exactly what pin 13 forbids. This crate is +//! therefore a separate, candidate-neutral packet from either Round 2 +//! candidate: it reads the already-committed, already-validated +//! `round2-textkit/fixtures.json` (`ROUND2_TEXT_RECIPE.md` §8; +//! `round2_textkit::a11y`) and derives every byte string a live AT-SPI +//! readback would need to compare against, so the verifier's classification +//! is a comparison against precommitted data, never a heuristic guess about +//! what a wrong name "looks like" (`ROUND2_TEXT_RECIPE.md` §8.1). +//! +//! ## What is derived, and what is restated +//! +//! `expected_name` / `expected_name_hex` / `expected_name_byte_len` and the +//! at-spi2 accepted/prohibited role sets are **restated** — they already +//! exist verbatim on each fixture's `SpikeAccessibilityExpectation` +//! (`round2_textkit::a11y`), computed and validated there. This crate does +//! not recompute them from `resolved.text` a second time; it reads the +//! oracle's own already-validated fields, the same discipline +//! `round2-textkit::output::FixtureFile::validate` uses for everything else. +//! +//! `alternative_forms` and `visual_order_name` are **derived** here, from +//! `SpikeResolvedText`'s own segment and cluster data — never hard-coded to a +//! particular codepoint or glyph id, so the derivation is reproducible from +//! `fixtures.json` alone and does not silently drift from it: +//! +//! * **`name-normalized`** — the NFC normalization of `text` +//! (`unicode-normalization`). Differs only for F-E (recipe §2: F-E is +//! deliberately NFD). +//! * **`name-drops-unresolved-codepoints`** — `text` with every segment whose +//! `face` is `None` removed (`SpikeShapedSegment::face`, `W3-F3`). Differs +//! only for F-C, whose U+0627 is covered by neither declared face. +//! * **`name-is-shaped-glyphs`** — two independently derived forms, both +//! modelling "the tree exposes what was drawn rather than what was said": +//! a cluster-collapse form ([`shaped_glyphs_form`]) and, where derivable, a +//! standard-ligature presentation-form substitution +//! ([`shaped_glyphs_presentation_form`]). F-A's `ff`/`fi` ligatures are the +//! case this fixture set exercises for both. See each function's doc +//! comment for exactly what it does and does not derive from the fixture +//! record. +//! * **`visual_order_name`** — concatenates every segment's source text in +//! the *stored* (logical) segment order, but reverses an `Rtl` segment's +//! own text by extended grapheme cluster before appending it. This +//! reproduces "a tree assembled by walking the visual runs left to right" +//! (recipe §8.1) for every fixture in this set, all of which nest at most +//! one `Rtl` run inside an `Ltr` base paragraph (recipe §4: base level 0, +//! Hebrew segments at level 1) — a single odd-level run does not change +//! the *order* of the run sequence under UAX#9 reordering, only the +//! *internal* order of that run's own text. **This is not a general bidi +//! run-reordering implementation**; it is correct for this fixture set and +//! would need revisiting for a fixture with nested embedding levels beyond +//! 0/1, which none of F-A..F-E have (measured, recipe §4). It also +//! diverges from the recipe's own claim about which fixture this +//! is unique to — see [`findings::RECIPE_F1_VISUAL_ORDER_NOT_UNIQUE_TO_F_D`]. +//! +//! ## Fail-closed on a colliding classification (O1) +//! +//! An earlier version of this crate could emit the *same string* under two +//! different `PROHIBITED_OUTCOMES` names for one fixture — F-C's unresolved +//! cluster produced `"Coro "` under both `name-drops-unresolved-codepoints` +//! and `name-is-shaped-glyphs`, because "drop the unresolved codepoint" and +//! "collapse a zero-glyph cluster" were, for that cluster, the same +//! operation. Which classification a verifier reported was then an artifact +//! of `BTreeMap` iteration (alphabetical) order, not a property of the +//! observation — the oracle was returning two different confident answers +//! for one input. That is fixed two ways, and both are load-bearing: +//! +//! 1. [`shaped_glyphs_form`] no longer collapses a *fully* unresolved +//! cluster (zero glyphs) — collapsing to "what was drawn" presumes +//! something was drawn; a wholly unresolved cluster's only legitimate +//! classification is `name-drops-unresolved-codepoints`. This is enough +//! to make F-C's two forms genuinely equal to `expected_name` again (no +//! codepoint was shape-collapsed), so `name-is-shaped-glyphs` is correctly +//! omitted for F-C by the ordinary omit-if-identical rule. +//! 2. [`build_expectation`] additionally **refuses to build** a fixture whose +//! candidate forms collide across two different outcome names, panicking +//! and naming the fixture and both outcomes — a generation-time backstop +//! for any future fixture or derivation that reintroduces the same +//! ambiguity, independent of whether fix 1 above happens to prevent it. + +use std::collections::BTreeMap; + +use round2_textkit::a11y::PROHIBITED_OUTCOMES; +use round2_textkit::output::{FixtureFile, FixtureRecord}; +use round2_textkit::types::{SpikeResolvedText, SpikeTextDirection}; +use serde::{Deserialize, Serialize}; +use unicode_normalization::UnicodeNormalization; +use unicode_segmentation::UnicodeSegmentation; + +pub mod findings; + +/// The one platform row this oracle emits: this machine's live AT client is +/// AT-SPI2 (recipe §8.2, round0-evidence's precedent). Candidates targeting +/// another platform stay covered by the recipe's own table; encoding all five +/// rows here would not make them checkable on a machine that cannot reach +/// them. +pub const PLATFORM: &str = "at-spi2"; + +/// One fixture's precommitted check-5 comparison data. +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct FixtureExpectation { + pub fixture_id: String, + /// The exact source string, restated from + /// `SpikeAccessibilityExpectation::name` (`round2_textkit::a11y`), not + /// recomputed — see the module doc comment. + pub expected_name: String, + pub expected_name_hex: String, + pub expected_name_byte_len: usize, + /// This machine's platform row (`at-spi2`) of recipe §8.2's accepted-role + /// table, restated from the fixture's own + /// `SpikeAccessibilityExpectation::accepted_roles`. + pub accepted_roles: Vec, + pub prohibited_roles: Vec, + /// D1: the per-segment source texts, in the segments' own stored + /// (logical, ascending-source) order — e.g. F-C's `["Coro ", "ا"]`, + /// F-D's `["Allegro ", "אבג", " con brio"]`. §8.1 explicitly permits "a + /// tree that exposes one text node per direction run," and §8.3 + /// requires an unresolved codepoint (F-C's `ا`) to appear in the name + /// regardless of whether it drew ink — but a lone unresolved segment can + /// be a single character, which falls below any reasonable + /// coincidence-guarded length floor a verifier-side substring rule would + /// use. This field lets the verifier match a node's name against a + /// precommitted exact component instead of guessing from length alone — + /// the same "precommitted comparison data, not a heuristic" discipline + /// this whole struct already uses everywhere else. `"".join(source_atoms) + /// == expected_name` always holds (see `source_atoms`'s own doc comment + /// and its test coverage). + pub source_atoms: Vec, + /// Keyed by a `PROHIBITED_OUTCOMES` name; every precommitted string that + /// classification would produce for this fixture, matched if the + /// observed name equals **any** entry in the list (O2: a single outcome + /// can have more than one plausible precommitted rendering — e.g. + /// `name-is-shaped-glyphs` carries both a cluster-collapse form and a + /// standard-ligature presentation-form substitution for F-A). An outcome + /// absent from this map produced no form distinguishable from + /// `expected_name` for this fixture (see the module doc comment) and so + /// cannot classify anything. The same string never appears under two + /// different outcome keys for one fixture — [`build_expectation`] + /// refuses to build a file where it would (O1). + pub alternative_forms: BTreeMap>, + /// The concatenation a tree assembled by walking visual runs left to + /// right would produce, only when it differs from `expected_name` (§8.1). + #[serde(skip_serializing_if = "Option::is_none")] + pub visual_order_name: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub visual_order_name_hex: Option, +} + +/// The complete artifact `a11y_expectations.json` carries. +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct ExpectationsFile { + pub contract: String, + pub recipe: String, + pub platform: String, + /// Traceability to the exact `fixtures.json` this file was derived from + /// (`round2_textkit::output::artifact_digest`) — so a verifier run + /// against a stale copy of either file is a detectable mismatch rather + /// than a silent one, the same discipline `fixtures.json` itself uses for + /// the two declared face hashes. + pub source_fixtures_digest: String, + pub fixtures: Vec, +} + +fn hex_lower(bytes: &[u8]) -> String { + use std::fmt::Write as _; + let mut s = String::with_capacity(bytes.len() * 2); + for b in bytes { + let _ = write!(s, "{b:02x}"); + } + s +} + +/// NFC normalization of `text`. §8.3's `name-normalized`: F-E's NFD source is +/// the only fixture where this differs from `text`. +pub fn nfc_form(text: &str) -> String { + text.nfc().collect() +} + +/// `text` with every segment whose `face` is `None` removed, in the +/// segments' own stored (logical, ascending-source) order. §8.3's +/// `name-drops-unresolved-codepoints`: F-C's U+0627 (covered by neither +/// declared face) is the only case in this fixture set. +/// +/// Derived entirely from `resolved.segments[*].face` and `.source` — never +/// from a hard-coded codepoint, so a future fixture with a different +/// uncovered span is handled the same way without a code change. +pub fn drop_unresolved_codepoints_form(resolved: &SpikeResolvedText) -> String { + let mut out = String::new(); + for seg in &resolved.segments { + if seg.face.is_none() { + continue; + } + let start = seg.source.start as usize; + let end = seg.source.end as usize; + out.push_str(&resolved.text[start..end]); + } + out +} + +/// The per-segment source texts, in the segments' own stored (logical, +/// ascending-source) order (D1). Every segment contributes an atom, +/// resolved or not — F-C's unresolved `ا` is included exactly like any +/// other segment, because the property this field exists to let a verifier +/// check ("does some node's name match one exact source component") is +/// just as true for an unresolved segment as a resolved one, and singling +/// it out would be exactly the kind of per-fixture special case this crate +/// avoids elsewhere. +/// +/// Derived entirely from `resolved.segments[*].source` — never from a +/// hard-coded codepoint or fixture id, so a future fixture's own segment +/// boundaries are picked up the same way without a code change. +/// `source_atoms(resolved).concat() == resolved.text` always holds, because +/// W3 invariant 2 (asserted elsewhere in this pipeline) requires segment +/// source ranges to partition the whole string totally, in logical order. +pub fn source_atoms(resolved: &SpikeResolvedText) -> Vec { + resolved + .segments + .iter() + .map(|seg| { + let start = seg.source.start as usize; + let end = seg.source.end as usize; + resolved.text[start..end].to_string() + }) + .collect() +} + +/// The run's text as a tree exposing "what was drawn" rather than "what was +/// said" would read it, by collapsing each cluster to as many leading +/// graphemes as it has glyphs. §8.3's `name-is-shaped-glyphs`: F-A's `ff`/`fi` +/// ligatures are the case this fixture set exercises. +/// +/// Walks `resolved.clusters.clusters` in its own documented ascending-source +/// order (`SpikeClusterMap`'s doc comment). For a cluster whose glyph count is +/// **strictly between zero and** its `grapheme_count` — a genuine ligature +/// drew fewer, but more than zero, glyphs than there are graphemes to +/// report — only that many leading graphemes of the cluster's own source text +/// are kept. A cluster with `glyphs == graphemes` (ordinary) or `glyphs == 0` +/// (**wholly unresolved** — O1: nothing was drawn, so there is no partial +/// "what was drawn" to report; that is `name-drops-unresolved-codepoints`'s +/// classification, not this one) contributes its whole source text unchanged. +/// Nothing here is specific to `ff`/`fi`: the rule is "one reportable unit per +/// glyph, when at least one glyph exists," derived purely from each cluster's +/// own `glyph_indices.len()` and `grapheme_count`. +pub fn shaped_glyphs_form(resolved: &SpikeResolvedText) -> String { + let mut out = String::new(); + for cluster in &resolved.clusters.clusters { + let start = cluster.source.start as usize; + let end = cluster.source.end as usize; + let chunk = &resolved.text[start..end]; + let glyph_count = cluster.glyph_indices.len() as u32; + if glyph_count > 0 && glyph_count < cluster.grapheme_count { + let kept: String = chunk.graphemes(true).take(glyph_count as usize).collect(); + out.push_str(&kept); + } else { + out.push_str(chunk); + } + } + out +} + +/// The standard Unicode Latin ligature presentation forms (Alphabetic +/// Presentation Forms block, U+FB00-U+FB06) that a shaper's default `liga` +/// feature can produce. This table is **fixed Unicode data, not derived from +/// `fixtures.json`** — this crate deliberately carries no font/cmap +/// dependency (see the crate doc comment on why: it never touches a live +/// tree, and adding one here would be the wrong layer for it), so there is no +/// way to derive "this glyph id denotes U+FB00" from the fixture record +/// alone. What **is** derived from the fixture, for every entry +/// [`shaped_glyphs_presentation_form`] produces, is *which* clusters this +/// table applies to (the same glyph-count-vs-grapheme-count ligature +/// detection [`shaped_glyphs_form`] uses) and *what source text* each one +/// spans; the table is only ever consulted as a lookup keyed by that +/// already-derived source text, never used to invent a cluster boundary of +/// its own. +const LATIN_LIGATURE_PRESENTATION_FORMS: &[(&str, char)] = &[ + ("ff", '\u{FB00}'), + ("fi", '\u{FB01}'), + ("fl", '\u{FB02}'), + ("ffi", '\u{FB03}'), + ("ffl", '\u{FB04}'), + ("st", '\u{FB06}'), +]; + +/// A second, independently plausible rendering of "the tree exposes what was +/// drawn" (§8.3's `name-is-shaped-glyphs`): a tree that reverse-mapped glyph +/// ids through a cmap would most plausibly emit the *standard ligature +/// presentation-form codepoint* (e.g. U+FB00 for `ff`) rather than +/// [`shaped_glyphs_form`]'s truncate-to-glyph-count text. Returns `None` if +/// this fixture has no ligature cluster, **or** if it has one whose source +/// text is not in [`LATIN_LIGATURE_PRESENTATION_FORMS`] — this function never +/// guesses a codepoint it cannot look up. +pub fn shaped_glyphs_presentation_form(resolved: &SpikeResolvedText) -> Option { + let mut out = String::new(); + let mut substituted_any = false; + for cluster in &resolved.clusters.clusters { + let start = cluster.source.start as usize; + let end = cluster.source.end as usize; + let chunk = &resolved.text[start..end]; + let glyph_count = cluster.glyph_indices.len() as u32; + let is_ligature = glyph_count > 0 && glyph_count < cluster.grapheme_count; + if is_ligature { + match LATIN_LIGATURE_PRESENTATION_FORMS + .iter() + .find(|(seq, _)| *seq == chunk) + { + Some((_, presentation_char)) => { + out.push(*presentation_char); + substituted_any = true; + } + // A ligature cluster whose source text has no known + // presentation-form codepoint: this function cannot honestly + // produce a full-string answer, so it produces none at all + // rather than a partially-substituted guess. + None => return None, + } + } else { + out.push_str(chunk); + } + } + substituted_any.then_some(out) +} + +/// The concatenation a tree assembled by walking the run's visual runs left +/// to right would produce (§8.1). See the module doc comment for exactly +/// what this does and does not model. +pub fn visual_order_form(resolved: &SpikeResolvedText) -> String { + let mut out = String::new(); + for seg in &resolved.segments { + let start = seg.source.start as usize; + let end = seg.source.end as usize; + let chunk = &resolved.text[start..end]; + match seg.direction { + SpikeTextDirection::Rtl => { + let reversed: String = chunk.graphemes(true).rev().collect(); + out.push_str(&reversed); + } + SpikeTextDirection::Ltr => out.push_str(chunk), + } + } + out +} + +/// Groups a fixture's candidate `(outcome, form)` pairs into +/// `alternative_forms`, in three steps: +/// +/// 1. drop any candidate whose form is byte-identical to `expected_name` (it +/// cannot classify anything — see the module doc comment); +/// 2. **refuse** (panic, naming `fixture_id` and both outcomes) if the same +/// remaining form string is produced by two *different* outcome names — +/// O1's fail-closed backstop, independent of whichever derivation bug did +/// or did not cause it; +/// 3. otherwise group by outcome, deduplicating repeated identical forms +/// within one outcome's own list (the same classification derived twice is +/// not a collision), and drop any outcome left with an empty list. +/// +/// Kept as its own function, separate from [`build_expectation`], so it has a +/// unit test that can hand-construct a collision without needing a real +/// `FixtureRecord` to provoke one. +fn group_alternative_forms( + fixture_id: &str, + expected_name: &str, + candidates: Vec<(&'static str, String)>, +) -> BTreeMap> { + let mut owner_of: BTreeMap = BTreeMap::new(); + let mut grouped: BTreeMap> = BTreeMap::new(); + + for (outcome, form) in candidates { + debug_assert!( + PROHIBITED_OUTCOMES.contains(&outcome), + "{outcome} must be one of round2_textkit::a11y::PROHIBITED_OUTCOMES" + ); + if form == expected_name { + continue; + } + match owner_of.get(&form) { + Some(&existing_outcome) if existing_outcome != outcome => { + panic!( + "{fixture_id}: alternative forms {existing_outcome:?} and {outcome:?} both \ + produce {form:?} — an oracle that returns two different classifications for \ + the same observed string must fail closed, not pick one by BTreeMap \ + iteration order (O1). Fix the derivation so the two outcomes do not collide, \ + or establish that only one of them legitimately applies to this fixture." + ); + } + Some(_) => { + // Same outcome producing an identical form a second time + // (e.g. two independent derivations that happen to agree) — + // not a collision, just redundant; skip the duplicate. + } + None => { + owner_of.insert(form.clone(), outcome); + grouped.entry(outcome.to_string()).or_default().push(form); + } + } + } + grouped +} + +/// Builds one fixture's [`FixtureExpectation`] from its already-validated +/// `FixtureRecord`. +/// +/// `expected_name` and the role sets are restated from +/// `record.accessibility`, not recomputed from `record.resolved.text` — that +/// field was already checked against the recipe §2 literal by +/// `FixtureFile::validate` (via `load_fixtures`) before this function ever +/// runs, so re-deriving it here would be a second, redundant source of +/// truth rather than a check. +/// +/// Panics (via [`group_alternative_forms`]) if two different outcome names +/// would classify the same observed string for this fixture (O1). +pub fn build_expectation(record: &FixtureRecord) -> FixtureExpectation { + let a = &record.accessibility; + let resolved = &record.resolved; + + let accepted_roles = a + .accepted_roles + .iter() + .find(|m| m.platform == PLATFORM) + .unwrap_or_else(|| panic!("{}: no {PLATFORM} row in accepted_roles", record.id)) + .tokens + .clone(); + let prohibited_roles = a + .prohibited_roles + .iter() + .find(|m| m.platform == PLATFORM) + .unwrap_or_else(|| panic!("{}: no {PLATFORM} row in prohibited_roles", record.id)) + .tokens + .clone(); + + let mut candidates: Vec<(&'static str, String)> = vec![ + ("name-normalized", nfc_form(&a.name)), + ( + "name-drops-unresolved-codepoints", + drop_unresolved_codepoints_form(resolved), + ), + ("name-is-shaped-glyphs", shaped_glyphs_form(resolved)), + ]; + if let Some(presentation) = shaped_glyphs_presentation_form(resolved) { + candidates.push(("name-is-shaped-glyphs", presentation)); + } + let alternative_forms = group_alternative_forms(&record.id, &a.name, candidates); + + let visual = visual_order_form(resolved); + let (visual_order_name, visual_order_name_hex) = if visual != a.name { + let hex = hex_lower(visual.as_bytes()); + (Some(visual), Some(hex)) + } else { + (None, None) + }; + + FixtureExpectation { + fixture_id: record.id.clone(), + expected_name: a.name.clone(), + expected_name_hex: a.name_bytes_hex.clone(), + expected_name_byte_len: a.name_byte_len, + accepted_roles, + prohibited_roles, + source_atoms: source_atoms(resolved), + alternative_forms, + visual_order_name, + visual_order_name_hex, + } +} + +/// Builds the complete [`ExpectationsFile`] from an already-loaded, already- +/// validated `FixtureFile` (`round2_textkit::output::load_fixtures`). +pub fn build_expectations_file(file: &FixtureFile) -> ExpectationsFile { + ExpectationsFile { + contract: "spec/CONTRACT_EDITOR_T4_SPIKE.md pin 13".to_string(), + recipe: "spikes/editor-toolkit/ROUND2_TEXT_RECIPE.md §8".to_string(), + platform: PLATFORM.to_string(), + source_fixtures_digest: round2_textkit::output::artifact_digest(file), + fixtures: file.fixtures.iter().map(build_expectation).collect(), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use round2_textkit::faces::{resolve_declared_chain, FaceResolution, LoadedFace}; + use round2_textkit::fixtures::{build_fixture, FIXTURES}; + use round2_textkit::output::build_fixture_file; + + /// Builds a real `FixtureFile` end to end against the actual declared + /// faces on this machine, the same path `round2-textkit`'s own tests and + /// `bin/generate.rs` take. `None` (test skipped, not failed — pin 14) if + /// either declared face is absent; on this development machine both are + /// present. + fn real_fixture_file() -> Option { + let resolved = resolve_declared_chain(); + let mut loaded: Vec = Vec::new(); + for r in resolved { + match r { + FaceResolution::Loaded(lf) => loaded.push(lf), + FaceResolution::Missing { .. } => return None, + } + } + let built: Vec<(String, String, SpikeResolvedText)> = FIXTURES + .iter() + .enumerate() + .map(|(i, def)| { + let rt = build_fixture(def, &loaded, i as u64); + (def.id.to_string(), def.purpose.to_string(), rt) + }) + .collect(); + Some(build_fixture_file(&loaded, built).expect("every fixture has a precommitted note")) + } + + fn require_file() -> FixtureFile { + real_fixture_file().expect( + "this test requires the two round2-textkit declared faces to be present on the \ + machine running it", + ) + } + + fn expectation_for<'a>(exp: &'a ExpectationsFile, id: &str) -> &'a FixtureExpectation { + exp.fixtures + .iter() + .find(|f| f.fixture_id == id) + .unwrap_or_else(|| panic!("no expectation built for {id}")) + } + + /// F-E's NFC form must differ from its (NFD) source text — the whole + /// reason F-E exists (recipe §2). If `nfc_form` stopped normalizing, or + /// F-E's source stopped being NFD, this fails. + #[test] + fn f_e_nfc_form_differs_from_its_text() { + let file = require_file(); + let f_e = file.fixtures.iter().find(|f| f.id == "F-E").unwrap(); + let nfc = nfc_form(&f_e.resolved.text); + assert_ne!( + nfc, f_e.resolved.text, + "F-E's NFC form must differ from its NFD source" + ); + // And it must actually surface in alternative_forms, keyed correctly. + let exp = build_expectations_file(&file); + let e = expectation_for(&exp, "F-E"); + assert_eq!( + e.alternative_forms.get("name-normalized"), + Some(&vec![nfc]), + "F-E must carry a name-normalized alternative form equal to its NFC" + ); + } + + /// F-C's dropped-codepoint form must be shorter than its source by + /// *exactly* the byte span of its unresolved (face: None) segment — not + /// merely shorter by some amount. + #[test] + fn f_c_dropped_codepoint_form_is_shorter_by_exactly_the_unresolved_span() { + let file = require_file(); + let f_c = file.fixtures.iter().find(|f| f.id == "F-C").unwrap(); + let unresolved_span: usize = f_c + .resolved + .segments + .iter() + .filter(|s| s.face.is_none()) + .map(|s| (s.source.end - s.source.start) as usize) + .sum(); + assert!( + unresolved_span > 0, + "anchor: F-C must have at least one unresolved segment" + ); + let dropped = drop_unresolved_codepoints_form(&f_c.resolved); + assert_eq!( + f_c.resolved.text.len() - dropped.len(), + unresolved_span, + "F-C's dropped-codepoint form must be shorter by exactly its unresolved span" + ); + assert_ne!(dropped, f_c.resolved.text); + + let exp = build_expectations_file(&file); + let e = expectation_for(&exp, "F-C"); + assert_eq!( + e.alternative_forms.get("name-drops-unresolved-codepoints"), + Some(&vec![dropped]) + ); + } + + /// O1's regression lock: F-C must carry exactly one alternative-outcome + /// classification (`name-drops-unresolved-codepoints`), never a second, + /// colliding `name-is-shaped-glyphs` entry for the same string. Before + /// the O1 fix, [`shaped_glyphs_form`] collapsed F-C's wholly-unresolved + /// cluster to nothing, which is byte-identical to the dropped-codepoint + /// form — this pins that `name-is-shaped-glyphs` is now correctly absent + /// for F-C (because it is byte-identical to `expected_name` once + /// zero-glyph clusters are left untouched), not merely that it happens + /// to agree with the other outcome. + #[test] + fn f_c_carries_no_shaped_glyphs_alternative_form() { + let file = require_file(); + let f_c = file.fixtures.iter().find(|f| f.id == "F-C").unwrap(); + assert_eq!( + shaped_glyphs_form(&f_c.resolved), + f_c.resolved.text, + "anchor: with the O1 fix, F-C's cluster-collapse form must equal its source text" + ); + let exp = build_expectations_file(&file); + let e = expectation_for(&exp, "F-C"); + assert!( + !e.alternative_forms.contains_key("name-is-shaped-glyphs"), + "F-C must not carry a name-is-shaped-glyphs alternative form: {:?}", + e.alternative_forms + ); + assert_eq!(e.alternative_forms.len(), 1); + } + + /// F-A's shaped-glyphs forms must differ from its source text — the + /// `ff`/`fi` ligature case §8.3 names — and both the cluster-collapse + /// form and the standard-ligature presentation-form substitution (O2) + /// must be present in the list. + #[test] + fn f_a_shaped_glyphs_forms_differ_from_its_text() { + let file = require_file(); + let f_a = file.fixtures.iter().find(|f| f.id == "F-A").unwrap(); + let has_ligature_cluster = f_a + .resolved + .clusters + .clusters + .iter() + .any(|c| (c.glyph_indices.len() as u32) < c.grapheme_count); + assert!( + has_ligature_cluster, + "anchor: F-A must have at least one cluster with fewer glyphs than graphemes" + ); + let collapsed = shaped_glyphs_form(&f_a.resolved); + assert_ne!(collapsed, f_a.resolved.text); + let presentation = shaped_glyphs_presentation_form(&f_a.resolved).expect( + "F-A's ligature clusters (ff, fi) are both in LATIN_LIGATURE_PRESENTATION_FORMS", + ); + assert_ne!(presentation, f_a.resolved.text); + assert_ne!( + presentation, collapsed, + "the two shaped-glyphs forms must be genuinely distinct renderings" + ); + assert!( + presentation.contains('\u{FB00}'), + "F-A's presentation form must substitute U+FB00 for the ff ligature: {presentation:?}" + ); + assert!( + presentation.contains('\u{FB01}'), + "F-A's presentation form must substitute U+FB01 for the fi ligature: {presentation:?}" + ); + + let exp = build_expectations_file(&file); + let e = expectation_for(&exp, "F-A"); + let forms = e + .alternative_forms + .get("name-is-shaped-glyphs") + .expect("F-A must carry a name-is-shaped-glyphs entry"); + assert!(forms.contains(&collapsed), "{forms:?}"); + assert!(forms.contains(&presentation), "{forms:?}"); + assert_eq!(forms.len(), 2, "{forms:?}"); + } + + /// F-D's visual-order form must differ from its logical text — the + /// composition trap §8.1 names. + #[test] + fn f_d_visual_order_form_differs_from_its_logical_text() { + let file = require_file(); + let f_d = file.fixtures.iter().find(|f| f.id == "F-D").unwrap(); + let has_rtl_segment = f_d + .resolved + .segments + .iter() + .any(|s| matches!(s.direction, SpikeTextDirection::Rtl)); + assert!(has_rtl_segment, "anchor: F-D must have an Rtl segment"); + let visual = visual_order_form(&f_d.resolved); + assert_ne!(visual, f_d.resolved.text); + + let exp = build_expectations_file(&file); + let e = expectation_for(&exp, "F-D"); + assert_eq!(e.visual_order_name.as_ref(), Some(&visual)); + assert_eq!( + e.visual_order_name_hex.as_deref(), + Some(hex_lower(visual.as_bytes()).as_str()) + ); + } + + /// D1: F-C's source atoms must be exactly its two segments, and the + /// unresolved one must stand alone as a single character — the specific + /// case a verifier-side length-2 substring rule cannot catch, and the + /// whole reason this field exists. + #[test] + fn f_c_source_atoms_are_its_two_segments_one_of_them_single_character() { + let file = require_file(); + let f_c = file.fixtures.iter().find(|f| f.id == "F-C").unwrap(); + assert_eq!( + f_c.resolved.segments.len(), + 2, + "anchor: F-C must have two segments" + ); + let expected: Vec = f_c + .resolved + .segments + .iter() + .map(|s| f_c.resolved.text[s.source.start as usize..s.source.end as usize].to_string()) + .collect(); + let atoms = source_atoms(&f_c.resolved); + assert_eq!(atoms, expected); + + let (_, unresolved_atom) = f_c + .resolved + .segments + .iter() + .zip(atoms.iter()) + .find(|(s, _)| s.face.is_none()) + .expect("anchor: F-C must have an unresolved segment"); + assert_eq!( + unresolved_atom.chars().count(), + 1, + "F-C's unresolved atom must be exactly one character: {unresolved_atom:?}" + ); + + let exp = build_expectations_file(&file); + let e = expectation_for(&exp, "F-C"); + assert_eq!(e.source_atoms, atoms); + } + + /// D1: F-D's source atoms must be exactly its three segments. + #[test] + fn f_d_source_atoms_are_its_three_segments() { + let file = require_file(); + let f_d = file.fixtures.iter().find(|f| f.id == "F-D").unwrap(); + assert_eq!( + f_d.resolved.segments.len(), + 3, + "anchor: F-D must have three segments" + ); + let expected: Vec = f_d + .resolved + .segments + .iter() + .map(|s| f_d.resolved.text[s.source.start as usize..s.source.end as usize].to_string()) + .collect(); + let atoms = source_atoms(&f_d.resolved); + assert_eq!(atoms, expected); + + let exp = build_expectations_file(&file); + let e = expectation_for(&exp, "F-D"); + assert_eq!(e.source_atoms, atoms); + } + + /// Every fixture's source atoms must concatenate back to its own + /// `expected_name` — the general partition property `source_atoms`'s own + /// doc comment claims, checked here on the real generated data rather + /// than only asserted in prose. + #[test] + fn source_atoms_concatenate_to_expected_name_for_every_fixture() { + let file = require_file(); + let exp = build_expectations_file(&file); + for f in &exp.fixtures { + let joined: String = f.source_atoms.concat(); + assert_eq!( + joined, f.expected_name, + "{}: source_atoms must concatenate to expected_name", + f.fixture_id + ); + } + } + + /// An alternative form byte-identical to `expected_name` must be omitted + /// entirely, never present with a value equal to the expectation — the + /// mutation this guards against is a verifier that reports a "match" as + /// a diagnosed FAIL because a no-op entry happened to be present. + #[test] + fn identical_alternative_forms_are_omitted_not_recorded_as_equal() { + let file = require_file(); + let exp = build_expectations_file(&file); + for f in &exp.fixtures { + for (outcome, forms) in &f.alternative_forms { + assert!( + !forms.is_empty(), + "{}: {outcome} must not be present with an empty list", + f.fixture_id + ); + for form in forms { + assert_ne!( + form, &f.expected_name, + "{}: alternative form {outcome} must not be recorded when byte-identical \ + to expected_name", + f.fixture_id + ); + } + } + if let Some(v) = &f.visual_order_name { + assert_ne!(v, &f.expected_name, "{}: visual_order_name", f.fixture_id); + } + } + } + + /// No fixture's `alternative_forms` may contain the same string under two + /// different outcome keys (O1) — re-checked here on the real, generated + /// data, in addition to [`group_alternative_forms_refuses_a_collision`]'s + /// synthetic unit test. + #[test] + fn no_fixture_has_the_same_form_under_two_outcomes() { + let file = require_file(); + let exp = build_expectations_file(&file); + for f in &exp.fixtures { + let mut seen: BTreeMap<&String, &String> = BTreeMap::new(); + for (outcome, forms) in &f.alternative_forms { + for form in forms { + if let Some(existing) = seen.insert(form, outcome) { + panic!( + "{}: {form:?} appears under both {existing:?} and {outcome:?}", + f.fixture_id + ); + } + } + } + } + } + + /// Every alternative-form key must be one of `PROHIBITED_OUTCOMES` — a + /// typo'd or invented key would silently fail to classify anything the + /// verifier actually checks for. + #[test] + fn every_alternative_form_key_is_a_prohibited_outcome() { + let file = require_file(); + let exp = build_expectations_file(&file); + for f in &exp.fixtures { + for outcome in f.alternative_forms.keys() { + assert!( + PROHIBITED_OUTCOMES.contains(&outcome.as_str()), + "{}: {outcome:?} is not in PROHIBITED_OUTCOMES", + f.fixture_id + ); + } + } + } + + /// The at-spi2 role rows restated here must equal + /// `round2_textkit::a11y`'s own at-spi2 row — this is the platform this + /// machine's live AT-SPI2 client actually queries (recipe §8.2, + /// round0-evidence's precedent). + #[test] + fn accepted_and_prohibited_roles_match_the_atspi2_row() { + let file = require_file(); + let exp = build_expectations_file(&file); + let expected_accepted: Vec = round2_textkit::a11y::ACCEPTED_ROLE_TABLE + .iter() + .find(|(p, _)| *p == PLATFORM) + .unwrap() + .1 + .iter() + .map(|s| s.to_string()) + .collect(); + let expected_prohibited: Vec = round2_textkit::a11y::PROHIBITED_ROLE_TABLE + .iter() + .find(|(p, _)| *p == PLATFORM) + .unwrap() + .1 + .iter() + .map(|s| s.to_string()) + .collect(); + for f in &exp.fixtures { + assert_eq!(f.accepted_roles, expected_accepted); + assert_eq!(f.prohibited_roles, expected_prohibited); + } + } + + /// JSON round-trips without loss — the shape a consumer other than this + /// crate (`a11y-verifier/verify.py`) will actually read. + #[test] + fn json_round_trip_preserves_the_expectations() { + let file = require_file(); + let exp = build_expectations_file(&file); + let json = serde_json::to_string_pretty(&exp).unwrap(); + let reloaded: ExpectationsFile = serde_json::from_str(&json).unwrap(); + assert_eq!(reloaded, exp); + } + + /// All five fixtures must be present, in order. + #[test] + fn all_five_fixtures_are_present_in_order() { + let file = require_file(); + let exp = build_expectations_file(&file); + let ids: Vec<&str> = exp.fixtures.iter().map(|f| f.fixture_id.as_str()).collect(); + assert_eq!(ids, vec!["F-A", "F-B", "F-C", "F-D", "F-E"]); + } + + /// B2: `source_fixtures_digest` must equal + /// `round2_textkit::output::expected_artifact_digest()` — the same + /// literal `round2-textkit`'s own `bin/generate` prints and its + /// `FixtureFile::validate` checks the *loaded* file against. This is the + /// generation-time half of B2's staleness guard: if `fixtures.json` ever + /// legitimately changes (a new frozen digest), this test catches that + /// `round2-a11y-oracle` was not regenerated against it, at test time, + /// before `a11y-verifier/verify.py`'s `--expect-source-digest` check + /// would ever catch it live. + #[test] + fn source_fixtures_digest_matches_round2_textkit_expected_digest() { + let file = require_file(); + let exp = build_expectations_file(&file); + assert_eq!( + exp.source_fixtures_digest, + round2_textkit::output::expected_artifact_digest() + ); + } + + // ---- O1: group_alternative_forms, exercised directly (no live fixture + // data required, so the collision-refusal logic itself is under test + // regardless of whether any current fixture happens to trigger it). ---- + + #[test] + fn group_alternative_forms_refuses_a_collision() { + let result = std::panic::catch_unwind(|| { + group_alternative_forms( + "F-TEST", + "expected", + vec![ + ("name-normalized", "same-string".to_string()), + ("name-is-shaped-glyphs", "same-string".to_string()), + ], + ) + }); + let err = result.expect_err("a collision between two outcomes must panic"); + let msg = err + .downcast_ref::() + .cloned() + .or_else(|| err.downcast_ref::<&str>().map(|s| s.to_string())) + .expect("panic payload must be a string"); + assert!(msg.contains("F-TEST"), "{msg}"); + assert!(msg.contains("name-normalized"), "{msg}"); + assert!(msg.contains("name-is-shaped-glyphs"), "{msg}"); + } + + /// Mutation guard: the same outcome producing the same form twice (e.g. + /// two derivations that happen to agree) must NOT panic — only a + /// cross-outcome collision is refused. Without this test, a mutation that + /// made the collision check fire on any duplicate (not just a + /// cross-outcome one) would still pass + /// `group_alternative_forms_refuses_a_collision` above. + #[test] + fn group_alternative_forms_deduplicates_a_same_outcome_repeat_without_panicking() { + let grouped = group_alternative_forms( + "F-TEST", + "expected", + vec![ + ("name-normalized", "same-string".to_string()), + ("name-normalized", "same-string".to_string()), + ], + ); + assert_eq!( + grouped.get("name-normalized"), + Some(&vec!["same-string".to_string()]) + ); + } + + #[test] + fn group_alternative_forms_omits_forms_identical_to_expected_name() { + let grouped = group_alternative_forms( + "F-TEST", + "expected", + vec![ + ("name-normalized", "expected".to_string()), + ("name-is-shaped-glyphs", "different".to_string()), + ], + ); + assert!(!grouped.contains_key("name-normalized")); + assert_eq!( + grouped.get("name-is-shaped-glyphs"), + Some(&vec!["different".to_string()]) + ); + } +} diff --git a/spikes/editor-toolkit/round2-candidatekit/Cargo.toml b/spikes/editor-toolkit/round2-candidatekit/Cargo.toml new file mode 100644 index 0000000..900dd04 --- /dev/null +++ b/spikes/editor-toolkit/round2-candidatekit/Cargo.toml @@ -0,0 +1,24 @@ +[package] +name = "round2-candidatekit" +version = "0.1.0" +edition.workspace = true +publish.workspace = true + +# Packet 2B-0 (ROUND2_TEXT_RECIPE.md, spec/CONTRACT_EDITOR_T4_SPIKE.md pins +# 8, 9, 10, 13, 14): the ONLY code shared between the two Round 2 text +# candidates (C1 = egui+lyon, C2 = vello). See src/lib.rs's crate doc +# comment for the neutrality boundary this crate exists to hold — it loads +# and validates Packet 2A's fixtures/probes/reference apparatus, and defines +# the shared report shape and scoring rule both candidates are measured +# against. It does NOT render, resolve hit tests, or build accessibility +# trees. +# +# tests/dependency_deny_list.rs enforces the boundary by reading THIS file +# at test time, not by convention: it fails if a rendering, windowing, GPU, +# or platform-accessibility crate is ever added to [dependencies] below. + +[dependencies] +serde = { version = "1", features = ["derive"] } +serde_json = "1" +round2-textkit = { path = "../round2-textkit" } +round2-diff = { path = "../round2-diff" } diff --git a/spikes/editor-toolkit/round2-candidatekit/src/inputs.rs b/spikes/editor-toolkit/round2-candidatekit/src/inputs.rs new file mode 100644 index 0000000..b471ac7 --- /dev/null +++ b/spikes/editor-toolkit/round2-candidatekit/src/inputs.rs @@ -0,0 +1,332 @@ +//! Loads and validates the candidate-neutral apparatus Packet 2A built: +//! fixtures, the hit-test probe table, and the per-fixture reference raster +//! + regions. Every failure here names the specific file and what was wrong +//! with it — see [`load_all`]. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; + +use round2_diff::GlyphRegion; +use round2_textkit::hittest::HitTestProbeFile; +use round2_textkit::output::FixtureFile; + +/// Pin 4's offscreen target, restated as a literal (the same discipline +/// every other crate in this workspace uses: a loader checks a file against +/// a stated constant, never trusts the file to agree with itself). +pub const WIDTH: u32 = 1920; +pub const HEIGHT: u32 = 1080; +const EXPECTED_RGBA_LEN: usize = (WIDTH as usize) * (HEIGHT as usize) * 4; + +/// The on-disk shape of one entry in `.regions.json` +/// (`round2-reference/output/`), matching the fields `round2-reference`'s +/// own `RegionRecord` writes. Deserialized here rather than depended on +/// directly, because `round2-reference` pulls in `round2-svgref`, which +/// pulls in `resvg`/`usvg`/`tiny-skia` — exactly the rendering dependencies +/// this crate's neutrality boundary forbids. The region *files* are neutral +/// data; the crate that produced them is not. +/// +/// **This is an implicit cross-crate schema with no shared type** — +/// `round2-reference`'s own `RegionRecord` and this one are two +/// independent hand-written structs that happen to agree on field names. +/// `deny_unknown_fields` is what turns a future drift between them into a +/// *named parse error at this crate's boundary* rather than a silently +/// ignored field: without it, serde drops unknown fields by default, and a +/// field `round2-reference` starts writing (or renames) would pass through +/// here unnoticed. +#[derive(Clone, Debug, serde::Deserialize)] +#[serde(deny_unknown_fields)] +struct RegionRecord { + label: String, + x0: u32, + y0: u32, + x1: u32, + y1: u32, +} + +impl From for GlyphRegion { + fn from(r: RegionRecord) -> Self { + GlyphRegion { + label: r.label, + x0: r.x0, + y0: r.y0, + x1: r.x1, + y1: r.y1, + } + } +} + +/// One fixture's reference apparatus: the rasterized reference image +/// (already length-checked), its D4 regions (already checked non-empty), +/// and the paths they were loaded from (traceability for a `FAIL`). +#[derive(Clone, Debug)] +pub struct ReferenceFixture { + pub fixture_id: String, + pub reference_rgba: Vec, + pub regions: Vec, + pub rgba_path: PathBuf, + pub regions_path: PathBuf, +} + +/// Every candidate-neutral input Packet 2A built, loaded and validated in +/// one call ([`load_all`]). +#[derive(Debug)] +pub struct NeutralInputs { + pub fixtures: FixtureFile, + pub hittest_probes: HitTestProbeFile, + /// Keyed by fixture id (`F-A`..`F-E`). + pub reference: BTreeMap, +} + +/// Loads `fixtures.json`, `hittest_probes.json`, and every fixture's +/// reference raster + regions, from the standard Packet 2A layout under +/// `spike_root` (`round2-textkit/fixtures.json`, +/// `round2-textkit/hittest_probes.json`, +/// `round2-reference/output/.rgba`, +/// `round2-reference/output/.regions.json`). +/// +/// Every failure names the specific file and what was wrong with it: +/// +/// - `fixtures.json` / `hittest_probes.json`: read/parse errors, or a +/// [`round2_textkit::output::FixtureFile::validate`] / +/// [`round2_textkit::hittest::HitTestProbeFile::validate`] failure +/// (digest mismatch, probe-table drift, ...) — propagated verbatim; those +/// loaders already name the path and the specific disagreement. +/// - `.rgba`: refused if its length is not exactly `1920 * 1080 * 4` +/// bytes ([`WIDTH`] x [`HEIGHT`] x 4 RGBA8), naming the file and the +/// actual length. +/// - `.regions.json`: refused if missing, unparsable, or **empty**. +/// This crate refuses an empty region list itself, naming the file, +/// rather than silently handing it to `round2_diff::diff` — which also +/// refuses an empty list (`diff` panics on nothing, it returns an `Err`), +/// but with a message that has no idea which file on disk was empty. +pub fn load_all(spike_root: &Path) -> Result { + let fixtures_path = spike_root.join("round2-textkit/fixtures.json"); + let fixtures = round2_textkit::output::load_fixtures(&fixtures_path)?; + + let hittest_path = spike_root.join("round2-textkit/hittest_probes.json"); + let hittest_probes = round2_textkit::hittest::load_hittest_probes(&hittest_path, &fixtures)?; + + let mut reference = BTreeMap::new(); + for f in &fixtures.fixtures { + let rgba_path = spike_root + .join("round2-reference/output") + .join(format!("{}.rgba", f.id)); + let rgba = std::fs::read(&rgba_path).map_err(|e| { + format!( + "{}: failed to read reference raster: {e}", + rgba_path.display() + ) + })?; + if rgba.len() != EXPECTED_RGBA_LEN { + return Err(format!( + "{}: reference raster is {} bytes, expected exactly {EXPECTED_RGBA_LEN} \ + ({WIDTH}x{HEIGHT} RGBA8) — a short or padded buffer cannot be sampled safely", + rgba_path.display(), + rgba.len() + )); + } + + let regions_path = spike_root + .join("round2-reference/output") + .join(format!("{}.regions.json", f.id)); + let regions_text = std::fs::read_to_string(®ions_path).map_err(|e| { + format!( + "{}: failed to read region file: {e}", + regions_path.display() + ) + })?; + let records: Vec = serde_json::from_str(®ions_text).map_err(|e| { + format!( + "{}: failed to parse region file: {e}", + regions_path.display() + ) + })?; + if records.is_empty() { + return Err(format!( + "{}: region list is empty — refusing here, before this could reach \ + round2_diff::diff (which also refuses an empty region list, but with a message \ + that does not name which file on disk was empty)", + regions_path.display() + )); + } + let regions: Vec = records.into_iter().map(GlyphRegion::from).collect(); + + reference.insert( + f.id.clone(), + ReferenceFixture { + fixture_id: f.id.clone(), + reference_rgba: rgba, + regions, + rgba_path, + regions_path, + }, + ); + } + + Ok(NeutralInputs { + fixtures, + hittest_probes, + reference, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// The real spike workspace root: this crate's manifest directory is + /// `spikes/editor-toolkit/round2-candidatekit`, one level below root. + fn real_spike_root() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("..") + } + + fn read_real(rel: &str) -> Vec { + std::fs::read(real_spike_root().join(rel)) + .unwrap_or_else(|e| panic!("failed to read real {rel}: {e}")) + } + + /// A fresh, uniquely named directory under the OS temp dir (never under + /// the repo working tree, so these tests cannot leave stray files for + /// `git status` to notice), laid out like a spike root's + /// `round2-textkit/` + `round2-reference/output/` — enough for + /// `load_all` to be pointed at it. + fn scratch_dir(name: &str) -> PathBuf { + let dir = std::env::temp_dir().join(format!( + "round2-candidatekit-test-{name}-{}", + std::process::id() + )); + let _ = std::fs::remove_dir_all(&dir); + std::fs::create_dir_all(dir.join("round2-textkit")).unwrap(); + std::fs::create_dir_all(dir.join("round2-reference/output")).unwrap(); + dir + } + + fn write(path: &Path, bytes: &[u8]) { + std::fs::write(path, bytes) + .unwrap_or_else(|e| panic!("failed to write {}: {e}", path.display())); + } + + /// Copies the real, committed, valid `fixtures.json` and + /// `hittest_probes.json` into `dir` — the two files every scenario + /// below needs unmutated so the failure under test is isolated to the + /// one file each test actually breaks. + fn seed_valid_fixtures_and_hittest(dir: &Path) { + write( + &dir.join("round2-textkit/fixtures.json"), + &read_real("round2-textkit/fixtures.json"), + ); + write( + &dir.join("round2-textkit/hittest_probes.json"), + &read_real("round2-textkit/hittest_probes.json"), + ); + } + + #[test] + fn load_all_succeeds_against_the_real_committed_apparatus() { + let inputs = load_all(&real_spike_root()).expect("real apparatus must load"); + assert_eq!(inputs.fixtures.fixtures.len(), 5); + assert_eq!(inputs.reference.len(), 5); + for id in ["F-A", "F-B", "F-C", "F-D", "F-E"] { + assert!(inputs.reference.contains_key(id), "missing {id}"); + let rf = &inputs.reference[id]; + assert_eq!(rf.reference_rgba.len(), EXPECTED_RGBA_LEN); + assert!(!rf.regions.is_empty()); + } + } + + /// Required kill: a `.rgba` of the wrong length is refused, naming the + /// file. + #[test] + fn a_wrong_length_rgba_is_refused_and_the_file_is_named() { + let dir = scratch_dir("wrong-length-rgba"); + seed_valid_fixtures_and_hittest(&dir); + write( + &dir.join("round2-reference/output/F-A.rgba"), + &vec![0u8; 100], + ); + let err = load_all(&dir).unwrap_err(); + assert!(err.contains("F-A.rgba"), "{err}"); + assert!(err.contains("100 bytes"), "{err}"); + assert!(err.contains("8294400"), "{err}"); + let _ = std::fs::remove_dir_all(&dir); + } + + /// Required kill: an empty region list is refused here — with a message + /// naming this file — rather than silently reaching + /// `round2_diff::diff`. + #[test] + fn an_empty_region_list_is_refused_before_it_could_reach_diff() { + let dir = scratch_dir("empty-regions"); + seed_valid_fixtures_and_hittest(&dir); + write( + &dir.join("round2-reference/output/F-A.rgba"), + &vec![0u8; EXPECTED_RGBA_LEN], + ); + write(&dir.join("round2-reference/output/F-A.regions.json"), b"[]"); + let err = load_all(&dir).unwrap_err(); + assert!(err.contains("F-A.regions.json"), "{err}"); + assert!(err.contains("empty"), "{err}"); + let _ = std::fs::remove_dir_all(&dir); + } + + /// A missing region file (never written at all, as opposed to written + /// empty) is refused and named — the other half of "missing region + /// file" in the required API's failure list, distinct from the + /// empty-but-present case above. + #[test] + fn a_missing_region_file_is_refused_and_named() { + let dir = scratch_dir("missing-regions"); + seed_valid_fixtures_and_hittest(&dir); + write( + &dir.join("round2-reference/output/F-A.rgba"), + &vec![0u8; EXPECTED_RGBA_LEN], + ); + // F-A.regions.json is deliberately never written. + let err = load_all(&dir).unwrap_err(); + assert!(err.contains("F-A.regions.json"), "{err}"); + let _ = std::fs::remove_dir_all(&dir); + } + + /// Required kill: a tampered `fixtures.json` digest is refused. Uses + /// the same mutation `round2-textkit`'s own + /// `validate_kills_a_changed_glyph_id` test does (change one glyph id + /// deep inside a fixture, leaving every named/counted field valid) — + /// only the whole-artifact digest catches it, which is exactly why this + /// crate's loader must not skip that check. + #[test] + fn a_tampered_fixtures_digest_is_refused() { + let dir = scratch_dir("tampered-digest"); + let mut tampered = round2_textkit::output::load_fixtures( + &real_spike_root().join("round2-textkit/fixtures.json"), + ) + .expect("real fixtures.json must load"); + let g = &mut tampered.fixtures[0].resolved.segments[0].glyphs[3]; + g.glyph_id = 9999; + let json = serde_json::to_string_pretty(&tampered).unwrap(); + write(&dir.join("round2-textkit/fixtures.json"), json.as_bytes()); + let err = load_all(&dir).unwrap_err(); + assert!(err.contains("digest"), "{err}"); + let _ = std::fs::remove_dir_all(&dir); + } + + /// F5: `.regions.json` is an implicit contract between + /// `round2-reference` (which writes it) and this crate (which reads + /// it), with no shared type. An extra field must be refused **by + /// name**, not silently dropped — that is what turns a future schema + /// drift into a named parse error here instead of quiet data loss. + #[test] + fn an_unknown_field_in_a_region_record_is_refused_by_name() { + let json = serde_json::json!([{ + "label": "x", + "x0": 0, + "y0": 0, + "x1": 1, + "y1": 1, + "smuggled_field": 1 + }]); + let err = serde_json::from_value::>(json) + .unwrap_err() + .to_string(); + assert!(err.contains("smuggled_field"), "{err}"); + } +} diff --git a/spikes/editor-toolkit/round2-candidatekit/src/lib.rs b/spikes/editor-toolkit/round2-candidatekit/src/lib.rs new file mode 100644 index 0000000..9f9ee66 --- /dev/null +++ b/spikes/editor-toolkit/round2-candidatekit/src/lib.rs @@ -0,0 +1,71 @@ +//! # round2-candidatekit — Packet 2B-0: the candidate-neutral apparatus, and +//! **nothing else**. +//! +//! `spec/CONTRACT_EDITOR_T4_SPIKE.md` Round 2 scores criterion 3 (text) via +//! the five checks `spec/ANALYSIS_TEXT_RUN_PRIMITIVES.md` (W3) §5 names. +//! Packet 2A built every piece of candidate-neutral apparatus those checks +//! are measured against (fixtures, the hit-test probe table, the reference +//! rasters and D4 regions, the accessibility oracle). This crate is Packet +//! 2B-0: it is what the two Round 2 candidates — **C1** (egui + lyon) and +//! **C2** (vello) — both depend on, so that neither one re-derives fixture +//! loading, and neither one gets to define the scoring rule for itself. +//! +//! ## The neutrality boundary — this is the point of the crate +//! +//! The user's ruling, verbatim: **"Share only neutral fixture/oracle +//! loading. Rendering, hit testing, and accessibility integration remain +//! candidate-owned."** +//! +//! This crate **MAY** contain: +//! +//! - Loading and validating fixtures, the probe table, the reference +//! rasters and region files, and the a11y expectations +//! ([`inputs::load_all`]). +//! - The shared *report* data shape both candidates emit, and its +//! serialization ([`report::CandidateReport`] and its constituent types). +//! - The scoring rule that turns per-check outcomes into the criterion cell +//! ([`scoring::criterion_cell`], [`scoring::is_eligible`]). +//! +//! This crate **MUST NOT** contain: +//! +//! - Any rendering, rasterization, path/outline conversion, or +//! tessellation. +//! - Any hit-test *resolution* — i.e. nothing that answers "which byte +//! offset does this device point select". Loading the expected answers +//! ([`round2_textkit::hittest::HitTestProbeFile`]) is neutral; computing +//! them is the candidate's job and the thing check 4 measures. This crate +//! only carries the *shape* of a recorded comparison +//! ([`report::HitTestProbeResult`]) — it never resolves one. +//! - Any accessibility node construction or platform-adapter code. This +//! crate only carries the *shape* of observed evidence +//! ([`report::A11yEvidence`]) against the precommitted oracle +//! ([`round2_textkit::a11y`]) — it never builds a tree. +//! +//! `tests/dependency_deny_list.rs` enforces what code review can miss: it +//! reads this crate's own `Cargo.toml` at test time and fails if `egui`, +//! `eframe`, `egui-wgpu`, `lyon`, `lyon_path`, `lyon_tessellation`, `vello`, +//! `wgpu`, `winit`, `accesskit`, `accesskit_winit`, `tiny-skia`, `resvg`, or +//! `usvg` is ever named in `[dependencies]`. +//! +//! ## What this crate does not decide +//! +//! [`scoring::criterion_cell`] implements the contract's outcome rule; it +//! does not implement W3 §5 itself, and it is not the place check 3's +//! `NOT RUN` ruling was *made* — that ruling is `ROUND2_TEXT_RECIPE.md` +//! §1.2, and this crate only encodes and enforces its consequences. + +pub mod inputs; +pub mod outcome; +pub mod report; +pub mod scoring; + +pub use inputs::{load_all, NeutralInputs, ReferenceFixture}; +pub use outcome::CheckOutcome; +pub use report::{ + A11yEvidence, AdapterStatus, BusUnreachableEvidence, CandidateReport, CostRecord, + DependencyDelta, DiffReportRecord, HitTestProbeResult, LocByPart, RegionMassRecord, ReportPart, +}; +pub use scoring::{ + criterion_cell, is_eligible, CellOutcome, CHECK_3_RULING, DISQUALIFYING_CHECKS, + ROUND0_READBACK_EVIDENCE, ROUND_PLATFORM, +}; diff --git a/spikes/editor-toolkit/round2-candidatekit/src/outcome.rs b/spikes/editor-toolkit/round2-candidatekit/src/outcome.rs new file mode 100644 index 0000000..faea637 --- /dev/null +++ b/spikes/editor-toolkit/round2-candidatekit/src/outcome.rs @@ -0,0 +1,270 @@ +//! The per-check outcome type both candidates report against. + +use serde::{Deserialize, Deserializer, Serialize}; + +/// One check's outcome. Exactly three states, and **both non-`Pass` states +/// carry a reason**: pin 14 requires an environmental `NotRun` to record +/// *why* it could not run, and a bare `Fail` with no reason would be +/// exactly the unfalsifiable report `round1-oracle`'s discipline exists to +/// forbid. There is deliberately no unit-only `NotRun` or `Fail` variant — +/// a candidate cannot report "did not pass" without saying why. +/// +/// **An empty or whitespace-only reason is a bare reason wearing a +/// string.** The checked constructors ([`CheckOutcome::fail`], +/// [`CheckOutcome::not_run`]) and this type's `Deserialize` impl both +/// reject one — those are the two paths a candidate actually uses to +/// produce a `CandidateReport` (build it in Rust, or read one back from +/// JSON). The variants' payloads stay `pub` because a fully private field +/// would need a getter/setter pair that adds ceremony without closing any +/// path a candidate is expected to take; the invalid state is +/// unconstructible through construction *and* deserialization, which is +/// what "a reason is required" needs to mean in practice. +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +pub enum CheckOutcome { + Pass, + Fail(String), + NotRun(String), +} + +impl CheckOutcome { + /// Checked constructor: rejects an empty-or-whitespace-only reason. + pub fn fail(reason: impl Into) -> Result { + let reason = reason.into(); + if reason.trim().is_empty() { + return Err( + "CheckOutcome::fail: reason must not be empty or whitespace-only — a Fail with \ + no reason is exactly the unfalsifiable report this type exists to forbid" + .to_string(), + ); + } + Ok(CheckOutcome::Fail(reason)) + } + + /// Checked constructor: rejects an empty-or-whitespace-only reason. + pub fn not_run(reason: impl Into) -> Result { + let reason = reason.into(); + if reason.trim().is_empty() { + return Err( + "CheckOutcome::not_run: reason must not be empty or whitespace-only — pin 14 \ + requires the environmental cause to be recorded, not merely gestured at" + .to_string(), + ); + } + Ok(CheckOutcome::NotRun(reason)) + } + + /// Ordering used by [`crate::scoring::criterion_cell`]'s worst-of-five + /// rule: `Pass` < `NotRun` < `Fail`. Higher is worse. + pub(crate) fn severity_rank(&self) -> u8 { + match self { + CheckOutcome::Pass => 0, + CheckOutcome::NotRun(_) => 1, + CheckOutcome::Fail(_) => 2, + } + } + + pub fn is_pass(&self) -> bool { + matches!(self, CheckOutcome::Pass) + } + + pub fn is_fail(&self) -> bool { + matches!(self, CheckOutcome::Fail(_)) + } + + pub fn is_not_run(&self) -> bool { + matches!(self, CheckOutcome::NotRun(_)) + } +} + +/// The wire shape `CheckOutcome` deserializes through — identical variants +/// and payloads, `#[serde(deny_unknown_fields)]` for the same structural- +/// drift reason every deserializable type in this workspace uses it, kept +/// as a **separate, private** type so [`CheckOutcome`]'s own `Deserialize` +/// impl can run [`CheckOutcome::fail`]/[`CheckOutcome::not_run`]'s +/// empty-reason check on the way through, which `#[derive(Deserialize)]` +/// on `CheckOutcome` directly could not do. +#[derive(Deserialize)] +#[serde(deny_unknown_fields)] +enum CheckOutcomeWire { + Pass, + Fail(String), + NotRun(String), +} + +impl TryFrom for CheckOutcome { + type Error = String; + + fn try_from(wire: CheckOutcomeWire) -> Result { + match wire { + CheckOutcomeWire::Pass => Ok(CheckOutcome::Pass), + CheckOutcomeWire::Fail(reason) => CheckOutcome::fail(reason), + CheckOutcomeWire::NotRun(reason) => CheckOutcome::not_run(reason), + } + } +} + +impl<'de> Deserialize<'de> for CheckOutcome { + fn deserialize(deserializer: D) -> Result + where + D: Deserializer<'de>, + { + let wire = CheckOutcomeWire::deserialize(deserializer)?; + CheckOutcome::try_from(wire).map_err(serde::de::Error::custom) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn severity_orders_pass_below_not_run_below_fail() { + assert!( + CheckOutcome::Pass.severity_rank() < CheckOutcome::NotRun("x".into()).severity_rank() + ); + assert!( + CheckOutcome::NotRun("x".into()).severity_rank() + < CheckOutcome::Fail("x".into()).severity_rank() + ); + } + + #[test] + fn predicates_agree_with_the_variant() { + assert!(CheckOutcome::Pass.is_pass()); + assert!(!CheckOutcome::Pass.is_fail()); + assert!(!CheckOutcome::Pass.is_not_run()); + + assert!(CheckOutcome::Fail("x".into()).is_fail()); + assert!(!CheckOutcome::Fail("x".into()).is_pass()); + + assert!(CheckOutcome::NotRun("x".into()).is_not_run()); + assert!(!CheckOutcome::NotRun("x".into()).is_pass()); + } + + #[test] + fn round_trips_through_json() { + for outcome in [ + CheckOutcome::Pass, + CheckOutcome::Fail("reason".to_string()), + CheckOutcome::NotRun("reason".to_string()), + ] { + let json = serde_json::to_string(&outcome).unwrap(); + let back: CheckOutcome = serde_json::from_str(&json).unwrap(); + assert_eq!(outcome, back); + } + } + + // ---- F6: a real distinguishing assertion, not `len() > 0` ---- + + /// A bare JSON string `"NotRun"` does not match the tuple-variant shape + /// `NotRun(String)` at all (that shape serializes as + /// `{"NotRun": "..."}`), so this is a **structural** deserialize + /// failure — distinct from the empty-reason rejection below, which + /// targets a `NotRun` that *does* carry a payload, just an empty one. + /// Asserts on serde's actual reported type mismatch (a unit-shaped + /// value where a payload-carrying variant was required), which is what + /// actually distinguishes this rejection from every other kind of + /// deserialize failure this file tests — not on "some error happened" + /// (measured: `err.to_string()` is `"invalid type: unit variant, + /// expected newtype variant"`, which names neither `NotRun` nor `Fail` + /// by name, so asserting on the variant name would itself have been + /// wrong). + #[test] + fn a_bare_string_not_run_with_no_payload_fails_to_deserialize() { + let bad = serde_json::json!("NotRun"); + let err = serde_json::from_value::(bad) + .unwrap_err() + .to_string(); + assert!(err.contains("unit variant"), "{err}"); + assert!(err.contains("newtype variant"), "{err}"); + } + + // ---- F3: an empty or whitespace-only reason is refused ---- + + #[test] + fn the_fail_constructor_rejects_an_empty_reason() { + let err = CheckOutcome::fail("").unwrap_err(); + assert!(err.contains("empty"), "{err}"); + } + + #[test] + fn the_fail_constructor_rejects_a_whitespace_only_reason() { + let err = CheckOutcome::fail(" \t ").unwrap_err(); + assert!(err.contains("empty"), "{err}"); + } + + #[test] + fn the_fail_constructor_accepts_a_real_reason() { + let outcome = CheckOutcome::fail("host-substituted the Hebrew segment").unwrap(); + assert_eq!( + outcome, + CheckOutcome::Fail("host-substituted the Hebrew segment".to_string()) + ); + } + + #[test] + fn the_not_run_constructor_rejects_an_empty_reason() { + let err = CheckOutcome::not_run("").unwrap_err(); + assert!(err.contains("empty"), "{err}"); + } + + #[test] + fn the_not_run_constructor_rejects_a_whitespace_only_reason() { + let err = CheckOutcome::not_run("\n").unwrap_err(); + assert!(err.contains("empty"), "{err}"); + } + + #[test] + fn the_not_run_constructor_accepts_a_real_reason() { + let outcome = CheckOutcome::not_run("no Arabic-capable face installed").unwrap(); + assert_eq!( + outcome, + CheckOutcome::NotRun("no Arabic-capable face installed".to_string()) + ); + } + + /// Guards the deserialize path the same way the constructors guard + /// direct construction: a `Fail` with an empty string payload must be + /// refused on the way in from JSON, not merely by a constructor a + /// candidate could route around by deserializing instead. + #[test] + fn deserializing_an_empty_reason_fail_is_refused() { + let bad = serde_json::json!({"Fail": ""}); + let err = serde_json::from_value::(bad) + .unwrap_err() + .to_string(); + assert!(err.contains("empty"), "{err}"); + } + + #[test] + fn deserializing_a_whitespace_only_reason_not_run_is_refused() { + let bad = serde_json::json!({"NotRun": " "}); + let err = serde_json::from_value::(bad) + .unwrap_err() + .to_string(); + assert!(err.contains("empty"), "{err}"); + } + + #[test] + fn deserializing_a_real_reason_still_works() { + let good = serde_json::json!({"Fail": "a real reason"}); + let outcome: CheckOutcome = serde_json::from_value(good).unwrap(); + assert_eq!(outcome, CheckOutcome::Fail("a real reason".to_string())); + } + + /// An unknown variant name must still be refused — `CheckOutcomeWire`'s + /// own shape carries forward through the custom `Deserialize` impl + /// rather than being silently lost when `CheckOutcome` stopped deriving + /// it directly. Measured: `err.to_string()` is `"unknown variant + /// \`Passed\`, expected one of \`Pass\`, \`Fail\`, \`NotRun\`"`, so the + /// specific bad name is named in the message. + #[test] + fn an_unknown_variant_name_is_refused() { + let bad = serde_json::json!({"Passed": null}); + let err = serde_json::from_value::(bad) + .unwrap_err() + .to_string(); + assert!(err.contains("unknown variant"), "{err}"); + assert!(err.contains("Passed"), "{err}"); + } +} diff --git a/spikes/editor-toolkit/round2-candidatekit/src/report.rs b/spikes/editor-toolkit/round2-candidatekit/src/report.rs new file mode 100644 index 0000000..3ed0b41 --- /dev/null +++ b/spikes/editor-toolkit/round2-candidatekit/src/report.rs @@ -0,0 +1,486 @@ +//! The shared report shape both candidates emit ([`CandidateReport`]), plus +//! serializable mirrors of `round2-diff`'s pass/fail types. +//! +//! `round2-diff` is a reviewed, frozen packet — its own `Cargo.toml` doc +//! comment states it is "deliberately zero dependencies", and this crate +//! does not modify it to add a `serde` derive it does not otherwise need. +//! [`DiffReportRecord`] and [`RegionMassRecord`] are lossless mirrors, with +//! an infallible `From` conversion, of `round2_diff::DiffReport` and +//! `round2_diff::RegionMass` — the same pattern `round2-reference`'s +//! `RegionRecord` uses for `round2_diff::GlyphRegion`. + +use std::collections::BTreeMap; + +use serde::{Deserialize, Serialize}; + +use round2_diff::{DiffReport, RegionMass}; +use round2_textkit::hittest::DevicePoint; +use round2_textkit::types::SpikeCaretAffinity; + +use crate::outcome::CheckOutcome; + +/// Serializable mirror of `round2_diff::RegionMass`. +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct RegionMassRecord { + pub label: String, + pub reference_mass: f64, + pub candidate_mass: f64, + pub relative_delta: f64, + pub pass: bool, +} + +impl From<&RegionMass> for RegionMassRecord { + fn from(r: &RegionMass) -> Self { + RegionMassRecord { + label: r.label.clone(), + reference_mass: r.reference_mass, + candidate_mass: r.candidate_mass, + relative_delta: r.relative_delta, + pass: r.pass, + } + } +} + +/// Serializable mirror of `round2_diff::DiffReport` — see this module's doc +/// comment for why this crate mirrors rather than modifies `round2-diff`. +/// `pass` is [`DiffReport::pass`]'s own computed verdict, stored rather than +/// re-derived, so a report read back from JSON does not need the four +/// D-rule fields recomputed by hand to know its own outcome. +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct DiffReportRecord { + pub width: u32, + pub height: u32, + pub band_pixel_count: u64, + pub d1_pixels_outside_band_differing: u64, + pub d1_pass: bool, + pub reference_ink_mass: f64, + pub candidate_ink_mass: f64, + pub d2_relative_delta: f64, + pub d2_pass: bool, + pub reference_centroid: Option<(f64, f64)>, + pub candidate_centroid: Option<(f64, f64)>, + pub d3_delta: Option<(f64, f64)>, + pub d3_pass: Option, + pub in_band_max_abs_delta_luma: u8, + pub in_band_count_delta_gt_report_threshold: u64, + pub d4_regions: Vec, + pub d4_pass: bool, + pub d4_worst: Option, + /// [`DiffReport::pass`]'s overall verdict: D1, D2, D4 must all hold, + /// and D3 must either hold or be inapplicable. + pub pass: bool, +} + +impl From<&DiffReport> for DiffReportRecord { + fn from(r: &DiffReport) -> Self { + DiffReportRecord { + width: r.width, + height: r.height, + band_pixel_count: r.band_pixel_count, + d1_pixels_outside_band_differing: r.d1_pixels_outside_band_differing, + d1_pass: r.d1_pass, + reference_ink_mass: r.reference_ink_mass, + candidate_ink_mass: r.candidate_ink_mass, + d2_relative_delta: r.d2_relative_delta, + d2_pass: r.d2_pass, + reference_centroid: r.reference_centroid, + candidate_centroid: r.candidate_centroid, + d3_delta: r.d3_delta, + d3_pass: r.d3_pass, + in_band_max_abs_delta_luma: r.in_band_max_abs_delta_luma, + in_band_count_delta_gt_report_threshold: r.in_band_count_delta_gt_report_threshold, + d4_regions: r.d4_regions.iter().map(RegionMassRecord::from).collect(), + d4_pass: r.d4_pass, + d4_worst: r.d4_worst.as_ref().map(RegionMassRecord::from), + pass: r.pass(), + } + } +} + +/// One hit-test probe's recorded comparison. The device point and expected +/// answer come straight from `round2-textkit`'s committed +/// `hittest_probes.json` (`round2_textkit::hittest::HitTestProbe`); +/// resolving *which* byte offset and affinity a candidate's renderer +/// actually returns for that point is the candidate's own job — check 4's +/// entire subject — so this type only carries the recorded outcome of that +/// resolution, never performs it. +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct HitTestProbeResult { + pub fixture_id: String, + pub point: DevicePoint, + pub expected_source_offset: u32, + pub expected_affinity: SpikeCaretAffinity, + pub actual_source_offset: u32, + pub actual_affinity: SpikeCaretAffinity, + pub pass: bool, +} + +/// One fixture's observed accessibility evidence — what the candidate's own +/// tree (or its absence) actually looked like, compared against +/// `round2-textkit`'s precommitted +/// `round2_textkit::a11y::SpikeAccessibilityExpectation`. Building the tree +/// is the candidate's job (recipe §8.4: "nothing here says *how* a +/// candidate builds the tree, on which thread, or through which crate"); +/// this type only carries what was observed. +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct A11yEvidence { + pub fixture_id: String, + /// The platform row (recipe §8.2 table key, e.g. `"accesskit-0.24"`) + /// this evidence was collected against — a candidate satisfies check 5 + /// by matching one row, the platform it actually exposes a tree on. + pub platform: String, + /// `None` when the run is absent from the tree entirely (the + /// `absent-from-tree` prohibited outcome) — a distinct state from an + /// empty-but-present name (`name-empty`), which is `Some("")`. + pub observed_name: Option, + pub observed_name_bytes_hex: Option, + pub observed_role: Option, + /// One of `round2_textkit::a11y::PROHIBITED_OUTCOMES`, or `None` if no + /// prohibited outcome applies. + pub prohibited_outcome: Option, + pub pass: bool, + pub notes: String, +} + +/// Positive evidence that the platform accessibility bus itself was +/// unreachable — the *only* thing that can make +/// [`CandidateReport::check5_accessibility`] `NotRun` admissible on the +/// round's own platform (AT-SPI2, on this machine); see +/// `crate::scoring::ROUND0_READBACK_EVIDENCE` for why "we did not build a +/// bridge" is not, by itself, an environmental cause here. +/// +/// A **typed** field rather than folding this into `CheckOutcome::NotRun`'s +/// free-text reason on purpose: a free-text reason is something a candidate +/// can write anything into ("bus unreachable" typed by hand proves +/// nothing), while this type asks for the specific thing that would make +/// the claim checkable — what was attempted, and what was actually +/// observed. +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct BusUnreachableEvidence { + /// How the candidate attempted to reach the platform accessibility bus + /// before concluding it was unreachable (e.g. "connected to the AT-SPI2 + /// session bus via `atspi::Bus::connect`"). + pub probe_description: String, + /// What was actually observed — the failure itself, not a restatement + /// of "unreachable" (e.g. the connection error message). + pub probe_output: String, +} + +/// One dependency added to the candidate's own crate(s) over the Round 1 +/// baseline. `reason` is a one-line justification a reader can check +/// against what the candidate actually needed to build. +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct DependencyDelta { + pub name: String, + pub version: String, + pub reason: String, +} + +/// One platform accessibility adapter's status. +/// +/// `NotBuilt` is a distinct variant from a failing status **on purpose** — +/// the user's ruling is that an adapter the candidate chose not to build is +/// **scope, not a hidden failure**. A string convention (e.g. a `notes` +/// field reading `"not built"`) could be typo'd, omitted, or silently +/// absorbed into a `PASS`; making it a variant the compiler enforces means +/// a report can never accidentally claim a platform is covered by leaving +/// its status ambiguous. +/// +/// **This variant covers *other* platforms only** (Windows UIA, macOS AX, +/// ...) — it must never be used to excuse an unbuilt bridge on the round's +/// own platform (AT-SPI2, on this machine); see +/// [`crate::scoring::ROUND0_READBACK_EVIDENCE`] and +/// [`CandidateReport::check5_bus_unreachable_evidence`] for the field that +/// actually governs whether check 5 is allowed to be `NotRun`. +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub enum AdapterStatus { + /// The candidate built and exercised an adapter for this platform. + Implemented { platform: String, notes: String }, + /// The candidate did not build an adapter for this platform. + /// **Scope not covered — not a failure.** + NotBuilt { platform: String, reason: String }, +} + +/// A shared part of the candidate's own integration work, common to both C1 +/// and C2 so their per-part LOC tables can be read **side by side** — the +/// one thing the user's ruling on cost tables asks of this record. +/// +/// Replaces an earlier free-text `part: String` design: free text let each +/// candidate invent its own vocabulary, which produced two tables that +/// could not be compared directly. `Other(String)` is the escape hatch for +/// a genuinely candidate-specific seam that none of the five shared rows +/// describes (e.g. egui's immediate-mode re-layout-per-frame glue, or +/// vello's scene-graph diffing) — the divergence between the two +/// candidates is still expressible, just visibly, instead of silently +/// fragmenting every row. +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub enum ReportPart { + TextRendering, + HitTestResolution, + AccessibilityTreeConstruction, + AccessibilityIntegrationWiring, + FixtureAndReportPlumbing, + /// A seam that is genuinely candidate-specific — not one of the five + /// shared rows above. + Other(String), +} + +/// LOC for one part of the candidate's own integration work. +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct LocByPart { + pub part: ReportPart, + pub lines: u64, +} + +/// Observed cost facts, reported at the same granularity by both +/// candidates — never a subjective score, only what was actually added, +/// built, or left as scope. +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct CostRecord { + /// The Round 1 baseline commit this delta is measured against. + pub baseline_commit: String, + pub dependencies_added: Vec, + /// One entry per platform row in `round2_textkit::a11y::ACCEPTED_ROLE_TABLE` + /// — every platform gets a status, `Implemented` or `NotBuilt`, never an + /// absent entry (an absent entry is indistinguishable from "forgot to + /// report", which is exactly what `NotBuilt` exists to make explicit). + pub adapters: Vec, + /// Free-text bullets describing integration/wiring the candidate wrote + /// itself — glue code, not vendored or generated. + pub integration_wiring: Vec, + pub loc_by_part: Vec, +} + +/// The shape both Round 2 text candidates emit. +/// +/// `check1`..`check5` are the five checks [`crate::scoring::criterion_cell`] +/// reduces to the criterion cell. `supplementary_f_d_bidi` is deliberately +/// **not** one of them — see that function's doc comment for why it is +/// structurally incapable of reaching the cell. +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct CandidateReport { + pub candidate_id: String, + + pub check1_faithful_consumption: CheckOutcome, + pub check2_fallback: CheckOutcome, + /// Must be `CheckOutcome::NotRun(_)` by the standing ruling + /// (`ROUND2_TEXT_RECIPE.md` §1.2) — enforced in + /// `crate::scoring::criterion_cell` (by panic, not silent acceptance), + /// not at construction time here, so a report can still be built and + /// inspected before that function ever runs. + pub check3_bidi: CheckOutcome, + pub check4_hit_testing: CheckOutcome, + pub check5_accessibility: CheckOutcome, + /// Present only when `check5_accessibility` is `NotRun` **and** that + /// `NotRun` is claimed to be caused by the platform accessibility bus + /// itself being unreachable — the only cause + /// `crate::scoring::criterion_cell`/`crate::scoring::is_eligible` + /// accept for a check-5 `NotRun` on this round's own platform. `None` + /// whenever `check5_accessibility` is `Pass` or `Fail`. + pub check5_bus_unreachable_evidence: Option, + + /// F-D's supplementary Hebrew/Latin bidi evidence (recipe §1.2) — a + /// separate field, never merged into the five above and never read by + /// `crate::scoring::criterion_cell`. + pub supplementary_f_d_bidi: CheckOutcome, + + /// Keyed by fixture id (`F-A`..`F-E`). + pub per_fixture_diffs: BTreeMap, + pub hittest_probe_results: Vec, + pub a11y_evidence: Vec, + pub cost: CostRecord, +} + +#[cfg(test)] +mod tests { + use super::*; + use round2_diff::GlyphRegion; + + fn solid(width: u32, height: u32, rgb: [u8; 3]) -> Vec { + let mut buf = vec![0u8; (width as usize) * (height as usize) * 4]; + for px in buf.chunks_mut(4) { + px[0] = rgb[0]; + px[1] = rgb[1]; + px[2] = rgb[2]; + px[3] = 255; + } + buf + } + + #[test] + fn diff_report_record_mirrors_every_field_and_the_computed_verdict() { + let reference = solid(8, 8, [255, 255, 255]); + let candidate = reference.clone(); + let region = GlyphRegion { + label: "x".to_string(), + x0: 2, + y0: 2, + x1: 6, + y1: 6, + }; + let report = round2_diff::diff(&reference, &candidate, 8, 8, &[region]).unwrap(); + let record = DiffReportRecord::from(&report); + assert_eq!(record.width, report.width); + assert_eq!(record.height, report.height); + assert_eq!(record.d1_pass, report.d1_pass); + assert_eq!(record.d2_pass, report.d2_pass); + assert_eq!(record.d3_pass, report.d3_pass); + assert_eq!(record.d4_pass, report.d4_pass); + assert_eq!(record.pass, report.pass()); + assert_eq!(record.d4_regions.len(), report.d4_regions.len()); + } + + fn empty_cost() -> CostRecord { + CostRecord { + baseline_commit: "abc1234".to_string(), + dependencies_added: Vec::new(), + adapters: vec![AdapterStatus::NotBuilt { + platform: "windows-uia".to_string(), + reason: "no Windows CI runner for this spike".to_string(), + }], + integration_wiring: Vec::new(), + loc_by_part: Vec::new(), + } + } + + fn base_candidate_report() -> CandidateReport { + CandidateReport { + candidate_id: "C-TEST".to_string(), + check1_faithful_consumption: CheckOutcome::Pass, + check2_fallback: CheckOutcome::Pass, + check3_bidi: CheckOutcome::NotRun("x".to_string()), + check4_hit_testing: CheckOutcome::Pass, + check5_accessibility: CheckOutcome::Pass, + check5_bus_unreachable_evidence: None, + supplementary_f_d_bidi: CheckOutcome::Pass, + per_fixture_diffs: BTreeMap::new(), + hittest_probe_results: Vec::new(), + a11y_evidence: Vec::new(), + cost: empty_cost(), + } + } + + #[test] + fn candidate_report_round_trips_through_json() { + let report = base_candidate_report(); + let json = serde_json::to_string_pretty(&report).unwrap(); + let reloaded: CandidateReport = serde_json::from_str(&json).unwrap(); + assert_eq!(reloaded.candidate_id, "C-TEST"); + assert!(matches!( + reloaded.cost.adapters[0], + AdapterStatus::NotBuilt { .. } + )); + assert!(reloaded.check5_bus_unreachable_evidence.is_none()); + } + + /// `check5_bus_unreachable_evidence` must round-trip when present, not + /// just when `None` — the field the review named is exactly the one a + /// lossy round trip would silently drop. + #[test] + fn bus_unreachable_evidence_round_trips_through_json() { + let mut report = base_candidate_report(); + report.check5_accessibility = CheckOutcome::not_run("bus unreachable").unwrap(); + report.check5_bus_unreachable_evidence = Some(BusUnreachableEvidence { + probe_description: "connected to the AT-SPI2 session bus".to_string(), + probe_output: "org.freedesktop.DBus.Error.ServiceUnknown".to_string(), + }); + let json = serde_json::to_string_pretty(&report).unwrap(); + let reloaded: CandidateReport = serde_json::from_str(&json).unwrap(); + let evidence = reloaded + .check5_bus_unreachable_evidence + .expect("evidence must survive the round trip"); + assert_eq!( + evidence.probe_output, + "org.freedesktop.DBus.Error.ServiceUnknown" + ); + } + + /// `NotBuilt` must not be interchangeable with `Implemented` — the + /// compiler-enforced distinction the doc comment claims. + #[test] + fn not_built_adapter_is_a_distinct_variant_from_implemented() { + let a = AdapterStatus::NotBuilt { + platform: "macos-nsaccessibility".to_string(), + reason: "no macOS runner".to_string(), + }; + assert!(matches!(a, AdapterStatus::NotBuilt { .. })); + assert!(!matches!(a, AdapterStatus::Implemented { .. })); + } + + /// An unknown field on the wire must be refused, not ignored — the same + /// discipline every deserializable type in this workspace uses. + #[test] + fn an_unknown_field_on_cost_record_is_refused() { + let mut v = serde_json::to_value(empty_cost()).unwrap(); + v.as_object_mut() + .unwrap() + .insert("smuggled_field".into(), serde_json::json!(1)); + let err = serde_json::from_value::(v) + .unwrap_err() + .to_string(); + assert!(err.contains("smuggled_field"), "{err}"); + } + + // ---- F4: the five shared ReportPart rows compare directly ---- + + /// The whole point of replacing free-text `part: String` with a fixed + /// enum: two candidates' `LocByPart` rows for the same shared part are + /// now directly comparable (`==`), which a free-text label (e.g. "text + /// rendering" vs. "rendering text") could never guarantee. + #[test] + fn the_same_shared_part_from_two_candidates_compares_equal() { + let c1_row = LocByPart { + part: ReportPart::HitTestResolution, + lines: 340, + }; + let c2_row = LocByPart { + part: ReportPart::HitTestResolution, + lines: 210, + }; + assert_eq!(c1_row.part, c2_row.part); + assert_ne!( + c1_row.lines, c2_row.lines, + "the LOC counts may legitimately differ" + ); + } + + /// `Other` stays the escape hatch: two candidate-specific seams with + /// different labels remain distinguishable, unlike the five fixed rows. + #[test] + fn other_parts_with_different_labels_are_not_conflated() { + let egui_seam = ReportPart::Other("immediate-mode re-layout per frame".to_string()); + let vello_seam = ReportPart::Other("scene-graph diffing".to_string()); + assert_ne!(egui_seam, vello_seam); + } + + /// All five shared rows round-trip, and `Other` carries its label + /// through — a lossy `Serialize`/`Deserialize` impl on the enum would + /// silently collapse rows that must stay comparable. + #[test] + fn every_report_part_round_trips_through_json() { + let parts = [ + ReportPart::TextRendering, + ReportPart::HitTestResolution, + ReportPart::AccessibilityTreeConstruction, + ReportPart::AccessibilityIntegrationWiring, + ReportPart::FixtureAndReportPlumbing, + ReportPart::Other("candidate-specific seam".to_string()), + ]; + for part in parts { + let json = serde_json::to_string(&part).unwrap(); + let back: ReportPart = serde_json::from_str(&json).unwrap(); + assert_eq!(part, back); + } + } +} diff --git a/spikes/editor-toolkit/round2-candidatekit/src/scoring.rs b/spikes/editor-toolkit/round2-candidatekit/src/scoring.rs new file mode 100644 index 0000000..0769b4f --- /dev/null +++ b/spikes/editor-toolkit/round2-candidatekit/src/scoring.rs @@ -0,0 +1,368 @@ +//! Turns a [`CandidateReport`]'s five check outcomes into the Round 2 +//! criterion cell, and reports eligibility separately +//! (`ROUND2_TEXT_RECIPE.md` §1.2). + +use crate::outcome::CheckOutcome; +use crate::report::CandidateReport; + +/// The round's own platform — `round2_textkit::a11y::ACCEPTED_ROLE_TABLE`'s +/// `"at-spi2"` row — restated here so [`ROUND0_READBACK_EVIDENCE`]'s doc +/// comment and [`require_check5_not_run_is_admissible`]'s panic can name it +/// precisely. +pub const ROUND_PLATFORM: &str = "at-spi2"; + +/// Round 0's own readback evidence, quoted verbatim in the panic +/// [`criterion_cell`]/[`is_eligible`] raise for a check-5 `NotRun` that +/// carries no [`crate::report::BusUnreachableEvidence`]. +/// +/// `round0-evidence/c1-egui-readback.txt` and +/// `round0-evidence/c2-vello-readback.txt` both record `READBACK: PASS` — a +/// live, out-of-process AT-SPI2 tree walk that succeeded for **both** +/// candidates on this machine. So on [`ROUND_PLATFORM`], "we did not build +/// an accessibility bridge" is not an environmental cause: the bus is +/// reachable, and check 5 is the round's own platform, not declared +/// out-of-scope adapter coverage. `AdapterStatus::NotBuilt` still covers +/// *other* platforms (Windows UIA, macOS AX, ...) as declared scope; it +/// must not be used to excuse the platform the round actually runs on. +pub const ROUND0_READBACK_EVIDENCE: &str = "round0-evidence/c1-egui-readback.txt and \ + round0-evidence/c2-vello-readback.txt both record READBACK: PASS — a live, out-of-process \ + AT-SPI2 tree walk succeeded for both candidates on this machine, so the platform \ + accessibility bus is reachable here and an unbuilt accessibility bridge is not \ + environmental NOT RUN on this platform. AdapterStatus::NotBuilt covers OTHER platforms \ + (Windows UIA, macOS AX, ...) as declared scope; it does not, by itself, excuse the \ + platform the round actually runs on."; + +/// Panics if `report.check5_accessibility` is `NotRun` without +/// `report.check5_bus_unreachable_evidence` present — see +/// [`ROUND0_READBACK_EVIDENCE`]. A no-op for `Pass`/`Fail`, and a no-op for +/// a `NotRun` that *does* carry evidence. Deliberately does **not** inspect +/// `report.cost.adapters` — an `AdapterStatus::NotBuilt` entry for +/// [`ROUND_PLATFORM`] must not, by itself, satisfy this check (that is the +/// exact loophole the review named). +fn require_check5_not_run_is_admissible(report: &CandidateReport) { + if report.check5_accessibility.is_not_run() && report.check5_bus_unreachable_evidence.is_none() + { + panic!( + "candidate {:?} reported check5_accessibility = NotRun(_) with no \ + check5_bus_unreachable_evidence — {ROUND0_READBACK_EVIDENCE}", + report.candidate_id + ); + } +} + +/// The Round 2 criterion cell for check 3 (bidi / text-run primitives) is +/// structurally identical to [`CheckOutcome`] — a cell is the worst of the +/// five checks, which is itself just a `CheckOutcome` — kept as a distinct +/// name so a reader is never unsure whether a value in hand is *one check's +/// own* outcome or *the criterion cell* five checks reduce to. +pub type CellOutcome = CheckOutcome; + +/// The standing ruling [`criterion_cell`] enforces (`ROUND2_TEXT_RECIPE.md` +/// §1.2, 2026-07-29): no Arabic-capable face is installed on the round's +/// declared machine, and pin 9 makes an absent required face environmental +/// `NOT RUN`. Quoted verbatim in the panic [`criterion_cell`] raises for a +/// report that disagrees with it. +pub const CHECK_3_RULING: &str = "ROUND2_TEXT_RECIPE.md §1.2 (2026-07-29 ruling): check 3 is \ + NOT RUN for every candidate, on both adapters — no Arabic-capable face is installed, and \ + pin 9 makes an absent required face environmental NOT RUN. F-D's supplementary Hebrew/Latin \ + bidi evidence is recorded separately and must never upgrade check 3 to PASS."; + +/// Reduces a [`CandidateReport`]'s five check outcomes to the Round 2 +/// criterion cell: the **worst of the five**, ordered `Pass` < `NotRun` < +/// `Fail` ([`CheckOutcome::severity_rank`]). +/// +/// The supplementary F-D bidi result +/// ([`CandidateReport::supplementary_f_d_bidi`]) is a separate field on +/// `CandidateReport` and this function never reads it — that is what makes +/// it **structurally** incapable of reaching the cell (recipe §1.2: "it +/// must not upgrade check 3 to PASS"), rather than merely conventionally +/// excluded by a check this function could someday grow to include by +/// accident. +/// +/// # Panics +/// +/// Panics if `report.check3_bidi` is anything other than `NotRun` — a +/// candidate reporting `Pass` or `Fail` for check 3 has violated the +/// standing ruling ([`CHECK_3_RULING`]), which this function treats as a +/// programming error in how the candidate assembled its report, not a value +/// a scoring rule is allowed to interpret. (Not every environmental +/// deviation deserves a panic; this one does, because pin 9's face-absence +/// fact does not vary between the two candidates or between runs — a +/// non-`NotRun` value here can only mean the report was built wrong.) +pub fn criterion_cell(report: &CandidateReport) -> CellOutcome { + if !report.check3_bidi.is_not_run() { + panic!( + "candidate {:?} reported check 3 as {:?}, not NotRun(_) — {CHECK_3_RULING}", + report.candidate_id, report.check3_bidi + ); + } + require_check5_not_run_is_admissible(report); + + let checks = [ + &report.check1_faithful_consumption, + &report.check2_fallback, + &report.check3_bidi, + &report.check4_hit_testing, + &report.check5_accessibility, + ]; + checks + .into_iter() + .max_by_key(|c| c.severity_rank()) + .cloned() + .expect("`checks` is a fixed non-empty array of five elements") +} + +/// The two disqualifying checks (recipe §1.2: "checks 2 and 5 are the +/// disqualifying set"). Named so the disqualifying set is a fact a reader +/// (and a grep) can find, not a claim buried in a comment beside +/// [`is_eligible`]. +pub const DISQUALIFYING_CHECKS: &str = "check2_fallback, check5_accessibility"; + +/// Whether `report` remains a candidate at all — reported **separately** +/// from [`criterion_cell`], because the two questions are different: the +/// cell is what the criterion 3 table shows, eligibility is whether the +/// candidate survives at all. +/// +/// Failing check 2 or check 5 disqualifies. Check 3's `NotRun` (the only +/// state it is ever allowed to carry — see [`criterion_cell`]) does **not** +/// disqualify, because check 3 is not in the disqualifying set +/// ([`DISQUALIFYING_CHECKS`]). +/// +/// # Panics +/// +/// Panics under the same condition [`criterion_cell`] does for check 5 — +/// see [`require_check5_not_run_is_admissible`] / [`ROUND0_READBACK_EVIDENCE`]. +/// Without this, a candidate that never wired an accessibility bridge could +/// report `check5_accessibility = NotRun("we did not build it")`, and +/// `is_eligible` would return `true` because `NotRun` is not `Fail` — the +/// exact loophole the ruling this function enforces exists to close. This +/// function does not merely return `false` for that case, because the +/// report itself is malformed (an inadmissible claim), not merely +/// disqualifying: a malformed report should not be silently readable as "at +/// least eligible." +pub fn is_eligible(report: &CandidateReport) -> bool { + require_check5_not_run_is_admissible(report); + !report.check2_fallback.is_fail() && !report.check5_accessibility.is_fail() +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::report::{AdapterStatus, BusUnreachableEvidence, CostRecord}; + use std::collections::BTreeMap; + + fn base_report() -> CandidateReport { + CandidateReport { + candidate_id: "C-TEST".to_string(), + check1_faithful_consumption: CheckOutcome::Pass, + check2_fallback: CheckOutcome::Pass, + check3_bidi: CheckOutcome::NotRun(CHECK_3_RULING.to_string()), + check4_hit_testing: CheckOutcome::Pass, + check5_accessibility: CheckOutcome::Pass, + check5_bus_unreachable_evidence: None, + supplementary_f_d_bidi: CheckOutcome::Pass, + per_fixture_diffs: BTreeMap::new(), + hittest_probe_results: Vec::new(), + a11y_evidence: Vec::new(), + cost: CostRecord { + baseline_commit: "0000000".to_string(), + dependencies_added: Vec::new(), + adapters: Vec::new(), + integration_wiring: Vec::new(), + loc_by_part: Vec::new(), + }, + } + } + + fn some_evidence() -> BusUnreachableEvidence { + BusUnreachableEvidence { + probe_description: "connected to the AT-SPI2 session bus".to_string(), + probe_output: "org.freedesktop.DBus.Error.ServiceUnknown".to_string(), + } + } + + // ---- criterion_cell: the worst-of-five rule ---- + + #[test] + fn all_pass_except_the_pinned_check_3_yields_a_not_run_cell() { + let cell = criterion_cell(&base_report()); + assert!(matches!(cell, CheckOutcome::NotRun(_)), "{cell:?}"); + } + + /// Required kill: a `CandidateReport` claiming check 3 `Pass` is + /// rejected, naming the §1.2 ruling. + #[test] + #[should_panic(expected = "§1.2")] + fn a_check_3_pass_is_rejected_naming_the_ruling() { + let mut report = base_report(); + report.check3_bidi = CheckOutcome::Pass; + let _ = criterion_cell(&report); + } + + /// Same requirement, the other disallowed value: a `Fail` for check 3 + /// is rejected exactly as a `Pass` is — the ruling pins check 3 to + /// `NotRun` specifically, not merely "not Pass". + #[test] + #[should_panic(expected = "§1.2")] + fn a_check_3_fail_is_also_rejected_naming_the_ruling() { + let mut report = base_report(); + report.check3_bidi = CheckOutcome::Fail("pretend Arabic shaping worked".to_string()); + let _ = criterion_cell(&report); + } + + /// Required kill: a supplementary F-D `Pass` does not move the cell off + /// `NotRun`. + #[test] + fn a_supplementary_f_d_pass_does_not_move_the_cell_off_not_run() { + let mut report = base_report(); + report.supplementary_f_d_bidi = CheckOutcome::Pass; + assert!(matches!(criterion_cell(&report), CheckOutcome::NotRun(_))); + } + + /// Required kill: a supplementary F-D `Fail` does not move the cell + /// either — in particular it must not turn `NotRun` into `Fail`, which + /// is the direction a naive "worst of six" implementation would break. + #[test] + fn a_supplementary_f_d_fail_does_not_move_the_cell_either() { + let mut report = base_report(); + report.supplementary_f_d_bidi = + CheckOutcome::Fail("Hebrew segment drawn in the wrong face".to_string()); + let cell = criterion_cell(&report); + assert!( + matches!(cell, CheckOutcome::NotRun(_)), + "a FAIL on the supplementary row must not reach the cell at all: got {cell:?}" + ); + } + + /// A genuine check-2 FAIL must still win the worst-of-five over the + /// pinned check-3 NotRun — confirms the ordering is real, not just + /// "always NotRun". + #[test] + fn a_check_2_failure_outranks_the_pinned_not_run_in_the_cell() { + let mut report = base_report(); + report.check2_fallback = + CheckOutcome::Fail("host-substituted the Hebrew segment".to_string()); + let cell = criterion_cell(&report); + assert!(matches!(cell, CheckOutcome::Fail(_)), "{cell:?}"); + } + + // ---- is_eligible: the disqualifying set is {check2, check5} only ---- + + /// Required kill: a candidate failing check 2 is ineligible. + #[test] + fn failing_check_2_makes_a_candidate_ineligible() { + let mut report = base_report(); + report.check2_fallback = CheckOutcome::Fail("...".to_string()); + assert!(!is_eligible(&report)); + } + + /// Required kill: a candidate failing check 5 is ineligible. + #[test] + fn failing_check_5_makes_a_candidate_ineligible() { + let mut report = base_report(); + report.check5_accessibility = CheckOutcome::Fail("...".to_string()); + assert!(!is_eligible(&report)); + } + + /// Required kill: a candidate whose only non-Pass is check 3 `NotRun` + /// is eligible. + #[test] + fn a_candidate_whose_only_non_pass_is_check_3_not_run_is_eligible() { + let report = base_report(); // check3 is NotRun; everything else Pass. + assert!( + is_eligible(&report), + "check 3 is not in the disqualifying set" + ); + } + + /// Checks 1 and 4 are not disqualifying either — only 2 and 5 are. This + /// distinguishes "affects the cell" from "affects eligibility": a + /// check-1 FAIL sinks the cell to FAIL but must not, by itself, remove + /// the candidate from the round. + #[test] + fn failing_check_1_or_4_sinks_the_cell_but_not_eligibility() { + let mut report = base_report(); + report.check1_faithful_consumption = CheckOutcome::Fail("...".to_string()); + assert!( + is_eligible(&report), + "checks 1 and 4 are not in the disqualifying set" + ); + assert!(matches!(criterion_cell(&report), CheckOutcome::Fail(_))); + } + + // ---- F1: a check-5 NotRun is admissible only with bus-unreachable evidence ---- + + /// Required kill: a check-5 `NotRun` with no unreachable-bus evidence is + /// rejected, naming Round 0's readback evidence. + #[test] + #[should_panic(expected = "READBACK: PASS")] + fn a_check_5_not_run_with_no_evidence_is_rejected_by_is_eligible() { + let mut report = base_report(); + report.check5_accessibility = CheckOutcome::not_run("we did not build it").unwrap(); + report.check5_bus_unreachable_evidence = None; + let _ = is_eligible(&report); + } + + /// Same rejection, reached through `criterion_cell` instead of + /// `is_eligible` — both are "the scoring path" the review named. + #[test] + #[should_panic(expected = "READBACK: PASS")] + fn a_check_5_not_run_with_no_evidence_is_rejected_by_criterion_cell() { + let mut report = base_report(); + report.check5_accessibility = CheckOutcome::not_run("we did not build it").unwrap(); + report.check5_bus_unreachable_evidence = None; + let _ = criterion_cell(&report); + } + + /// Required kill: a check-5 `NotRun` **with** unreachable-bus evidence + /// is accepted, and does not disqualify the candidate (NotRun is not in + /// the disqualifying set — see [`DISQUALIFYING_CHECKS`]). + #[test] + fn a_check_5_not_run_with_evidence_is_accepted_and_does_not_disqualify() { + let mut report = base_report(); + report.check5_accessibility = CheckOutcome::not_run("bus unreachable").unwrap(); + report.check5_bus_unreachable_evidence = Some(some_evidence()); + assert!( + is_eligible(&report), + "a legitimately NotRun check 5 must not disqualify" + ); + let cell = criterion_cell(&report); + assert!(matches!(cell, CheckOutcome::NotRun(_)), "{cell:?}"); + } + + /// Required kill: an `AdapterStatus::NotBuilt` entry for the round's own + /// platform does not, by itself, make a check-5 `NotRun` admissible — + /// `NotBuilt` covers *other* platforms as declared scope, and must not + /// be usable to excuse AT-SPI2, the platform this round actually runs + /// on. The admissibility check must ignore `cost.adapters` entirely. + #[test] + #[should_panic(expected = "READBACK: PASS")] + fn a_not_built_adapter_for_the_rounds_own_platform_does_not_grant_admissibility() { + let mut report = base_report(); + report.check5_accessibility = CheckOutcome::not_run("we did not build it").unwrap(); + report.check5_bus_unreachable_evidence = None; + report.cost.adapters.push(AdapterStatus::NotBuilt { + platform: ROUND_PLATFORM.to_string(), + reason: "ran out of time".to_string(), + }); + let _ = is_eligible(&report); + } + + /// A check-5 `Pass` or `Fail` never triggers the admissibility check at + /// all — it exists only to gate `NotRun`, and must not fire on a report + /// that never claimed environmental absence. + #[test] + fn a_check_5_pass_or_fail_never_needs_bus_unreachable_evidence() { + let mut report = base_report(); + report.check5_accessibility = CheckOutcome::Pass; + report.check5_bus_unreachable_evidence = None; + assert!(is_eligible(&report)); + let _ = criterion_cell(&report); + + let mut report = base_report(); + report.check5_accessibility = CheckOutcome::fail("absent-from-tree").unwrap(); + report.check5_bus_unreachable_evidence = None; + assert!(!is_eligible(&report)); + let _ = criterion_cell(&report); + } +} diff --git a/spikes/editor-toolkit/round2-candidatekit/tests/dependency_deny_list.rs b/spikes/editor-toolkit/round2-candidatekit/tests/dependency_deny_list.rs new file mode 100644 index 0000000..5897d33 --- /dev/null +++ b/spikes/editor-toolkit/round2-candidatekit/tests/dependency_deny_list.rs @@ -0,0 +1,223 @@ +//! Enforces the neutrality boundary `src/lib.rs`'s crate doc comment states: +//! `round2-candidatekit` must not depend on any rendering, windowing, GPU, +//! or platform-accessibility crate — those stay candidate-owned (C1 = +//! egui + lyon, C2 = vello). This test reads this crate's own `Cargo.toml` +//! **at test time** rather than hard-coding "the current dependency list is +//! X" — the point is to catch a *future* dependency add, not merely to +//! assert today's file is fine. + +/// Rendering, windowing, GPU, and platform-accessibility crates that must +/// never appear in `round2-candidatekit`'s own `[dependencies]`. This list +/// is the thing under test — it is deliberately hard-coded, unlike the +/// dependency names it is checked against, which are always read fresh from +/// the manifest. +const DENY_LIST: &[&str] = &[ + "egui", + "eframe", + "egui-wgpu", + "lyon", + "lyon_path", + "lyon_tessellation", + "vello", + "wgpu", + "winit", + "accesskit", + "accesskit_winit", + "tiny-skia", + "resvg", + "usvg", +]; + +/// If `header` (the contents of a `[...]` line, already trimmed) names a +/// dependency **sub-table** — TOML's `[dependencies.name]` form, or the +/// same thing nested under a target, `[target.'cfg(...)'.dependencies.name]` +/// — returns `name`. `Cargo.toml` lets a single dependency spread across +/// its own `[...]` header when it needs more than a version string (e.g. +/// `[dependencies.wgpu]\nversion = "0.19"`), and that header names the +/// dependency directly rather than introducing a block of `key = value` +/// lines the way `[dependencies]` does — a scanner that only recognizes the +/// block form misses this shape entirely (confirmed empirically: it +/// returned `[]` for a manifest whose only dependency used this form). +fn dependency_subtable_name(header: &str) -> Option { + let rest = if let Some(r) = header.strip_prefix("dependencies.") { + r + } else if let Some(idx) = header.find(".dependencies.") { + &header[idx + ".dependencies.".len()..] + } else { + return None; + }; + // A dependency's own sub-table (e.g. hand-spread build metadata) would + // add a further dot, as in `dependencies.foo.metadata`; only the first + // segment is the crate name. + let name = rest.split('.').next().unwrap_or(rest); + Some(name.trim_matches('"').trim_matches('\'').to_string()) +} + +/// True if `header` opens a **block** of `key = value` dependency lines — +/// `[dependencies]` itself, or the same thing nested under a target +/// (`[target.'cfg(unix)'.dependencies]`). Deliberately does not match +/// `dev-dependencies` or `build-dependencies`: both end in "dependencies" +/// but with a hyphen, not a dot, immediately before it, so +/// `.ends_with(".dependencies")` is false for them — those tables are out +/// of scope for this guard on purpose (see +/// `the_line_scanner_finds_dependencies_and_ignores_other_sections`). +fn opens_dependency_block(header: &str) -> bool { + header == "dependencies" || header.ends_with(".dependencies") +} + +/// Extracts dependency names from a `Cargo.toml`, covering both shapes +/// Cargo accepts: the block form (`[dependencies]` followed by `key = +/// value` lines) and the sub-table form (`[dependencies.name]`), each +/// optionally nested under `[target.'cfg(...)'. ...]`. Deliberately not a +/// TOML parser — pulling one in as a dependency of a crate whose whole +/// point is a short, auditable dependency list would be self-defeating — +/// but a plain line scan that recognizes both header shapes, not just the +/// block one. +fn dependency_names(manifest: &str) -> Vec { + let mut names = Vec::new(); + let mut in_dependency_block = false; + for raw_line in manifest.lines() { + let line = raw_line.trim(); + if let Some(header) = line.strip_prefix('[').and_then(|s| s.strip_suffix(']')) { + let header = header.trim(); + if let Some(name) = dependency_subtable_name(header) { + names.push(name); + in_dependency_block = false; + continue; + } + in_dependency_block = opens_dependency_block(header); + continue; + } + if !in_dependency_block || line.is_empty() || line.starts_with('#') { + continue; + } + if let Some((key, _)) = line.split_once('=') { + names.push(key.trim().trim_matches('"').to_string()); + } + } + names +} + +fn manifest_path() -> std::path::PathBuf { + std::path::PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("Cargo.toml") +} + +#[test] +fn dependencies_do_not_include_a_denied_rendering_windowing_or_a11y_crate() { + let manifest = std::fs::read_to_string(manifest_path()) + .unwrap_or_else(|e| panic!("failed to read {}: {e}", manifest_path().display())); + let names = dependency_names(&manifest); + assert!( + !names.is_empty(), + "the line scanner found zero dependencies in {} — that means this test is vacuous, not \ + that the crate has no dependencies (it depends on round2-textkit, round2-diff, serde, \ + serde_json); check the scanner, not the crate", + manifest_path().display() + ); + let violations: Vec<&String> = names + .iter() + .filter(|n| DENY_LIST.contains(&n.as_str())) + .collect(); + assert!( + violations.is_empty(), + "round2-candidatekit/Cargo.toml [dependencies] names denied crate(s) {violations:?} — \ + this crate is candidate-neutral apparatus only (rendering, hit-test resolution, and \ + accessibility integration are candidate-owned; see src/lib.rs's crate doc comment for \ + the ruling this enforces). Denied list: {DENY_LIST:?}" + ); +} + +/// Sanity check on the scanner itself, against a synthetic manifest +/// fragment: if this fails, the test above could be silently vacuous no +/// matter what `[dependencies]` actually contains. Also confirms +/// `[dev-dependencies]` is not scanned — a denied crate under dev-only use +/// (impossible here, since this crate declares none, but stated as a +/// property of the scanner) must not trip the production-dependency check. +#[test] +fn the_line_scanner_finds_dependencies_and_ignores_other_sections() { + let synthetic = "[package]\nname = \"x\"\nversion = \"0.1.0\"\n\n[dependencies]\nserde = \ + \"1\"\nwgpu = \"0.19\"\n\n[dev-dependencies]\nwgpu = \"0.19\"\n"; + let names = dependency_names(synthetic); + assert_eq!(names, vec!["serde".to_string(), "wgpu".to_string()]); +} + +/// Confirms the scanner (and by extension the test above) actually flags a +/// denied name when one is present — otherwise `violations.is_empty()` +/// could be vacuously true because the scanner finds nothing, not because +/// the manifest is clean. +#[test] +fn a_synthetic_manifest_with_a_denied_dependency_is_flagged() { + let synthetic = "[dependencies]\nserde = \"1\"\ntiny-skia = \"0.11\"\n"; + let names = dependency_names(synthetic); + let violations: Vec<&String> = names + .iter() + .filter(|n| DENY_LIST.contains(&n.as_str())) + .collect(); + assert_eq!(violations, vec![&"tiny-skia".to_string()]); +} + +// ---- F2: the dotted sub-table form, confirmed empirically to be missed ---- +// +// Feeding the original scanner +// `"[dependencies]\nserde = \"1\"\n\n[dependencies.wgpu]\nversion = \"0.19\"\n"` +// returned `["serde"]` — `wgpu` never appeared, because the scanner only +// recognized `[dependencies]` as a block header and had no notion of a +// dependency named directly by its own `[...]` header. Each test below +// would fail if `dependency_subtable_name`'s handling were removed (i.e. +// if `dependency_names` fell back to the old block-only logic). + +/// The bare sub-table form: `[dependencies.wgpu]`. +#[test] +fn the_scanner_detects_a_dependency_named_via_a_dotted_subtable_header() { + let synthetic = "[package]\nname = \"x\"\n\n[dependencies]\nserde = \"1\"\n\n\ + [dependencies.wgpu]\nversion = \"0.19\"\n"; + let names = dependency_names(synthetic); + assert!( + names.contains(&"wgpu".to_string()), + "sub-table form missed: {names:?}" + ); + let violations: Vec<&String> = names + .iter() + .filter(|n| DENY_LIST.contains(&n.as_str())) + .collect(); + assert_eq!(violations, vec![&"wgpu".to_string()]); +} + +/// The block form nested under a target: `[target.'cfg(unix)'.dependencies]`. +#[test] +fn the_scanner_detects_a_dependency_block_under_a_target_cfg_table() { + let synthetic = + "[dependencies]\nserde = \"1\"\n\n[target.'cfg(unix)'.dependencies]\nwgpu = \"0.19\"\n"; + let names = dependency_names(synthetic); + assert!( + names.contains(&"wgpu".to_string()), + "target-cfg block form missed: {names:?}" + ); + let violations: Vec<&String> = names + .iter() + .filter(|n| DENY_LIST.contains(&n.as_str())) + .collect(); + assert_eq!(violations, vec![&"wgpu".to_string()]); +} + +/// Both forms combined: the sub-table form nested under a target, +/// `[target.'cfg(windows)'.dependencies.tiny-skia]`. +#[test] +fn the_scanner_detects_a_dotted_subtable_header_under_a_target_cfg_table() { + let synthetic = "[target.'cfg(windows)'.dependencies.tiny-skia]\nversion = \"0.11\"\n"; + let names = dependency_names(synthetic); + assert!( + names.contains(&"tiny-skia".to_string()), + "target-cfg sub-table form missed: {names:?}" + ); +} + +/// A dependency's own further sub-table (e.g. a spread-out `package` +/// rename) must still resolve to the crate name, the first dotted segment +/// after `dependencies.`, not the whole trailing path. +#[test] +fn a_deeper_dotted_path_still_resolves_to_the_leading_crate_name() { + let synthetic = "[dependencies.serde.metadata]\nfoo = 1\n"; + let names = dependency_names(synthetic); + assert_eq!(names, vec!["serde".to_string()]); +}