//! A committed completeness gate on the Text Projection companion's grammar. //! //! Version 0.3.0 claimed a "machine-checked" grammar. It was checked, once, by a //! script that was never committed — true of that run and of nothing durable. //! This is the durable form, and it is exactly the class of evidence P2–P4 //! established: a claim about coverage is a lock, and an uncommitted check is not //! one. //! //! What it enforces: //! //! 1. Every nonterminal the grammar references is defined, and every nonterminal //! it defines is reachable from `projection`. //! 2. The `kind` alternatives are **exactly** the operation vocabulary, *derived* //! from `OperationKindTag::catalog_name` rather than transcribed. That name is //! generated by the same `operation_kind_tag_vocabulary!` list as the //! discriminant and the decoder, so a tag added to the enum and not to the //! vocabulary fails to compile, and one added to the vocabulary but not the //! grammar fails here. //! 3. The `chunk-kind` alternatives are exactly the `ChunkKind` vocabulary. //! 4. The escape rule excludes the four codepoints //! `req:textproj:string-escapes` requires a writer to escape — the //! contradiction that 0.4.0 fixed. //! 5. `value` admits a bare parenthesised list. Without it the grammar cannot //! derive a sequence, so it cannot derive an ordinary pitched note — the //! omission 0.5.0 fixed. And it must *not* re-introduce a symbol-headed //! struct alternative: shape does not distinguish a struct from a sequence, //! and `req:textproj:schema-directed` says so rather than pretending //! otherwise. //! 6. The Chapter 6 operation vocabulary is explicitly grammar-directed, and //! the requirement carrying that boundary is cited at both reliance points. //! 7. `transpose-interval` takes target bytes followed by a `value`; `interval` //! is not allowed to become a grammar production parallel to the canonical //! `TranspositionInterval` value spelling. //! 8. The worked example's header is the one implemented companion version. //! 9. Unreferenced blob lines have an explicitly labelled, cited rejection rule //! with the round-trip and version-gating rationale that makes rejection //! necessary. use std::collections::{BTreeSet, VecDeque}; use epiphany_bundle::ChunkKind; use epiphany_ops::OperationKindTag; const SPEC: &str = include_str!("../../../spec/text_projection.tex"); /// The grammar block: from the `projection` production to the end of its listing. /// /// Located by name, never by column. The 0.4.0 reflow of the production headers /// broke a column-anchored needle, and a checker that cannot find the grammar /// silently checks nothing. fn grammar() -> &'static str { let start = SPEC .lines() .scan(0usize, |acc, l| { let here = *acc; *acc += l.len() + 1; Some((here, l)) }) .find(|(_, l)| { l.split_once("::=") .is_some_and(|(lhs, _)| lhs.trim() == "projection") }) .map(|(at, _)| at) .expect("the grammar block begins with the `projection` production"); let end = SPEC[start..] .find("\\end{lstlisting}") .expect("the grammar block is a listing"); &SPEC[start..start + end] } /// The worked example's complete projection listing. fn worked_example() -> &'static str { SPEC.split_once("\\chapter{A Worked Example}") .expect("the specification contains the worked example") .1 .split_once("\\begin{lstlisting}\n") .expect("the worked example contains a projection listing") .1 .split_once("\\end{lstlisting}") .expect("the worked projection listing is closed") .0 } /// Strips `;` line comments, then removes `<...>` prose spans, which may run /// across lines. A per-line stripper leaks the second line's characters, and the /// leaked fragments look like nonterminals. fn uncommented(g: &str) -> String { let no_comments: String = g .lines() .map(|l| l.split(';').next().unwrap_or("")) .collect::>() .join("\n"); let mut out = String::new(); let mut depth = 0usize; for c in no_comments.chars() { match c { '<' => depth += 1, '>' if depth > 0 => depth -= 1, '\n' if depth > 0 => out.push('\n'), _ if depth == 0 => out.push(c), _ => {} } } out } fn is_name_char(c: char) -> bool { c.is_ascii_lowercase() || c.is_ascii_digit() || c == '-' } /// The left-hand side of every production. fn defined(g: &str) -> BTreeSet { uncommented(g) .lines() .filter_map(|l| { let (lhs, _) = l.split_once("::=")?; let name = lhs.trim(); (!name.is_empty() && name.chars().all(is_name_char)).then(|| name.to_string()) }) .collect() } /// Removes quoted terminals, angle-bracket prose, and character classes, leaving /// only nonterminal references. fn strip_terminals(line: &str) -> String { let mut out = String::new(); let mut chars = line.chars().peekable(); while let Some(c) = chars.next() { match c { '"' => { for d in chars.by_ref() { if d == '"' { break; } } } '\'' => { for d in chars.by_ref() { if d == '\'' { break; } } } '[' => { for d in chars.by_ref() { if d == ']' { break; } } } _ => out.push(c), } } out } /// Every nonterminal referenced on a right-hand side, per production. fn references(g: &str) -> Vec<(String, BTreeSet)> { let text = uncommented(g); // A production may continue onto `|` continuation lines, which have no `::=`. let mut out: Vec<(String, BTreeSet)> = Vec::new(); for line in text.lines() { let (lhs, rhs) = match line.split_once("::=") { Some((l, r)) if l.trim().chars().all(is_name_char) && !l.trim().is_empty() => { out.push((l.trim().to_string(), BTreeSet::new())); (l.trim().to_string(), r) } _ => match out.last() { Some((name, _)) => (name.clone(), line), None => continue, }, }; let _ = lhs; let bare = strip_terminals(rhs); let mut token = String::new(); let entry = &mut out.last_mut().expect("a production is open").1; for c in bare.chars().chain(std::iter::once(' ')) { if is_name_char(c) { token.push(c); } else { // A nonterminal is `[a-z] [a-z0-9-]*`, so a token that does not // begin with a letter is not one. Without this, the `U+0022` in // the escape production reads as a nonterminal named `0022`. if token.starts_with(|c: char| c.is_ascii_lowercase()) { entry.insert(std::mem::take(&mut token)); } token.clear(); } } } out } #[test] fn every_nonterminal_is_defined_and_reachable() { let g = grammar(); let defined = defined(g); assert!(defined.contains("projection"), "the start symbol exists"); let refs = references(g); let used: BTreeSet = refs.iter().flat_map(|(_, r)| r.iter().cloned()).collect(); let undefined: Vec<&String> = used.difference(&defined).collect(); assert!( undefined.is_empty(), "grammar references undefined nonterminals: {undefined:?}" ); // Reachability from `projection`. let edges: std::collections::BTreeMap> = refs.into_iter().collect(); let mut seen = BTreeSet::new(); let mut queue = VecDeque::from(vec!["projection".to_string()]); while let Some(n) = queue.pop_front() { if !seen.insert(n.clone()) { continue; } for next in edges.get(&n).into_iter().flatten() { queue.push_back(next.clone()); } } let unreachable: Vec<&String> = defined.difference(&seen).collect(); assert!( unreachable.is_empty(), "grammar defines unreachable nonterminals: {unreachable:?}" ); } /// The right-hand side of `production`, spanning its continuation lines. /// /// Located by name, not by column: a grammar reflow must not silently turn a /// check off. Version 0.3.0's checks were column-anchored and did exactly that. fn production_block(production: &str) -> &'static str { let g = grammar(); let start = g .lines() .scan(0usize, |acc, l| { let here = *acc; *acc += l.len() + 1; Some((here, l)) }) .find(|(_, l)| { l.split_once("::=") .is_some_and(|(lhs, _)| lhs.trim() == production) }) .map(|(at, l)| at + l.find("::=").expect("has ::=") + 3) .unwrap_or_else(|| panic!("the grammar defines `{production}`")); let rest = &g[start..]; let end = rest .lines() .scan(0usize, |acc, l| { let here = *acc; *acc += l.len() + 1; Some((here, l)) }) .find(|(_, l)| l.contains("::=") && !l.trim_start().starts_with('|')) .map(|(at, _)| at) .unwrap_or(rest.len()); &rest[..end] } /// The alternatives of `production`, as the constructor symbol each opens with. fn alternatives(production: &str) -> BTreeSet { let block = production_block(production); let mut out = BTreeSet::new(); for (i, _) in block.match_indices("\"(") { let name: String = block[i + 2..] .chars() .take_while(|c| is_name_char(*c)) .collect(); if !name.is_empty() { out.insert(name); } } // Bare-symbol alternatives, e.g. `"not-in-tuplet"`. for (i, _) in block.match_indices('"') { let tail = &block[i + 1..]; if tail.starts_with('(') { continue; } let name: String = tail.chars().take_while(|c| is_name_char(*c)).collect(); if !name.is_empty() && tail[name.len()..].starts_with('"') { out.insert(name); } } out } #[test] fn the_kind_productions_are_the_operation_vocabulary() { let expected: BTreeSet = OperationKindTag::PAYLOAD_FREE .iter() .copied() .chain(std::iter::once(OperationKindTag::Registered( epiphany_ops::OperationKindRegistryId(0), ))) // `catalog_name` is production code, generated by the same // `operation_kind_tag_vocabulary!` list that generates the discriminant and // the decoder. It was a test-local `match` until 0.5.0 — a hand-maintained // list parallel to an enum, which is the exact shape that cost this project // four bugs. .map(|t| t.catalog_name().to_string()) .collect(); assert_eq!( expected.len(), 31, "30 payload-free kinds plus `Registered`" ); let actual = alternatives("kind"); assert_eq!( actual, expected, "the grammar's `kind` alternatives must be exactly the operation vocabulary.\n\ missing from the grammar: {:?}\n\ present but not a kind: {:?}", expected.difference(&actual).collect::>(), actual.difference(&expected).collect::>() ); } /// The operation grammar is the authority for Chapter 6 vocabulary; the /// mechanical Chapter-5 value rule begins only where a production says `value`. /// The two places that rely on that boundary must cite the requirement rather /// than leaving implementors to infer it from the grammar's shape. #[test] fn the_operation_vocabulary_requirement_is_cited_where_it_applies() { const LABEL: &str = "req:textproj:operation-vocabulary"; assert!( SPEC.contains(&format!("\\label{{{LABEL}}}")), "the grammar/value boundary must be a labeled normative requirement" ); let operation_preamble = grammar() .split_once("; --- Operation kinds") .expect("the grammar introduces its operation kinds") .1 .split_once("\nkind") .expect("the operation preamble precedes the `kind` production") .0; assert!( operation_preamble.contains(LABEL), "the operation-kind grammar must cite `{LABEL}` at the point it relies \ on the grammar/value boundary" ); let after_grammar = SPEC .split_once(grammar()) .expect("the specification contains the located grammar") .1; let grammar_explanation = after_grammar .split_once("\\chapter{A Worked Example}") .expect("the grammar chapter precedes the worked example") .0; assert!( grammar_explanation.contains(LABEL), "the prose explaining why Chapter-5 field lists are not restated must \ cite `{LABEL}`" ); } /// `TranspositionInterval` is a Chapter-5 value with the canonical /// `(transposition-interval ...)` spelling. Giving `transpose-interval` a second, /// inline `(interval ...)` spelling creates two names for one type; defining an /// `interval` nonterminal merely moves that same special case behind another /// production head. #[test] fn transpose_interval_delegates_to_value_without_an_interval_production() { let kind = uncommented(production_block("kind")) .split_whitespace() .collect::>() .join(" "); assert!( kind.contains("\"(transpose-interval (\" bytes* \") \" value \")\""), "`transpose-interval` must take target bytes followed by exactly one \ Chapter-5 `value`; the `kind` production reads: {kind}" ); assert!( !defined(grammar()).contains("interval"), "`interval` must not be a production head: TranspositionInterval uses \ the ordinary `value` projection" ); } #[test] fn the_chunk_kind_productions_are_the_chunk_vocabulary() { let expected: BTreeSet = (0u8..=8) .map(|d| { let kind = ChunkKind::from_discriminant(d).unwrap_or_else(|| panic!("chunk kind {d} exists")); kebab(&format!("{kind:?}")) }) .collect(); assert_eq!(expected.len(), 9); assert_eq!(alternatives("chunk-kind"), expected); } /// `OperationEnvelopeBlock` -> `operation-envelope-block`. fn kebab(camel: &str) -> String { let mut out = String::new(); for (i, c) in camel.chars().enumerate() { if c.is_ascii_uppercase() { if i != 0 { out.push('-'); } out.push(c.to_ascii_lowercase()); } else { out.push(c); } } out } /// `req:textproj:string-escapes` requires a writer to escape exactly the /// quotation mark, the backslash, U+000A and U+0009, and a parser to reject a /// literal one. Version 0.3.0's `unescaped` production admitted the backslash, /// contradicting the requirement it sits beneath. #[test] fn the_escape_grammar_agrees_with_the_escape_requirement() { assert!( defined(grammar()).contains("escape"), "the escape sequences must be a production of their own" ); // `unescaped` must exclude every character the requirement obliges a writer to // escape. Version 0.3.0 admitted the backslash here while requiring it escaped. let unescaped = production_block("unescaped"); for codepoint in ["U+0022", "U+005C", "U+000A", "U+0009"] { assert!( unescaped.contains(codepoint), "`unescaped` must exclude {codepoint} by codepoint; it reads: {unescaped}" ); } // Each escape is *two* characters: the U+005C introducer and one more. Spelling // the introducer as a quoted terminal `"\\"` reads as two backslashes and makes // every escape three characters long -- the requirement says two. let escapes: Vec> = uncommented(production_block("escape")) .split('|') .map(|alt| alt.split_whitespace().map(str::to_string).collect()) .collect(); let tails: BTreeSet = escapes .iter() .map(|alt| { assert_eq!( alt.len(), 2, "an escape is exactly two characters, the introducer and one more; \ found {alt:?}" ); assert_eq!( alt[0], "U+005C", "every escape is introduced by U+005C, written as a codepoint so no \ quoting convention can make it ambiguous; found {:?}", alt[0] ); alt[1].clone() }) .collect(); let expected: BTreeSet = ["U+0022", "U+005C", "\"n\"", "\"t\""] .iter() .map(|s| s.to_string()) .collect(); assert_eq!( tails, expected, "the escapes are exactly \\\", \\\\, \\n and \\t -- no more, no fewer" ); } /// `value` must admit a bare parenthesised list, or the grammar cannot derive any /// sequence — a `PitchedEvent`'s `articulations` among them, which makes an /// ordinary `insert-event` line underivable. It must equally *not* carry a /// symbol-headed struct alternative: a sequence whose first element is a fieldless /// variant has exactly that shape, so a grammar claiming to tell them apart would /// be lying. `req:textproj:schema-directed` carries the distinction instead. #[test] fn value_admits_a_bare_list_and_claims_no_shape_it_cannot_distinguish() { let block = uncommented(production_block("value")); let alts: Vec = block .split('|') .map(|a| a.split_whitespace().collect::>().join(" ")) .collect(); assert!( alts.iter().any(|a| a == "\"(\" value* \")\""), "`value` must admit a bare parenthesised list, else no sequence is \ derivable; alternatives were {alts:?}" ); assert!( !alts.iter().any(|a| a.contains("symbol \" \" value*")), "`value` must not claim to distinguish a struct from a sequence by shape; \ alternatives were {alts:?}" ); // The requirement that carries the distinction must exist and be cited where // it is relied on: the value rule, the strict-parse rule, and the grammar. assert!(SPEC.contains("\\label{req:textproj:schema-directed}")); let citations = SPEC.matches("req:textproj:schema-directed").count(); assert!( citations >= 4, "expected the label plus citations from the value rule, strict parsing \ and the grammar; found {citations}" ); } /// The mono font must not apply TeX ligatures. `tlig` rewrites `\"` as a right /// curly quote and `--` as an en dash, so the grammar -- which delimits terminals /// with U+0022 and builds escapes from U+005C -- would render characters other /// than the ones it specifies. A syntax document cannot misprint its own syntax. #[test] fn the_mono_font_does_not_substitute_glyphs_in_the_grammar() { let mono = SPEC .lines() .find(|l| l.starts_with("\\setmonofont")) .expect("the document sets a mono font"); assert!( !mono.contains("Ligatures"), "the mono font must not enable ligatures; it reads: {mono}" ); } /// The two sequences whose binary order reads erased physical attributes must be /// named by the derived-ordering requirement, and it must be cited where they are /// defined. A rule nobody points at is a rule nobody applies. #[test] fn the_derived_ordering_requirement_is_cited_where_it_applies() { assert!(SPEC.contains("\\label{req:textproj:derived-ordering}")); let citations = SPEC.matches("req:textproj:derived-ordering").count(); assert!( citations >= 4, "expected the label plus citations from the value rule, the blob \ requirement and the extension requirement; found {citations}" ); } /// The envelope already has a byte-exact production-code lock in /// `epiphany-ops`; the companion header above it must be locked as well. Derive /// the expected spelling from the title version so a future bump cannot update /// only one of the two. #[test] fn worked_example_header_is_the_implemented_companion_version() { let title_version = SPEC .lines() .find_map(|line| { line.split_once("Version ") .and_then(|(_, rest)| rest.split_once(" ---")) .map(|(version, _)| version) }) .expect("the title page declares the companion version"); assert_eq!( title_version, "0.7.0", "this implementation targets exactly companion 0.7.0" ); let expected = format!("(text-projection ({}))", title_version.replace('.', " ")); let actual = worked_example() .lines() .next() .expect("the worked projection has a header"); assert_eq!( actual, expected, "the worked example header must match the implemented companion version" ); } /// At 0.7.0 no canonical state can reference a blob, so accepting a blob line /// would make the text round trip lossy. Lock both the normative rejection and /// the rationale/citations that prevent a future reader from treating it as an /// accidental incompatibility. #[test] fn unreferenced_blob_rejection_is_normative_labelled_and_cited() { const LABEL: &str = "req:textproj:reject-unreferenced-blobs"; let label = format!("\\label{{{LABEL}}}"); let label_at = SPEC .find(&label) .unwrap_or_else(|| panic!("the blob rejection must be labelled `{LABEL}`")); let requirement_start = SPEC[..label_at] .rfind("\\begin{requirement}") .expect("the blob-rejection label is on a normative requirement"); let section_end = SPEC[label_at..] .find("\\section{Profile Declarations}") .map(|offset| label_at + offset) .expect("the blob rejection precedes profile declarations"); let rule_and_rationale = &SPEC[requirement_start..section_end]; assert!( rule_and_rationale .contains("A parser \\MUST{} reject every \\texttt{(blob ...)} line whose blob is"), "the labelled requirement must reject every unreferenced blob line" ); assert!( rule_and_rationale.contains("req:textproj:canonical-blobs"), "the rejection rule must cite the canonical-blob definition" ); assert!( SPEC.matches(LABEL).count() >= 2, "`{LABEL}` must be both declared and cited" ); for rationale_anchor in [ "necessarily non-canonical", "next projection silently drops", "causing data loss and falsifying", "\\textrm{project}(\\textrm{serialize}(\\textrm{parse}(T))) = T", "Forward compatibility belongs to header-version gating", "req:textproj:header-version", ] { assert!( rule_and_rationale.contains(rationale_anchor), "the blob-rejection rationale must retain `{rationale_anchor}`" ); } } /// The companion version declared on the title page. fn title_version() -> &'static str { SPEC.lines() .find_map(|line| { line.split_once("Version ") .and_then(|(_, rest)| rest.split_once(" ---")) .map(|(version, _)| version) }) .expect("the title page declares the companion version") } /// Every `\begin{requirement}`…`\end{requirement}` body. fn requirement_blocks() -> Vec<&'static str> { let mut out = Vec::new(); let mut rest = SPEC; while let Some((_, after)) = rest.split_once("\\begin{requirement}") { let (body, tail) = after .split_once("\\end{requirement}") .expect("every requirement block is closed"); out.push(body); rest = tail; } out } /// Version literals in `0.7.0` and `(0 7 0)` form, as `(major, minor, patch)`. fn version_literals(text: &str) -> Vec { let bytes: Vec = text.chars().collect(); let mut out = Vec::new(); for i in 0..bytes.len() { // `D.D.D` if bytes[i].is_ascii_digit() { let mut j = i; let mut parts = Vec::new(); let mut cur = String::new(); while j < bytes.len() && (bytes[j].is_ascii_digit() || bytes[j] == '.') { if bytes[j] == '.' { if cur.is_empty() { break; } parts.push(std::mem::take(&mut cur)); } else { cur.push(bytes[j]); } j += 1; } if !cur.is_empty() { parts.push(cur); } if parts.len() == 3 && (i == 0 || !bytes[i - 1].is_ascii_digit()) { out.push(parts.join(".")); } } // `(D D D)` if bytes[i] == '(' { let mut j = i + 1; let mut parts = Vec::new(); let mut cur = String::new(); while j < bytes.len() && (bytes[j].is_ascii_digit() || bytes[j] == ' ') { if bytes[j] == ' ' { if !cur.is_empty() { parts.push(std::mem::take(&mut cur)); } } else { cur.push(bytes[j]); } j += 1; } if !cur.is_empty() { parts.push(cur); } if j < bytes.len() && bytes[j] == ')' && parts.len() == 3 { out.push(parts.join(".")); } } } out } /// A version number named inside a **normative** block must be *this* companion's /// version. /// /// The version lives in six places in this document: the title page, two /// requirements, the worked example, and two revision-history cells. Only the /// title and the example were locked. The two requirements are the dangerous /// ones — they are normative, so a bump that misses them leaves the companion /// *requiring* parsers to accept a version it no longer is. The revision history /// is deliberately not covered: old rows name old versions, which is the point of /// a history. #[test] fn requirements_name_only_this_companion_version() { let title = title_version(); let mut found = 0usize; for block in requirement_blocks() { for literal in version_literals(block) { found += 1; assert_eq!( literal, title, "a requirement names version `{literal}` but the companion is \ `{title}`; normative text must not outlive a bump" ); } } assert!( found >= 2, "expected the header-version and blob-rejection requirements to name a \ version; found {found} — this lock has stopped reaching them" ); }