Merge pull request #209 from levineuwirth/ci-crdt-coverage

Compile and run the CRDT half of the test corpus in CI
This commit is contained in:
Levi Neuwirth 2026-08-01 13:35:35 -04:00 committed by GitHub
commit c5f750105c
No known key found for this signature in database
GPG Key ID: B5690EEEBB952194
10 changed files with 1420 additions and 146 deletions

View File

@ -249,6 +249,112 @@ jobs:
# off the shared cache). # off the shared cache).
- run: cargo test -p pmacs-protocol --all-targets - run: cargo test -p pmacs-protocol --all-targets
# crdt-test: the reason this job exists is that NOTHING in this
# workflow ever enabled the `crdt` feature. Every
# `#[cfg(feature = "crdt")]` test was therefore NOT COMPILED — not
# skipped, not filtered, not reported. Measured at 4223dd3: 3,467
# tests under the `test` job's flags versus 3,746 with `crdt`, so 279
# tests had never executed in CI. 186 of them are in the library,
# whose `cargo test --lib --features crdt` invocation CLAUDE.md lists
# as a REQUIRED pre-PR gate — a required gate CI had never run.
#
# Eight test binaries also contained zero tests under the old flags.
# They built, ran, and reported `ok` with nothing in them.
#
# ONE JOB, NOT TWO, and deliberately so. The obvious split — non-GPU
# suites here, GPU-requiring ones onto `gpu-render` — was rejected
# because `gpu-render` runs `cargo test -p pmacs-gpu`, a DIFFERENT
# PACKAGE from the root-package suites that would move there. Splitting
# would also require enumerating which suites are GPU-requiring, and a
# suite added later would silently land in whichever job did not need
# a GPU and skip there forever. Running the whole corpus in one place
# with a working adapter cannot develop that hole.
#
# ubuntu-only and luajit-only to start, per the standing guidance to
# take macOS from evidence rather than assumption; `crdt` is
# orthogonal to the Lua flavor, the same reasoning m5/m6-perf-gates
# already apply.
#
# The external-tool block from the `test` job is deliberately NOT
# duplicated here. It gates m4_acceptance, m6_5_repl_acceptance and
# m6_8_multi_repl_acceptance, and MEASURED: none of those has a single
# dark test, so installing clangd/zsh/fish/lua/rust-analyzer/gopls/npm
# servers again would cost minutes to change nothing. The only
# tool-gated code in the dark set is src/process.rs, whose two
# variables need no install on this runner and are set below.
#
# OBSERVED EXECUTION (local, 2026-08-01): the full serialized sweep —
# this job's exact test command plus PMACS_REQUIRE_GPU=1 — ran in 366s
# (6.1 min) for 3,715 passed / 0 failed / 30 ignored. The ceiling is
# 35 rather than something near that for two reasons: a hosted runner
# is slower per-core than the measuring machine and renders the GPU
# suites through lavapipe rather than a real adapter, and a cold cache
# adds a full debug workspace build. 35 matches the `test` job, which
# runs the same corpus without `crdt` and whose own observed max is
# 17 min — the closest available comparison, and the reason this
# number is not smaller.
crdt-test:
name: Test (crdt)
runs-on: ubuntu-latest
timeout-minutes: 35
steps:
- uses: actions/checkout@v4
- uses: dtolnay/rust-toolchain@stable
- uses: Swatinem/rust-cache@v2
# Several dark suites drive a real GPU frontend. Without an adapter
# they do not fail — a37 and its siblings `eprintln!` a skip and
# return `ok`, measured at 9/9 in 0.17s versus ~4s for a real run.
# lavapipe plus PMACS_REQUIRE_GPU below is what makes that skip
# fatal instead of silent.
- name: Install lavapipe (Vulkan software rasterizer)
run: |
sudo apt-get update
sudo apt-get install -y mesa-vulkan-drivers vulkan-tools
- name: Confirm a Vulkan adapter is present
run: vulkaninfo --summary || true
# Lint the crdt targets HERE rather than in the `clippy` job.
# Clearing these lints once is not enough: the `clippy` job matrixes
# over Lua flavor and never enables `crdt`, so without this step the
# crdt targets would drift straight back out of compliance and the
# next job to compile them would be red on arrival — which is
# exactly the state this lane found and fixed. Keeping it beside
# the crdt build and test means one job owns "crdt compiles, lints,
# and passes."
#
# `crdt` is orthogonal to the Lua flavor, so this runs once rather
# than matrixed. --keep-going reports every finding in one run
# instead of aborting at the first failing target, which is what
# made the original inventory of these lints a lower bound rather
# than a list.
- run: cargo clippy --workspace --all-targets --no-default-features --features luajit,crdt --keep-going -- -D warnings
# --workspace, not the root package: a37 and the gpu_* suites
# locate the `pmacs-gpu` binary beside `pmacs`, and a root-only
# build leaves it absent. This is the documented cause of twelve
# gpu_invocation_acceptance failures on a crdt sweep.
- run: cargo build --workspace --no-default-features --features luajit,crdt
# Serial for the same reason the `test` job is: these suites spawn
# real daemons and real PTYs, and this job adds MORE of them.
- run: cargo test --all-targets --no-default-features --features luajit,crdt -- --test-threads=1
env:
PMACS_REQUIRE_GPU: "1"
PMACS_REQUIRE_SETSID: "1"
PMACS_REQUIRE_BASH: "1"
- run: cargo test --doc --no-default-features --features luajit,crdt
# pmacs-protocol under `crdt`, which nothing else runs. Its own
# feature is NOT inert: it gates no tests, so a test census sees
# 17 either way, but it changes `cfg!(feature = "crdt")`
# EXPRESSIONS inside `InstanceCapabilities::default` and
# `FrontendCapabilities::default` — so the same 17 tests exercise
# different runtime values. CI had only ever run the non-crdt
# ones.
#
# This is also a blind spot of scripts/feature-census by
# construction: it lists the workspace DEFAULT MEMBER (`pmacs`),
# so no per-test census of the root package can ever surface a
# sibling crate's coverage. The `test` job carries the same
# explicit invocation for the same underlying reason.
- run: cargo test -p pmacs-protocol --all-targets --features crdt
acceptance: acceptance:
name: M1 Acceptance Gates name: M1 Acceptance Gates
runs-on: ubuntu-latest runs-on: ubuntu-latest
@ -319,3 +425,57 @@ jobs:
PMACS_M6_CANCEL_TRIALS: "30" PMACS_M6_CANCEL_TRIALS: "30"
PMACS_M6_CANCEL_MAX_DELAY_MS: "500" PMACS_M6_CANCEL_MAX_DELAY_MS: "500"
run: cargo test --release --test m6_perf_acceptance -- --ignored --nocapture --test-threads=1 run: cargo test --release --test m6_perf_acceptance -- --ignored --nocapture --test-threads=1
# m10-perf-gates: the M10 suites, which were dark for TWO independent
# reasons. Both are fixed here.
#
# 1. They are `crdt`-gated, and nothing in this workflow enabled the
# feature, so they were never compiled. That is this lane's subject.
# 2. Even setting `crdt` aside, NO job named them. Grepping this file
# for `--test` before this job existed yielded exactly four suites:
# acceptance, m4_acceptance, m5_perf_acceptance, m6_perf_acceptance.
# Their `#[ignore]` is deliberate; their absence from CI was not.
#
# luajit-only, for the same reason m5-perf-gates and m6-perf-gates are:
# the measured paths (CRDT buffer mutation, socket round-trips) do not
# enter the Lua VM, so matrixing over flavors doubles cost for no
# signal.
#
# WHAT EACH SUITE ACTUALLY GATES — these differ, and the difference
# matters for how a red run is read:
#
# * m10_11_perf asserts ONE budget: cross-frontend propagation p99
# under 50ms. Observed 1.47ms locally (2026-08-01), a ~34x margin,
# so a red here is a real regression rather than runner noise.
#
# * m10_2_perf asserts NOTHING. It is six measurement benches that
# print throughput numbers — the baselines M10.2's 391x unicode
# finding and v0.2+ optimization work compare against. It cannot
# fail a budget because it has none.
#
# It is here anyway, and NOT as a perf gate: `run_workload` drives
# 30 seconds of randomized mixed edits against both the v0.1 and
# CRDT buffer paths, and nothing else in the corpus exercises a
# sustained randomized CRDT workload. Its CI value is soak and
# panic detection. Do not "fix" a future silent run by adding
# budget assertions — the numbers are deliberately reported, not
# enforced, and asserting throughput on shared runners is how perf
# jobs become flaky.
#
# OBSERVED EXECUTION (local, 2026-08-01, release): m10_2_perf 79s
# (6 tests), m10_11_perf 5s (1 test) — about 85s combined. The ceiling
# is 25 rather than something near that, because a perf job's cost is
# dominated by its COLD-CACHE RELEASE BUILD, not its tests; this is
# the same reasoning that gives m5-perf-gates 25.
m10-perf-gates:
name: M10 Perf Gates (crdt)
runs-on: ubuntu-latest
timeout-minutes: 25
steps:
- uses: actions/checkout@v4
- uses: dtolnay/rust-toolchain@stable
- uses: Swatinem/rust-cache@v2
- name: CRDT buffer throughput baselines (soak; asserts nothing)
run: cargo test --release --features crdt --test m10_2_perf -- --ignored --nocapture
- name: cross-frontend propagation p99 over a real socket
run: cargo test --release --features crdt --test m10_11_perf -- --ignored --nocapture

View File

@ -5,6 +5,15 @@ landed on `main`. Read it after `docs/agent-handoff.md`. Remove completed
entries when their PR merges; do not let this become a second permanent entries when their PR merges; do not let this become a second permanent
backlog. backlog.
**Updated later the same day, on a new machine.** Development moved to
the laptop; the recovery path in "Repository authority" below was
exercised from this checkout and the `githubsucks` alias was absent and
had to be added, exactly as that section anticipates. **One lane opened:
CI CRDT coverage**, which had been sitting under "NEEDS A LANE" with no
branch and no owner since #166. It is implemented on
`ci-crdt-coverage` and its block replaces the old one below. Everything
else in this file is unchanged.
**This snapshot is an absorption pass, taken with ZERO open PRs** — **This snapshot is an absorption pass, taken with ZERO open PRs** —
the one window in which a ledger refresh has nothing to re-conflict the one window in which a ledger refresh has nothing to re-conflict
with, and taken deliberately before a machine move. Nine PRs landed with, and taken deliberately before a machine move. Nine PRs landed
@ -139,140 +148,181 @@ form. All four steps ran clean. **The two-argument form still does not
work** for a remote-only branch (`fatal: invalid reference`), which is work** for a remote-only branch (`fatal: invalid reference`), which is
why every lane below spells out the `-b` form. why every lane below spells out the `-b` form.
## The CRDT half of the test corpus is dark in CI — NEEDS A LANE ## CI CRDT coverage lane — **PR #209 OPEN, all 14 checks green**
- **No branch, no framing yet.** Found while gating #166, then measured **This lane had no branch and no owner from #166 until 2026-08-01.** It
properly during the vterm as-framed audit. Deliberately kept out of #166 so now has both, and a PR. Framing:
a CI change would not arrive after review approval. `docs/ci-crdt-coverage-framing.md` revision 4 (approved at revision 2).
- **Root cause:** `.github/workflows/ci.yml` never enables the `crdt` feature Branch `ci-crdt-coverage` off `githubsucks/main` @ `4223dd3`, developed
anywhere — zero hits across the workflow directory. The `test` job runs in the primary checkout on the laptop, not a worktree.
`cargo test --all-targets --no-default-features --features luajit|lua54`.
Every `#[cfg(feature = "crdt")]` test is therefore **not compiled** in CI,
not merely skipped.
- **Measured, `--list` under CI's exact flags versus the same flags plus
`crdt`: 3,176 vs 3,449 — 273 tests dark.** Re-measured at `74301d1`
(2026-07-26; at `fe8b8ba` it read 3,170 vs 3,443, the same 273 dark —
#176 added six tests, none of them `crdt`-gated). **The number moves
with every merge and must be
re-measured, not quoted.** #168 reported 3,024 vs 3,288 — 264 dark,
177 in the library — at `1b6a084`; #178 then added CRDT-only
generated-buffer coverage, and other lanes landed CRDT tests in
between. Per target:
| dark | CI | full | target | - **PR: <https://github.com/levineuwirth/pmacs/pull/209>**, opened
|---:|---:|---:|---| 2026-08-01, awaiting user review. Six commits: `7a9cf5b` clippy,
| 185 | 1,848 | 2,033 | **the library itself** (`src/lib.rs`) | `06abbac` m10-perf-gates, `7a8746d` crdt-test, `a776bc3`
| 21 | 15 | 36 | `m5_5_acceptance` | feature-census, `57abcd9` docs, plus a review-round commit for the
| 13 | 1 | 14 | `gpu_invocation_acceptance` | pmacs-protocol gap below.
| 13 | 1 | 14 | `gpu_initial_target_acceptance` | - **First CI run: all 14 checks green** (run `30705124856`), including
| 8 | 0 | 8 | `m10_11_acceptance` | both new jobs — `Test (crdt)` 12m20s and `M10 Perf Gates (crdt)`
| 6 | 0 | 6 | `auto_pair_crdt_acceptance` | 5m40s — and the macOS/luajit leg that is the usual flake surface.
| 6 | 0 | 6 | `m10_2_perf` | **This was the first time in the project's history that any of these
| 4 | 5 | 9 | `vterm_stage3_acceptance` | tests executed in CI.**
| 4 | 0 | 4 | `m10_10_perf` | - **Acceptance criterion 8 — the count reconciliation — HOLDS against
| 3 | 0 | 3 | `compile_mode_crdt_acceptance` | the real run.** `Test (crdt)` reported **3,717 passed / 0 failed / 30
| 2 | 22 | 24 | `theme_faces_acceptance` | ignored**. That is 3,746 (the `--all-targets` census, with
| 2 | 0 | 2 | `m11_5_semantic_acceptance` | `basedpyright` not skipped as it is locally) plus 1 doc test, minus
| 1 | 14 | 15 | `terminal_copy_mode_acceptance` | the 30 ignored: 3,716 + 1 = 3,717. The job demonstrably compiled and
| 1 | 9 | 10 | `vterm_stage1_acceptance` | ran the crdt corpus rather than reporting green over nothing.
| 1 | 7 | 8 | `statusline_segments_acceptance` | - **Do not compare the two jobs' raw totals.** In the first run
| 1 | 10 | 11 | `gpu_font_acceptance` | `Test (ubuntu/luajit)` reported 3,485 and `Test (crdt)` 3,747 — a
| 1 | 0 | 1 | `auto_indent_crdt_acceptance` | difference of 262, not 279, because the jobs ran different *sets*:
| 1 | 0 | 1 | `m10_11_perf` | 3,467 + 1 doc + 17 protocol = 3,485, against 3,746 + 1 doc = 3,747.
**The dark count is the all-targets comparison, 3,746 3,467 = 279.**
A reviewer who subtracts job totals gets a wrong number that looks
entirely plausible.
- **Those two figures are superseded** and are kept only to explain
the trap. Round 1 added a pmacs-protocol step to `crdt-test` and
round 2 added two capability tests to that crate. The current
totals were **predicted from the census and then confirmed exactly**
by run `30706324644` @ `71a1ebd`: `Test (crdt)` **3,766**
(3,746 + 1 doc + 19 protocol) and `Test (ubuntu/luajit)` **3,487**
(3,467 + 1 + 19). Predicting the count *before* the run and matching
it is a stronger reading of acceptance 8 than reconciling afterwards.
- The root-package census is untouched at 3,467 / 3,746 — the new
tests live in a sibling crate, exactly the region
`scripts/feature-census` cannot see.
- **The macOS legs report 3,474, thirteen fewer than ubuntu's 3,487**,
and that is expected: the Linux-gated process tests
(`setsid`, the `bash -m` job-control corroboration) are
`cfg`-compiled out rather than skipped. Do not read it as macOS
coverage loss.
- **Verify CI by `head_sha`, never by the check summary — it bit again
here.** Round 1's run (`30705916037` @ `6519bc3`) was **cancelled**,
not green: round 2's push superseded it, which is the concurrency
group working as designed. A `gh pr checks` summary polled around that
moment reported the *previous* run's results, with plausible timings,
and was briefly reported as round 1 passing. The ledger already
carried this lesson from #178 ("twelve checks green on head `1b44c69`
— verified by `head_sha`, not by the check summary"); it recurs
because the wrong answer looks exactly like the right one.
The rows sum to 273; the table is the whole census, not its head. - **Root cause, unchanged:** `.github/workflows/ci.yml` never enabled
the `crdt` feature anywhere. Every `#[cfg(feature = "crdt")]` test was
therefore **not compiled** in CI, not merely skipped — including the
186 library tests behind `cargo test --lib --features crdt`, which
`CLAUDE.md` lists as a REQUIRED pre-PR gate. CI had never run a
required gate.
- **Re-measured at `4223dd3`: 3,467 vs 3,746 — 279 dark**, up from the
273 recorded at `74301d1`. **Do not quote this number either.** It
moves with every merge, and there is now a tool: `scripts/feature-census
luajit luajit,crdt` reproduces the whole per-target table, the ignored
split, and the zero-test-target count in one command.
- **The 279 are fully dispositioned; 275 are recovered.** 268 by the new
`crdt-test` job, 7 by the new `m10-perf-gates` job, and 4 deliberately
excluded: `m10_11_acceptance`'s three PTY-doubled tests (marked
operator-invoked before tagging) and the #157 CRDT undo repro, an
`#[ignore]`d known-defect marker whose arming belongs to that defect's
lane.
- **The deliberate/accidental classification the old lane text called
"the lane's first task" is finished**, and it found THREE dispositions
rather than two: benches awaiting a job (`m10_2_perf`,
`m10_11_perf`), deliberately-manual operator tests
(`m10_11_acceptance`), and known-defect markers. Collapsing the second
into the first would have given a CI job to tests whose `#[ignore]`
reason says not to.
- **`m10_10_perf` is NOT a perf gate**, and the framing got this wrong
first. Its header says its bounds are "generous ... to catch
catastrophic regressions, not to verify a tight perf claim," so its
lack of `#[ignore]` is the design. Adding one to give it a job would
have shipped a coverage reduction inside a coverage lane. It stays
untouched and rides the plain leg.
- **The clippy obstacle is cleared, and the old inventory was stale in
both directions.** `cargo clippy --workspace --all-targets --features
crdt` had never passed. The previous seven-item list was correctly
labelled "a lower bound, not an inventory" — `--keep-going` is what
converts it. The real set was eight findings across four files; the
`unneeded mut` at `daemon.rs:4965` had been fixed incidentally, a
finding in `bottom_panel_stage2b_gpu_acceptance.rs` was new, and every
`daemon.rs` line number had moved.
- **One job, not two.** The fix-shape recorded here previously put the
GPU-requiring suites onto `gpu-render` "which already has lavapipe and
PMACS_REQUIRE_GPU". **That job runs `cargo test -p pmacs-gpu` — a
different package** from the four root-package suites, so co-locating
them means a new invocation rather than an extension. Splitting also
requires classifying every future suite as GPU-requiring or not, and a
misclassified one skips forever. `crdt-test` installs lavapipe and
sets `PMACS_REQUIRE_GPU=1` for the whole corpus instead.
- **`PMACS_REQUIRE_GPU` is not uniform and cannot serve as blanket
proof**: it appears in `vterm_stage3_acceptance` (twice) and
`bottom_panel_stage2b_gpu_acceptance` (once), and **not at all** in
`gpu_invocation_acceptance` or `gpu_initial_target_acceptance`.
- **The external-tool install block is deliberately not duplicated.**
Measured: it gates `m4_acceptance`, `m6_5_repl_acceptance` and
`m6_8_multi_repl_acceptance`, none of which has a single dark test.
The only tool-gated code in the dark set is `src/process.rs`, whose
two variables need no install.
- **Crdt clippy runs in `crdt-test`, not the `clippy` job.** The `clippy`
job matrixes over Lua flavor and never enables `crdt`, so clearing
those lints once would let them drift straight back and the next job
to compile them would be red on arrival — the exact state this lane
found.
- **Verification on the laptop (integrated Radeon, 16 threads), all at
the exact commands CI runs:** clippy green with and without `crdt`;
fmt; diff-check; `--lib` 1,896; `--lib --features crdt` 2,081; doc
tests; `m10_2_perf` 6/6 in 79s and `m10_11_perf` 1/1 in 5s under
release; and the full serialized sweep with `PMACS_REQUIRE_GPU=1` at
**3,715 passed / 0 failed / 30 ignored in 366s**, reconciling exactly
to the 3,746 census (3,715 + 30 + 1 basedpyright-skipped). The
reconciliation is the point: a sweep that does not reconcile is the
a37 vacuum at corpus scale.
- **What is NOT established, and it is the lane's whole remaining risk:**
the sweep is green *serialized on a developer machine*. The failures
this lane expects are hosted-runner timing and concurrency — real PTY
on CI runners, wgpu under lavapipe, daemon sockets at unfamiliar
concurrency. A green local run removes the "tests are wrong"
explanation and leaves the expected one untested. **Do not quote it as
evidence the CI leg will be green.**
- **Red-first-run policy, decided:** fix in-lane by default, with one
escape hatch keyed on cause class. Lane-configuration failures and
locally-reproducing failures are fixed here; an environment-only
failure (green locally and on Universum, red only on a hosted runner)
becomes a named follow-on rather than an unbounded investigation
inside a workflow-config PR. The local green makes that third class
the most likely red, which is why the hatch exists.
- **Universum (7900 XTX, remote) is where the GPU ambiguity settles.**
The laptop renders the GPU suites but is thermally constrained, which
is the documented condition for a37's "all spaces with nonzero
rendered_nonuniform_frames" signature. Universum can establish the
tests are sound; it cannot establish that the lavapipe CI job passes.
- **The single worst line is the library.** `cargo test --lib --features crdt` Recovery from a clean checkout:
is a REQUIRED local gate in `CLAUDE.md`, and CI has never run it. 185
library tests — the whole CRDT half — are developer-machine-only, and ```sh
that count grows with every merged branch that adds a `crdt`-gated git fetch githubsucks
unit test. git worktree add ../pmacs-ci-crdt \
- **Ten suites run zero or one test in CI**, including `gpu_initial_target` -b ci-crdt-coverage \
(#148's entire acceptance, 1/14), `gpu_invocation` (#141's, 1/14), and githubsucks/ci-crdt-coverage
`a37`, the Vterm Stage 3 real-daemon/real-PTY/real-wgpu path that #135 ```
built specifically because "a decoded-message fixture would prove none of
the three fit together". ### Still owned by this lane, not yet done
- **`a37` will report green in the new job without running, unless the
job builds `pmacs-gpu` AND sets `PMACS_REQUIRE_GPU=1`.** Measured - **The `--lib --features crdt` flake.**
2026-07-26 while gating #173. `a37_real_daemon_real_pty_and_headless_gpu_
render_one_terminal_session` derives its sibling binary path from
`CARGO_BIN_EXE_pmacs`, and on a missing binary it `eprintln!`s a skip and
**returns `ok`**. A fresh worktree running
`cargo test --features crdt --test vterm_stage3_acceptance` reports **9/9
in 0.17 s having never run it**; a real run takes ~4 s. Only
`PMACS_REQUIRE_GPU=1` promotes that skip to a failure, and `CLAUDE.md`
applies that flag to `cargo test -p pmacs-gpu` — a **different package**,
so the required local gate does not cover a37 either. The `gpu-render`
job already sets the flag, which is what makes fix-shape part 2 sound;
state it as a **requirement** of that job rather than inheriting it by
luck, because a `crdt` leg added to the plain `test` job would run a37
vacuously.
- **`a37` is also load-sensitive, which changes how to read the expected
first-run failures.** It passed at `d152120` and failed at that *same
commit* twenty minutes later, with a second agent saturating the machine
with `rustc` in between; it then failed identically on `d152120`,
`04c5ad1`, and the #173 merge commit, which is how #173 established the
failure was not its own. The signature is `last_frame_text` all spaces
with `rendered_nonuniform_frames` nonzero — frames arrive, content does
not. `pmacs-gpu`'s own suite flaked the same way under the same load
(201/202, then 202/202 on immediate rerun). **So a red a37 on the first
CI run is ambiguous by construction**: before treating it as a real
failure, run the same command on the merge base, and prefer serialized
execution for this suite over retry-until-green.
- **Sort deliberate from accidental before proposing a fix.** Some of the 264
are perf suites that are `#[ignore]`d by default and belong to their own
jobs (`m10_2_perf` 6, `m10_11_perf` 1). `m10_10_perf` has **no** `#[ignore]`
and no CI job naming it, so it looks accidental. This classification is not
finished and is the lane's first task.
- **Fix shape, two parts** (the flag combination is verified to work:
`--no-default-features --features luajit,crdt` lists 10 vterm Stage 1 tests
versus 9 without):
1. a `crdt` leg on the `test` job for the non-GPU suites and the library;
2. the GPU-requiring `crdt` suites onto the existing `gpu-render` job, which
already has lavapipe and `PMACS_REQUIRE_GPU=1`
`vterm_stage3_acceptance`, `gpu_invocation_acceptance`,
`gpu_initial_target_acceptance`, `gpu_font_acceptance`.
- **Expect first-run failures, and budget for them.** These would execute in
CI for the first time ever: real PTY timing on CI runners, wgpu under
lavapipe, and daemon-socket tests at unfamiliar concurrency. Start
ubuntu-only and decide about macOS from evidence. A red first run is the
lane working, not the lane failing.
- Mitigating fact, verified rather than assumed: #166's three unit pins are
**not** `crdt`-gated and do run under CI's exact flags, including the
controller-release pin whose only job is catching the plausible wrong fix.
- **This lane also owns a `--lib --features crdt` flake, observed and
scoped without overclaiming its cause** (inherited from #178's gating,
where the terminal lane recorded it). `cargo test --lib --features
crdt` failed ~1 run in 5 on
`process::tests::setsid_escapee_is_not_reaped_and_teardown_reclaims_readers` `process::tests::setsid_escapee_is_not_reaped_and_teardown_reclaims_readers`
`active_reader_probe` returning `None` at `process.rs:3179` ("live failed ~1 run in 5 with `active_reader_probe` returning `None`. **It
runtime probe"). **Pre-existing and unrelated to #178:** that branch did not reproduce in this lane's runs** — `--lib --features crdt` was
did not touch `src/process.rs` at all, and the test passed 10/10 2,081/2,081 and the full serialized sweep was clean — but a
standalone; the observed non-reproduction under serial execution is consistent with the
failures were during parallel full-suite runs. That localizes the leading hypothesis rather than evidence against it: the trigger was
trigger to suite load or interaction, but does **not** distinguish observed under *parallel* full-suite load, and every run here was
parallelism from another full-suite effect — no serial full-suite bite `--test-threads=1`. The `drain_until` explanation (draining for
was run. The leading code-path explanation is the known `drain_until` `Started` also ticks, and a tick can reap the leader before the
trap: draining for `Started` also ticks, and a tick can reap the leader following probe) remains an inference from control flow, not a
before the following `active_reader_probe`. That is an inference from falsified root cause. Discriminating it is its own PR — it is a
the failure site and control flow, not yet a falsified root cause. product-defect hypothesis and everything else here is workflow
Discriminating it belongs here. Two unnamed CRDT failures in #178's configuration.
round-2 gating are a plausible match but remain **unattributed** — no - **The two unattributed CRDT failures from #178's round-2 gating.** No
test names were captured. test names were captured, so there is nothing to reproduce.
- **A second standing obstacle for this lane:** `cargo clippy --workspace - **macOS.** Ubuntu-only first, deliberately; a macOS `crdt` leg is a
--all-targets --features crdt -- -D warnings` **fails on `main`** follow-on decided from the first run's evidence.
measured at `74301d1`: seven errors before the build aborts, four in
`src/daemon.rs` (`useless_conversion` at 3996, missing doc backticks at
4076, `too_many_lines` 112/100 at 4083, an unneeded `mut` at 4965) and
three in `tests/vterm_stage3_acceptance.rs` (`too_many_lines` at 637
and 793, a redundant `continue` at 843). **Treat that as a lower
bound, not an inventory:** Clippy abandons the remaining targets once
one fails, and a run on an older tree surfaced a further doc-backticks
error in `tests/auto_indent_crdt_acceptance.rs:42` that this run never
reached. The
standing gate list runs Clippy without `crdt`, so these lints have
never been enforced. Any CI job that compiles the `crdt` targets has to
fix them first or it will be red on arrival.
## Discovery lane (P4) — STAGE 1 MERGED (#207); STAGE 2 IS NEXT ## Discovery lane (P4) — STAGE 1 MERGED (#207); STAGE 2 IS NEXT

View File

@ -1,6 +1,6 @@
# Agent handoff — cross-machine continuity # Agent handoff — cross-machine continuity
**Last updated: 2026-07-29, as bottom-panel Stage 2B-3 — the GPU panel **Last updated: 2026-08-01 (CI CRDT coverage lane opened on `ci-crdt-coverage`; development moved to the laptop). Previously 2026-07-29, as bottom-panel Stage 2B-3 — the GPU panel
band, compatible protocol-v21 activation, and the negotiated band, compatible protocol-v21 activation, and the negotiated
`panel_capable` flip, completing Arc 7 Stage 2 — opens atop `e003b81`. `panel_capable` flip, completing Arc 7 Stage 2 — opens atop `e003b81`.
Beneath it, bottom-panel Stage 2B-2 (#187) — the Beneath it, bottom-panel Stage 2B-2 (#187) — the
@ -94,10 +94,24 @@ anchor, so every item is startable.
#### Open lanes (branch exists, work not finished) #### Open lanes (branch exists, work not finished)
- **CRDT half of the corpus is dark in CI — still no lane, no owner.** - **CRDT half of the corpus is dark in CI — PR #209 OPEN, all 14 checks
CI never enables `crdt`, so those tests are *not compiled*, not merely green** (`ci-crdt-coverage`, `docs/ci-crdt-coverage-framing.md` rev 4;
skipped. **Re-measure before quoting a number**; the ledger's figure opened 2026-08-01, awaiting user review). CI never enabled `crdt`, so
moves with every merge. those tests were *not compiled*, not merely skipped — including 186
library tests behind a **required** `CLAUDE.md` gate CI had never run.
The first run reported **3,717 passed / 0 failed / 30 ignored** in
`Test (crdt)`, reconciling exactly to the census, so the job is
demonstrably not vacuous.
**Re-measure before quoting a number**, and there is now a tool:
`scripts/feature-census luajit luajit,crdt`. Measured 279 dark at
`4223dd3`; 275 recovered, 4 excluded with stated reasons. Two
corrections it forced are worth carrying: **`m10_10_perf` is a
CI-default regression tripwire, not a bench** (its bounds are
deliberately generous, so `#[ignore]`ing it to give it a perf job
would *reduce* coverage), and **`gpu-render` runs a different
package** (`pmacs-gpu`) from the root-package GPU suites, so the
long-recorded "move them onto gpu-render" fix-shape does not work as
written.
- **Generated-buffer immutability** — Stage 1 merged (#191); Stage 2 - **Generated-buffer immutability** — Stage 1 merged (#191); Stage 2
not started. Four writer mechanisms have still not adopted not started. Four writer mechanisms have still not adopted
`set_generated_contents`; key the inventory by *writer*, not buffer. `set_generated_contents`; key the inventory by *writer*, not buffer.
@ -156,7 +170,17 @@ someone forgot.
`cargo test --test m4_acceptance -- --skip basedpyright`. `cargo test --test m4_acceptance -- --skip basedpyright`.
- **The crdt sweep needs `cargo build --workspace` first**, or twelve - **The crdt sweep needs `cargo build --workspace` first**, or twelve
`gpu_invocation_acceptance` tests fail on a missing `pmacs-gpu` `gpu_invocation_acceptance` tests fail on a missing `pmacs-gpu`
binary. binary. `cargo build --workspace --no-default-features --features
luajit,crdt` is the invocation that produces both binaries.
- **Never hand-roll the dark-test census — use
`scripts/feature-census`.** libtest prints `name: test` with **no
space before the colon**, so a filter written `/ : test$/` matches
nothing and reports a clean zero; a target with zero tests prints its
`Running` line and nothing else, so counting only test lines drops it
from the diff; and both configurations need an `--ignored` pass, or
pre-existing ignores get attributed to the feature. Each of those was
hit while writing the script, and the second one survived two
revisions of a framing doc.
- **A green a37 means nothing on its own** — the vterm real-daemon - **A green a37 means nothing on its own** — the vterm real-daemon
acceptance returns `ok` without running unless `pmacs-gpu` is built. acceptance returns `ok` without running unless `pmacs-gpu` is built.

View File

@ -0,0 +1,710 @@
# Framing — the CRDT half of the test corpus is dark in CI
**Revision 5.** Status: **PR #209 open, all 14 checks green on the first
run.** Branch `ci-crdt-coverage`, based on `githubsucks/main` @
`4223dd3` (#208). Approved at revision 2.
**Revision 3 → 4** records the first CI run and one review finding:
- **Acceptance 8 holds against the real run.** `Test (crdt)` reported
**3,717 passed / 0 failed / 30 ignored** — 3,746 census (with
`basedpyright` not skipped as it is locally) + 1 doc test 30
ignored. The job compiled and ran the corpus; it did not report green
over nothing. §1.7's warning that the local green proved nothing about
hosted runners is now **discharged by evidence** rather than still
outstanding.
- **A gap `feature-census` structurally could not see**: pmacs-protocol
has its own `crdt` feature. It gates no tests — the census is 17
either way — but it changes `cfg!(feature = "crdt")` *expressions*
inside the capability defaults, so those 17 tests exercise different
runtime values under it, and CI had only ever run the non-crdt ones.
Closed by an explicit `-p pmacs-protocol --features crdt` step. **A
feature can matter to a crate a per-test census scores as
unaffected**, and the script's header now says so.
**Revision 4 → 5** closes review round 2, and it is this lane's own
defect class one level down.
- **Running the code was not testing it.** Round 1's
`-p pmacs-protocol --features crdt` step *executed*
`InstanceCapabilities::default` in both configurations, but the crate's
only use of that value was a **transport round-trip** — and a
round-trip is invariant to the values. An all-false default, or one
whose three fields disagreed, encodes and decodes just as happily and
passes in both builds. The step added execution and no assertion.
- **Three tests now pin it**, and their split matters: one asserts all
three fields `true` under `crdt`, one asserts all three `false`
without it, and a third — **deliberately not feature-gated** — asserts
`FrontendCapabilities::default` is all-false in *both* builds.
- **That third test pins an asymmetry nothing else did.**
`FrontendCapabilities` derives `Default` and is feature-**invariant**;
`InstanceCapabilities` is feature-**dependent**. This is load-bearing,
not an oversight: an instance advertises what it can do, a frontend
**opts in** through negotiation, and a v1 frontend has no local CRDT
state regardless of how the crate it links was compiled.
- **Bite-verified rather than assumed**: mutating `multi_frontend` to a
literal `false` produced `FAILED. 18 passed; 1 failed` with the
expected assertion message; restoring returned 19/19.
- **Left untested, deliberately:** `InstanceCapabilities::crdt_replica`
carries `#[serde(default = "default_true")]`, a *third* default
mechanism that is unconditional and therefore disagrees with the
`Default` impl in a non-CRDT build. Exercising it needs a
self-describing format, and this crate's only serde dependency is
postcard, which is not one. Adding `serde_json` as a dev-dependency to
test a divergence this lane did not introduce is scope creep; it is
recorded here instead.
**Revision 2 → 3** records implementation findings, not a new design
round. Three things changed:
- **§1.1's target-column claim was wrong**, and `scripts/feature-census`
found it. "Eight test binaries contain zero tests" merged two
different true statements; the corrected reading is in §1.1.
- **The census now has a tool** (§1.9). The ledger's standing "re-measure,
don't quote" instruction had never had one.
- **Acceptance 9 is revised** — the deliberate-break bite is replaced by
a structural coverage assertion plus the CI count reconciliation, with
what each does and does not prove stated explicitly. See §4.
**Revision 1 → 2** recorded the decisions on Q#CC3, Q#CC4 and Q#CC7, and
**corrected revision 1's classification of `m10_10_perf`**, which was
wrong in a way that would have made the lane worse (§1.3a). It also
completed the disposition accounting: all 279 dark tests assigned, 275
recovered and 4 excluded for named reasons (§1.2a).
`.github/workflows/ci.yml` never enables the `crdt` feature anywhere.
Every `#[cfg(feature = "crdt")]` test is therefore **not compiled** in
CI — not skipped, not filtered, not reported. **279 tests have never
executed in CI, and 186 of them are in the library**, whose
`cargo test --lib --features crdt` invocation `CLAUDE.md` lists as a
**required** pre-PR gate. CI has never once run a required gate.
This lane was named in `docs/active-work.md` under "NEEDS A LANE" and
has had no branch and no owner since it was found while gating #166.
---
## 0. Coherence impact (COHERENCE §20)
- **Journey steps touched:** none directly. This lane adds no user-facing
surface and no command.
- **Interaction islands added:** none.
- **Config registry adoption:** none.
- **Background-work attribution:** none.
- **Why it belongs on the board anyway:** it is the release-readiness
prerequisite for **P8 (Distribution, §17)**. §17's grade is "missing —
zero release machinery exists," and the first thing release machinery
must do is produce an artifact from a tree whose tests actually ran.
Shipping binaries over a corpus where half of the library's tests have
never been compiled by CI converts an invisible gap into a shipped one.
This lane does not advance a coherence concern; it protects the one
that comes next.
---
## 1. Ground truth (measured at `4223dd3`, 2026-08-01)
Every number below was **re-measured on this tree**, not quoted forward.
The ledger is explicit that the figure moves with every merge, and it
did: the previous reading was 273 at `74301d1`.
### 1.1 The census
Under CI's exact flags versus the same flags plus `crdt`:
| | tests | targets with ≥1 test |
|---|---:|---:|
| `--no-default-features --features luajit` | 3,467 | 93 |
| `--no-default-features --features luajit,crdt` | 3,746 | 101 |
| **dark** | **279** | **8 targets gain tests** |
**Corrected in revision 3.** Revisions 1 and 2 read the target column as
"eight test binaries contain zero tests under CI's flags." That is not
what the column measures, and `scripts/feature-census` (§1.9) found the
error by reporting both figures separately. Two different true
statements had been merged:
- **Eleven** targets run under CI's flags with **zero tests**. They
build, run, and report `ok` with nothing in them.
- **Eight** of those eleven **gain tests under `crdt`** — the ones this
lane recovers. The other three are helper binaries (`pmacs_audit`,
`pmacs_fake_lsp`, `pmacs_fake_mcp`) with no tests in either
configuration, which is correct and not a gap.
A green result from the eight is not weak evidence, it is no evidence.
The three are fine. Conflating them inflates the defect.
Per target, every row with a nonzero delta:
| dark | CI | full | target |
|---:|---:|---:|---|
| 186 | 1,899 | 2,085 | **the library itself** (`src/lib.rs`) |
| 21 | 15 | 36 | `m5_5_acceptance` |
| 14 | 1 | 15 | `gpu_invocation_acceptance` |
| 14 | 1 | 15 | `gpu_initial_target_acceptance` |
| 8 | 0 | 8 | `m10_11_acceptance` |
| 6 | 0 | 6 | `m10_2_perf` |
| 6 | 0 | 6 | `auto_pair_crdt_acceptance` |
| 4 | 5 | 9 | `vterm_stage3_acceptance` |
| 4 | 0 | 4 | `m10_10_perf` |
| 3 | 2 | 5 | `bottom_panel_stage2b_gpu_acceptance` |
| 3 | 0 | 3 | `compile_mode_crdt_acceptance` |
| 2 | 22 | 24 | `theme_faces_acceptance` |
| 2 | 0 | 2 | `m11_5_semantic_acceptance` |
| 1 | 18 | 19 | `terminal_copy_mode_acceptance` |
| 1 | 10 | 11 | `gpu_font_acceptance` |
| 1 | 9 | 10 | `vterm_stage1_acceptance` |
| 1 | 7 | 8 | `statusline_segments_acceptance` |
| 1 | 0 | 1 | `m10_11_perf` |
| 1 | 0 | 1 | `auto_indent_crdt_acceptance` |
The rows sum to 279. This is the whole census, not its head.
`bottom_panel_stage2b_gpu_acceptance` is new since the previous reading.
### 1.2 "279 dark" overstates what a plain `crdt` leg recovers
**Eleven of the 279 are `#[ignore]`d**, so adding `--features crdt` to
the `test` job does not run them — `--ignored` does, and the `test` job
does not pass it. Measured by listing the ignored set under the crdt
build and differencing against the CI build:
| dark | of which ignored | recovered by a plain leg | target |
|---:|---:|---:|---|
| 186 | 1 | 185 | the library |
| 8 | 3 | 5 | `m10_11_acceptance` |
| 6 | 6 | **0** | `m10_2_perf` |
| 1 | 1 | **0** | `m10_11_perf` |
| — | — | — | all other rows: nothing ignored |
**A plain `crdt` leg recovers 268 tests, not 279.** The remaining 11
need either an `--ignored` invocation or a deliberate exclusion.
Quoting 279 as the lane's deliverable would overstate it by the exact
set that is hardest to place.
### 1.2a Full disposition of all 279
Every dark test is assigned. Nothing is residue.
| count | disposition | mechanism |
|---:|---|---|
| 268 | **recovered by the plain `crdt` leg** — including `m10_10_perf`'s 4, which stay unignored (§1.3a) | new ubuntu/luajit `crdt` job |
| 7 | **recovered by a new `m10-perf-gates` job**: `m10_2_perf` (6) + `m10_11_perf` (1) | `--release --ignored --features crdt`, per the `m5`/`m6` precedent |
| 3 | **deliberately excluded**: `m10_11_acceptance`'s PTY-doubled tests, marked *"operator-invoked before tagging, not CI-default"* | unchanged; documented, not a gap |
| 1 | **deliberately excluded**: `buffer::tests::proptests::crdt_undo_of_an_identity_replace_reports_a_no_op_edit_carrying_an_op` — the **#157 CRDT undo repro**, an `#[ignore]`d marker for a known open defect | unchanged; arming it is that defect's lane, not this one |
| **279** | | |
**The lane recovers 275 of 279.** The other four are excluded with
stated reasons. This distinction matters for the PR description: a lane
that says "268 of 279" invites the question of what the 11 are, and two
of the four answers are "already correct."
### 1.3 The deliberate/accidental classification — the lane's stated first task
`docs/active-work.md` says this classification "is not finished and is
the lane's first task." It is finished here.
**Deliberate, and belonging in a perf job** — `#[ignore]`d because they
are release-mode benches or budget gates, exactly like
`m5_perf_acceptance` and `m6_perf_acceptance`:
- `m10_2_perf` — 6 dark, all 6 ignored: *"perf bench; release-mode-only
via `--ignored --nocapture`"*.
- `m10_11_perf` — 1 dark, ignored: *"perf gate; requires release build"*.
**But their placement is accidental even so.** Grepping `ci.yml` for
`--test` yields exactly four named suites: `acceptance`,
`m4_acceptance`, `m5_perf_acceptance`, `m6_perf_acceptance`.
**`m10_2_perf` and `m10_11_perf` have no CI job at all**, with or
without `crdt`. Their being `#[ignore]`d is deliberate; their being
unreferenced by any workflow is not. Q#CC4 fixes the second.
**Deliberate, and belonging nowhere in CI** — a third disposition,
distinct from the above and easy to collapse into it:
- `m10_11_acceptance` — 3 of its 8 dark are ignored, marked *"PTY-doubled
tests are operator-invoked before tagging, not CI-default"*. These are
not benches awaiting a job; they are **deliberately manual**, run by an
operator before tagging a release. Giving them a job would contradict
the reason they are ignored. Its other 5 dark tests are ordinary
coverage and are recovered by the plain leg.
**Accidental** — `m10_10_perf`: 4 dark, **zero `#[ignore]` markers**, and
no CI job naming it. Accidental in that *nothing runs it*. See §1.3a for
why its unignored state is nevertheless deliberate and must be
preserved.
Everything else in the table is accidental: ordinary correctness
coverage that has simply never been compiled.
### 1.3a `m10_10_perf` is not a perf gate — revision 1 got this wrong
Revision 1 classified `m10_10_perf` as a perf suite that "a plain `crdt`
leg would start running inside the general correctness job... a hazard
the fix must handle," and the natural remedy — add `#[ignore]`, give it
a perf job — was proposed and accepted on that basis. **The suite's own
header falsifies it:**
> The numbers are recorded to stdout via eprintln (visible under
> `cargo test -- --nocapture`) and asserted against **generous bounds
> that exist to catch catastrophic regressions, not to verify a tight
> perf claim.**
Its four tests are catastrophic-regression tripwires with deliberately
loose bounds. **The absence of `#[ignore]` is the design, not an
oversight**, and the contrast with its siblings is explicit in their
ignore reasons:
| suite | `#[ignore]` reason | what it is |
|---|---|---|
| `m10_2_perf` | *"perf bench; release-mode-only via `--ignored --nocapture`"* | benchmark |
| `m10_11_perf` | *"perf gate; requires release build"* | budget gate |
| `m10_10_perf` | **none** | regression tripwire |
Three consequences:
1. **Adding `#[ignore]` would demote a deliberate CI-default tripwire
into an operator-invoked bench** — a coverage *reduction* shipped
inside a coverage lane.
2. **The hazard revision 1 named does not apply to this suite.** Perf
assertions are dangerous in a shared correctness job when their
bounds are tight; generous bounds designed to catch only catastrophic
regressions are precisely what is safe there.
3. **`m10_10_perf` therefore needs no job of its own.** It is recovered
by the plain `crdt` leg, and its 4 tests are inside the 268.
The general lesson, which is the reusable part: *a suite's `#[ignore]`
state is a claim about how it should be invoked, and the file that makes
the claim is the authority.* Revision 1 classified three suites by their
filename suffix (`_perf`) and their marker counts, and got the one whose
name and markers disagreed with its purpose exactly backwards.
### 1.4 The clippy obstacle is real, and the ledger's inventory is stale
`cargo clippy --workspace --all-targets --features crdt -- -D warnings`
**fails on `main`.** The ledger recorded this and correctly warned that
its list was "a lower bound, not an inventory," because clippy abandons
remaining targets once one fails.
**`--keep-going` is what converts the lower bound into an inventory.**
That flag was not used before; with it, the complete set at `4223dd3` is
**eight findings across four files**:
| file:line | lint |
|---|---|
| `src/daemon.rs:4464` | `useless_conversion` to the same type: `u64` |
| `src/daemon.rs:4544` | item in documentation is missing backticks |
| `src/daemon.rs:4551` | `too_many_lines` (112/100) |
| `tests/auto_indent_crdt_acceptance.rs:42` | missing doc backticks |
| `tests/bottom_panel_stage2b_gpu_acceptance.rs:509` | `too_many_lines` (104/100) |
| `tests/vterm_stage3_acceptance.rs:637` | `too_many_lines` (122/100) |
| `tests/vterm_stage3_acceptance.rs:816` | `too_many_lines` (132/100) |
| `tests/vterm_stage3_acceptance.rs:866` | redundant `continue` |
**The ledger's list is wrong in both directions**, which is why it had to
be re-measured rather than carried forward: the `unneeded mut` at
`src/daemon.rs:4965` is **gone** (fixed incidentally by later work), a
finding in `bottom_panel_stage2b_gpu_acceptance.rs` is **new**, and every
`src/daemon.rs` line number has moved. A stale lint inventory is worse
than none, because it invites fixing lines that no longer exist.
None of the eight is a correctness defect. All are lint-policy findings,
and none requires a behavioral change — which is what makes them safe to
clear in a preparatory commit rather than a design round.
### 1.5 `PMACS_REQUIRE_GPU` does not cover the suites the fix wants to move
The proposed fix routes four GPU-requiring `crdt` suites onto the
existing `gpu-render` job, on the grounds that it already has lavapipe
and `PMACS_REQUIRE_GPU=1`. **That job runs `cargo test -p pmacs-gpu` — a
different package.** All four target suites live in the root `pmacs`
package's `tests/`, so moving them means adding a *new* root-package
invocation to that job, not extending an existing one.
And the guard is **not uniform across the four**. Grepping every
reference to `PMACS_REQUIRE_GPU`:
- `tests/vterm_stage3_acceptance.rs` — two sites, both binary-presence
skips promoted to failures.
- `tests/bottom_panel_stage2b_gpu_acceptance.rs` — one site, same shape.
- `pmacs-gpu/src/main.rs` — adapter presence, a different condition.
- **`gpu_invocation_acceptance` and `gpu_initial_target_acceptance`
reference it nowhere.** Setting the variable does not arm them.
So `PMACS_REQUIRE_GPU=1` is necessary for a37 and the panel suite and
**insufficient as a blanket guarantee** that all four ran. Whatever
proves these suites executed has to be per-suite, not one environment
variable assumed to cover the set.
### 1.6 The a37 vacuum, restated with its current mechanism
`a37_real_daemon_real_pty_and_headless_gpu_render_one_terminal_session`
derives the frontend binary path from `CARGO_BIN_EXE_pmacs`'s sibling and
**returns `ok` after an `eprintln!` when it is absent**. The ledger
measured 9/9 in 0.17 s having never run it, against ~4 s for a real run.
`PMACS_REQUIRE_GPU` is the only thing that promotes that skip to a
failure, and `CLAUDE.md` applies that flag to `cargo test -p pmacs-gpu`,
a different package — so **the required local gate does not cover a37
either.** This is unchanged and re-confirmed at the grep above.
### 1.7 The local sweep is green, and what that does and does not mean
A full serialized sweep under CI's flags plus `crdt`
`cargo test --all-targets --no-default-features --features luajit,crdt -- --test-threads=1 --skip basedpyright`
— completed on this machine at `4223dd3`:
**104 suites reporting, 3,715 passed, 0 failed, 30 ignored.**
It **reconciles exactly** with §1.1's census, which is the check that
matters: 3,715 passed + 30 ignored + 1 `basedpyright` test filtered by
`--skip` = **3,746**, the crdt-flags census figure. No test binary was
silently absent from the sweep, and no suite reported `ok` for a
population smaller than the census predicted. A sweep that did not
reconcile would be the a37 vacuum at corpus scale.
Seven targets report `ok. 0 passed`. Three are helper binaries with no
tests (`pmacs_audit`, `pmacs_fake_lsp`, `pmacs_fake_mcp`) — expected.
**The other four make §1.3's asymmetry visible in the run itself:**
`m5_perf_acceptance` and `m6_perf_acceptance` report zero *and have
`--ignored` CI jobs*; `m10_2_perf` and `m10_11_perf` report identically
and **have none**. The same output line means "covered elsewhere" for two
of them and "covered nowhere" for the other two, which is exactly why
Q#CC4 cannot be answered by looking at a test run.
**What this establishes:** the 268 recoverable tests are not hiding
correctness defects. Bet 1 holds locally.
**What it does not establish, and the distinction is the lane's whole
risk:** the failures the ledger predicts are *hosted-runner timing and
concurrency* failures — real PTY behavior on CI runners, wgpu under
lavapipe, daemon sockets at unfamiliar concurrency. A green serialized
run on a 16-thread developer machine removes one class of explanation
(the tests are wrong) and leaves the class actually expected (the
environment differs) entirely untested. **This result must not be quoted
as evidence that the CI leg will be green.**
### 1.8 What is NOT established
- **No macOS reading exists.** Every measurement here is Linux. The
ledger's expectation that these suites will be flakier on hosted
runners than on a developer machine is an expectation, not a
measurement.
- **This machine is a laptop with an integrated Radeon (Phoenix1), 16
threads.** It is a real Vulkan device rather than lavapipe, but it is
shared-memory and thermally constrained, which is precisely the
condition under which the ledger records a37 producing
`last_frame_text` all spaces with nonzero `rendered_nonuniform_frames`.
**A red a37 here is ambiguous by construction.** Universum (7900 XTX,
available remotely) is where that ambiguity gets settled — and settling
it there proves the *test* is sound, not that the *CI job* will pass,
because the CI job runs on lavapipe regardless.
---
### 1.9 The census now has a tool
`docs/active-work.md` says the dark-test figure "moves with every merge
and must be re-measured, not quoted." **That instruction had no tool**,
so every re-measurement was a hand-rolled `--list` pipeline written from
scratch — including this lane's, whose first attempt filtered on
`/ : test$/` when libtest prints `name: test` with no space before the
colon. It matched nothing, reported zero targets, and looked like a
clean run.
`scripts/feature-census <config-a> <config-b> [--covers <test>]` is that
tool. It reproduces every figure in §1.1 and §1.2 independently, and its
header records each parsing trap, all of which were hit while writing
it. Two are worth repeating here because they change results rather than
merely breaking a run:
- **A target with zero tests prints its `Running` line and nothing
else.** Counting only test lines drops it from the diff entirely —
losing precisely the finding worth surfacing. This is how the §1.1
error above survived two revisions.
- **Both configurations need an `--ignored` pass, not just the richer
one.** Counting only B's ignored set attributes pre-existing ignores
to the feature: `rope::tests::perf_smoke_*` are ignored under both and
are not "dark and ignored." The difference is what turns a flat
"279 dark" into §1.2's "268 recovered, 11 needing `--ignored`."
It is **fail-closed on a build failure** (exit 3). A configuration that
does not compile yields no test list, which is indistinguishable by
counting from "this configuration has no tests" and would render as an
entirely false "every test is dark." That is not a small error; it is a
number that would get quoted.
## 2. Questions
- **Q#CC1 — does the `crdt` leg go on the existing `test` matrix or its
own job?** The matrix is already 4 legs (2 OS × 2 Lua flavors) and is
the CI critical path at an observed 17-minute max. A `crdt` leg on all
four doubles the most expensive job. A separate ubuntu-only job costs
one leg. **Leaning: separate job, ubuntu-only, per the ledger's "start
ubuntu-only and decide about macOS from evidence."**
- **Q#CC2 — is `crdt` matrixed over Lua flavor?** The feature is
orthogonal to the Lua VM. `m5-perf-gates` and `m6-perf-gates` already
set the precedent of luajit-only with a written justification.
**Leaning: luajit only, with the reason recorded in the workflow the
way the existing perf jobs do.**
- **Q#CC3 — where does `m10_10_perf` go? DECIDED: it stays unignored and
runs in the plain `crdt` leg.** The decision was first taken as "give
it a perf job," on revision 1's classification; §1.3a then established
that classification was wrong — the suite is a deliberate CI-default
regression tripwire with generous bounds, not a bench, and adding
`#[ignore]` would reduce coverage inside a coverage lane. **The
decision was reversed on the evidence, not on preference**, and it is
a one-line change to reinstate the original answer if the tripwire
reading is rejected.
- **Q#CC4 — do `m10_2_perf` and `m10_11_perf` get a job in this lane, or
a follow-on? DECIDED: in this lane.** They are dark for a second,
independent reason (unreferenced by any workflow) that predates the
`crdt` gap, so this PR does fix two causes at once. That is accepted
deliberately: leaving them would ship a lane headlined "the dark tests
now run" with 7 still dark, and the second cause is one job block, not
a second investigation. They go in a new `m10-perf-gates` job shaped
like the existing `m5-perf-gates` / `m6-perf-gates`.
- **Q#CC5 — what proves each GPU suite actually executed?** §1.5
establishes `PMACS_REQUIRE_GPU` covers only two of the four. Does the
lane add the guard to the other two, or assert execution some other
way (a test-count floor per suite)?
- **Q#CC6 — do the clippy fixes ride this PR or precede it?** Eight
lint-policy findings across four files, none behavioral. Riding along
means the PR that adds CI coverage also touches `src/daemon.rs`.
**Leaning: a separate preparatory commit on the same branch**, so the
diff reads as two intents, but not a separate PR — the lints are
unreachable-by-CI today and have no independent reason to be fixed.
- **Q#CC7 — is a red first run acceptable to merge behind? DECIDED:
fix in-lane by default, with one bounded escape hatch (below).**
First, one of the three options offered in revision 1 was not real.
**"Land the leg non-required until it is green" assumed the first CI
run happens after merge. It does not** — the new job runs on the pull
request, so the entire first-run failure set is visible *during
review*, before anything reaches `main`. There is no window in which
`main` carries a red required check, and therefore nothing to protect
against by landing non-required. The genuine choice is only
fix-in-lane versus quarantine-with-follow-on, taken per failure once
observed.
**Default: fix in-lane.** The escape hatch is keyed on *cause class*,
not on effort, because effort is what turns a scoped lane into an
unbounded one:
1. **The lane's own configuration** (wrong flags, missing build step,
a suite that skipped vacuously) — **always in-lane.** It is this
PR's defect.
2. **Reproduces locally** under §1.7's sweep or on Universum —
**in-lane.** A real test defect the lane surfaced is the lane
working as intended.
3. **Environment-only** — green serialized here, green on Universum,
red only on a hosted runner — **named follow-on lane.** This is the
class that historically consumed multiple review rounds on this
project: the signal lane had three tolerance rules rejected across
three revisions, each concluding something about a process from
something that was not about that process. Committing in advance to
resolve that class *inside a workflow-configuration PR* is
committing to an unbounded investigation in a lane whose whole
value is being small and mergeable.
**§1.7 makes class 3 the most likely red and class 2 the least**, which
is the uncomfortable direction: the corpus is already green serialized
on a developer machine, so a CI red is by elimination an environment
difference. The escape hatch exists precisely because the measurement
points that way.
---
## 3. Bets
- **Bet 1 — the 268 recoverable tests mostly pass. RESOLVED: they all
do, locally.** The serialized sweep is 3,715/3,715 with an exact
census reconciliation (§1.7). This bet is settled for the developer
machine and **explicitly not settled for CI**, which is the
environment the lane is actually changing.
- **Bet 2 — the failures that do appear concentrate in the process/PTY
and GPU suites**, not in the library. The library's 185 are pure logic
over a CRDT backend; the flake surface the ledger documents is
uniformly real-process and real-device. *Unresolved: with zero local
failures there is nothing yet to concentrate. This bet now resolves
only on the first CI run, and is the reason §7 sequences the GPU
commit last.*
- **Bet 3 — the eight clippy findings are the complete blocker.** With
`--keep-going` there is no remaining truncation, so no further lint
surprises appear once these are cleared. *Falsified if clearing them
reveals findings in targets that failed to build for a non-lint
reason.*
- **Bet 4 — `PMACS_REQUIRE_GPU` alone will not prove the GPU suites ran.**
§1.5 already all but establishes this; the bet is that a per-suite
execution assertion is needed and that adding it finds at least one
suite silently skipping.
---
## 4. Acceptance
Written against §1's measurements, with the local sweep (§1.7) in hand.
Criteria 8 and 9 exist specifically because that sweep came back green:
a green pre-measurement is the condition under which a vacuous CI job is
easiest to ship unnoticed.
1. `cargo clippy --workspace --all-targets --features crdt -- -D warnings`
exits zero on the branch, and the check is run **with `--keep-going`**
so its success is an inventory rather than a first-failure abort.
2. The dark census is **re-measured on the branch** and the workflow runs
a leg that compiles them. **275 of 279 recovered**, with the four
exclusions named individually (§1.2a) — never one headline number
with an unexplained remainder.
2a. `m10_10_perf` is **unmodified** by this lane: no `#[ignore]` added,
no job of its own, its 4 tests recovered by the plain leg. A diff
touching `tests/m10_10_perf.rs` fails this criterion (§1.3a).
3. Every GPU-requiring suite added to a job **proves it executed**, per
Q#CC5 — a suite that skips its body reports failure, not `ok`.
`PMACS_REQUIRE_GPU` alone does not satisfy this, because it is absent
from two of the four suites (§1.5).
4. `a37` specifically: a run that does not build `pmacs-gpu` **fails**
rather than reporting 9/9 in 0.17 s.
5. The perf suites' placement is explicit and each is justified in the
workflow text: `m10_2_perf` and `m10_11_perf` in a new
`m10-perf-gates` job matching the `m5`/`m6` precedent;
`m10_10_perf` deliberately in the correctness leg, **with its
generous-bounds rationale written into the workflow comment** so a
later reader does not "fix" the inconsistency by ignoring it.
6. The new job carries `timeout-minutes`, per the workflow's own standing
rule that every job does.
7. `docs/active-work.md`'s "NEEDS A LANE" block is replaced by this
lane's state, and its stale figures (273 dark, the seven-item clippy
list) are corrected rather than left beside the new ones.
8. **The new job's test count reconciles.** Its reported passed +
ignored + filtered must equal the branch's re-measured crdt census
for the suites it runs, the way §1.7 reconciles to 3,746. A job that
runs fewer tests than the census predicts has found a silently-absent
binary, and that is the defect class this whole lane exists to end.
9. **The leg is proven load-bearing, structurally rather than by
breakage.** *Revised in revision 3 — the original criterion asked for
a deliberately-broken test, and it is worth saying plainly why that
was dropped rather than quietly meeting a weaker bar.*
The original: break a `crdt`-gated library test, confirm the job goes
red, revert. It proves two things at once — that the leg compiles
crdt tests, and that a failure propagates to a red job. The second is
generic cargo/libtest behavior, not anything this lane changes, and
buying it costs a mutation of tracked source restored by a shell
trap, plus a broken commit in the PR's history.
What replaces it, in two parts:
- **Structural, and stronger on the point that matters:**
`scripts/feature-census luajit luajit,crdt --covers <test>` asserts
a named test is present under the new job's flags and **absent
under the old job's**. That is a claim the break test does *not*
make — a red job proves the leg caught something, but not that the
existing job could never have. Verified for
`crdt_apply_edit_keeps_invariant_basic`: present under B, absent
under A, exit 0. The negative cases are exercised too (a test in
both configs, and a misspelled name, both exit 1).
- **Empirical, from the first CI run:** criterion 8's count
reconciliation. A job reporting ~3,715 passed has demonstrably
compiled and run the crdt corpus; a vacuous leg reports a number
near the `test` job's.
**What is no longer proven, stated rather than glossed:** that a
*failing* crdt test turns this specific job red. Nothing on this
branch demonstrates it. It rests on cargo returning non-zero on test
failure and GitHub Actions failing a step on non-zero — both
universal, neither lane-specific. If review wants that proven
directly, the break test is still the way, and it belongs on a
throwaway PR rather than in this one's history.
---
## 5. Parked
- **macOS.** Ubuntu-only first, by the ledger's own instruction. A macOS
`crdt` leg is a follow-on decided from the first run's evidence.
- **The `--lib --features crdt` flake.** `process::tests::setsid_escapee_is_not_reaped_and_teardown_reclaims_readers`
failing ~1 in 5 under parallel full-suite load, with
`active_reader_probe` returning `None`. The ledger's leading
explanation — `drain_until`'s tick reaping the leader before the probe
— is an inference from control flow, **not a falsified root cause**, and
no serial full-suite bite has been run to separate parallelism from
another whole-suite effect. Discriminating it is named as belonging to
this lane; it should be its own PR, because it is a product defect
hypothesis and everything else here is workflow configuration.
- **The two unattributed CRDT failures from #178's round-2 gating.** No
test names were captured, so there is nothing to reproduce.
- **`basedpyright`.** Still hangs forever, still `--skip`ped, still
deliberately not installed in CI. Unchanged by this lane.
---
## 6. Gates
The standing suite from `CLAUDE.md`, plus the two this lane exists to
make meaningful:
- `cargo fmt --check`
- `cargo clippy --workspace --all-targets -- -D warnings`
- **`cargo clippy --workspace --all-targets --features crdt --keep-going -- -D warnings`** (new, and the lane's own subject)
- **`scripts/feature-census luajit luajit,crdt`** (new; re-measures the
census rather than quoting it forward, per the ledger's own rule)
- `cargo test --lib`
- `cargo test --lib --features crdt`
- the touched acceptance suites
- `cargo test --test m4_acceptance -- --skip basedpyright`
- `PMACS_REQUIRE_GPU=1 cargo test -p pmacs-gpu`
- `git diff --check`
Verified green on this machine at `4223dd3` **before any change**: fmt,
diff-check, `--lib` (1,896 passed), `--lib --features crdt` (2,081
passed, 4 ignored), required GPU (221 passed), and the full serialized
crdt sweep (3,715 passed, 0 failed, 30 ignored, reconciling exactly to
the 3,746 census — §1.7).
Recording the pre-change baseline matters here more than usual: this
lane's subject *is* the test corpus, so without a baseline any red on
the branch is unattributable between the lane's changes and the tests it
newly compiles.
---
## 7. Branch plan
One branch, `ci-crdt-coverage`. Commits in this order, because each
earlier one is a precondition for the next being observable. **As
landed:**
1. **Clear the eight clippy findings** (§1.4) — `7a9cf5b`. Nothing can
compile the `crdt` targets under `-D warnings` until this lands, so
no CI change was testable before it.
2. **Add the `m10-perf-gates` job** for `m10_2_perf` and `m10_11_perf`
(Q#CC4) — `06abbac`. **No source file changes**: per Q#CC3 and
§1.3a, `m10_10_perf` is not touched and gets no `#[ignore]`.
3. **Add the `crdt-test` job**`7a8746d`. Recovers 268 tests,
`m10_10_perf`'s 4 among them.
4. **Give the census a tool**`a776bc3`, `scripts/feature-census`
(§1.9). This replaced the planned break-test as the non-vacuity
proof; see acceptance 9 for what was traded away and why.
5. **Update `docs/active-work.md` and `docs/agent-handoff.md`** per
acceptance 7.
**Two departures from the plan as approved, both deliberate:**
- **The GPU suites did not get a separate commit against `gpu-render`.**
The plan's step 5 assumed they could be added to that job; §1.5
established it runs a different package, so co-locating them there
would have meant a new invocation rather than an extension, plus a
standing requirement to classify every future suite as GPU-requiring
or not. They are covered by `crdt-test` instead, which installs
lavapipe and sets `PMACS_REQUIRE_GPU=1` for the whole corpus. **One
job cannot develop the hole that splitting invites.**
- **The break-test became a coverage assertion plus a tool** (acceptance
9). The proof got stronger on the claim specific to this lane — that
the old job *structurally cannot* see these tests — and weaker on a
generic one it no longer makes.
**Universum is still where the GPU question settles.** Everything above
was verified on this laptop, whose integrated Radeon renders the GPU
suites but is thermally constrained, so a red `a37` here is ambiguous by
construction (§1.8). What Universum's 7900 XTX can establish is that the
tests are sound; it cannot establish that the lavapipe CI job will pass,
and the first PR run is the only thing that can. Budget for that gap
rather than assuming it away.

View File

@ -2296,3 +2296,79 @@ pub enum InitialTargetResult {
message: String, message: String,
}, },
} }
#[cfg(test)]
mod capability_default_tests {
use super::{FrontendCapabilities, InstanceCapabilities};
// These exist because running code is not testing it.
//
// The `crdt` feature reaches this crate for exactly one purpose:
// `InstanceCapabilities::default` reads `cfg!(feature = "crdt")`,
// so the same source produces different advertised capabilities in
// the two builds. Until these tests, the only thing exercising that
// default was a transport round-trip, and a round-trip is INVARIANT
// TO THE VALUES — an all-false default, or one where the three
// fields disagreed with each other, encodes and decodes just as
// happily and passes in both configurations.
//
// So the crate was being COMPILED both ways without either set of
// values being asserted. That is the same defect class this lane
// exists to close, one level down: the CI step that runs
// pmacs-protocol under `crdt` executed this code but checked
// nothing about it.
/// The instance advertises CRDT capability exactly when it was
/// built with the feature. Advertising `true` on a non-CRDT build
/// would be wire-protocol false advertising — the CRDT paths are
/// conditionally compiled out — and advertising `false` on a CRDT
/// build would strand every frontend in single-frontend mode.
#[cfg(feature = "crdt")]
#[test]
fn instance_capability_defaults_are_enabled_under_crdt() {
let caps = InstanceCapabilities::default();
assert!(caps.multi_frontend, "multi_frontend must default true");
assert!(caps.crdt_replica, "crdt_replica must default true");
assert!(caps.semantic_render, "semantic_render must default true");
}
/// The non-CRDT counterpart. All three track one `cfg!`, so a
/// change that flipped only some of them would leave the daemon
/// advertising a capability whose code paths are compiled out.
#[cfg(not(feature = "crdt"))]
#[test]
fn instance_capability_defaults_are_disabled_without_crdt() {
let caps = InstanceCapabilities::default();
assert!(!caps.multi_frontend, "multi_frontend must default false");
assert!(!caps.crdt_replica, "crdt_replica must default false");
assert!(
!caps.semantic_render,
"semantic_render must default false; a semantic session is \
necessarily a text replica and a non-CRDT build hosts neither"
);
}
/// **Deliberately NOT feature-gated** — this asserts the same thing
/// in both builds, which is the point.
///
/// `FrontendCapabilities` derives `Default`, so it is
/// feature-INVARIANT while `InstanceCapabilities` is
/// feature-DEPENDENT. That asymmetry is load-bearing rather than an
/// oversight: an instance advertises what it can do, while a
/// frontend OPTS IN through the negotiation handshake, and a v1
/// frontend has no local CRDT state regardless of how the crate it
/// links was compiled. Making this one track the feature would have
/// frontends claiming support they do not have.
#[test]
fn frontend_capability_defaults_do_not_track_the_crdt_feature() {
let caps = FrontendCapabilities::default();
assert!(
!caps.multi_frontend,
"frontend multi_frontend must default false in BOTH builds"
);
assert!(
!caps.crdt_replica,
"frontend crdt_replica must default false in BOTH builds"
);
}
}

242
scripts/feature-census Executable file
View File

@ -0,0 +1,242 @@
#!/bin/sh
# scripts/feature-census --- measure which tests a feature flag HIDES.
#
# scripts/feature-census <config-a> <config-b> [--covers <test-name>]
#
# Both arguments are `--features` strings, applied with
# `--no-default-features`. Config A is the baseline (typically CI's
# flags); config B is the richer one. The script reports every test that
# exists under B and NOT under A --- the "dark" set.
#
# scripts/feature-census luajit luajit,crdt
# scripts/feature-census luajit luajit,crdt --covers crdt_undo_keeps_invariant
#
# WHY THIS EXISTS. A cargo feature is a COMPILE-TIME gate, so a test
# behind one is not skipped or filtered --- it does not exist. Nothing in
# a test run says so: the suite reports `ok`, the counts look plausible,
# and the absent tests are absent from the report too. `.github/
# workflows/ci.yml` never enabled `crdt` for the project's whole life,
# and 279 tests --- including 186 library tests behind a REQUIRED
# CLAUDE.md gate --- had never executed in CI. Nobody was ignoring a red
# signal; there was no signal.
#
# docs/active-work.md therefore says the figure "moves with every merge
# and must be re-measured, not quoted." That instruction had no tool, so
# every re-measurement was a hand-rolled `--list` pipeline. This is that
# tool.
#
# EXIT STATUS: 0 when the census completes (and, with --covers, the
# claim holds); 1 when a --covers claim FAILS; 2 on usage errors; 3 when
# either configuration FAILS TO BUILD.
#
# Exit 3 is the important one, and it is fail-closed on purpose. A
# configuration that does not compile produces no test list, which is
# indistinguishable by counting from "this configuration contains no
# tests" --- and would render as a spectacular, entirely false, "every
# test is dark under A." A census whose baseline did not build is not a
# small error; it is a number that will be quoted.
#
# PARSING NOTES, each one a mistake made while writing this:
#
# * libtest prints `some::path::name: test` with NO SPACE before the
# colon. A filter written as `/ : test$/` matches nothing, reports
# zero targets, and looks like a clean run. Cost: one silently empty
# census. The pattern below is anchored to end-of-line instead.
#
# * `--list` also emits `: benchmark` lines. Only `: test` is counted.
#
# * Target attribution comes from cargo's `Running` lines, which have
# TWO shapes: `Running unittests src/lib.rs (...)` for a lib/bin
# target and `Running tests/foo.rs (...)` for an integration test.
# Reading a fixed field number handles one and mangles the other.
#
# * A target with zero tests still prints its `Running` line and then
# nothing. Counting only test lines DROPS such targets entirely, so
# they never appear in the diff --- which is precisely the finding
# worth surfacing (eight binaries reported `ok` with nothing in
# them). They are tracked from the `Running` line, not inferred from
# having tests.
#
# * `--list` includes #[ignore]d tests. They are dark in the same
# sense but are NOT recovered by adding the feature to an ordinary
# test job --- that needs `--ignored`. Conflating the two overstates
# what a fix delivers, so ignored tests are counted separately via a
# second `--list --ignored` pass.
#
# SCOPE LIMIT, and it is structural rather than an omission: this
# censuses the workspace DEFAULT MEMBER only (`pmacs`), because that is
# what a bare `cargo test --all-targets` builds. Sibling crates
# --- pmacs-protocol, pmacs-gpu --- are invisible to it no matter what
# configs are passed.
#
# That blind spot is not hypothetical. pmacs-protocol has its own
# `crdt` feature which gates NO tests, so a census of it would report
# zero dark either way --- yet the feature changes
# `cfg!(feature = "crdt")` expressions inside the capability defaults,
# so its 17 tests exercise different runtime values under it. A feature
# can therefore matter to a crate this script would score as
# unaffected. Check sibling crates by hand.
#
# CARGO_TERM_COLOR is pinned off so escape codes cannot break the
# anchored matches, the same precaution and for the same reason as
# scripts/bite.
set -eu
usage() {
echo "usage: scripts/feature-census <config-a> <config-b> [--covers <test-name>]" >&2
exit 2
}
[ "$#" -ge 2 ] || usage
config_a=$1
config_b=$2
shift 2
covers=""
while [ "$#" -gt 0 ]; do
case $1 in
--covers)
[ "$#" -ge 2 ] || usage
covers=$2
shift 2
;;
*) usage ;;
esac
done
export CARGO_TERM_COLOR=never
work=$(mktemp -d "${TMPDIR:-/tmp}/feature-census.XXXXXX")
trap 'rm -rf -- "$work"' EXIT INT TERM
# Emit "<count> <target>" per target. Targets with zero tests are
# emitted with a count of 0 rather than omitted.
parse_list() {
awk '
/^ *Running /{
t = $2
if (t == "unittests") t = $3
sub(/.*\//, "", t)
sub(/\.rs$/, "", t)
seen[t] = 1
next
}
/: test$/ { c[t]++ }
END { for (k in seen) printf "%d %s\n", c[k] + 0, k }
' | sort -k2
}
# $1 = features, $2 = output basename, $3... = extra libtest args
run_list() {
features=$1
out=$2
shift 2
if ! cargo test --all-targets --no-default-features --features "$features" \
-- --list "$@" > "$work/$out.raw" 2>&1; then
echo "feature-census: CONFIGURATION FAILED TO BUILD: --features $features" >&2
echo "feature-census: refusing to report a census against a config that does not" >&2
echo "feature-census: compile --- every test would read as absent. Last output:" >&2
tail -n 20 "$work/$out.raw" >&2
exit 3
fi
parse_list < "$work/$out.raw" > "$work/$out.counts"
}
echo "feature-census: listing A (--features $config_a)" >&2
run_list "$config_a" a
echo "feature-census: listing B (--features $config_b)" >&2
run_list "$config_b" b
# BOTH sides need an ignored pass. A test that is #[ignore]d under A as
# well as B is not "dark and ignored" --- it was already there. Counting
# only B's ignored set attributes pre-existing ignores to the feature
# and overstates the untouchable remainder.
echo "feature-census: listing A ignored" >&2
run_list "$config_a" a_ignored --ignored
echo "feature-census: listing B ignored" >&2
run_list "$config_b" b_ignored --ignored
total_a=$(awk '{s+=$1} END {print s+0}' "$work/a.counts")
total_b=$(awk '{s+=$1} END {print s+0}' "$work/b.counts")
targets_a=$(wc -l < "$work/a.counts" | tr -d ' ')
targets_b=$(wc -l < "$work/b.counts" | tr -d ' ')
echo
echo " config A: --features $config_a"
echo " config B: --features $config_b"
echo
# One pass over all four count files, keyed by target, so every derived
# figure comes from the same table rather than a chain of joins whose
# defaulting rules have to agree.
awk '
FILENAME ~ /a\.counts$/ { a[$2] = $1; seen[$2] = 1; next }
FILENAME ~ /b\.counts$/ { b[$2] = $1; seen[$2] = 1; next }
FILENAME ~ /a_ignored\.counts$/{ ai[$2] = $1; next }
FILENAME ~ /b_ignored\.counts$/{ bi[$2] = $1; next }
END {
for (t in seen) {
d = b[t] - a[t]
if (d <= 0) continue
# A dark test is ignored only if B ignores MORE than A does.
di = bi[t] - ai[t]
if (di < 0) di = 0
if (di > d) di = d
printf "row %6d %6d %6d %6d %s\n", d, a[t], b[t], di, t
dark += d; darkt++; darkig += di
}
printf "sum %d %d %d\n", dark + 0, darkt + 0, darkig + 0
}
' "$work/a.counts" "$work/b.counts" "$work/a_ignored.counts" "$work/b_ignored.counts" \
> "$work/table"
awk '$1 == "row" { printf "%6d dark | %6d A | %6d B | %s%s\n", $2, $3, $4, ($5 > 0 ? "(" $5 " ignored) " : ""), $6 }' \
"$work/table" | sort -rn
read_sum() { awk -v f="$1" '$1 == "sum" { print $(f + 1) }' "$work/table"; }
dark_total=$(read_sum 1)
dark_targets=$(read_sum 2)
dark_ignored=$(read_sum 3)
# Targets with zero tests under A. These build, run, and report `ok`
# with nothing in them. Split by whether B gives them any, because the
# two mean different things: one is a target the feature unlocks, the
# other is a binary that simply has no tests at all.
empty_a=$(awk '$1 == 0 { n++ } END { print n+0 }' "$work/a.counts")
empty_a_filled_by_b=$(awk '
FILENAME ~ /a\.counts$/ { a[$2] = $1; next }
FILENAME ~ /b\.counts$/ { if (a[$2] == 0 && $1 > 0) n++ }
END { print n+0 }
' "$work/a.counts" "$work/b.counts")
echo
echo " A: $total_a tests across $targets_a targets"
echo " B: $total_b tests across $targets_b targets"
echo " DARK: $dark_total tests across $dark_targets targets"
echo " of which #[ignore]d under B but not A: $dark_ignored"
echo " recovered by adding B's features to a plain test job: $((dark_total - dark_ignored))"
echo " (the remainder needs an --ignored invocation, not just the feature)"
echo
echo " $empty_a target(s) run under A with ZERO tests --- they report ok with nothing in them."
echo " $empty_a_filled_by_b of those gain tests under B; the rest have no tests in either."
if [ -n "$covers" ]; then
echo
in_b=no
in_a=no
grep -q "^${covers}: test$\|::${covers}: test$" "$work/b.raw" && in_b=yes
grep -q "^${covers}: test$\|::${covers}: test$" "$work/a.raw" && in_a=yes
echo " --covers $covers"
echo " present under B: $in_b"
echo " present under A: $in_a"
if [ "$in_b" = yes ] && [ "$in_a" = no ]; then
echo " => B-only. A configuration running only A cannot see this test."
elif [ "$in_b" = no ]; then
echo " => CLAIM FAILED: not present under B at all (misspelled?)." >&2
exit 1
else
echo " => CLAIM FAILED: present under A too, so B is not load-bearing for it." >&2
exit 1
fi
fi

View File

@ -4461,8 +4461,7 @@ mod tests {
.expect("export snapshot") .expect("export snapshot")
}; };
let peer = loro::LoroDoc::new(); let peer = loro::LoroDoc::new();
peer.set_peer_id(u64::from(FrontendId::LOCAL.0)) peer.set_peer_id(FrontendId::LOCAL.0).expect("set peer id");
.expect("set peer id");
peer.import(&snapshot_bytes).expect("import snapshot"); peer.import(&snapshot_bytes).expect("import snapshot");
let v_before = peer.oplog_vv(); let v_before = peer.oplog_vv();
peer.get_text("body").insert(0, "x").expect("peer insert"); peer.get_text("body").insert(0, "x").expect("peer insert");
@ -4541,13 +4540,17 @@ mod tests {
} }
/// Kill ring Q#KR2 — GPU typing arrives here without touching /// Kill ring Q#KR2 — GPU typing arrives here without touching
/// dispatch_key, so it must update the source frontend's command /// `dispatch_key`, so it must update the source frontend's command
/// boundary or `C-k x C-k` on the GPU would append across the typed /// boundary or `C-k x C-k` on the GPU would append across the typed
/// character. A single-codepoint insert classifies as /// character. A single-codepoint insert classifies as
/// `buffer.self-insert` (the input-origin signal for signature /// `buffer.self-insert` (the input-origin signal for signature
/// help); anything else breaks the chain outright. /// help); anything else breaks the chain outright.
#[cfg(feature = "crdt")] #[cfg(feature = "crdt")]
#[test] #[test]
#[allow(
clippy::too_many_lines,
reason = "one end-to-end classification scenario per kill-chain case"
)]
fn handle_remote_crdt_op_classifies_typed_input_and_ends_kill_chains() { fn handle_remote_crdt_op_classifies_typed_input_and_ends_kill_chains() {
use crate::editor::EditorState; use crate::editor::EditorState;
use crate::protocol::FrontendId; use crate::protocol::FrontendId;

View File

@ -39,7 +39,7 @@ fn read_initial_snapshot(
} }
/// Mutate the local replica, export the delta, and ship it as an /// Mutate the local replica, export the delta, and ship it as an
/// optimistic `FrontendEvent::CrdtOp` (the m10_11 idiom). /// optimistic `FrontendEvent::CrdtOp` (the `m10_11` idiom).
fn send_optimistic_op_from<F>( fn send_optimistic_op_from<F>(
stream: &mut std::os::unix::net::UnixStream, stream: &mut std::os::unix::net::UnixStream,
replica: &CrdtState, replica: &CrdtState,

View File

@ -506,6 +506,10 @@ fn decode_hex(encoded: &str) -> String {
/// out its safety deadline cannot read as a pass. /// out its safety deadline cannot read as a pass.
#[cfg(feature = "crdt")] #[cfg(feature = "crdt")]
#[test] #[test]
#[allow(
clippy::too_many_lines,
reason = "one real-daemon/real-PTY/real-wgpu scenario; splitting it would hide the fit it exists to prove"
)]
fn a54_real_daemon_real_pty_and_headless_gpu_render_one_panel_hosted_terminal() { fn a54_real_daemon_real_pty_and_headless_gpu_render_one_panel_hosted_terminal() {
use std::path::{Path, PathBuf}; use std::path::{Path, PathBuf};

View File

@ -634,6 +634,10 @@ pmacs.keymap.bind { scope = "global", sequence = "C-M-t", command = "vterm-probe
/// prove nothing about the three fitting together. /// prove nothing about the three fitting together.
#[cfg(feature = "crdt")] #[cfg(feature = "crdt")]
#[test] #[test]
#[allow(
clippy::too_many_lines,
reason = "one real-daemon/real-PTY/real-wgpu scenario; a decoded-message fixture is deliberately not a substitute"
)]
fn a37_real_daemon_real_pty_and_headless_gpu_render_one_terminal_session() { fn a37_real_daemon_real_pty_and_headless_gpu_render_one_terminal_session() {
use std::path::{Path, PathBuf}; use std::path::{Path, PathBuf};
@ -813,6 +817,10 @@ fn a37_real_daemon_real_pty_and_headless_gpu_render_one_terminal_session() {
/// property of the dispatcher loop, not of any function it calls. /// property of the dispatcher loop, not of any function it calls.
#[cfg(feature = "crdt")] #[cfg(feature = "crdt")]
#[test] #[test]
#[allow(
clippy::too_many_lines,
reason = "real daemon, real wire, two real frontends in one dispatcher-loop scenario"
)]
fn terminal_mode_keeps_reporting_presence_so_peers_drop_the_stale_caret() { fn terminal_mode_keeps_reporting_presence_so_peers_drop_the_stale_caret() {
use pmacs::protocol::{ use pmacs::protocol::{
AttachRequest, FrontendCapabilities, Hello, Key, KeyEvent, PROTOCOL_VERSION, AttachRequest, FrontendCapabilities, Hello, Key, KeyEvent, PROTOCOL_VERSION,
@ -857,13 +865,10 @@ fn terminal_mode_keeps_reporting_presence_so_peers_drop_the_stale_caret() {
) -> T { ) -> T {
let deadline = Instant::now() + Duration::from_secs(20); let deadline = Instant::now() + Duration::from_secs(20);
while Instant::now() < deadline { while Instant::now() < deadline {
match read_message::<InstanceMessage>(stream) { if let Ok(msg) = read_message::<InstanceMessage>(stream)
Ok(msg) => { && let Some(found) = want(&msg)
if let Some(found) = want(&msg) { {
return found; return found;
}
}
Err(_) => continue,
} }
} }
panic!("timed out waiting for {what}"); panic!("timed out waiting for {what}");