From c6066926782a70685c49a9a5a49e95a28e025455 Mon Sep 17 00:00:00 2001 From: Sadig Akhund Date: Sat, 12 Sep 2026 16:30:56 +0400 Subject: [PATCH 1/9] [S-162] drop a leading option-rejection diagnostic before layout analysis fuser, Xvfb and nfsidmap print one diagnostic line then their real document; it no longer fuses into the root description or a usage label. New corpus fixtures: fuser, Xvfb (partial, still xfail). Co-Authored-By: Claude Fable 5.1 --- CHANGELOG.md | 1 + corpus/README.md | 18 +++ corpus/Xvfb/audit-seed/help.stderr.txt | 107 +++++++++++++ corpus/Xvfb/audit-seed/help.stdout.txt | 0 corpus/Xvfb/audit-seed/meta.toml | 32 ++++ corpus/fuser/audit-seed/expected.snap | 144 ++++++++++++++++++ corpus/fuser/audit-seed/help.stderr.txt | 26 ++++ corpus/fuser/audit-seed/help.stdout.txt | 0 corpus/fuser/audit-seed/meta.toml | 29 ++++ corpus/lsof/4.95.0/expected.snap | 2 +- corpus/nfsidmap/audit-seed/expected.snap | 2 +- corpus/nfsidmap/audit-seed2/expected.snap | 2 +- corpus/wpa_cli/audit-seed/expected.snap | 2 +- docs/shapes.md | 26 ++++ .../src/help_text/sections/mod.rs | 16 +- .../src/help_text/sections/preamble.rs | 48 ++++-- .../src/help_text/sections/usage.rs | 19 +++ xtask/src/corpus/contract.rs | 36 +++++ xtask/src/corpus/mod.rs | 14 ++ xtask/src/coverage/mod.rs | 1 + xtask/src/coverage/round10.rs | 20 +++ xtask/src/coverage/score.rs | 1 + xtask/src/detector/leading_diagnostic_line.rs | 142 +++++++++++++++++ xtask/src/detector/mod.rs | 2 + 24 files changed, 674 insertions(+), 16 deletions(-) create mode 100644 corpus/Xvfb/audit-seed/help.stderr.txt create mode 100644 corpus/Xvfb/audit-seed/help.stdout.txt create mode 100644 corpus/Xvfb/audit-seed/meta.toml create mode 100644 corpus/fuser/audit-seed/expected.snap create mode 100644 corpus/fuser/audit-seed/help.stderr.txt create mode 100644 corpus/fuser/audit-seed/help.stdout.txt create mode 100644 corpus/fuser/audit-seed/meta.toml create mode 100644 xtask/src/coverage/round10.rs create mode 100644 xtask/src/detector/leading_diagnostic_line.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index a6bbc073..8b8dd3e8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -33,6 +33,7 @@ once it reaches a published 0.1.0 release. - [S-145] A single-dash long option in a table with no double-dash row anywhere keeps its whole name and case now, so `mandible mksquashfs` and `mandible sqfstar` show `-pf`, `-ef`, `-Xhelp`, `-Xstrategy`, `-Xhc` and `-Xbcj` instead of splitting each one. - [S-145] A tab-separated value name on a single-dash long option survives the same repair now, so `mandible mksquashfs` and `mandible sqfstar` keep `-mem`, `-comp` and `-mkfs-time` with their placeholders instead of losing them. - [S-146] A heading with no indent step to its own rows, and a bare label above a nested option block, now name a group instead of leaving every row underneath ungrouped, so `mandible mksquashfs` shows its ten option headings and its five compressor names. +- [S-162] A leading option-rejection diagnostic no longer fuses into the root description, so `mandible fuser` and `mandible nfsidmap` show their real description and usage instead of the tool's own complaint about the probe. ## [0.7.0] - 2026-09-05 diff --git a/corpus/README.md b/corpus/README.md index 469c4832..286926b7 100644 --- a/corpus/README.md +++ b/corpus/README.md @@ -291,6 +291,24 @@ no root satisfies this vacuously, the same reasoning `must_not_contain_flags` uses. Dropping an entry is a weakening exactly as dropping a `must_not_contain_flags` entry is. +### Stating that the root description carries text it must not: `must_not_describe_root` + +`must_not_describe` only ever checks a *flag's* own description. Nothing +before this field could say the *root's* own `description` is +contaminated — `Xvfb`'s leading option-rejection diagnostic +(`Unrecognized option: --help`) used to fuse into the root description +alongside its whole eighty-row option table (docs/shapes.md S-162). + +```toml +must_not_describe_root = ["Unrecognized option"] +``` + +Every listed string is checked as a substring of `root.description`, +whitespace-collapsed to a single space on both sides, `must_describe`'s +own rule. `cargo xtask corpus` fails when any listed text is still +present, naming it. Satisfied vacuously by a tree with no root or no +description at all, the same reasoning `must_not_contain_flags` uses. + ### Stating that a flag group is not an invocation line: `must_not_contain_flag_group_prefixes` `must_not_contain_flags` and `must_not_contain_usage_text` say nothing diff --git a/corpus/Xvfb/audit-seed/help.stderr.txt b/corpus/Xvfb/audit-seed/help.stderr.txt new file mode 100644 index 00000000..a93cedc5 --- /dev/null +++ b/corpus/Xvfb/audit-seed/help.stderr.txt @@ -0,0 +1,107 @@ +Unrecognized option: --help +use: X [:] [option] +-a # default pointer acceleration (factor) +-ac disable access control restrictions +-audit int set audit trail level +-auth file select authorization file +-br create root window with black background ++bs enable any backing store support +-bs disable any backing store support ++byteswappedclients Allow clients with endianess different to that of the server +-byteswappedclients Prohibit clients with endianess different to that of the server +-c turns off key-click +c # key-click volume (0-100) +-cc int default color visual class +-nocursor disable the cursor +-core generate core dump on fatal error +-displayfd fd file descriptor to write display number to when ready to connect +-dpi int screen resolution in dots per inch +-dpms disables VESA DPMS monitor control +-deferglyphs [none|all|16] defer loading of [no|all|16-bit] glyphs +-f # bell base (0-100) +-fakescreenfps # fake screen default fps (1-600) +-fp string default font path +-help prints message with these options ++iglx Allow creating indirect GLX contexts +-iglx Prohibit creating indirect GLX contexts (default) +-I ignore all remaining arguments +-ld int limit data space to N Kb +-lf int limit number of open files to N +-ls int limit stack space to N Kb +-nolock disable the locking mechanism +-maxclients n set maximum number of clients (power of two) +-nolisten string don't listen on protocol +-listen string listen on protocol +-noreset don't reset after last client exists +-background [none] create root window with no background +-reset reset after last client exists +-p # screen-saver pattern duration (minutes) +-pn accept failure to listen on all ports +-nopn reject failure to listen on all ports +-r turns off auto-repeat +r turns on auto-repeat +-render [default|mono|gray|color] set render color alloc policy +-retro start with classic stipple and cursor +-s # screen-saver timeout (minutes) +-seat string seat to run on +-t # default pointer threshold (pixels/t) +-terminate [delay] terminate at server reset (optional delay in sec) +-tst disable testing extensions +ttyxx server started from init on /dev/ttyxx +v video blanking for screen-saver +-v screen-saver without video blanking +-wr create root window with white background +-maxbigreqsize set maximal bigrequest size ++xinerama Enable XINERAMA extension +-xinerama Disable XINERAMA extension +-dumbSched Disable smart scheduling and threaded input, enable old behavior +-schedInterval int Set scheduler interval in msec +-sigstop Enable SIGSTOP based startup ++extension name Enable extension +-extension name Disable extension + Only the following extensions can be run-time enabled/disabled: + Generic Event Extension + MIT-SHM + XTEST + SECURITY + XINERAMA + XFIXES + RENDER + RANDR + COMPOSITE + DAMAGE + MIT-SCREEN-SAVER + DOUBLE-BUFFER + RECORD + DPMS + X-Resource + XVideo + XVideo-MotionCompensation + SELinux + GLX +-query host-name contact named host for XDMCP +-broadcast broadcast for XDMCP +-multicast [addr [hops]] IPv6 multicast for XDMCP +-indirect host-name contact named host for indirect XDMCP +-port port-num UDP port number to send messages to +-from local-address specify the local address to connect from +-once Terminate server after one session +-class display-class specify display class to send in manage +-cookie xdm-auth-bits specify the magic cookie for XDMCP +-displayID display-id manufacturer display ID for request +[+-]accessx [ timeout [ timeout_mask [ feedback [ options_mask] ] ] ] + enable/disable accessx key sequences +-ardelay set XKB autorepeat delay +-arinterval set XKB autorepeat interval +-screen scrn WxHxD set screen's width, height, depth +-pixdepths list-of-int support given pixmap depths ++/-render turn on/off RENDER extension support(default on) +-linebias n adjust thin line pixelization +-blackpixel n pixel value for black +-whitepixel n pixel value for white +-fbdir directory put framebuffers in mmap'ed files in directory +-shmem put framebuffers in shared memory +(EE) +Fatal server error: +(EE) Unrecognized option: --help +(EE) diff --git a/corpus/Xvfb/audit-seed/help.stdout.txt b/corpus/Xvfb/audit-seed/help.stdout.txt new file mode 100644 index 00000000..e69de29b diff --git a/corpus/Xvfb/audit-seed/meta.toml b/corpus/Xvfb/audit-seed/meta.toml new file mode 100644 index 00000000..9e306e37 --- /dev/null +++ b/corpus/Xvfb/audit-seed/meta.toml @@ -0,0 +1,32 @@ +# Xvfb rejects `--help`, prints a leading diagnostic line, then a headingless +# table of over eighty option rows with no recognized heading at all. Three +# defects share one specimen: the leading diagnostic (S-162), `+word` and +# `+/-name`/`[+-]name` rows (S-163), and the headingless table landing in the +# root description (S-165). + +[bless] +provenance = "agent" + +[tool] +name = "Xvfb" +version = "audit-seed" +platform = "ubuntu-24.04" +captured_with = "frozen capture, round 10 W5" + +[[capture]] +argv = ["Xvfb", "--help"] +stdout = "help.stdout.txt" +stderr = "help.stderr.txt" + +[contract] +expected_framework = "generic" +# The F4/S-162 half is fixed: the leading diagnostic no longer survives +# into the root description. This assertion is what makes that checkable; +# the fixture stays [xfail] below because the headingless option table +# (S-165) and the `+word`/`+/-name` rows (S-163) still land as prose in +# that same description. +must_not_describe_root = ["Unrecognized option", "disable access control restrictions"] + +[xfail] +broken = true +reason = "a headingless option table lands in the root description as prose (S-165); +word and +/-name rows are unrecovered (S-163)" diff --git a/corpus/fuser/audit-seed/expected.snap b/corpus/fuser/audit-seed/expected.snap new file mode 100644 index 00000000..714f0ac2 --- /dev/null +++ b/corpus/fuser/audit-seed/expected.snap @@ -0,0 +1,144 @@ +name: fuser +description: Show which processes use the named files, sockets, or filesystems. +usage: +- 'Usage: fuser [-fIMuvw] [-a|-s] [-4|-6] [-c|-m|-n SPACE] [-k [-i] [-SIGNAL]] NAME...' +- ' fuser -l' +- ' fuser -V' +positionals: +- name: NAME + required: true + variadic: true + provenance: + sources: + - help-text +flags: +- spellings: + - -a + - --all + description: display unused files too + provenance: + sources: + - help-text +- spellings: + - -i + - --interactive + description: ask before killing (ignored without -k) + provenance: + sources: + - help-text +- spellings: + - -I + - --inode + description: use always inodes to compare files + provenance: + sources: + - help-text +- spellings: + - -k + - --kill + description: kill processes accessing the named file + provenance: + sources: + - help-text +- spellings: + - -l + - --list-signals + description: list available signal names + provenance: + sources: + - help-text +- spellings: + - -m + - --mount + description: show all processes using the named filesystems or block device + provenance: + sources: + - help-text +- spellings: + - -M + - --ismountpoint + description: fulfill request only if NAME is a mount point + provenance: + sources: + - help-text +- spellings: + - -n + - --namespace + value_name: SPACE + value_kind: Required + description: search in this name space (file, udp, or tcp) + provenance: + sources: + - help-text +- spellings: + - -s + - --silent + description: silent operation + provenance: + sources: + - help-text +- spellings: + - -S + value_name: IGNAL + value_kind: Required + description: send this signal instead of SIGKILL + provenance: + sources: + - help-text +- spellings: + - -u + - --user + description: display user IDs + provenance: + sources: + - help-text +- spellings: + - -v + - --verbose + description: verbose output + provenance: + sources: + - help-text +- spellings: + - -w + - --writeonly + description: kill only processes with write access + provenance: + sources: + - help-text +- spellings: + - -V + - --version + description: display version information + provenance: + sources: + - help-text +- spellings: + - '-4' + - --ipv4 + description: search IPv4 sockets only + provenance: + sources: + - help-text +- spellings: + - '-6' + - --ipv6 + description: search IPv6 sockets only + provenance: + sources: + - help-text +- spellings: + - -f + provenance: + sources: + - help-text-synopsis +- spellings: + - -c + provenance: + sources: + - help-text-synopsis +provenance: + sources: + - help-text + confidence: 0.5 +children_filled: true diff --git a/corpus/fuser/audit-seed/help.stderr.txt b/corpus/fuser/audit-seed/help.stderr.txt new file mode 100644 index 00000000..042ddefe --- /dev/null +++ b/corpus/fuser/audit-seed/help.stderr.txt @@ -0,0 +1,26 @@ +/usr/bin/fuser: Invalid option --help +Usage: fuser [-fIMuvw] [-a|-s] [-4|-6] [-c|-m|-n SPACE] + [-k [-i] [-SIGNAL]] NAME... + fuser -l + fuser -V +Show which processes use the named files, sockets, or filesystems. + + -a,--all display unused files too + -i,--interactive ask before killing (ignored without -k) + -I,--inode use always inodes to compare files + -k,--kill kill processes accessing the named file + -l,--list-signals list available signal names + -m,--mount show all processes using the named filesystems or + block device + -M,--ismountpoint fulfill request only if NAME is a mount point + -n,--namespace SPACE search in this name space (file, udp, or tcp) + -s,--silent silent operation + -SIGNAL send this signal instead of SIGKILL + -u,--user display user IDs + -v,--verbose verbose output + -w,--writeonly kill only processes with write access + -V,--version display version information + -4,--ipv4 search IPv4 sockets only + -6,--ipv6 search IPv6 sockets only + udp/tcp names: [local_port][,[rmt_host][,[rmt_port]]] + diff --git a/corpus/fuser/audit-seed/help.stdout.txt b/corpus/fuser/audit-seed/help.stdout.txt new file mode 100644 index 00000000..e69de29b diff --git a/corpus/fuser/audit-seed/meta.toml b/corpus/fuser/audit-seed/meta.toml new file mode 100644 index 00000000..055cc37f --- /dev/null +++ b/corpus/fuser/audit-seed/meta.toml @@ -0,0 +1,29 @@ +# fuser refuses `--help` (spec §6 rule 0's list) and printed a leading +# diagnostic line before its real usage and flag table. Fixed: +# `strip_leading_diagnostic_line` drops the line before any layout +# analysis. See docs/shapes.md S-162. + +[bless] +provenance = "agent" + +[tool] +name = "fuser" +version = "audit-seed" +platform = "ubuntu-24.04" +captured_with = "frozen capture, round 10 W5" + +[[capture]] +argv = ["fuser", "--help"] +stdout = "help.stdout.txt" +stderr = "help.stderr.txt" + +[contract] +expected_framework = "generic" +min_status = "ok" +must_not_describe_root = ["Invalid option", "/usr/bin/fuser:"] +must_contain_flags = [ + "-a", "-i", "-I", "-k", "-l", "-m", "-M", "-n", + "-s", "-u", "-v", "-w", "-V", "-4", "-6", +] +[contract.must_describe] +"-a" = "display unused files too" diff --git a/corpus/lsof/4.95.0/expected.snap b/corpus/lsof/4.95.0/expected.snap index a2a0036a..e526efb7 100644 --- a/corpus/lsof/4.95.0/expected.snap +++ b/corpus/lsof/4.95.0/expected.snap @@ -1,5 +1,5 @@ name: lsof -description: 'lsof: illegal option character: - lsof: -e not followed by a file system path: "lp" lsof 4.95.0 Defaults in parentheses; comma-separated set (s) items; dash-separated ranges. Anyone can list all files; /dev warnings disabled; kernel ID check disabled.' +description: 'lsof: -e not followed by a file system path: "lp" lsof 4.95.0 Defaults in parentheses; comma-separated set (s) items; dash-separated ranges. Anyone can list all files; /dev warnings disabled; kernel ID check disabled.' usage: - 'usage: [-?abhKlnNoOPRtUvVX] [+|-c c] [+|-d s] [+D D] [+|-E] [+|-e s] [+|-f[gG]] [-F [f]] [-g [s]] [-i [i]] [+|-L [l]] [+m [m]] [+|-M] [-o [o]] [-p s] [+|-r [t]] [-s [p:s]] [-S [t]] [-T [t]] [-u s] [+|-w] [-x [fl]] [--] [names]' flags: diff --git a/corpus/nfsidmap/audit-seed/expected.snap b/corpus/nfsidmap/audit-seed/expected.snap index 204a558f..e9012097 100644 --- a/corpus/nfsidmap/audit-seed/expected.snap +++ b/corpus/nfsidmap/audit-seed/expected.snap @@ -1,6 +1,6 @@ name: nfsidmap usage: -- 'nfsidmap: Usage: nfsidmap [-vh] [-c || [-u|-g|-r key] || -d || -l || [-t timeout] key desc]' +- 'Usage: nfsidmap [-vh] [-c || [-u|-g|-r key] || -d || -l || [-t timeout] key desc]' flags: - spellings: - -v diff --git a/corpus/nfsidmap/audit-seed2/expected.snap b/corpus/nfsidmap/audit-seed2/expected.snap index 204a558f..e9012097 100644 --- a/corpus/nfsidmap/audit-seed2/expected.snap +++ b/corpus/nfsidmap/audit-seed2/expected.snap @@ -1,6 +1,6 @@ name: nfsidmap usage: -- 'nfsidmap: Usage: nfsidmap [-vh] [-c || [-u|-g|-r key] || -d || -l || [-t timeout] key desc]' +- 'Usage: nfsidmap [-vh] [-c || [-u|-g|-r key] || -d || -l || [-t timeout] key desc]' flags: - spellings: - -v diff --git a/corpus/wpa_cli/audit-seed/expected.snap b/corpus/wpa_cli/audit-seed/expected.snap index 35688bd0..ef8cd5ae 100644 --- a/corpus/wpa_cli/audit-seed/expected.snap +++ b/corpus/wpa_cli/audit-seed/expected.snap @@ -1,5 +1,5 @@ name: wpa_cli -description: 'wpa_cli: invalid option -- ''-'' commands:' +description: 'commands:' usage: - wpa_cli [-p] [-i] [-hvBr] [-a] [-P] [-g] [-G] [-s] [command..] flags: diff --git a/docs/shapes.md b/docs/shapes.md index ad8e4642..0b0e69f6 100644 --- a/docs/shapes.md +++ b/docs/shapes.md @@ -2817,3 +2817,29 @@ entry's `tools` field and nothing else. It does not get a new entry. not gated: 3 are fail2ban-client's own still-open `set`/`add` gap (S-141's name rule, not this fix), the rest are false alarms on text that merely resembles a label followed by a row. 2026-09-07. + +### S-162: a leading diagnostic line inside the chosen document + +- id: S-162 +- looks like: | + /usr/bin/fuser: Invalid option --help + Usage: fuser [-fIMuvw] [-a|-s] [-4|-6] [-c|-m|-n SPACE] +- tools: fuser, Xvfb, nfsidmap +- handling: Fixed (the diagnostic-drop half). A tool that refuses `--help` often + prints one option-rejection line first, then its document anyway. The + chosen stream's own first non-empty line is dropped before any layout + analysis when it names an option-rejection (`invalid`/`unrecognized`/ + `unknown`/`illegal option`, case-insensitive), optionally preceded by the + program's own name or path and `": "`. No later line is ever dropped this + way. A `: ` prefix glued in front of a usage label + (`nfsidmap: Usage: ...`) is stripped the same way, wherever it sits, not + only on the first line. Same hazard class as S-029 and S-091: a diagnostic + preamble merged into the document is how banner text becomes fabricated + structure. +- fleet: `leading-diagnostic-line` (`xtask/src/detector/leading_diagnostic_line.rs`) + is family `None`: no DEFECT_FAMILIES label covers this shape, so its + calibration reads NOT EVALUABLE rather than a score. Self-checks hold + (5/5). Raw-shape grep count: 197 tools / 198 findings for the leading + diagnostic over both streams; the detector reads the tree's chosen stream + only, and that count is an upper bound, not this family's own fleet + count. 2026-09-12. diff --git a/mandible-extract/src/help_text/sections/mod.rs b/mandible-extract/src/help_text/sections/mod.rs index 54495ee5..5611de37 100644 --- a/mandible-extract/src/help_text/sections/mod.rs +++ b/mandible-extract/src/help_text/sections/mod.rs @@ -286,6 +286,14 @@ pub fn parse_with_profile( // fuses into one alphanumeric run that matches no recognized heading // word. See S-002. let raw = strip_escapes(raw); + // A leading option-rejection diagnostic (`fuser`'s `Invalid option + // --help`, `Xvfb`'s `Unrecognized option: --help`, `nfsidmap`'s + // `invalid option -- '-'`) is the tool's own complaint about the probe, + // not part of its document, and merging it into the root description or + // a heading is the same S-029/S-091 hazard a banner line already is + // (spec §7 Tier B rule 11's Why paragraph). Dropped once, here, before + // any layout analysis sees it. See docs/shapes.md S-162. + let raw = strip_leading_diagnostic_line(&raw); // lowdown's man-page-like rendering (nix/Lix, issue #138) writes // every entry, command or option alike, as a `·`-led bullet row and // sometimes wraps a group label across two physical lines. Rewritten @@ -326,8 +334,12 @@ fn scan_usage_section( ) -> UsageScan { let mut i = start; let base_indent = leading_whitespace(lines[i]); - usage_lines.push(lines[i].trim().to_string()); - let mut usage_entries = vec![lines[i].trim().to_string()]; + // Drop a `: ` prefix in front of this line's own usage label + // (S-162): the C fprintf idiom's diagnostic prefix, never the label + // itself. + let head = strip_name_prefixed_usage_label(lines[i], tool_name); + usage_lines.push(head.clone()); + let mut usage_entries = vec![head]; // Parallel to `usage_lines`: which `usage_entries` index each // physical line was folded into — a wrapped entry (sg_sanitize's // five-line synopsis) spans several lines but is one entry, and diff --git a/mandible-extract/src/help_text/sections/preamble.rs b/mandible-extract/src/help_text/sections/preamble.rs index 7d870a49..f0fe1b93 100644 --- a/mandible-extract/src/help_text/sections/preamble.rs +++ b/mandible-extract/src/help_text/sections/preamble.rs @@ -194,6 +194,34 @@ pub(super) fn option_error_tail_is_shapely(tail: &str) -> bool { }) } +/// Drop `raw`'s own leading option-rejection diagnostic line, whole +/// document unaffected otherwise. See docs/shapes.md S-162. +/// +/// Narrower than [`is_option_error_line`]: consulted only on the +/// document's own first non-empty physical line, before any paragraph or +/// section boundary is known, and returns the whole document with that one +/// line (and its line terminator) removed rather than a verdict on a +/// paragraph. Every other line is untouched, including a later paragraph +/// [`is_option_error_paragraph`] would still drop on its own terms. +pub(super) fn strip_leading_diagnostic_line(raw: &str) -> String { + let mut consumed = 0usize; + for line in raw.split_inclusive('\n') { + let content = line.trim_end_matches(['\n', '\r']); + if content.trim().is_empty() { + consumed += line.len(); + continue; + } + if is_option_error_line(content) { + let mut out = String::with_capacity(raw.len() - line.len()); + out.push_str(&raw[..consumed]); + out.push_str(&raw[consumed + line.len()..]); + return out; + } + break; + } + raw.to_string() +} + #[cfg(test)] mod tests { use super::*; @@ -325,22 +353,22 @@ mod tests { ); } - /// A leading complaint followed by unrelated content in the same - /// paragraph must not be dropped — `sshd`'s real shape: its version - /// banner sits directly under the complaint. + /// `sshd`'s real shape: its version banner sits directly under its own + /// leading complaint, no blank line between them. `S-162` (the + /// document's first non-empty line, checked before any paragraph + /// boundary is known) now drops the complaint line on its own, so only + /// the real banner text survives as the description — never re-fused + /// with the diagnostic the way a whole-paragraph check would have kept + /// it, since the rule cares about the first line alone, never what + /// follows it. See docs/shapes.md S-162. #[test] - fn a_mixed_paragraph_with_real_content_is_kept_whole() { + fn a_leading_complaint_is_dropped_even_when_real_content_follows_on_the_next_line() { let raw = "unknown option -- -\nOpenSSH_9.6p1 Ubuntu, OpenSSL 3.0.13\n\n\ usage: sshd [-46DdeGiqTtV]\n"; let parsed = parse_named(raw, "sshd"); assert_eq!( parsed.description.as_deref(), - Some("unknown option -- -\nOpenSSH_9.6p1 Ubuntu, OpenSSL 3.0.13") - ); - // Neither line is structural, so the pair reflows once sanitized. - assert_eq!( - mandible_core::Text::sanitize(parsed.description.as_deref().unwrap()).as_str(), - "unknown option -- - OpenSSH_9.6p1 Ubuntu, OpenSSL 3.0.13" + Some("OpenSSH_9.6p1 Ubuntu, OpenSSL 3.0.13") ); } diff --git a/mandible-extract/src/help_text/sections/usage.rs b/mandible-extract/src/help_text/sections/usage.rs index 50c27047..3982bfb6 100644 --- a/mandible-extract/src/help_text/sections/usage.rs +++ b/mandible-extract/src/help_text/sections/usage.rs @@ -88,6 +88,25 @@ pub fn starts_with_name_prefixed_usage(t: &str, name: &str) -> bool { .is_some_and(starts_with_usage_prefix) } +/// Drop the `: ` prefix [`starts_with_name_prefixed_usage`] +/// recognizes, wherever it sits in front of a usage label — not only the +/// document's first line. `nfsidmap`'s C `fprintf(stderr, "%s: Usage: +/// ...", argv[0])` idiom keeps that prefix glued to its own usage line +/// rendered; the diagnostic prefix is not the label, and a reader wants the +/// label. Returns `t` trimmed and unchanged when the prefix isn't present. +/// See docs/shapes.md S-162 and S-001. +pub(super) fn strip_name_prefixed_usage_label(t: &str, tool_name: Option<&str>) -> String { + let trimmed = t.trim(); + if let Some(name) = tool_name { + if starts_with_name_prefixed_usage(trimmed, name) { + // `starts_with_name_prefixed_usage` already confirmed `trimmed` + // opens with exactly `"{name}: "`. + return trimmed[name.len() + 2..].to_string(); + } + } + trimmed.to_string() +} + /// True if `t` opens with `name` at a word boundary and its remainder /// reads as usage-synopsis grammar rather than prose — the unlabelled /// synopsis convention (`wpa_cli --help` opens `wpa_cli [-p] diff --git a/xtask/src/corpus/contract.rs b/xtask/src/corpus/contract.rs index 88c6f280..8c7478a3 100644 --- a/xtask/src/corpus/contract.rs +++ b/xtask/src/corpus/contract.rs @@ -99,6 +99,11 @@ pub(crate) fn contract_weakened_lines(current: &[Fixture], baseline: &[Fixture]) &b.must_not_contain_flag_group_prefixes, &n.must_not_contain_flag_group_prefixes, ), + ( + "must_not_describe_root", + &b.must_not_describe_root, + &n.must_not_describe_root, + ), ( "must_contain_positionals", &b.must_contain_positionals, @@ -435,6 +440,35 @@ fn check_contract_missing_root(contract: &ContractMeta) -> Vec failures } +/// The root-level mirror of `must_not_describe`: text the root's own +/// `description` must not carry. Whitespace-collapsed substring match, +/// `must_describe`'s own rule (a real description wraps). Built for +/// `Xvfb`'s leading option-rejection diagnostic, which used to fuse into +/// the root description alongside its whole option table. See +/// docs/shapes.md S-162. +fn check_must_not_describe_root( + contract: &ContractMeta, + root: &CommandNode, +) -> Vec { + let Some(description) = root.description.as_ref().map(|t| t.as_str()) else { + return Vec::new(); + }; + let description_collapsed = collapse_whitespace(description); + let present: Vec<&str> = contract + .must_not_describe_root + .iter() + .filter(|text| description_collapsed.contains(&collapse_whitespace(text))) + .map(|s| s.as_str()) + .collect(); + if present.is_empty() { + return Vec::new(); + } + vec![ContractFailure(format!( + "must_not_describe_root: present {}", + present.join(", ") + ))] +} + /// The scalar `[contract]` fields: `expected_framework`, `min_status`, /// `min_subcommands`, `must_contain_flags`, `must_not_contain_flags`. fn check_contract_scalar_fields( @@ -551,6 +585,8 @@ fn check_contract_scalar_fields( ))); } + failures.extend(check_must_not_describe_root(contract, root)); + // The group-label mirror of the negative claim above: no root flag's // own `group` may start with one of these spellings — the invented // group `must_not_contain_flags` cannot see, since a fabricated diff --git a/xtask/src/corpus/mod.rs b/xtask/src/corpus/mod.rs index 5faee7f8..f5434004 100644 --- a/xtask/src/corpus/mod.rs +++ b/xtask/src/corpus/mod.rs @@ -23,6 +23,7 @@ //! `must_contain_positionals`, `must_contain_modifiers`, //! `must_not_contain_flags`, `must_not_contain_positionals`, //! `must_not_contain_flags`, `must_not_contain_usage_text`, +//! `must_not_describe_root`, //! `must_keep_separate`, `must_attach_choices`, //! `must_describe`, `must_usage_forms_min`, //! `must_not_contain_flag_group_prefixes`, `must_flag_group`, @@ -199,6 +200,19 @@ pub(crate) struct ContractMeta { /// same reasoning `must_not_contain_flags` uses. #[serde(default)] must_not_contain_usage_text: Vec, + /// Text the root's own `description` must **not** carry — the root-level + /// analogue of `must_not_describe`, which only ever checks a *flag's* + /// description. `Xvfb`'s leading option-rejection diagnostic + /// (`Unrecognized option: --help`) used to fuse into the root + /// description alongside its whole eighty-row option table; nothing + /// before this field could state that the diagnostic itself must not + /// survive into the tree. Substring match after collapsing runs of + /// whitespace to a single space on both sides, the same rule + /// `must_describe` uses. Satisfied vacuously by a tree with no root or + /// no description at all, the same reasoning `must_not_contain_flags` + /// uses. See docs/shapes.md S-162. + #[serde(default)] + must_not_describe_root: Vec, /// Spellings no root flag's own `group` may **start with** — the /// group-label mirror of `must_not_contain_flags`, added for /// `lvcreate`'s own shape (docs/shapes.md S-137): the unfixed parser diff --git a/xtask/src/coverage/mod.rs b/xtask/src/coverage/mod.rs index 468324bd..5bc80212 100644 --- a/xtask/src/coverage/mod.rs +++ b/xtask/src/coverage/mod.rs @@ -13,6 +13,7 @@ mod render_text; mod round7; mod round8; mod round9; +mod round10; mod score; use aggregate::compute_aggregate; diff --git a/xtask/src/coverage/round10.rs b/xtask/src/coverage/round10.rs new file mode 100644 index 00000000..19ffa693 --- /dev/null +++ b/xtask/src/coverage/round10.rs @@ -0,0 +1,20 @@ +//! Round-10 family detectors. Atlas S-162: a leading option-rejection +//! diagnostic line fused into the root description. + +use crate::detector::{Detector, ToolEvidence}; +use mandible_core::CommandNode; + +pub(super) fn round10_family_counts( + raw: &str, + root: &CommandNode, +) -> Vec<(&'static str, usize, Vec)> { + let cap = super::score::FAMILY_DETECTOR_SAMPLES_PER_ROW; + let evidence = ToolEvidence { raw, root }; + let leading_diagnostic = + crate::detector::leading_diagnostic_line::LeadingDiagnosticLine.hits(&evidence); + vec![( + "leading-diagnostic-line", + leading_diagnostic.len(), + leading_diagnostic.into_iter().take(cap).collect(), + )] +} diff --git a/xtask/src/coverage/score.rs b/xtask/src/coverage/score.rs index ba7466c3..cba33bae 100644 --- a/xtask/src/coverage/score.rs +++ b/xtask/src/coverage/score.rs @@ -497,6 +497,7 @@ fn vim_family_counts( counts.extend(super::round7::round7_usage_family_counts(&raw, root)); counts.extend(super::round8::round8_family_counts(&raw, root)); counts.extend(super::round9::round9_family_counts(&raw, root)); + counts.extend(super::round10::round10_family_counts(&raw, root)); counts } diff --git a/xtask/src/detector/leading_diagnostic_line.rs b/xtask/src/detector/leading_diagnostic_line.rs new file mode 100644 index 00000000..7174cab2 --- /dev/null +++ b/xtask/src/detector/leading_diagnostic_line.rs @@ -0,0 +1,142 @@ +//! `leading-diagnostic-line` (atlas S-162): the chosen stream's own first +//! non-empty line is an option-rejection diagnostic (`fuser`'s `Invalid +//! option --help`, `Xvfb`'s `Unrecognized option: --help`, `nfsidmap`'s +//! `invalid option -- '-'`), and it survives into the root's own +//! `description`. Mirrors `mandible_extract::help_text::sections::preamble`'s +//! (private) `is_option_error_line`, checked independently here since a +//! detector reads only `raw`+`root` — see `bare_or_usage_separator.rs`'s own +//! doc comment for why that duplication is the accepted shape. + +use crate::detector::{Detector, Expect, Scope, SelfCheck, ToolEvidence}; +use mandible_core::{CommandNode, Provenance, Source, Text}; + +/// True if `line` (trimmed) opens with one of the four option-rejection +/// phrases, after an optional single-token `: ` prefix. Deliberately +/// simpler than the parser's own `is_option_error_line`: no trailer-shape +/// bound, since a detector counts a raw shape, it does not have to decide +/// whether the line's tail is "shapely" the way the parser's containment +/// fence does. +fn looks_like_option_rejection(line: &str) -> bool { + let trimmed = line.trim(); + if trimmed.is_empty() { + return false; + } + let body = match trimmed.split_once(": ") { + Some((prefix, rest)) if !prefix.is_empty() && !prefix.contains(char::is_whitespace) => { + rest + } + _ => trimmed, + }; + let lower = body.to_ascii_lowercase(); + ["invalid option", "unrecognized option", "unknown option", "illegal option"] + .iter() + .any(|kw| lower.starts_with(kw)) +} + +/// `raw`'s own first non-empty physical line, or `None` for an empty +/// document. +fn first_nonblank_line(raw: &str) -> Option<&str> { + raw.lines().find(|l| !l.trim().is_empty()) +} + +pub struct LeadingDiagnosticLine; + +impl Detector for LeadingDiagnosticLine { + fn name(&self) -> &'static str { + "leading-diagnostic-line" + } + + fn family(&self) -> Option<&'static str> { + None + } + + fn describes(&self) -> &'static str { + "the chosen stream's own first non-empty line reads as an option-rejection diagnostic, \ + and it still occurs in the root's own description" + } + + fn hits(&self, evidence: &ToolEvidence<'_>) -> Vec { + let Some(first) = first_nonblank_line(evidence.raw) else { + return Vec::new(); + }; + if !looks_like_option_rejection(first) { + return Vec::new(); + } + let description = evidence + .root + .description + .as_ref() + .map(|t| t.as_str()) + .unwrap_or(""); + if description.contains(first.trim()) { + vec![format!( + "the leading diagnostic {first:?} still occurs in the root description" + )] + } else { + Vec::new() + } + } + + fn scope(&self) -> Scope { + Scope::full() + } + + fn self_checks(&self) -> Vec { + fn node_with_description(description: Option<&str>) -> CommandNode { + let mut root = CommandNode::new("prog", Provenance::single(Source::HelpText)); + root.description = description.map(Text::sanitize); + root + } + + vec![ + SelfCheck { + name: "fuser's own bytes, diagnostic fused into the description", + why: "the defect itself: the diagnostic line is still contained in the root \ + description", + expect: Expect::Fires(1), + raw: "/usr/bin/fuser: Invalid option --help\nUsage: fuser [-fIMuvw]\n".to_string(), + root: node_with_description(Some( + "/usr/bin/fuser: Invalid option --help Show which processes use the named \ + files", + )), + }, + SelfCheck { + name: "fuser's own bytes, diagnostic dropped before the description", + why: "once the diagnostic line is stripped before layout analysis, the same raw \ + bytes must go silent", + expect: Expect::Silent, + raw: "/usr/bin/fuser: Invalid option --help\nUsage: fuser [-fIMuvw]\n".to_string(), + root: node_with_description(Some( + "Show which processes use the named files, sockets, or filesystems.", + )), + }, + SelfCheck { + name: "Xvfb's own bytes, no program-name prefix at all", + why: "the bare-phrase shape, no `: ` prefix, must fire the same way", + expect: Expect::Fires(1), + raw: "Unrecognized option: --help\nuse: X [:] [option]\n".to_string(), + root: node_with_description(Some( + "Unrecognized option: --help use: X [:] [option]", + )), + }, + SelfCheck { + name: "a real sentence merely mentioning the phrase mid-clause", + why: "the phrase must open the line, not merely appear in it, so an ordinary \ + sentence must never fire", + expect: Expect::Silent, + raw: "An invalid option combination here raises an error.\n\nUsage: mytool\n" + .to_string(), + root: node_with_description(Some( + "An invalid option combination here raises an error.", + )), + }, + SelfCheck { + name: "a document with no leading diagnostic at all", + why: "an ordinary tool's first line is never claimed", + expect: Expect::Silent, + raw: "usage: mytool [OPTIONS]\n\n -a do a thing\n".to_string(), + root: node_with_description(None), + }, + ] + } +} diff --git a/xtask/src/detector/mod.rs b/xtask/src/detector/mod.rs index 15c03797..4b26f933 100644 --- a/xtask/src/detector/mod.rs +++ b/xtask/src/detector/mod.rs @@ -55,6 +55,7 @@ pub(crate) mod generic_option_placeholder_flag; pub(crate) mod glued_optional_group_spelling; pub(crate) mod glued_uppercase_shared_prefix; pub(crate) mod hash_in_spelling; +pub(crate) mod leading_diagnostic_line; pub(crate) mod multi_operand_usage_tail; pub(crate) mod nested_bracket_value; pub(crate) mod numbered_variadic_usage_tail; @@ -748,6 +749,7 @@ pub fn registry() -> Vec> { Box::new(crate::centered_label_baseline::LabelPrecedesShallowerLine), Box::new(crate::centered_label_baseline::MissingRowAfterLabel), Box::new(option_table_multiword_value_name::OptionTableMultiwordValueName), + Box::new(leading_diagnostic_line::LeadingDiagnosticLine), ] } From 62fecc5f262d3cb4ce1958266db5fca5346000a5 Mon Sep 17 00:00:00 2001 From: Sadig Akhund Date: Sat, 12 Sep 2026 16:42:53 +0400 Subject: [PATCH 2/9] parser: read a header-declared three-column env-var table at its own offsets MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit [S-166] A header row naming Argument/Env-variable/Description columns is read at those offsets instead of one column gap, so the env-variable cell becomes the flag's own env_var cross-reference (spec §4.5) instead of gluing onto the description. -cpu and -dfilter keep their own value name now, and the prose paragraph below the table ("The following lines are equivalent:") no longer fabricates a second -E row and group, fixing all 42 qemu-*-static tools (mandible qemu-riscv64-static). The TUI now renders a flag's own env_var as an "env: FOO" line next to values:, since it carried no rendering before this shape existed (docs/design.md §9.3 rule 8). corpus/qemu-arm64-static/audit-seed2/expected.snap moved for the same reason: it already exhibited this shape and was already green, so re-blessing it after the fix is the ordinary case, not a promotion. Every flag on it keeps its own spelling, value_name and description; each row's env_var is new. New fixture corpus/qemu-riscv64-static/8.2.2. New detector header-declared-env-column (xtask/src/coverage/round10.rs), family() None: the seed-7 audit labels qemu-riscv64-static "incomplete" and describes this exact shape, but no DEFECT_FAMILIES entry names it and the entry carries no derived family label. Co-Authored-By: Claude Fable 5.1 --- CHANGELOG.md | 1 + .../audit-seed2/expected.snap | 108 +++++--- .../qemu-riscv64-static/8.2.2/expected.snap | 227 +++++++++++++++++ corpus/qemu-riscv64-static/8.2.2/help.txt | 50 ++++ corpus/qemu-riscv64-static/8.2.2/meta.toml | 45 ++++ docs/design.md | 30 ++- docs/shapes.md | 25 ++ .../src/help_text/sections/emit.rs | 38 +++ .../src/help_text/sections/mod.rs | 30 ++- .../src/help_text/sections/scan.rs | 137 +++++++++++ mandible-tui/src/render/detail_pane/entity.rs | 25 +- xtask/src/coverage/mod.rs | 1 + xtask/src/coverage/round10.rs | 23 ++ xtask/src/coverage/score.rs | 1 + .../detector/header_declared_env_column.rs | 230 ++++++++++++++++++ xtask/src/detector/mod.rs | 5 + 16 files changed, 925 insertions(+), 51 deletions(-) create mode 100644 corpus/qemu-riscv64-static/8.2.2/expected.snap create mode 100644 corpus/qemu-riscv64-static/8.2.2/help.txt create mode 100644 corpus/qemu-riscv64-static/8.2.2/meta.toml create mode 100644 xtask/src/coverage/round10.rs create mode 100644 xtask/src/detector/header_declared_env_column.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index a6bbc073..f7ad2204 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -33,6 +33,7 @@ once it reaches a published 0.1.0 release. - [S-145] A single-dash long option in a table with no double-dash row anywhere keeps its whole name and case now, so `mandible mksquashfs` and `mandible sqfstar` show `-pf`, `-ef`, `-Xhelp`, `-Xstrategy`, `-Xhc` and `-Xbcj` instead of splitting each one. - [S-145] A tab-separated value name on a single-dash long option survives the same repair now, so `mandible mksquashfs` and `mandible sqfstar` keep `-mem`, `-comp` and `-mkfs-time` with their placeholders instead of losing them. - [S-146] A heading with no indent step to its own rows, and a bare label above a nested option block, now name a group instead of leaving every row underneath ungrouped, so `mandible mksquashfs` shows its ten option headings and its five compressor names. +- [S-166] A header row that names its own columns as `Argument`, `Env-variable` and `Description` is now read at those exact offsets, so `mandible qemu-riscv64-static` and the rest of the `qemu-*-static` fleet show a clean description and each flag's own environment variable instead of the two glued together, keep `-cpu` and `-dfilter`'s own value names, and no longer invent an `-E` row from the prose paragraph below the table. ## [0.7.0] - 2026-09-05 diff --git a/corpus/qemu-arm64-static/audit-seed2/expected.snap b/corpus/qemu-arm64-static/audit-seed2/expected.snap index cdbfac2c..675e737b 100644 --- a/corpus/qemu-arm64-static/audit-seed2/expected.snap +++ b/corpus/qemu-arm64-static/audit-seed2/expected.snap @@ -5,12 +5,14 @@ usage: flags: - spellings: - -h + group: 'Options and associated environment variables:' description: print this help provenance: sources: - help-text - spellings: - -help + group: 'Options and associated environment variables:' provenance: sources: - help-text @@ -18,7 +20,9 @@ flags: - -g value_name: port value_kind: Required - description: QEMU_GDB wait gdb connection to 'port' + group: 'Options and associated environment variables:' + description: wait gdb connection to 'port' + env_var: QEMU_GDB provenance: sources: - help-text @@ -26,7 +30,9 @@ flags: - -L value_name: path value_kind: Required - description: QEMU_LD_PREFIX set the elf interpreter prefix to 'path' + group: 'Options and associated environment variables:' + description: set the elf interpreter prefix to 'path' + env_var: QEMU_LD_PREFIX provenance: sources: - help-text @@ -34,13 +40,19 @@ flags: - -s value_name: size value_kind: Required - description: QEMU_STACK_SIZE set the stack size to 'size' bytes + group: 'Options and associated environment variables:' + description: set the stack size to 'size' bytes + env_var: QEMU_STACK_SIZE provenance: sources: - help-text - spellings: - -cpu - description: QEMU_CPU select CPU (-cpu help for list) + value_name: model + value_kind: Required + group: 'Options and associated environment variables:' + description: select CPU (-cpu help for list) + env_var: QEMU_CPU provenance: sources: - help-text @@ -48,7 +60,9 @@ flags: - -E value_name: var=value value_kind: Required - description: QEMU_SET_ENV sets targets environment variable (see below) + group: 'Options and associated environment variables:' + description: sets targets environment variable (see below) + env_var: QEMU_SET_ENV provenance: sources: - help-text @@ -56,7 +70,9 @@ flags: - -U value_name: var value_kind: Required - description: QEMU_UNSET_ENV unsets targets environment variable (see below) + group: 'Options and associated environment variables:' + description: unsets targets environment variable (see below) + env_var: QEMU_UNSET_ENV provenance: sources: - help-text @@ -64,7 +80,9 @@ flags: - '-0' value_name: argv0 value_kind: Required - description: QEMU_ARGV0 forces target process argv[0] to be 'argv0' + group: 'Options and associated environment variables:' + description: forces target process argv[0] to be 'argv0' + env_var: QEMU_ARGV0 provenance: sources: - help-text @@ -72,7 +90,9 @@ flags: - -r value_name: uname value_kind: Required - description: QEMU_UNAME set qemu uname release string to 'uname' + group: 'Options and associated environment variables:' + description: set qemu uname release string to 'uname' + env_var: QEMU_UNAME provenance: sources: - help-text @@ -80,7 +100,9 @@ flags: - -B value_name: address value_kind: Required - description: QEMU_GUEST_BASE set guest_base address to 'address' + group: 'Options and associated environment variables:' + description: set guest_base address to 'address' + env_var: QEMU_GUEST_BASE provenance: sources: - help-text @@ -88,7 +110,9 @@ flags: - -R value_name: size value_kind: Required - description: QEMU_RESERVED_VA reserve 'size' bytes for guest virtual address space + group: 'Options and associated environment variables:' + description: reserve 'size' bytes for guest virtual address space + env_var: QEMU_RESERVED_VA provenance: sources: - help-text @@ -96,13 +120,19 @@ flags: - -d value_name: item[,...] value_kind: Required - description: QEMU_LOG enable logging of specified items (use '-d help' for a list of items) + group: 'Options and associated environment variables:' + description: enable logging of specified items (use '-d help' for a list of items) + env_var: QEMU_LOG provenance: sources: - help-text - spellings: - -dfilter + value_name: range[,...] + value_kind: Required + group: 'Options and associated environment variables:' description: filter logging based on address range + env_var: QEMU_DFILTER provenance: sources: - help-text @@ -110,7 +140,9 @@ flags: - -D value_name: logfile value_kind: Required - description: QEMU_LOG_FILENAME write logs to 'logfile' (default stderr) + group: 'Options and associated environment variables:' + description: write logs to 'logfile' (default stderr) + env_var: QEMU_LOG_FILENAME provenance: sources: - help-text @@ -118,71 +150,73 @@ flags: - -p value_name: pagesize value_kind: Required - description: QEMU_PAGESIZE set the host page size to 'pagesize' + group: 'Options and associated environment variables:' + description: set the host page size to 'pagesize' + env_var: QEMU_PAGESIZE provenance: sources: - help-text - spellings: - -one-insn-per-tb - description: QEMU_ONE_INSN_PER_TB run with one guest instruction per emulated TB + group: 'Options and associated environment variables:' + description: run with one guest instruction per emulated TB + env_var: QEMU_ONE_INSN_PER_TB provenance: sources: - help-text - spellings: - -singlestep - description: QEMU_SINGLESTEP deprecated synonym for -one-insn-per-tb + group: 'Options and associated environment variables:' + description: deprecated synonym for -one-insn-per-tb + env_var: QEMU_SINGLESTEP provenance: sources: - help-text - spellings: - -strace - description: QEMU_STRACE log system calls + group: 'Options and associated environment variables:' + description: log system calls + env_var: QEMU_STRACE provenance: sources: - help-text - spellings: - -seed - description: QEMU_RAND_SEED Seed for pseudo-random number generator + group: 'Options and associated environment variables:' + description: Seed for pseudo-random number generator + env_var: QEMU_RAND_SEED provenance: sources: - help-text - spellings: - -trace - description: QEMU_TRACE [[enable=]][,events=][,file=] + group: 'Options and associated environment variables:' + description: '[[enable=]][,events=][,file=]' + env_var: QEMU_TRACE provenance: sources: - help-text - spellings: - -version - description: QEMU_VERSION display version information and exit + group: 'Options and associated environment variables:' + description: display version information and exit + env_var: QEMU_VERSION provenance: sources: - help-text - spellings: - -perfmap - description: QEMU_PERFMAP Generate a /tmp/perf-${pid}.map file for perf + group: 'Options and associated environment variables:' + description: Generate a /tmp/perf-${pid}.map file for perf + env_var: QEMU_PERFMAP provenance: sources: - help-text - spellings: - -jitdump - description: QEMU_JITDUMP Generate a jit-${pid}.dump file for perf - provenance: - sources: - - help-text -- spellings: - - -E - value_name: var1=val2 - value_kind: Required - group: 'The following lines are equivalent:' - provenance: - sources: - - help-text -- spellings: - - -E - value_name: var1=val2,var2=val2 - value_kind: Required - group: 'The following lines are equivalent:' + group: 'Options and associated environment variables:' + description: Generate a jit-${pid}.dump file for perf + env_var: QEMU_JITDUMP provenance: sources: - help-text diff --git a/corpus/qemu-riscv64-static/8.2.2/expected.snap b/corpus/qemu-riscv64-static/8.2.2/expected.snap new file mode 100644 index 00000000..66b65fa6 --- /dev/null +++ b/corpus/qemu-riscv64-static/8.2.2/expected.snap @@ -0,0 +1,227 @@ +name: qemu-riscv64-static +description: Linux CPU emulator (compiled for riscv64 emulation) +usage: +- 'usage: qemu-riscv64 [options] program [arguments...]' +flags: +- spellings: + - -h + group: 'Options and associated environment variables:' + description: print this help + provenance: + sources: + - help-text +- spellings: + - -help + group: 'Options and associated environment variables:' + provenance: + sources: + - help-text +- spellings: + - -g + value_name: port + value_kind: Required + group: 'Options and associated environment variables:' + description: wait gdb connection to 'port' + env_var: QEMU_GDB + provenance: + sources: + - help-text +- spellings: + - -L + value_name: path + value_kind: Required + group: 'Options and associated environment variables:' + description: set the elf interpreter prefix to 'path' + env_var: QEMU_LD_PREFIX + provenance: + sources: + - help-text +- spellings: + - -s + value_name: size + value_kind: Required + group: 'Options and associated environment variables:' + description: set the stack size to 'size' bytes + env_var: QEMU_STACK_SIZE + provenance: + sources: + - help-text +- spellings: + - -cpu + value_name: model + value_kind: Required + group: 'Options and associated environment variables:' + description: select CPU (-cpu help for list) + env_var: QEMU_CPU + provenance: + sources: + - help-text +- spellings: + - -E + value_name: var=value + value_kind: Required + group: 'Options and associated environment variables:' + description: sets targets environment variable (see below) + env_var: QEMU_SET_ENV + provenance: + sources: + - help-text +- spellings: + - -U + value_name: var + value_kind: Required + group: 'Options and associated environment variables:' + description: unsets targets environment variable (see below) + env_var: QEMU_UNSET_ENV + provenance: + sources: + - help-text +- spellings: + - '-0' + value_name: argv0 + value_kind: Required + group: 'Options and associated environment variables:' + description: forces target process argv[0] to be 'argv0' + env_var: QEMU_ARGV0 + provenance: + sources: + - help-text +- spellings: + - -r + value_name: uname + value_kind: Required + group: 'Options and associated environment variables:' + description: set qemu uname release string to 'uname' + env_var: QEMU_UNAME + provenance: + sources: + - help-text +- spellings: + - -B + value_name: address + value_kind: Required + group: 'Options and associated environment variables:' + description: set guest_base address to 'address' + env_var: QEMU_GUEST_BASE + provenance: + sources: + - help-text +- spellings: + - -R + value_name: size + value_kind: Required + group: 'Options and associated environment variables:' + description: reserve 'size' bytes for guest virtual address space + env_var: QEMU_RESERVED_VA + provenance: + sources: + - help-text +- spellings: + - -d + value_name: item[,...] + value_kind: Required + group: 'Options and associated environment variables:' + description: enable logging of specified items (use '-d help' for a list of items) + env_var: QEMU_LOG + provenance: + sources: + - help-text +- spellings: + - -dfilter + value_name: range[,...] + value_kind: Required + group: 'Options and associated environment variables:' + description: filter logging based on address range + env_var: QEMU_DFILTER + provenance: + sources: + - help-text +- spellings: + - -D + value_name: logfile + value_kind: Required + group: 'Options and associated environment variables:' + description: write logs to 'logfile' (default stderr) + env_var: QEMU_LOG_FILENAME + provenance: + sources: + - help-text +- spellings: + - -p + value_name: pagesize + value_kind: Required + group: 'Options and associated environment variables:' + description: set the host page size to 'pagesize' + env_var: QEMU_PAGESIZE + provenance: + sources: + - help-text +- spellings: + - -one-insn-per-tb + group: 'Options and associated environment variables:' + description: run with one guest instruction per emulated TB + env_var: QEMU_ONE_INSN_PER_TB + provenance: + sources: + - help-text +- spellings: + - -singlestep + group: 'Options and associated environment variables:' + description: deprecated synonym for -one-insn-per-tb + env_var: QEMU_SINGLESTEP + provenance: + sources: + - help-text +- spellings: + - -strace + group: 'Options and associated environment variables:' + description: log system calls + env_var: QEMU_STRACE + provenance: + sources: + - help-text +- spellings: + - -seed + group: 'Options and associated environment variables:' + description: Seed for pseudo-random number generator + env_var: QEMU_RAND_SEED + provenance: + sources: + - help-text +- spellings: + - -trace + group: 'Options and associated environment variables:' + description: '[[enable=]][,events=][,file=]' + env_var: QEMU_TRACE + provenance: + sources: + - help-text +- spellings: + - -version + group: 'Options and associated environment variables:' + description: display version information and exit + env_var: QEMU_VERSION + provenance: + sources: + - help-text +- spellings: + - -perfmap + group: 'Options and associated environment variables:' + description: Generate a /tmp/perf-${pid}.map file for perf + env_var: QEMU_PERFMAP + provenance: + sources: + - help-text +- spellings: + - -jitdump + group: 'Options and associated environment variables:' + description: Generate a jit-${pid}.dump file for perf + env_var: QEMU_JITDUMP + provenance: + sources: + - help-text +provenance: + sources: + - help-text + confidence: 0.5 +children_filled: true diff --git a/corpus/qemu-riscv64-static/8.2.2/help.txt b/corpus/qemu-riscv64-static/8.2.2/help.txt new file mode 100644 index 00000000..93a461dd --- /dev/null +++ b/corpus/qemu-riscv64-static/8.2.2/help.txt @@ -0,0 +1,50 @@ +usage: qemu-riscv64 [options] program [arguments...] +Linux CPU emulator (compiled for riscv64 emulation) + +Options and associated environment variables: + +Argument Env-variable Description +-h print this help +-help +-g port QEMU_GDB wait gdb connection to 'port' +-L path QEMU_LD_PREFIX set the elf interpreter prefix to 'path' +-s size QEMU_STACK_SIZE set the stack size to 'size' bytes +-cpu model QEMU_CPU select CPU (-cpu help for list) +-E var=value QEMU_SET_ENV sets targets environment variable (see below) +-U var QEMU_UNSET_ENV unsets targets environment variable (see below) +-0 argv0 QEMU_ARGV0 forces target process argv[0] to be 'argv0' +-r uname QEMU_UNAME set qemu uname release string to 'uname' +-B address QEMU_GUEST_BASE set guest_base address to 'address' +-R size QEMU_RESERVED_VA reserve 'size' bytes for guest virtual address space +-d item[,...] QEMU_LOG enable logging of specified items (use '-d help' for a list of items) +-dfilter range[,...] QEMU_DFILTER filter logging based on address range +-D logfile QEMU_LOG_FILENAME write logs to 'logfile' (default stderr) +-p pagesize QEMU_PAGESIZE set the host page size to 'pagesize' +-one-insn-per-tb QEMU_ONE_INSN_PER_TB run with one guest instruction per emulated TB +-singlestep QEMU_SINGLESTEP deprecated synonym for -one-insn-per-tb +-strace QEMU_STRACE log system calls +-seed QEMU_RAND_SEED Seed for pseudo-random number generator +-trace QEMU_TRACE [[enable=]][,events=][,file=] +-version QEMU_VERSION display version information and exit +-perfmap QEMU_PERFMAP Generate a /tmp/perf-${pid}.map file for perf +-jitdump QEMU_JITDUMP Generate a jit-${pid}.dump file for perf + +Defaults: +QEMU_LD_PREFIX = /usr/gnemul/qemu-riscv64 +QEMU_STACK_SIZE = 8388608 byte + +You can use -E and -U options or the QEMU_SET_ENV and +QEMU_UNSET_ENV environment variables to set and unset +environment variables for the target process. +It is possible to provide several variables by separating them +by commas in getsubopt(3) style. Additionally it is possible to +provide the -E and -U options multiple times. +The following lines are equivalent: + -E var1=val2 -E var2=val2 -U LD_PRELOAD -U LD_DEBUG + -E var1=val2,var2=val2 -U LD_PRELOAD,LD_DEBUG + QEMU_SET_ENV=var1=val2,var2=val2 QEMU_UNSET_ENV=LD_PRELOAD,LD_DEBUG +Note that if you provide several changes to a single variable +the last change will stay in effect. + +See for how to report bugs. +More information on the QEMU project at . diff --git a/corpus/qemu-riscv64-static/8.2.2/meta.toml b/corpus/qemu-riscv64-static/8.2.2/meta.toml new file mode 100644 index 00000000..a4d48f38 --- /dev/null +++ b/corpus/qemu-riscv64-static/8.2.2/meta.toml @@ -0,0 +1,45 @@ +[bless] +provenance = "agent" + +[tool] +name = "qemu-riscv64-static" +version = "8.2.2" +platform = "ubuntu-24.04" +captured_with = "mandible 0.7.0" + +[[capture]] +argv = ["qemu-riscv64-static", "--help"] +stdout = "help.txt" + +[contract] +expected_framework = "generic" +min_status = "ok" +min_subcommands = 0 + +# S-166: the header row names its own three columns, `Argument`, +# `Env-variable`, `Description`. The unfixed parser read every row at one +# column gap, so the env-variable cell glued onto the description +# (`-g` carried "QEMU_GDB wait gdb connection to 'port'") and two rows +# lost their own value name outright once the glued text was later +# repaired away (`-cpu` lost `model`, `-dfilter` lost `range[,...]`). A +# fabricated group and a fabricated `-E` row also appeared, folded in +# from the prose paragraph below the table; the table now ends at the +# blank line before `Defaults:`, so neither survives. +must_contain_flags = ["-cpu", "-dfilter", "-g"] + +# The prose paragraph below the table ("The following lines are +# equivalent:") used to fold onto the last real row and fabricate a +# second, differently-valued `-E` entity under itself as a group label. +must_not_contain_flag_group_prefixes = ["The following lines are equivalent"] + +verdict_scope = ["flags", "descriptions"] + +[contract.must_value_name] +"-cpu" = "model" +"-dfilter" = "range[,...]" + +[contract.must_describe] +"-g" = "wait gdb connection" + +[contract.must_not_describe] +"-g" = "QEMU_GDB" diff --git a/docs/design.md b/docs/design.md index a6364d82..3899c0ff 100644 --- a/docs/design.md +++ b/docs/design.md @@ -1855,30 +1855,36 @@ sections. within one flag's list. A tool's own scope-flag columns (ffmpeg's `ED.VAS.....`) stay verbatim inside the description; mandible parses no meaning out of them. -8. Capped shared column, per section. Every list section computes its own +8. A flag's own `env_var` cross-reference (§4.5) renders as its own + `env: FOO` line, indented the same two columns past the description + column as `values:`, never folded into the description. Distinct from + the `ENVIRONMENT` section below: this is one flag's own row-level + relation, not a variable documented as an item in its own right + (docs/shapes.md S-166). +9. Capped shared column, per section. Every list section computes its own column, fitted to roughly the p90 row width, measured from the pane's left edge through the placeholder's end. Every description line in the section, first line and continuation alike, begins at that column. Never a per-row column, never a global uncapped one. A wrapped entry is one logical row for selection and scroll math. -9. A head that reaches the column pushes its own first line, and only - that, never truncated and never moving the column for the section. A - head too wide for the pane wraps within the head area, each line at - its own spelling's column, description beginning on the line beneath - at the shared column. -10. A narrow pane moves the column, not the layout (§9.1a): clamped down +10. A head that reaches the column pushes its own first line, and only + that, never truncated and never moving the column for the section. A + head too wide for the pane wraps within the head area, each line at + its own spelling's column, description beginning on the line beneath + at the shared column. +11. A narrow pane moves the column, not the layout (§9.1a): clamped down until the description has its 28 columns, never below two past the long column. A 90-column terminal's 41-column detail pane clamps the column to 13, still holding a short-and-long pair. -11. POSITIONALS is inset by two columns; the flag-shaped sections are not. -12. The vertical gaps are the container hierarchy: two blank rows above a +12. POSITIONALS is inset by two columns; the flag-shaped sections are not. +13. The vertical gaps are the container hierarchy: two blank rows above a section header, one above a ruled group divider, none below either, none above the first header on the page. Each count is exact, not a minimum, and belongs to the block that opens, never to the one that closes. -13. ENVIRONMENT is display-only: documented vars under an explicit heading +14. ENVIRONMENT is display-only: documented vars under an explicit heading only, no probing, no inferred cross-references (§4.5). -14. Group dividers are label-first, like the headers above them. A `group` +15. Group dividers are label-first, like the headers above them. A `group` renders once as its label at column 0 followed by a rule to the pane's edge, mixed case; rows beneath sit at the section's normal margin. Section headers are CAPS with a count, group dividers @@ -1899,7 +1905,7 @@ sections. - A divider that opens its section drops its rule and its blank row, rendering its label alone at column 0 directly beneath the header. A divider later in the same section keeps both. -15. Descriptions always wrap. Sections are mandible's own layout, so +16. Descriptions always wrap. Sections are mandible's own layout, so nothing in them is ever clipped or horizontally scrolled. USAGE is mandible's own reconstruction too (§9 rule 9) and wraps the same way. `[ui] horizontal_scroll` governs only content whose layout is not ours diff --git a/docs/shapes.md b/docs/shapes.md index ad8e4642..60f4d7c0 100644 --- a/docs/shapes.md +++ b/docs/shapes.md @@ -2817,3 +2817,28 @@ entry's `tools` field and nothing else. It does not get a new entry. not gated: 3 are fail2ban-client's own still-open `set`/`add` gap (S-141's name rule, not this fix), the rest are false alarms on text that merely resembles a label followed by a row. 2026-09-07. + +### S-166: header-declared three-column option table, env-variable column + +- id: S-166 +- looks like: | + Argument Env-variable Description + -h print this help + -g port QEMU_GDB wait gdb connection to 'port' + -cpu model QEMU_CPU select CPU (-cpu help for list) +- tools: the whole `qemu-*-static` fleet (one help template, 42 tools) +- handling: A header row whose cells name its own columns is read at the column + offsets that header declares, stronger evidence than a heading. The middle + column, named as an environment variable, becomes the matching flag's own + `Entity::env_var` cross-reference (spec §4.5), never folded into the + description. `-cpu` and `-dfilter` also lost their own value name outright; + a header-declared table's own argument field is read directly rather than + through the general single-dash-long repair, whose bare-word value recovery + regressed `dbiprof`'s `-match=K=V` when tried document-wide. A fabricated + `-E` row and group, folded in from the prose paragraph below the table, + stop appearing once the table is read as ending at the header's own column + structure, at the first blank line. +- fleet: `header-declared-env-column` reads 0 findings post-fix on + `qemu-riscv64-static` and `qemu-arm64-static`; raw-shape grep over the + seed's own captures reads 42 tools / 42 findings, the whole `qemu-*-static` + set, 2026-09-12. diff --git a/mandible-extract/src/help_text/sections/emit.rs b/mandible-extract/src/help_text/sections/emit.rs index 3a03eb16..b935de31 100644 --- a/mandible-extract/src/help_text/sections/emit.rs +++ b/mandible-extract/src/help_text/sections/emit.rs @@ -303,6 +303,44 @@ pub(super) fn emit_env_vars( (seen, seen) } +/// Emit a header-declared three-column option table's rows as flags +/// (docs/shapes.md S-166). The middle column becomes [`Entity::env_var`], +/// the flag's own cross-reference to the variable that row names for it +/// (spec §4.5) — never folded into `description`, and never a standalone +/// [`EntityKind::EnvVar`] item, since this is a per-row relation a named +/// column header states, not a variable documented as an item in its own +/// right. +pub(super) fn emit_three_column_env_table( + group: Option, + rows: Vec, + out: &mut ParsedHelp, +) -> (usize, usize) { + let mut seen = 0usize; + let mut clean = 0usize; + for (argument, env_var, description) in rows { + if out.flags.len() >= MAX_RECOVERED_ENTRIES { + break; + } + seen += 1; + let spec = three_column_argument_spec(&argument); + if spec.spellings.is_empty() { + continue; + } + if spec.fully_consumed { + clean += 1; + } + let mut flag = Entity::new(EntityKind::Flag, Provenance::single(Source::HelpText)); + flag.spellings = spec.spellings; + flag.value_name = spec.value_name; + flag.value_kind = spec.value_kind; + flag.group = group.clone(); + flag.description = non_empty_text(&description); + flag.env_var = env_var; + out.flags.push(flag); + } + (seen, clean) +} + /// True when `rest` is nothing but argument placeholders: uppercase /// metavariables (`UNIT`, `PATTERN`), optionally bracketed (`[UNIT...]`), /// `...`-repeated, `|`-alternated (`PATTERN...|PID...`), or diff --git a/mandible-extract/src/help_text/sections/mod.rs b/mandible-extract/src/help_text/sections/mod.rs index 54495ee5..5b7a5dc3 100644 --- a/mandible-extract/src/help_text/sections/mod.rs +++ b/mandible-extract/src/help_text/sections/mod.rs @@ -147,7 +147,14 @@ fn is_ignorable_heading(heading: &str) -> bool { // Deliberately not matching "see also": git's own command group // headings legitimately carry that phrase as a parenthetical aside. let lower = heading.to_lowercase(); - lower.starts_with("example") || lower.contains("report bugs") + // "are equivalent" introduces worked invocation-line comparisons, the + // same class as "example" — qemu's own "The following lines are + // equivalent:" (docs/shapes.md S-166), whose indented rows repeat a + // real flag's own spelling with a different value on each line and + // would otherwise read as further, fabricated rows of that flag. + lower.starts_with("example") + || lower.contains("report bugs") + || lower.contains("are equivalent") } /// True when `heading` positively names a section whose rows describe CLI @@ -1184,6 +1191,27 @@ fn emit_flush_heading( return i; } } + // A header-declared three-column option table (`Argument + // Env-variable Description`, the whole `qemu-*-static` fleet): + // checked before the word-grid reading below, which would otherwise + // read this same header row as a one-row grid and silently discard + // it (docs/design.md §7 Tier B rule 16). See docs/shapes.md S-166. + if i < lines.len() && leading_whitespace(lines[i]) == heading_indent { + if let Some((env_col, desc_col)) = three_column_env_table_header(lines[i]) { + let (end, rows) = scan_three_column_env_table(lines, i + 1, env_col, desc_col); + i = end; + st.in_ignorable_section = false; + st.command_mode = false; + let (seen, clean) = emit_three_column_env_table( + meaningful_flag_group(heading.clone()), + rows, + st.result, + ); + st.total_entries += seen; + st.clean_entries += clean; + return i; + } + } // Nothing more-indented follows. openssl and BSD-style // listings generally present a command list as a same-indent // word grid: a heading followed by lines of several bare diff --git a/mandible-extract/src/help_text/sections/scan.rs b/mandible-extract/src/help_text/sections/scan.rs index 2ead8f34..5a558e36 100644 --- a/mandible-extract/src/help_text/sections/scan.rs +++ b/mandible-extract/src/help_text/sections/scan.rs @@ -888,6 +888,143 @@ pub(super) fn scan_env_var_table(lines: &[&str], start: usize) -> Option<(usize, (rows.len() >= MIN_ENV_VAR_TABLE_ROWS).then_some((i, rows)) } +/// A header-declared three-column option table's own row, read at the +/// header's own offsets: the argument field (a flag spec, parsed the +/// ordinary way), the environment variable that row names for it (empty +/// when the row names none), and the description. See docs/shapes.md +/// S-166. +pub(super) type ThreeColumnRow = (String, Option, String); + +fn is_argument_column_label(cell: &str) -> bool { + matches!( + cell.trim().to_lowercase().as_str(), + "argument" | "arguments" | "option" | "options" | "flag" | "flags" + ) +} + +/// Env-var column labels, compared with hyphens and spaces removed so +/// `Env-variable`, `Env variable` and `Environment Variable` all match the +/// same shape rather than three separate literals. +fn is_env_var_column_label(cell: &str) -> bool { + let compact: String = cell + .trim() + .to_lowercase() + .chars() + .filter(|c| !matches!(c, '-' | ' ')) + .collect(); + matches!( + compact.as_str(), + "envvariable" | "environmentvariable" | "envvar" + ) +} + +fn is_description_column_label(cell: &str) -> bool { + cell.trim().eq_ignore_ascii_case("description") +} + +/// Column offsets a three-column option table declares for itself on its +/// own header row (`Argument Env-variable Description`, qemu's own +/// `--help`): byte offsets of the env-var and description columns. A +/// named column header is stronger evidence than a heading (docs/design.md +/// §7 Tier B rule 16), so every row under it is read at these exact +/// offsets rather than by the generic single-gap description split, which +/// otherwise glues the middle column onto the description. See +/// docs/shapes.md S-166. +pub(super) fn three_column_env_table_header(line: &str) -> Option<(usize, usize)> { + let cols = split_columns(line); + let [c0, c1, c2]: [&str; 3] = cols.try_into().ok()?; + if !is_argument_column_label(c0) + || !is_env_var_column_label(c1) + || !is_description_column_label(c2) + { + return None; + } + let env_col = line.find(c1)?; + let desc_col = line[env_col..].find(c2)? + env_col; + Some((env_col, desc_col)) +} + +/// [`parse_flag_spec`] reads a multi-char single-dash name plus a spaced +/// bare word as a short flag with a glued value (`-cpu model` becomes +/// `-c` valued `"pu"`), and the post-pass that untangles this elsewhere +/// only recovers a bracket-delimited spaced value, dropping a bare word +/// like `model`. Widening that post-pass regressed `dbiprof` fleet-wide, +/// so this instead reads two tokens straight off a header-declared +/// table's own row, whose boundary is already fixed by the header's own +/// offsets: a single-dash name (2+ chars) then a lowercase-led bare word +/// (an optional `[...]` suffix stays glued). See docs/shapes.md S-166. +pub(super) fn three_column_argument_spec(argument: &str) -> FlagSpec { + let mut words = argument.split_whitespace(); + if let (Some(name_tok), Some(value_tok), None) = (words.next(), words.next(), words.next()) { + if let Some(name) = name_tok + .strip_prefix('-') + .filter(|n| n.len() > 1 && !n.starts_with('-')) + { + let mut chars = value_tok.chars(); + let starts_lowercase = chars.next().is_some_and(|c| c.is_ascii_lowercase()); + if starts_lowercase + && chars.all(|c| { + c.is_ascii_alphanumeric() || matches!(c, '-' | '_' | '[' | ']' | ',' | '.') + }) + { + return FlagSpec { + spellings: vec![Spelling::single_dash(name)], + value_name: Some(value_tok.to_string()), + value_kind: ValueKind::Required, + fully_consumed: true, + ..FlagSpec::default() + }; + } + } + } + parse_flag_spec(argument) +} + +/// Split one row of a header-declared three-column option table at the +/// header's own offsets. A row shorter than `env_col`/`desc_col` (`-h` +/// with no env var and a short description) reads the missing columns as +/// empty rather than panicking off a byte boundary — `get` never a raw +/// index (AGENTS.md §2). See docs/shapes.md S-166. +fn split_three_column_row(line: &str, env_col: usize, desc_col: usize) -> ThreeColumnRow { + let argument = line.get(..env_col).unwrap_or(line).trim().to_string(); + let env_var = line + .get(env_col..desc_col) + .map(str::trim) + .filter(|s| !s.is_empty()) + .map(str::to_string); + let description = line.get(desc_col..).unwrap_or("").trim().to_string(); + (argument, env_var, description) +} + +/// Scan the rows of a header-declared three-column option table, +/// immediately after its own header row. Stops at the first blank line or +/// the first row whose argument field does not open with a dash — the +/// table's own column structure ends there, so qemu's own trailing prose +/// about `-E`/`-U` (`examples-block-contaminates-last-flag`'s own family) +/// is never read as a further row. See docs/shapes.md S-166. +pub(super) fn scan_three_column_env_table( + lines: &[&str], + start: usize, + env_col: usize, + desc_col: usize, +) -> (usize, Vec) { + let mut rows = Vec::new(); + let mut i = start; + while i < lines.len() { + let line = lines[i]; + if line.trim().is_empty() { + break; + } + let row = split_three_column_row(line, env_col, desc_col); + if !row.0.starts_with('-') { + break; + } + i += 1; + rows.push(row); + } + (i, rows) +} + #[cfg(test)] mod modifier_tests { use super::*; diff --git a/mandible-tui/src/render/detail_pane/entity.rs b/mandible-tui/src/render/detail_pane/entity.rs index ad67d011..b71f7149 100644 --- a/mandible-tui/src/render/detail_pane/entity.rs +++ b/mandible-tui/src/render/detail_pane/entity.rs @@ -244,7 +244,17 @@ pub(super) fn entity_line( format!("values: {joined}") }); - if description_text.is_none() && values_line.is_none() && !has_choice_descriptions { + // A flag's own environment-variable cross-reference (spec §4.5) + // renders the same way `values:` does: its own line, two columns past + // the description column, never folded into the description text. See + // docs/shapes.md S-166. + let env_line = flag.env_var.as_ref().map(|v| format!("env: {v}")); + + if description_text.is_none() + && values_line.is_none() + && env_line.is_none() + && !has_choice_descriptions + { if !head.is_empty() { return head; } @@ -320,6 +330,19 @@ pub(super) fn entity_line( lines.extend(choice_detail_lines(flag, column, width, color_enabled)); } + if let Some(env_line) = env_line { + let env_column = column + 2; + let env_width = width.saturating_sub(env_column).max(1); + let env_indent = " ".repeat(env_column); + let env_style = style::muted(color_enabled); + for chunk in wrap_words(&env_line, env_width) { + lines.push(Line::from(Span::styled( + format!("{env_indent}{chunk}"), + env_style, + ))); + } + } + lines } diff --git a/xtask/src/coverage/mod.rs b/xtask/src/coverage/mod.rs index 468324bd..b4fb95ee 100644 --- a/xtask/src/coverage/mod.rs +++ b/xtask/src/coverage/mod.rs @@ -10,6 +10,7 @@ mod aggregate; mod fingerprint; mod render_markdown; mod render_text; +mod round10; mod round7; mod round8; mod round9; diff --git a/xtask/src/coverage/round10.rs b/xtask/src/coverage/round10.rs new file mode 100644 index 00000000..43c81b1a --- /dev/null +++ b/xtask/src/coverage/round10.rs @@ -0,0 +1,23 @@ +//! The round-10 family detectors. `header-declared-env-column` is atlas +//! S-166, W6's own header-declared three-column option table. Split into +//! its own file for the same line-count reason `round7.rs`, `round8.rs` +//! and `round9.rs` are. + +use super::score::FAMILY_DETECTOR_SAMPLES_PER_ROW; +use crate::detector::{Detector, ToolEvidence}; +use mandible_core::CommandNode; + +pub(super) fn round10_family_counts( + raw: &str, + root: &CommandNode, +) -> Vec<(&'static str, usize, Vec)> { + let cap = FAMILY_DETECTOR_SAMPLES_PER_ROW; + let evidence = ToolEvidence { raw, root }; + let env_col = + crate::detector::header_declared_env_column::HeaderDeclaredEnvColumn.hits(&evidence); + vec![( + "header-declared-env-column", + env_col.len(), + env_col.into_iter().take(cap).collect(), + )] +} diff --git a/xtask/src/coverage/score.rs b/xtask/src/coverage/score.rs index ba7466c3..cba33bae 100644 --- a/xtask/src/coverage/score.rs +++ b/xtask/src/coverage/score.rs @@ -497,6 +497,7 @@ fn vim_family_counts( counts.extend(super::round7::round7_usage_family_counts(&raw, root)); counts.extend(super::round8::round8_family_counts(&raw, root)); counts.extend(super::round9::round9_family_counts(&raw, root)); + counts.extend(super::round10::round10_family_counts(&raw, root)); counts } diff --git a/xtask/src/detector/header_declared_env_column.rs b/xtask/src/detector/header_declared_env_column.rs new file mode 100644 index 00000000..e2becb02 --- /dev/null +++ b/xtask/src/detector/header_declared_env_column.rs @@ -0,0 +1,230 @@ +//! `header-declared-env-column` (atlas S-166): a header-declared three- +//! column option table (`Argument`/`Env-variable`/`Description`, the +//! whole `qemu-*-static` fleet) whose own header row names its middle +//! column as an environment variable. The unfixed parser glues that +//! column onto the matching flag's description instead of reading it as +//! the flag's own [`Entity::env_var`] cross-reference (spec §4.5, §7 +//! Tier B rule 16). +//! +//! Independent re-implementation of the header/row shape — no shared +//! code with `mandible_extract::help_text::sections`, so the detector +//! cannot agree with the parser by construction. +//! +//! The seed-7 audit labels `qemu-riscv64-static` "incomplete" and +//! describes this exact shape ("a triple column help text ... flags +//! being an alias of the env vars"), but no `DEFECT_FAMILIES` entry +//! names it yet and the entry carries no derived family, so +//! [`Detector::family`] returns `None` (spec §13.1e rule 6) rather than +//! forcing it onto an unrelated family. + +use crate::detector::{Detector, Expect, Scope, SelfCheck, ToolEvidence}; +use mandible_core::CommandNode; + +pub struct Finding { + pub argument: String, + pub env_var: String, +} + +pub struct Report { + pub findings: Vec, +} + +fn is_argument_label(cell: &str) -> bool { + matches!( + cell.trim().to_lowercase().as_str(), + "argument" | "arguments" | "option" | "options" | "flag" | "flags" + ) +} + +fn is_env_var_label(cell: &str) -> bool { + let compact: String = cell + .trim() + .to_lowercase() + .chars() + .filter(|c| !matches!(c, '-' | ' ')) + .collect(); + matches!( + compact.as_str(), + "envvariable" | "environmentvariable" | "envvar" + ) +} + +fn is_description_label(cell: &str) -> bool { + cell.trim().eq_ignore_ascii_case("description") +} + +/// Column offsets a header row declares for itself, an independent copy +/// of `mandible_extract`'s own reading. +fn header_offsets(line: &str) -> Option<(usize, usize)> { + let cols: Vec<&str> = line + .trim() + .split(" ") + .map(str::trim) + .filter(|s| !s.is_empty()) + .collect(); + let [c0, c1, c2]: [&str; 3] = cols.try_into().ok()?; + if !is_argument_label(c0) || !is_env_var_label(c1) || !is_description_label(c2) { + return None; + } + let env_col = line.find(c1)?; + let desc_col = line[env_col..].find(c2)? + env_col; + Some((env_col, desc_col)) +} + +/// True when `root` carries the row's own env var as a cross-reference on +/// the matching flag, and that flag's description does not also repeat +/// it — the fixed shape. False for both directions of the defect: the +/// contaminated description, and a flag that never gained `env_var` at +/// all. +fn row_is_clean(root: &CommandNode, arg_name: &str, env_var: &str) -> bool { + let Some(flag) = root + .flags() + .find(|f| f.spellings.iter().any(|s| s.name == arg_name)) + else { + return false; + }; + let carries_reference = flag.env_var.as_deref() == Some(env_var); + let description_clean = !flag + .description + .as_ref() + .is_some_and(|d| d.as_str().contains(env_var)); + carries_reference && description_clean +} + +pub fn detect(raw: &str, root: &CommandNode) -> Report { + let mut findings = Vec::new(); + let lines: Vec<&str> = raw.lines().collect(); + let mut i = 0; + while i < lines.len() { + let Some((env_col, desc_col)) = header_offsets(lines[i]) else { + i += 1; + continue; + }; + let mut j = i + 1; + while j < lines.len() && !lines[j].trim().is_empty() { + let line = lines[j]; + let argument = line.get(..env_col).unwrap_or(line).trim(); + if !argument.starts_with('-') { + break; + } + let env_var = line + .get(env_col..desc_col) + .map(str::trim) + .unwrap_or_default(); + if !env_var.is_empty() { + let name = argument + .split_whitespace() + .next() + .unwrap_or(argument) + .trim_start_matches('-'); + if !row_is_clean(root, name, env_var) { + findings.push(Finding { + argument: argument.to_string(), + env_var: env_var.to_string(), + }); + } + } + j += 1; + } + i = j; + } + Report { findings } +} + +pub struct HeaderDeclaredEnvColumn; + +impl Detector for HeaderDeclaredEnvColumn { + fn name(&self) -> &'static str { + "header-declared-env-column" + } + + fn family(&self) -> Option<&'static str> { + None + } + + fn describes(&self) -> &'static str { + "a header-declared three-column option table (Argument/Env-variable/Description) whose \ + middle column never reached the matching flag's own env_var cross-reference" + } + + fn hits(&self, evidence: &ToolEvidence<'_>) -> Vec { + detect(evidence.raw, evidence.root) + .findings + .iter() + .map(|f| { + format!( + "{:?} never carried {:?} as its own env_var", + f.argument, f.env_var + ) + }) + .collect() + } + + fn scope(&self) -> Scope { + Scope::full() + } + + fn self_checks(&self) -> Vec { + self_checks() + } +} + +// ---------------------------------------------------------------------- +// Self-checks +// ---------------------------------------------------------------------- + +use mandible_core::{Entity, Provenance, Spelling, Text}; + +const QEMU_HEADER_AND_ROW: &str = "Argument Env-variable Description\n\ + -cpu model QEMU_CPU select CPU (-cpu help for list)\n"; + +fn flag_with(short: char, description: Option<&str>, env_var: Option<&str>) -> Entity { + let mut e = Entity::flag_spelled(Some(short), None, false, false, Provenance::default()); + e.spellings = vec![Spelling::single_dash("cpu")]; + e.description = description.map(Text::sanitize); + e.env_var = env_var.map(str::to_string); + e +} + +fn node_with(flags: Vec) -> CommandNode { + let mut root = CommandNode::new("qemu-riscv64-static", Provenance::default()); + root.set_entities_of(mandible_core::EntityKind::Flag, flags); + root +} + +pub(crate) fn self_checks() -> Vec { + vec![ + SelfCheck { + name: "qemu's own contaminated row, env var glued onto the description", + why: "the defect itself: QEMU_CPU never reached its own env_var and still sits in \ + the description", + expect: Expect::Fires(1), + raw: QEMU_HEADER_AND_ROW.to_string(), + root: node_with(vec![flag_with( + 'c', + Some("QEMU_CPU select CPU (-cpu help for list)"), + None, + )]), + }, + SelfCheck { + name: "qemu's own row, already read at the header's own offsets", + why: "once env_var carries the cross-reference and the description is clean, the \ + same row must go silent", + expect: Expect::Silent, + raw: QEMU_HEADER_AND_ROW.to_string(), + root: node_with(vec![flag_with( + 'c', + Some("select CPU (-cpu help for list)"), + Some("QEMU_CPU"), + )]), + }, + SelfCheck { + name: "an ordinary two-column table, no env-variable header at all", + why: "a table whose header never names an environment-variable column carries no \ + finding to report", + expect: Expect::Silent, + raw: "Option Description\n-cpu model select CPU\n".to_string(), + root: node_with(vec![flag_with('c', Some("select CPU"), None)]), + }, + ] +} diff --git a/xtask/src/detector/mod.rs b/xtask/src/detector/mod.rs index 15c03797..5afcfa75 100644 --- a/xtask/src/detector/mod.rs +++ b/xtask/src/detector/mod.rs @@ -102,6 +102,10 @@ pub(crate) mod lowdown_bullet_option_row; // parser change ships for it. pub(crate) mod option_table_multiword_value_name; +// Round-10 family detector (atlas S-166, W6's header-declared +// three-column option table), same direct-`Detector`-impl shape. +pub(crate) mod header_declared_env_column; + pub(crate) use calibration::*; pub(crate) use commands::*; pub(crate) use detectors_families::*; @@ -748,6 +752,7 @@ pub fn registry() -> Vec> { Box::new(crate::centered_label_baseline::LabelPrecedesShallowerLine), Box::new(crate::centered_label_baseline::MissingRowAfterLabel), Box::new(option_table_multiword_value_name::OptionTableMultiwordValueName), + Box::new(header_declared_env_column::HeaderDeclaredEnvColumn), ] } From 4843b914d0cd38161358ae80bc51cf1daf250818 Mon Sep 17 00:00:00 2001 From: Sadig Akhund Date: Sat, 12 Sep 2026 20:02:29 +0400 Subject: [PATCH 3/9] extract: admit a `+word` option row The `+` sigil gate reads a whole letter-led run as the spelling, so a `+bs` or `+i` row becomes an option instead of ending its block. docs/shapes.md S-095, S-163. Co-Authored-By: Claude Fable 5.1 --- .../src/help_text/sections/flag_rows.rs | 70 +++++-- xtask/src/coverage/round10.rs | 21 +- xtask/src/detector/mod.rs | 2 + xtask/src/detector/plus_word_option.rs | 194 ++++++++++++++++++ 4 files changed, 266 insertions(+), 21 deletions(-) create mode 100644 xtask/src/detector/plus_word_option.rs diff --git a/mandible-extract/src/help_text/sections/flag_rows.rs b/mandible-extract/src/help_text/sections/flag_rows.rs index 94deb83c..784b4d14 100644 --- a/mandible-extract/src/help_text/sections/flag_rows.rs +++ b/mandible-extract/src/help_text/sections/flag_rows.rs @@ -426,15 +426,28 @@ pub(super) enum FlagsBlockRow<'a> { // do not share code). /// True when `token` is a `+`-prefixed option spelling this family -/// claims: bare `+`, or `+` followed by a bracketed placeholder -/// (`+`, `+`) — never `++` or a token with a real letter -/// straight after the sigil (`+d`, which [`is_flag_shaped`] already -/// reads). See docs/shapes.md S-095. +/// claims: bare `+`, `+` followed by a bracketed placeholder (`+`, +/// `+`), or `+word` — a whole run of letters/digits/`-` opening with +/// a letter (`+bs`, `+byteswappedclients`, `+i`) — never `++` and never a +/// token whose run is punctuation-only. `looks_like_flag_start` never +/// reads a `+`-led line as an ordinary entry on its own (no signal tells +/// it apart from prose that merely starts with `+`), so this is the only +/// gate a `+word` row ever passes through. See docs/shapes.md S-095, +/// S-163. pub(super) fn is_claimed_plus_token(token: &str) -> bool { let Some(rest) = token.strip_prefix('+') else { return false; }; - rest.is_empty() || rest.starts_with('<') + if rest.is_empty() || rest.starts_with('<') { + return true; + } + let mut chars = rest.chars(); + match chars.next() { + Some(c) if c.is_ascii_alphabetic() => { + chars.all(|c| c.is_ascii_alphanumeric() || c == '-') + } + _ => false, + } } /// True when `line`'s own leading token is flag-shaped evidence for the @@ -491,22 +504,49 @@ pub(super) fn parse_plus_sigil_spec(spec_text: &str) -> FlagSpec { let Some(rest) = trimmed.strip_prefix('+') else { return FlagSpec::default(); }; - if !(rest.is_empty() || rest.starts_with('<')) { + if rest.is_empty() { + return FlagSpec { + spellings: vec![Spelling::bare("+")], + fully_consumed: true, + ..FlagSpec::default() + }; + } + if rest.starts_with('<') { + let mut spec = FlagSpec { + spellings: vec![Spelling::bare("+")], + ..FlagSpec::default() + }; + let tail = parse_flag_spec(rest); + spec.value_name = tail.value_name; + spec.value_kind = tail.value_kind; + spec.spellings.extend(tail.spellings); + spec.fully_consumed = tail.fully_consumed; + return spec; + } + // `+word` (S-163): the whole run is the spelling, verbatim, `+` + // included — a second, distinct entity from any `-word` sibling the + // table also carries, never a value glued onto a bare `+`. + let word_end = rest + .char_indices() + .find(|(_, c)| !(c.is_ascii_alphanumeric() || *c == '-')) + .map_or(rest.len(), |(i, _)| i); + if word_end == 0 { return FlagSpec::default(); } + let word = &rest[..word_end]; + let after = rest[word_end..].trim_start(); let mut spec = FlagSpec { - spellings: vec![Spelling::bare("+")], + spellings: vec![Spelling::bare(format!("+{word}"))], + fully_consumed: after.is_empty(), ..FlagSpec::default() }; - if rest.is_empty() { - spec.fully_consumed = true; - return spec; + if !after.is_empty() { + let tail = parse_flag_spec(after); + spec.value_name = tail.value_name; + spec.value_kind = tail.value_kind; + spec.spellings.extend(tail.spellings); + spec.fully_consumed = tail.fully_consumed; } - let tail = parse_flag_spec(rest); - spec.value_name = tail.value_name; - spec.value_kind = tail.value_kind; - spec.spellings.extend(tail.spellings); - spec.fully_consumed = tail.fully_consumed; spec } diff --git a/xtask/src/coverage/round10.rs b/xtask/src/coverage/round10.rs index 19ffa693..34b4cdb8 100644 --- a/xtask/src/coverage/round10.rs +++ b/xtask/src/coverage/round10.rs @@ -1,5 +1,6 @@ //! Round-10 family detectors. Atlas S-162: a leading option-rejection -//! diagnostic line fused into the root description. +//! diagnostic line fused into the root description. S-163: a `+word` +//! option row. use crate::detector::{Detector, ToolEvidence}; use mandible_core::CommandNode; @@ -12,9 +13,17 @@ pub(super) fn round10_family_counts( let evidence = ToolEvidence { raw, root }; let leading_diagnostic = crate::detector::leading_diagnostic_line::LeadingDiagnosticLine.hits(&evidence); - vec![( - "leading-diagnostic-line", - leading_diagnostic.len(), - leading_diagnostic.into_iter().take(cap).collect(), - )] + let plus_word = crate::detector::plus_word_option::PlusWordOption.hits(&evidence); + vec![ + ( + "leading-diagnostic-line", + leading_diagnostic.len(), + leading_diagnostic.into_iter().take(cap).collect(), + ), + ( + "plus-word-option", + plus_word.len(), + plus_word.into_iter().take(cap).collect(), + ), + ] } diff --git a/xtask/src/detector/mod.rs b/xtask/src/detector/mod.rs index 4b26f933..73b0d24c 100644 --- a/xtask/src/detector/mod.rs +++ b/xtask/src/detector/mod.rs @@ -56,6 +56,7 @@ pub(crate) mod glued_optional_group_spelling; pub(crate) mod glued_uppercase_shared_prefix; pub(crate) mod hash_in_spelling; pub(crate) mod leading_diagnostic_line; +pub(crate) mod plus_word_option; pub(crate) mod multi_operand_usage_tail; pub(crate) mod nested_bracket_value; pub(crate) mod numbered_variadic_usage_tail; @@ -750,6 +751,7 @@ pub fn registry() -> Vec> { Box::new(crate::centered_label_baseline::MissingRowAfterLabel), Box::new(option_table_multiword_value_name::OptionTableMultiwordValueName), Box::new(leading_diagnostic_line::LeadingDiagnosticLine), + Box::new(plus_word_option::PlusWordOption), ] } diff --git a/xtask/src/detector/plus_word_option.rs b/xtask/src/detector/plus_word_option.rs new file mode 100644 index 00000000..52c97bab --- /dev/null +++ b/xtask/src/detector/plus_word_option.rs @@ -0,0 +1,194 @@ +//! `plus-word-option` (atlas S-163): a `+word` option row (`+bs`, `+i`, +//! `+s`) — a letter-led run after the sigil, distinct from S-095's own +//! `plus-prefixed-option` (bare `+`/`+` only) — reaches no +//! entity in the tree. Requires indentation, the same evidence +//! `mandible_core::family_row::leading_token` and the parser's own +//! `scan_flags_block` both require, so an unindented specimen (`Xvfb`'s +//! own column-0 rows) is out of this family's reach until S-165's +//! headingless-table defect is fixed. +//! +//! A separate detector rather than a widening of `plus-prefixed-option`: +//! that family is already `REPAIRED` and ratchet-gated at zero +//! (docs/shapes.md S-095), so folding a still-open shape into it would +//! break the gate on tools this fix has not reached. Mirrors +//! `mandible_extract::help_text::sections::flag_rows::is_claimed_plus_token`'s +//! `+word` arm, checked independently here since a detector reads only +//! `raw`+`root`. + +use crate::detector::{Detector, Expect, Scope, SelfCheck, ToolEvidence}; +use crate::family_row::{leading_token, opens_description_column}; +use mandible_core::{CommandNode, Provenance, Source}; + +/// True when `token` is a `+word` spelling this family claims: `+` +/// followed by a run opening with a letter, every later character +/// alphanumeric or `-` — never a bare `+`, a bracketed placeholder +/// (S-095's own claim), or a token with no real letter run at all. +fn is_plus_word_token(token: &str) -> bool { + let Some(rest) = token.strip_prefix('+') else { + return false; + }; + let mut chars = rest.chars(); + match chars.next() { + Some(c) if c.is_ascii_alphabetic() => { + chars.all(|c| c.is_ascii_alphanumeric() || c == '-') + } + _ => false, + } +} + +fn tree_has_spelling(root: &CommandNode, token: &str) -> bool { + root.flags() + .any(|e| e.spellings.iter().any(|s| s.name == token)) +} + +/// True when `line`'s own leading token is flag-shaped evidence: a real +/// `-`-prefixed flag, or this same family's own `+word` claim. +fn is_flag_shaped_neighbor(line: &str) -> bool { + let Some((token, _)) = leading_token(line) else { + return false; + }; + let token = token.trim_end_matches(','); + if let Some(rest) = token.strip_prefix("--") { + return rest.is_empty() || rest.chars().next().is_some_and(|c| c.is_ascii_alphabetic()); + } + if let Some(rest) = token.strip_prefix('-') { + return rest + .chars() + .next() + .is_some_and(|c| c.is_ascii_alphanumeric()); + } + is_plus_word_token(token) +} + +fn has_flag_shaped_neighbor(lines: &[&str], i: usize) -> bool { + let above = lines[..i].iter().rev().find(|l| !l.trim().is_empty()); + let below = lines[i + 1..].iter().find(|l| !l.trim().is_empty()); + above.is_some_and(|l| is_flag_shaped_neighbor(l)) + || below.is_some_and(|l| is_flag_shaped_neighbor(l)) +} + +pub struct PlusWordOption; + +impl Detector for PlusWordOption { + fn name(&self) -> &'static str { + "plus-word-option" + } + + fn family(&self) -> Option<&'static str> { + None + } + + fn describes(&self) -> &'static str { + "an indented `+word` option row, opening a real description column beside a flag-shaped \ + neighbor, with no entity anywhere spelled that exact token" + } + + fn hits(&self, evidence: &ToolEvidence<'_>) -> Vec { + let mut findings = Vec::new(); + let lines: Vec<&str> = evidence.raw.lines().collect(); + for (i, line) in lines.iter().enumerate() { + let Some((token, rest)) = leading_token(line) else { + continue; + }; + let token = token.trim_end_matches(','); + if !is_plus_word_token(token) || !opens_description_column(rest) { + continue; + } + if !has_flag_shaped_neighbor(&lines, i) { + continue; + } + if !tree_has_spelling(evidence.root, token) { + findings.push(format!("{token:?} never became a flag, from {line:?}")); + } + } + findings + } + + fn scope(&self) -> Scope { + Scope::full() + } + + fn self_checks(&self) -> Vec { + fn node_with_flags(name: &str, flags: Vec) -> CommandNode { + let mut root = CommandNode::new(name, Provenance::single(Source::HelpText)); + root.set_entities_of(mandible_core::EntityKind::Flag, flags); + root + } + fn plus_word_flag(word: &str) -> mandible_core::Entity { + let mut e = mandible_core::Entity::new( + mandible_core::EntityKind::Flag, + Provenance::single(Source::HelpText), + ); + e.spellings + .push(mandible_core::Spelling::bare(format!("+{word}"))); + e + } + + let fzf_raw = " -i Case-insensitive match (default: smart-case \ + match)\n +i Case-sensitive match\n" + .to_string(); + // `Xvfb`'s own row shape (multi-letter, `-word` neighbor) at its + // own indent (two spaces) — column 0, `Xvfb`'s real indentation, + // carries no evidence at all for `leading_token` (S-165's own + // headingless-table defect, not this family's to fix). + let indented_multiletter_raw = " -br create root window with black \ + background\n +bs enable any backing \ + store support\n -bs disable any \ + backing store support\n" + .to_string(); + + vec![ + SelfCheck { + name: "fzf's own bytes, `+i` dropped", + why: "the defect itself: `+i`'s leading token is no entity's spelling", + expect: Expect::Fires(1), + raw: fzf_raw.clone(), + root: node_with_flags("fzf", vec![]), + }, + SelfCheck { + name: "fzf's own bytes, `+i` recovered as its own spelling", + why: "once the tree carries an entity spelled `+i`, the same raw row goes silent", + expect: Expect::Silent, + raw: fzf_raw, + root: node_with_flags("fzf", vec![plus_word_flag("i")]), + }, + SelfCheck { + name: "an indented multi-letter `+word` row, `Xvfb`'s own spelling shape", + why: "the multi-letter shape, beside a real `-`-prefixed neighbor, at a real \ + indent", + expect: Expect::Fires(1), + raw: indented_multiletter_raw.clone(), + root: node_with_flags("Xvfb", vec![]), + }, + SelfCheck { + name: "the same row recovered as its own spelling", + why: "once recovered, the same raw row goes silent", + expect: Expect::Silent, + raw: indented_multiletter_raw, + root: node_with_flags("Xvfb", vec![plus_word_flag("bs")]), + }, + SelfCheck { + name: "a bare `+` row, S-095's own claim, not this family's", + why: "a bare `+` has no letter run at all and must never be claimed here", + expect: Expect::Silent, + raw: " + Start at end of file\n".to_string(), + root: node_with_flags("vim.basic", vec![]), + }, + SelfCheck { + name: "a `+` row, S-095's own claim, not this family's", + why: "a bracketed placeholder is S-095's own shape, distinct from a plain word", + expect: Expect::Silent, + raw: " +\t\tStart at line \n".to_string(), + root: node_with_flags("vim.basic", vec![]), + }, + SelfCheck { + name: "an unindented `+word`-led heading line, no flag-shaped neighbor", + why: "no flag-shaped neighbor is present, so the positive evidence this family \ + requires is absent", + expect: Expect::Silent, + raw: "+bs this is not a table row\n".to_string(), + root: node_with_flags("prog", vec![]), + }, + ] + } +} From 52b2cc115eaf28b380c9a4f72e099d6757985f6b Mon Sep 17 00:00:00 2001 From: Sadig Akhund Date: Sat, 12 Sep 2026 20:56:02 +0400 Subject: [PATCH 4/9] [S-163, S-164, S-165] finish the document-shape branch A headingless table no longer duplicates into the root description (S-165), which let `+word` and `+/-name`/`[+-]name` alternation rows reach a column-0 table (S-163). A root flag group equal to the node's own description is refused as a second copy of it (S-164). Xvfb: 69 to 78 flags, fzf: 63 to 65. fc-scan, fc-validate, grub-macbless and lto-dump stop showing their description sentence twice. Co-Authored-By: Claude Fable 5.1 --- CHANGELOG.md | 3 + corpus/Xvfb/audit-seed/expected.snap | 495 ++++++++++++++++++ corpus/Xvfb/audit-seed/meta.toml | 24 +- corpus/fc-validate/audit-seed2/expected.snap | 5 - corpus/fc-validate/audit-seed2/meta.toml | 4 + corpus/fzf/0.44.1/expected.snap | 24 + corpus/fzf/0.44.1/meta.toml | 12 +- corpus/lto-dump/13.3.0/expected.snap | 1 - corpus/lto-dump/13.3.0/meta.toml | 5 + docs/shapes.md | 106 ++++ .../src/help_text/sections/emit.rs | 136 +++-- .../src/help_text/sections/flag_rows.rs | 165 +++++- .../src/help_text/sections/heading.rs | 2 +- .../src/help_text/sections/mod.rs | 57 +- xtask/src/coverage/mod.rs | 2 +- xtask/src/coverage/round10.rs | 30 +- .../description_reused_as_group_label.rs | 108 ++++ .../headingless_table_in_root_description.rs | 160 ++++++ xtask/src/detector/leading_diagnostic_line.rs | 15 +- xtask/src/detector/mod.rs | 8 +- .../detector/plus_minus_alternation_option.rs | 182 +++++++ xtask/src/detector/plus_word_option.rs | 71 ++- 22 files changed, 1505 insertions(+), 110 deletions(-) create mode 100644 corpus/Xvfb/audit-seed/expected.snap create mode 100644 xtask/src/detector/description_reused_as_group_label.rs create mode 100644 xtask/src/detector/headingless_table_in_root_description.rs create mode 100644 xtask/src/detector/plus_minus_alternation_option.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index 8b8dd3e8..cf8c9516 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -34,6 +34,9 @@ once it reaches a published 0.1.0 release. - [S-145] A tab-separated value name on a single-dash long option survives the same repair now, so `mandible mksquashfs` and `mandible sqfstar` keep `-mem`, `-comp` and `-mkfs-time` with their placeholders instead of losing them. - [S-146] A heading with no indent step to its own rows, and a bare label above a nested option block, now name a group instead of leaving every row underneath ungrouped, so `mandible mksquashfs` shows its ten option headings and its five compressor names. - [S-162] A leading option-rejection diagnostic no longer fuses into the root description, so `mandible fuser` and `mandible nfsidmap` show their real description and usage instead of the tool's own complaint about the probe. +- [S-163] A `+word` option row and a `+/-name`/`[+-]name` alternation row now reach the tree, the second expanding to its own `+name` and `-name` entities, so `mandible fzf` keeps `+i` and `+s, --no-sort` with the rest of its `Search` group, and `mandible Xvfb` recovers every `+`-prefixed row. +- [S-164] A root flag group that only repeated the node's own description verbatim is dropped now, so `mandible fc-scan`, `mandible fc-validate`, `mandible grub-macbless` and `mandible lto-dump` no longer show the same sentence twice. +- [S-165] A headingless option table no longer duplicates into the root description, so `mandible Xvfb` shows its real description once and its whole flag table instead of the same text rendered twice. ## [0.7.0] - 2026-09-05 diff --git a/corpus/Xvfb/audit-seed/expected.snap b/corpus/Xvfb/audit-seed/expected.snap new file mode 100644 index 00000000..e01e8c9e --- /dev/null +++ b/corpus/Xvfb/audit-seed/expected.snap @@ -0,0 +1,495 @@ +name: Xvfb +description: 'use: X [:] [option]' +flags: +- spellings: + - -a + value_name: '#' + value_kind: Required + description: default pointer acceleration (factor) + provenance: + sources: + - help-text +- spellings: + - -ac + description: disable access control restrictions + provenance: + sources: + - help-text +- spellings: + - -audit + description: set audit trail level + provenance: + sources: + - help-text +- spellings: + - -auth + description: select authorization file + provenance: + sources: + - help-text +- spellings: + - -br + description: create root window with black background + provenance: + sources: + - help-text +- spellings: + - +bs + description: enable any backing store support + provenance: + sources: + - help-text +- spellings: + - -bs + description: disable any backing store support + provenance: + sources: + - help-text +- spellings: + - +byteswappedclients + description: Allow clients with endianess different to that of the server + provenance: + sources: + - help-text +- spellings: + - -byteswappedclients + description: Prohibit clients with endianess different to that of the server + provenance: + sources: + - help-text +- spellings: + - -c + description: turns off key-click + provenance: + sources: + - help-text +- spellings: + - -cc + description: default color visual class + provenance: + sources: + - help-text +- spellings: + - -nocursor + description: disable the cursor + provenance: + sources: + - help-text +- spellings: + - -core + description: generate core dump on fatal error + provenance: + sources: + - help-text +- spellings: + - -displayfd + description: file descriptor to write display number to when ready to connect + provenance: + sources: + - help-text +- spellings: + - -dpi + description: screen resolution in dots per inch + provenance: + sources: + - help-text +- spellings: + - -dpms + description: disables VESA DPMS monitor control + provenance: + sources: + - help-text +- spellings: + - -deferglyphs + description: defer loading of [no|all|16-bit] glyphs + provenance: + sources: + - help-text +- spellings: + - -f + value_name: '#' + value_kind: Required + description: bell base (0-100) + provenance: + sources: + - help-text +- spellings: + - -fakescreenfps + description: fake screen default fps (1-600) + provenance: + sources: + - help-text +- spellings: + - -fp + description: default font path + provenance: + sources: + - help-text +- spellings: + - -help + description: prints message with these options + provenance: + sources: + - help-text +- spellings: + - +iglx + description: Allow creating indirect GLX contexts + provenance: + sources: + - help-text +- spellings: + - -iglx + description: Prohibit creating indirect GLX contexts (default) + provenance: + sources: + - help-text +- spellings: + - -I + description: ignore all remaining arguments + provenance: + sources: + - help-text +- spellings: + - -ld + description: limit data space to N Kb + provenance: + sources: + - help-text +- spellings: + - -lf + description: limit number of open files to N + provenance: + sources: + - help-text +- spellings: + - -ls + description: limit stack space to N Kb + provenance: + sources: + - help-text +- spellings: + - -nolock + description: disable the locking mechanism + provenance: + sources: + - help-text +- spellings: + - -maxclients + description: set maximum number of clients (power of two) + provenance: + sources: + - help-text +- spellings: + - -nolisten + description: don't listen on protocol + provenance: + sources: + - help-text +- spellings: + - -listen + description: listen on protocol + provenance: + sources: + - help-text +- spellings: + - -noreset + description: don't reset after last client exists + provenance: + sources: + - help-text +- spellings: + - -background + value_name: '[none]' + value_kind: Optional + description: create root window with no background + provenance: + sources: + - help-text +- spellings: + - -reset + description: reset after last client exists + provenance: + sources: + - help-text +- spellings: + - -p + value_name: '#' + value_kind: Required + description: screen-saver pattern duration (minutes) + provenance: + sources: + - help-text +- spellings: + - -pn + description: accept failure to listen on all ports + provenance: + sources: + - help-text +- spellings: + - -nopn + description: reject failure to listen on all ports + provenance: + sources: + - help-text +- spellings: + - -r + description: turns off auto-repeat + provenance: + sources: + - help-text +- spellings: + - -render + description: set render color alloc policy + provenance: + sources: + - help-text +- spellings: + - -retro + description: start with classic stipple and cursor + provenance: + sources: + - help-text +- spellings: + - -s + value_name: '#' + value_kind: Required + description: screen-saver timeout (minutes) + provenance: + sources: + - help-text +- spellings: + - -seat + description: seat to run on + provenance: + sources: + - help-text +- spellings: + - -t + value_name: '#' + value_kind: Required + description: default pointer threshold (pixels/t) + provenance: + sources: + - help-text +- spellings: + - -terminate + value_name: '[delay]' + value_kind: Optional + description: terminate at server reset (optional delay in sec) + provenance: + sources: + - help-text +- spellings: + - -tst + description: disable testing extensions + provenance: + sources: + - help-text +- spellings: + - -v + description: screen-saver without video blanking + provenance: + sources: + - help-text +- spellings: + - -wr + description: create root window with white background + provenance: + sources: + - help-text +- spellings: + - -maxbigreqsize + description: set maximal bigrequest size + provenance: + sources: + - help-text +- spellings: + - +xinerama + description: Enable XINERAMA extension + provenance: + sources: + - help-text +- spellings: + - -xinerama + description: Disable XINERAMA extension + provenance: + sources: + - help-text +- spellings: + - -dumbSched + description: Disable smart scheduling and threaded input, enable old behavior + provenance: + sources: + - help-text +- spellings: + - -schedInterval + description: Set scheduler interval in msec + provenance: + sources: + - help-text +- spellings: + - -sigstop + description: Enable SIGSTOP based startup + provenance: + sources: + - help-text +- spellings: + - +extension + value_name: name + value_kind: Required + description: Enable extension + provenance: + sources: + - help-text +- spellings: + - -extension + description: 'Disable extension Only the following extensions can be run-time enabled/disabled: Generic Event Extension MIT-SHM XTEST SECURITY XINERAMA XFIXES RENDER RANDR COMPOSITE DAMAGE MIT-SCREEN-SAVER DOUBLE-BUFFER RECORD DPMS X-Resource XVideo XVideo-MotionCompensation SELinux GLX' + provenance: + sources: + - help-text +- spellings: + - -query + description: contact named host for XDMCP + provenance: + sources: + - help-text +- spellings: + - -broadcast + description: broadcast for XDMCP + provenance: + sources: + - help-text +- spellings: + - -multicast + description: IPv6 multicast for XDMCP + provenance: + sources: + - help-text +- spellings: + - -indirect + description: contact named host for indirect XDMCP + provenance: + sources: + - help-text +- spellings: + - -port + description: UDP port number to send messages to + provenance: + sources: + - help-text +- spellings: + - -from + description: specify the local address to connect from + provenance: + sources: + - help-text +- spellings: + - -once + description: Terminate server after one session + provenance: + sources: + - help-text +- spellings: + - -class + description: specify display class to send in manage + provenance: + sources: + - help-text +- spellings: + - -cookie + description: specify the magic cookie for XDMCP + provenance: + sources: + - help-text +- spellings: + - -displayID + description: manufacturer display ID for request + provenance: + sources: + - help-text +- spellings: + - +accessx + value_name: '[ timeout [ timeout_mask [ feedback [ options_mask] ] ] ]' + value_kind: Optional + description: enable/disable accessx key sequences + provenance: + sources: + - help-text +- spellings: + - -accessx + value_name: '[ timeout [ timeout_mask [ feedback [ options_mask] ] ] ]' + value_kind: Optional + description: enable/disable accessx key sequences + provenance: + sources: + - help-text +- spellings: + - -ardelay + description: set XKB autorepeat delay + provenance: + sources: + - help-text +- spellings: + - -arinterval + description: set XKB autorepeat interval + provenance: + sources: + - help-text +- spellings: + - -screen + description: set screen's width, height, depth + provenance: + sources: + - help-text +- spellings: + - -pixdepths + provenance: + sources: + - help-text +- spellings: + - +render + description: turn on/off RENDER extension support(default on) + provenance: + sources: + - help-text +- spellings: + - -render + description: turn on/off RENDER extension support(default on) + provenance: + sources: + - help-text +- spellings: + - -linebias + description: adjust thin line pixelization + provenance: + sources: + - help-text +- spellings: + - -blackpixel + description: pixel value for black + provenance: + sources: + - help-text +- spellings: + - -whitepixel + description: pixel value for white + provenance: + sources: + - help-text +- spellings: + - -fbdir + description: put framebuffers in mmap'ed files in directory + provenance: + sources: + - help-text +- spellings: + - -shmem + description: put framebuffers in shared memory + provenance: + sources: + - help-text +provenance: + sources: + - help-text + confidence: 0.5 +children_filled: true diff --git a/corpus/Xvfb/audit-seed/meta.toml b/corpus/Xvfb/audit-seed/meta.toml index 9e306e37..a7622e88 100644 --- a/corpus/Xvfb/audit-seed/meta.toml +++ b/corpus/Xvfb/audit-seed/meta.toml @@ -1,8 +1,8 @@ # Xvfb rejects `--help`, prints a leading diagnostic line, then a headingless # table of over eighty option rows with no recognized heading at all. Three -# defects share one specimen: the leading diagnostic (S-162), `+word` and +# defects shared one specimen: the leading diagnostic (S-162), `+word` and # `+/-name`/`[+-]name` rows (S-163), and the headingless table landing in the -# root description (S-165). +# root description (S-165). All three are fixed. [bless] provenance = "agent" @@ -20,13 +20,17 @@ stderr = "help.stderr.txt" [contract] expected_framework = "generic" -# The F4/S-162 half is fixed: the leading diagnostic no longer survives -# into the root description. This assertion is what makes that checkable; -# the fixture stays [xfail] below because the headingless option table -# (S-165) and the `+word`/`+/-name` rows (S-163) still land as prose in -# that same description. +min_status = "ok" +# The leading diagnostic never survives into the root description (S-162). must_not_describe_root = ["Unrecognized option", "disable access control restrictions"] +# `+word` (S-163 rule 1) and the alternation-sigil rows (S-163 rules 2/3) +# all reach the tree, both halves each. +must_contain_flags = ["+bs", "-bs", "+accessx", "-accessx", "+render", "-render"] +# Neither the `use:` usage-label line nor `[+-]accessx`'s own wrapped +# description survives as a fabricated group heading (S-164/S-165). +must_not_contain_flag_group_prefixes = ["use: X", "Enable/disable accessx"] -[xfail] -broken = true -reason = "a headingless option table lands in the root description as prose (S-165); +word and +/-name rows are unrecovered (S-163)" +[contract.must_describe] +# The headingless table no longer duplicates into the root description +# (S-165): each flag keeps its own row's own description. +"-a" = "default pointer acceleration" diff --git a/corpus/fc-validate/audit-seed2/expected.snap b/corpus/fc-validate/audit-seed2/expected.snap index a53a8f97..ff0c4634 100644 --- a/corpus/fc-validate/audit-seed2/expected.snap +++ b/corpus/fc-validate/audit-seed2/expected.snap @@ -8,7 +8,6 @@ flags: - --index value_name: INDEX value_kind: Required - group: Validate font files and print result description: display the INDEX face of each font file only provenance: sources: @@ -18,7 +17,6 @@ flags: - --lang value_name: LANG value_kind: Required - group: Validate font files and print result description: set LANG instead of current locale provenance: sources: @@ -26,7 +24,6 @@ flags: - spellings: - -v - --verbose - group: Validate font files and print result description: show more detailed information provenance: sources: @@ -34,7 +31,6 @@ flags: - spellings: - -V - --version - group: Validate font files and print result description: display font config version and exit provenance: sources: @@ -42,7 +38,6 @@ flags: - spellings: - -h - --help - group: Validate font files and print result description: display this help and exit provenance: sources: diff --git a/corpus/fc-validate/audit-seed2/meta.toml b/corpus/fc-validate/audit-seed2/meta.toml index 4b9ff4d3..02cc5e86 100644 --- a/corpus/fc-validate/audit-seed2/meta.toml +++ b/corpus/fc-validate/audit-seed2/meta.toml @@ -25,3 +25,7 @@ expected_framework = "generic" min_status = "ok" min_subcommands = 0 must_contain_flags = ["--index", "--lang", "--verbose", "--version", "--help"] +# The root description ("Validate font files and print result") no longer +# doubles as every flag's own group (docs/shapes.md S-164). +[contract.must_flag_group] +"--index" = "" diff --git a/corpus/fzf/0.44.1/expected.snap b/corpus/fzf/0.44.1/expected.snap index 253d8819..474fda41 100644 --- a/corpus/fzf/0.44.1/expected.snap +++ b/corpus/fzf/0.44.1/expected.snap @@ -25,16 +25,25 @@ flags: provenance: sources: - help-text +- spellings: + - +i + group: Search + description: Case-sensitive match + provenance: + sources: + - help-text - spellings: - --scheme value_name: SCHEME value_kind: Required + group: Search description: Scoring scheme [default|path|history] provenance: sources: - help-text - spellings: - --literal + group: Search description: Do not normalize latin script letters before matching provenance: sources: @@ -44,6 +53,7 @@ flags: - --nth value_name: N[,..] value_kind: Required + group: Search description: Comma-separated list of field index expressions for limiting search scope. Each can be a non-zero integer or a range expression ([BEGIN]..[END]). provenance: sources: @@ -52,6 +62,7 @@ flags: - --with-nth value_name: N[,..] value_kind: Required + group: Search description: Transform the presentation of each line using field index expressions provenance: sources: @@ -61,24 +72,36 @@ flags: - --delimiter value_name: STR value_kind: Required + group: Search description: 'Field delimiter regex (default: AWK-style)' provenance: sources: - help-text +- spellings: + - +s + - --no-sort + group: Search + description: Do not sort the result + provenance: + sources: + - help-text - spellings: - --track + group: Search description: Track the current selection when the result is updated provenance: sources: - help-text - spellings: - --tac + group: Search description: Reverse the order of the input provenance: sources: - help-text - spellings: - --disabled + group: Search description: Do not perform search provenance: sources: @@ -87,6 +110,7 @@ flags: - --tiebreak value_name: CRI[,..] value_kind: Required + group: Search description: 'Comma-separated list of sort criteria to apply when the scores are tied [length|chunk|begin|end|index] (default: length)' provenance: sources: diff --git a/corpus/fzf/0.44.1/meta.toml b/corpus/fzf/0.44.1/meta.toml index 108a366a..3b68f9cc 100644 --- a/corpus/fzf/0.44.1/meta.toml +++ b/corpus/fzf/0.44.1/meta.toml @@ -29,4 +29,14 @@ must_contain_env_vars = ["FZF_DEFAULT_COMMAND", "FZF_DEFAULT_OPTS", "FZF_API_KEY # the environment-heading recognizer's placement in `parse_body` can never # silently start swallowing (or splitting) the grouped flags blocks that # sit directly above it under the same layout convention. -must_contain_flags = ["--extended", "--multi", "--height", "--history", "--preview", "--query"] +# `+i` and `+s` (S-163) sit inside the `Search` group and used to end the +# block outright, splitting the group's own later rows away from their +# label (docs/shapes.md S-163) — asserted here alongside their neighbors. +must_contain_flags = [ + "--extended", "--multi", "--height", "--history", "--preview", "--query", + "+i", "+s", "--scheme", +] +[contract.must_flag_group] +"+i" = "Search" +"+s" = "Search" +"--scheme" = "Search" diff --git a/corpus/lto-dump/13.3.0/expected.snap b/corpus/lto-dump/13.3.0/expected.snap index 881a4361..49884578 100644 --- a/corpus/lto-dump/13.3.0/expected.snap +++ b/corpus/lto-dump/13.3.0/expected.snap @@ -3,7 +3,6 @@ description: 'The following options are specific to just the language Ada:' flags: - spellings: - -fdump-scos - group: 'The following options are specific to just the language Ada:' description: '[available in Ada]' provenance: sources: diff --git a/corpus/lto-dump/13.3.0/meta.toml b/corpus/lto-dump/13.3.0/meta.toml index dba2a4e5..82683707 100644 --- a/corpus/lto-dump/13.3.0/meta.toml +++ b/corpus/lto-dump/13.3.0/meta.toml @@ -25,3 +25,8 @@ stdout = "help.txt" [contract] must_keep_separate = [["-C", "-CC"]] +# The root description ("The following options are specific to just the +# language Ada:") no longer doubles as `-fdump-scos`'s own group +# (docs/shapes.md S-164). +[contract.must_flag_group] +"-fdump-scos" = "" diff --git a/docs/shapes.md b/docs/shapes.md index 0b0e69f6..9f55a243 100644 --- a/docs/shapes.md +++ b/docs/shapes.md @@ -2843,3 +2843,109 @@ entry's `tools` field and nothing else. It does not get a new entry. diagnostic over both streams; the detector reads the tree's chosen stream only, and that count is an upper bound, not this family's own fleet count. 2026-09-12. + +### S-163: `+word`, `+/-name` and `[+-]name` option rows + +- id: S-163 +- looks like: | + +bs enable any backing store support + -bs disable any backing store support + +/-render turn on/off RENDER extension support(default on) + [+-]accessx [ timeout [ ttb [ tpo [ ctrls ]]]] enable/disable accessx +- tools: Xvfb, fzf, lsof +- handling: Fixed. Three rules. (1) A `+word` row — a letter-led run after + the sigil (`+bs`, `+byteswappedclients`) — is admitted beside a + flag-shaped neighbor (`has_flag_shaped_plus_neighbor`), the same evidence + bare `+`/`+` already required (S-095), extended to a whole + word. That neighbor check no longer requires the neighbor row itself to + carry leading whitespace: a headingless table (S-165) sits flush at + column 0, and `+bs`'s own neighbor `-bs` does too. The gate this rides on + (`scan_flags_block`'s own "indented, or already inside an open block" + test) is widened the same way, so a `+word` row is admitted at column 0 + once a real entry has already opened the block. (2) `+/-word`, `-/+word`, + `[+-]word` and `[-+]word` each expand to two entities, `+word` and + `-word`, sharing the row's own description and any trailing value spec + verbatim. (3) Neither rule fires where the sigil is not the row's own + leading token: `xxd`'s `-s [+][-]seek` opens with `-s`, and stays + refused, matching S-097's own counter-case. +- fleet: `plus-word-option` (`xtask/src/detector/plus_word_option.rs`, rule 1) + and `plus-minus-alternation-option` + (`xtask/src/detector/plus_minus_alternation_option.rs`, rules 2/3) are + both family `None`, so calibration against the seed-7 labelled set reads + NOT EVALUABLE for each. Self-checks hold (7/7 and 7/7). Raw-shape count: + 3 tools / 7 findings for `+word` (Xvfb, fzf, lsof), 1 tool / 2 findings + for `+/-name` (Xvfb); both are upper bounds, not the tree-level count. + Full-`PATH` sweep-diff of 2269 tools against `origin/main` 0b30c15, + 2026-09-12: 0 flags lost anywhere, 11 flags gained across 2 tools — + Xvfb 69 to 78 (+bs, +byteswappedclients, +iglx, +xinerama, +extension, + +render, +accessx, -accessx), fzf 63 to 65 (+i, `+s, --no-sort`). Every + gain checked against the tool's own `--help` text. + +### S-164: the root description reused as a flag group's own label + +- id: S-164 +- looks like: | + usage: fc-scan [-bcVh] ... font-file... + Scan font files and directories, and print resulting pattern(s) + + -b, --brief display font pattern briefly +- tools: fc-scan, fc-validate, grub-macbless, lto-dump, Xvfb +- handling: Fixed. A sentence directly above a flags block, with no + recognized heading word, is read two ways at once: once as the node's + own root `description` (the leading-prose rule), and a second time as + that block's own group label (`set_pending_bare_label`'s flush-heading + shortcut, S-146, and the "recognized heading" flags-block path's + `meaningful_flag_group` fallback). A label equal, verbatim (trimmed), to + the root description is now refused at both sites: a sentence already + spent as the description is not available a second time as a group. + `gcc-ranlib-13`'s own `The options are` label is unaffected, since its + text differs from the description. +- fleet: `description-reused-as-group-label` + (`xtask/src/detector/description_reused_as_group_label.rs`) is family + `None`, so calibration reads NOT EVALUABLE. Self-checks hold (4/4). + `fc-scan` and `grub-macbless` are the maintainer-named specimens; + `fc-validate` and `lto-dump` are pre-existing, previously-`ok` corpus + fixtures the fix also silently repaired (both re-blessed, group lines + removed, nothing else changed). Full-`PATH` sweep-diff of 2269 tools + against `origin/main` 0b30c15, 2026-09-12: 0 flag/subcommand losses. The + detector itself still reads 2 tools / 2 findings fleet-wide after this + fix (`"where possible options include:"`, `"where options include:"`), + a different pair of tools this round's brief did not name; left as a + future finding, not chased here. + +### S-165: a headingless table lands in the root description + +- id: S-165 +- looks like: | + Unrecognized option: --help + use: X [:] [option] + -a # default pointer acceleration (factor) + -ac disable access control restrictions +- tools: Xvfb +- handling: Fixed. `extract_description`'s own bound + (`leading_prose_bound`) is a blank-line search with no notion of a + flags block at all; with no recognized `usage:` line and no blank line + anywhere in the document, it returns the whole document, so a + headingless table's rows land in the description as well as being + independently recovered by `scan_entries`. Narrowly bounded: only when + no blank line exists at all and no usage line was recognized does the + description scan now also stop at the first line + `starts_attested_headingless_flag_block` (S-052's own recognizer) + accepts as a real option row, so an ordinary document's already-correct, + cheap bound pays nothing extra. Xvfb's own `use: X [:] + [option]` line, an unusual `use:` label rather than `usage:`, is the + root cause `flags_block_start` never reaches on its own. This is the + same specimen S-164 fixes the group-duplication half of; landing this + fix first is what let S-163's `+word` and alternation rows reach column + 0 at all. +- fleet: `headingless-table-in-root-description` + (`xtask/src/detector/headingless_table_in_root_description.rs`) is + family `None`, so calibration reads NOT EVALUABLE. Self-checks hold + (4/4). One tool, below the five-tool bar, maintainer-named (carried + item 23). Full-`PATH` sweep-diff of 2269 tools against `origin/main` + 0b30c15, 2026-09-12: 0 flag/subcommand losses anywhere. The detector's + own word-boundary heuristic still reads 30 tools fleet-wide after this + fix — a broader symptom (a description repeating several of its own + tree's flag spellings) than the narrow structural cause this fix + closes (no blank line anywhere, no recognized usage line); left as a + future finding, not chased here. diff --git a/mandible-extract/src/help_text/sections/emit.rs b/mandible-extract/src/help_text/sections/emit.rs index 3a03eb16..9f336dfe 100644 --- a/mandible-extract/src/help_text/sections/emit.rs +++ b/mandible-extract/src/help_text/sections/emit.rs @@ -11,20 +11,29 @@ use super::*; /// [`parse_plus_sigil_spec`]'s grammar, and would otherwise keep the /// literal `+`/`+` text as a bare "spelling". See docs/shapes.md /// S-095. +/// Three-way now, not two: a packed block's own alternation-sigil rows +/// (S-163) need the same detour around [`emit_packed_flags`] a plus-sigil +/// row does, for the identical reason — that reader assumes a +/// `-wholename` shape [`parse_plus_sigil_spec`]/ +/// [`parse_plus_minus_alternation_spec`] never produce. fn partition_plus_sigil_entries( entries: Vec, is_plus_sigil: &[bool], -) -> (Vec, Vec) { + is_alternation: &[bool], +) -> (Vec, Vec, Vec) { let mut ordinary = Vec::new(); let mut plus_sigil = Vec::new(); + let mut alternation = Vec::new(); for (idx, entry) in entries.into_iter().enumerate() { - if is_plus_sigil.get(idx).copied().unwrap_or(false) { + if is_alternation.get(idx).copied().unwrap_or(false) { + alternation.push(entry); + } else if is_plus_sigil.get(idx).copied().unwrap_or(false) { plus_sigil.push(entry); } else { ordinary.push(entry); } } - (ordinary, plus_sigil) + (ordinary, plus_sigil, alternation) } /// Emit everything one [`scan_flags_block`] call recovered — the packed @@ -38,11 +47,13 @@ pub(super) fn emit_flags_block( entries: Vec, packed: bool, is_plus_sigil: &[bool], + is_alternation: &[bool], argfile_entry: Option, out: &mut ParsedHelp, ) -> (usize, usize) { let (mut seen, mut clean) = if packed { - let (ordinary, plus_sigil) = partition_plus_sigil_entries(entries, is_plus_sigil); + let (ordinary, plus_sigil, alternation) = + partition_plus_sigil_entries(entries, is_plus_sigil, is_alternation); let ordinary_seen = ordinary.len(); emit_packed_flags( group.clone(), @@ -50,11 +61,27 @@ pub(super) fn emit_flags_block( out, ); let plus_seen = plus_sigil.len(); - let (_, plus_clean) = - emit_flags_with(group.clone(), plus_sigil, &vec![true; plus_seen], out); - (ordinary_seen + plus_seen, ordinary_seen + plus_clean) + let (_, plus_clean) = emit_flags_with( + group.clone(), + plus_sigil, + &vec![true; plus_seen], + &vec![false; plus_seen], + out, + ); + let alt_seen = alternation.len(); + let (_, alt_clean) = emit_flags_with( + group.clone(), + alternation, + &vec![false; alt_seen], + &vec![true; alt_seen], + out, + ); + ( + ordinary_seen + plus_seen + alt_seen, + ordinary_seen + plus_clean + alt_clean, + ) } else { - emit_flags_with(group.clone(), entries, is_plus_sigil, out) + emit_flags_with(group.clone(), entries, is_plus_sigil, is_alternation, out) }; if let Some(entry) = argfile_entry { seen += 1; @@ -96,6 +123,7 @@ pub(super) fn emit_flags_with( group: Option, entries: Vec, is_plus_sigil: &[bool], + is_alternation: &[bool], out: &mut ParsedHelp, ) -> (usize, usize) { let mut seen = 0usize; @@ -105,6 +133,24 @@ pub(super) fn emit_flags_with( break; } seen += 1; + // The alternation-sigil row (S-163) expands to two entities, + // `+word` and `-word`, both sharing this row's own description + // and choices — built directly from the two `FlagSpec`s rather + // than through the loop's single-`spec` path below, since one + // source row producing two entities is the one shape this loop + // otherwise never has. + if is_alternation.get(idx).copied().unwrap_or(false) { + if let Some((plus_spec, minus_spec)) = parse_plus_minus_alternation_spec(&spec_text) { + clean += 1; + for spec in [plus_spec, minus_spec] { + if out.flags.len() >= MAX_RECOVERED_ENTRIES { + break; + } + push_flag_entity(spec, &desc_text, &choice_names, group.clone(), out); + } + } + continue; + } let mut spec = if is_plus_sigil.get(idx).copied().unwrap_or(false) { parse_plus_sigil_spec(&spec_text) } else { @@ -132,40 +178,54 @@ pub(super) fn emit_flags_with( } } } - let mut flag = Entity::new(EntityKind::Flag, Provenance::single(Source::HelpText)); - flag.spellings = spec.spellings; - flag.value_name = spec.value_name; - flag.value_kind = spec.value_kind; - flag.group = group.clone(); - flag.description = non_empty_text(&description); - // Sub-rows nested directly under this flag's own row (llvm-ar's - // bare `=value` shape and ffmpeg/ffplay's described AVOption shape, - // see `choices_sub_row_value`/`choice_description_sub_row`) share - // this same `choices` field with clap's `[possible values: …]`. - flag.choices = choice_names - .into_iter() - .map(|(name, desc)| Choice { - name, - description: desc.map(|d| Text::sanitize(&d)), - }) - .collect(); - // A docopt bracket row's own trailing `|`-list (`trailing_choice_list`, - // S-120) already carries every value as `choices`; when no bracketed - // placeholder introduced it (`--configreport log|vg|lv|pv|pvseg|seg`, - // unlike `--units [Number]r|R|...`), the same list is *also* what - // grammar read as `value_name`, so the rendered screen prints it - // twice. Dropped rather than replaced with a generic placeholder, - // since `choices` already carries the full enumeration and a - // placeholder here would tell the reader nothing new. See - // docs/shapes.md S-130. - if value_name_duplicates_its_own_choices(flag.value_name.as_deref(), &flag.choices) { - flag.value_name = None; - } - out.flags.push(flag); + push_flag_entity(spec, &description, &choice_names, group.clone(), out); } (seen, clean) } +/// Build and push one [`Entity`] from an already-parsed [`FlagSpec`], +/// shared by the ordinary/plus-sigil path above and the alternation-sigil +/// row's two-entity expansion — the exact steps every flag entity needs +/// regardless of which spec produced it. See docs/shapes.md S-163. +fn push_flag_entity( + spec: FlagSpec, + description: &str, + choice_names: &[(String, Option)], + group: Option, + out: &mut ParsedHelp, +) { + let mut flag = Entity::new(EntityKind::Flag, Provenance::single(Source::HelpText)); + flag.spellings = spec.spellings; + flag.value_name = spec.value_name; + flag.value_kind = spec.value_kind; + flag.group = group; + flag.description = non_empty_text(description); + // Sub-rows nested directly under this flag's own row (llvm-ar's + // bare `=value` shape and ffmpeg/ffplay's described AVOption shape, + // see `choices_sub_row_value`/`choice_description_sub_row`) share + // this same `choices` field with clap's `[possible values: …]`. + flag.choices = choice_names + .iter() + .map(|(name, desc)| Choice { + name: name.clone(), + description: desc.clone().map(|d| Text::sanitize(&d)), + }) + .collect(); + // A docopt bracket row's own trailing `|`-list (`trailing_choice_list`, + // S-120) already carries every value as `choices`; when no bracketed + // placeholder introduced it (`--configreport log|vg|lv|pv|pvseg|seg`, + // unlike `--units [Number]r|R|...`), the same list is *also* what + // grammar read as `value_name`, so the rendered screen prints it + // twice. Dropped rather than replaced with a generic placeholder, + // since `choices` already carries the full enumeration and a + // placeholder here would tell the reader nothing new. See + // docs/shapes.md S-130. + if value_name_duplicates_its_own_choices(flag.value_name.as_deref(), &flag.choices) { + flag.value_name = None; + } + out.flags.push(flag); +} + /// Emit the argfile sigil flag [`super::flag_rows::argfile_row_value_name`] /// recovered (spec §4.5), built directly through /// [`mandible_core::Entity::argfile_sigil`] rather than diff --git a/mandible-extract/src/help_text/sections/flag_rows.rs b/mandible-extract/src/help_text/sections/flag_rows.rs index 784b4d14..f6949c97 100644 --- a/mandible-extract/src/help_text/sections/flag_rows.rs +++ b/mandible-extract/src/help_text/sections/flag_rows.rs @@ -408,6 +408,10 @@ pub(super) enum FlagsBlockRow<'a> { /// A `+`/`+` row admitted only by /// [`has_flag_shaped_plus_neighbor`] — see docs/shapes.md S-095. PlusSigil(&'a str), + /// A `+/-word`/`-/+word`/`[+-]word`/`[-+]word` alternation-sigil row + /// (`plus_minus_alternation_word`), expanded to two entities later. + /// See docs/shapes.md S-163. + AlternationSigil(&'a str), /// A continuation of the previous entry's description (`trim_end`ed /// text only — the row's own indentation has already done its job). Continuation(&'a str), @@ -443,9 +447,7 @@ pub(super) fn is_claimed_plus_token(token: &str) -> bool { } let mut chars = rest.chars(); match chars.next() { - Some(c) if c.is_ascii_alphabetic() => { - chars.all(|c| c.is_ascii_alphanumeric() || c == '-') - } + Some(c) if c.is_ascii_alphabetic() => chars.all(|c| c.is_ascii_alphanumeric() || c == '-'), _ => false, } } @@ -455,8 +457,13 @@ pub(super) fn is_claimed_plus_token(token: &str) -> bool { /// this same family's own claimed `+`-token shape. See docs/shapes.md /// S-095. fn plus_neighbor_row_is_flag_shaped(line: &str) -> bool { + // No indentation requirement: a headingless table (S-165) sits flush + // at column 0 throughout, so a real neighbor row there carries no + // leading whitespace either. The row's own shape — a `-`-led token, + // `--`, or this family's own claimed `+`-token — is evidence enough + // on its own regardless of indent. let trimmed = line.trim_start(); - if trimmed.is_empty() || trimmed == line { + if trimmed.is_empty() { return false; } let Some(token) = trimmed.split_whitespace().next() else { @@ -550,6 +557,71 @@ pub(super) fn parse_plus_sigil_spec(spec_text: &str) -> FlagSpec { spec } +// --- S-163: the `+/-word`/`[+-]word` alternation-sigil row ------------- +// +// `+/-render` and `[+-]accessx` name two opposite flags, `+word` and +// `-word`, on one row with one shared description — a different shape +// from the neighbor-gated bare `+word` above, and unambiguous on its own +// four-character sigil, so it needs no neighbor evidence. Anchored to the +// row's own leading token, so `xxd`'s `-s [+][-]seek` (whose leading +// token is `-s`) is never in scope. See docs/shapes.md S-163. + +/// The base word right after a `+/-`/`-/+`/`[+-]`/`[-+]` alternation +/// sigil at the very start of `token`, when the sigil is immediately +/// followed by a letter-led run of letters/digits/`-`. `None` for every +/// other token, including a bare `+`/`-` or a sigil with nothing +/// word-shaped after it. +pub(super) fn plus_minus_alternation_word(token: &str) -> Option<&str> { + let rest = token + .strip_prefix("+/-") + .or_else(|| token.strip_prefix("-/+")) + .or_else(|| token.strip_prefix("[+-]")) + .or_else(|| token.strip_prefix("[-+]"))?; + let word_end = rest + .char_indices() + .find(|(_, c)| !(c.is_ascii_alphanumeric() || *c == '-')) + .map_or(rest.len(), |(i, _)| i); + if word_end == 0 || !rest.starts_with(|c: char| c.is_ascii_alphabetic()) { + return None; + } + Some(&rest[..word_end]) +} + +/// Parse an alternation-sigil row's spec text into its two entities' +/// specs, `+word` and `-word`, sharing whatever trailing value spec +/// follows the sigil verbatim (kept as one optional value name rather +/// than run through the ordinary grammar, which assumes a leading dash). +/// `None` when `spec_text` does not open with the claimed shape — +/// defensive only, since every caller already gated on +/// [`plus_minus_alternation_word`] before routing a row here. See +/// docs/shapes.md S-163. +pub(super) fn parse_plus_minus_alternation_spec(spec_text: &str) -> Option<(FlagSpec, FlagSpec)> { + let trimmed = spec_text.trim_start(); + let leading = first_word(trimmed); + let word = plus_minus_alternation_word(leading)?.to_string(); + let rest = trimmed[leading.len()..].trim_start(); + let (value_name, value_kind) = if rest.is_empty() { + (None, ValueKind::None) + } else { + (Some(rest.to_string()), ValueKind::Optional) + }; + let plus = FlagSpec { + spellings: vec![Spelling::bare(format!("+{word}"))], + value_name: value_name.clone(), + value_kind, + fully_consumed: true, + ..FlagSpec::default() + }; + let minus = FlagSpec { + spellings: vec![Spelling::single_dash(word)], + value_name, + value_kind, + fully_consumed: true, + ..FlagSpec::default() + }; + Some((plus, minus)) +} + /// One recovered flag-table row: its spec text, its description, and any /// enumerated values nested directly under it (`llvm-ar`'s bare `=value` /// sub-rows, see [`choices_sub_row_value`]; ffmpeg/ffplay's described @@ -729,17 +801,31 @@ pub(super) fn trailing_choice_list(content: &str) -> Vec { Vec::new() } -pub(super) fn scan_flags_block<'a>( - lines: &[&'a str], - start: usize, - heading_is_bnf: bool, -) -> ( +/// `scan_flags_block`'s own return: the next line index, the recovered +/// entries, whether the block is `packed` (S-047), the argfile row if +/// present (S-021), and two `entries`-length parallel flag vectors +/// marking which rows are plus-sigil (S-095) or alternation-sigil +/// (S-163) rows. Named, not a plain tuple, so every call site stays +/// readable and clippy's `type_complexity` lint never trips. +pub(super) type FlagsBlockScan = ( usize, Vec, bool, Option, Vec, -) { + Vec, +); + +/// Phase 1 of [`scan_flags_block`]: walk `lines` from `start`, classifying +/// each into a [`FlagsBlockRow`] (or capturing it as the block's own +/// argfile row, S-021) until the block ends. Split out on its own so +/// `scan_flags_block` itself stays under the size ceilings — this is the +/// row-classification half; the entry-building half stays in the parent. +fn collect_flags_block_rows<'a>( + lines: &[&'a str], + start: usize, + heading_is_bnf: bool, +) -> (usize, Vec>, Option) { const ENTRY_INDENT_TOLERANCE: usize = 10; let mut i = start; let mut rows: Vec> = Vec::new(); @@ -785,21 +871,38 @@ pub(super) fn scan_flags_block<'a>( && min_entry_indent.is_none_or(|min| indent <= min + ENTRY_INDENT_TOLERANCE); // The neighbor-gated `+`/`+` row (S-095): indented - // (a heading has none), the claimed shape, and beside a + // (a heading has none) **or** inside a block this scan has + // already opened (`min_entry_indent.is_some()` — Xvfb's own + // headingless table sits flush at column 0 throughout, so no row + // in it is ever indented; S-165), the claimed shape, and beside a // flag-shaped neighbor — see `has_flag_shaped_plus_neighbor`. // Checked only once the ordinary shapes above have refused the // row, and independently of the indent-tolerance gate above (a // block's own plus row may open no more indented than its first // real flag). + let inside_open_block = indent > 0 || min_entry_indent.is_some(); let is_plus_sigil_start = !is_entry_start - && indent > 0 + && inside_open_block && is_claimed_plus_token(first_word(trimmed).trim_end_matches(',')) && min_entry_indent.is_none_or(|min| indent <= min + ENTRY_INDENT_TOLERANCE) && has_flag_shaped_plus_neighbor(lines, i); - if is_entry_start || is_plus_sigil_start { + // The alternation-sigil row (S-163): same "indented, or already + // inside an open block" evidence as the plus-sigil row above, no + // neighbor gate needed — the four-character sigil is unambiguous + // on its own (see `plus_minus_alternation_word`'s own doc + // comment). + let is_alternation_start = !is_entry_start + && !is_plus_sigil_start + && inside_open_block + && plus_minus_alternation_word(first_word(trimmed).trim_end_matches(',')).is_some() + && min_entry_indent.is_none_or(|min| indent <= min + ENTRY_INDENT_TOLERANCE); + + if is_entry_start || is_plus_sigil_start || is_alternation_start { rows.push(if is_plus_sigil_start { FlagsBlockRow::PlusSigil(line) + } else if is_alternation_start { + FlagsBlockRow::AlternationSigil(line) } else { FlagsBlockRow::Entry(line) }); @@ -831,6 +934,16 @@ pub(super) fn scan_flags_block<'a>( break; } + (i, rows, argfile_entry) +} + +pub(super) fn scan_flags_block( + lines: &[&str], + start: usize, + heading_is_bnf: bool, +) -> FlagsBlockScan { + let (i, rows, argfile_entry) = collect_flags_block_rows(lines, start, heading_is_bnf); + // Whether this block packs several flag+description pairs per line // (spec §7 Tier B, `lsof`'s options table) is a property of the block, // decided once from every entry row together — never per line, which @@ -840,7 +953,9 @@ pub(super) fn scan_flags_block<'a>( .iter() .filter_map(|r| match r { FlagsBlockRow::Entry(l) => Some(*l), - FlagsBlockRow::PlusSigil(_) | FlagsBlockRow::Continuation(_) => None, + FlagsBlockRow::PlusSigil(_) + | FlagsBlockRow::AlternationSigil(_) + | FlagsBlockRow::Continuation(_) => None, }) .collect(); let multi_column = block_is_multi_column(&entry_lines); @@ -864,11 +979,17 @@ pub(super) fn scan_flags_block<'a>( // `parse_flag_spec`/`try_bare_sigil` grammar every other entry goes // through. See docs/shapes.md S-095. let mut is_plus_sigil: Vec = Vec::new(); + // Parallel to `entries` the same way `is_plus_sigil` is: `true` where + // that entry came from a `FlagsBlockRow::AlternationSigil` row, so + // `emit_flags_block` knows which single recovered entry to expand + // into the row's own `+word`/`-word` pair. See docs/shapes.md S-163. + let mut is_alternation: Vec = Vec::new(); for row in rows { let plus_sigil_row = matches!(row, FlagsBlockRow::PlusSigil(_)); + let alt_sigil_row = matches!(row, FlagsBlockRow::AlternationSigil(_)); let before = entries.len(); match row { - FlagsBlockRow::PlusSigil(line) => { + FlagsBlockRow::PlusSigil(line) | FlagsBlockRow::AlternationSigil(line) => { let (spec, desc) = split_single_column_entry(line); entries.push((spec, desc, Vec::new())); } @@ -995,12 +1116,24 @@ pub(super) fn scan_flags_block<'a>( } } is_plus_sigil.resize(entries.len(), plus_sigil_row); + is_alternation.resize(entries.len(), alt_sigil_row); debug_assert!( !plus_sigil_row || entries.len() == before + 1, "a PlusSigil row must produce exactly one entry" ); + debug_assert!( + !alt_sigil_row || entries.len() == before + 1, + "an AlternationSigil row must produce exactly one entry" + ); } - (i, entries, packed, argfile_entry, is_plus_sigil) + ( + i, + entries, + packed, + argfile_entry, + is_plus_sigil, + is_alternation, + ) } /// The fewest name/description pairs a deeper-indented run must show before diff --git a/mandible-extract/src/help_text/sections/heading.rs b/mandible-extract/src/help_text/sections/heading.rs index 69cd706d..7c7d820d 100644 --- a/mandible-extract/src/help_text/sections/heading.rs +++ b/mandible-extract/src/help_text/sections/heading.rs @@ -526,7 +526,7 @@ pub(super) fn starts_attested_headingless_flag_block(lines: &[&str], idx: usize) { return false; } - let (_, entries, _, _, _) = scan_flags_block(lines, idx, false); + let (_, entries, _, _, _, _) = scan_flags_block(lines, idx, false); entries.len() >= MIN_ATTESTED_SECTION_FLAGS } diff --git a/mandible-extract/src/help_text/sections/mod.rs b/mandible-extract/src/help_text/sections/mod.rs index 5611de37..478ee176 100644 --- a/mandible-extract/src/help_text/sections/mod.rs +++ b/mandible-extract/src/help_text/sections/mod.rs @@ -210,7 +210,7 @@ fn starts_attested_flag_section(lines: &[&str], heading_idx: usize) -> bool { // at least two independently parsed rows plus the heading vocabulary // above is the minimum evidence to reopen a same-indent section. See // S-071. - let (_, entries, _, _, _) = scan_flags_block(lines, flags_start, false); + let (_, entries, _, _, _, _) = scan_flags_block(lines, flags_start, false); entries.len() >= MIN_ATTESTED_SECTION_FLAGS } @@ -860,6 +860,7 @@ fn set_pending_bare_label(st: &mut BodyScan, label: Option, lines: &[&st if !st.in_ignorable_section && heading_can_name_a_group(&label) && !label.trim_end().ends_with(" :") + && !text_is_already_root_description(&label, st.result) && pending_label_names_a_real_table(lines, next) { st.pending_bare_label = Some(label); @@ -867,6 +868,21 @@ fn set_pending_bare_label(st: &mut BodyScan, label: Option, lines: &[&st } } +/// True when `text` is, verbatim (trimmed), the node's own root +/// `description` — already decided by [`extract_description`] before +/// this body scan runs. A sentence already spent as the root description +/// is not available a second time as a group label (S-164): Xvfb's own +/// `use: X [:] [option]` line is both, and a group repeating the +/// description word for word is never new information (AGENTS.md §3.9's +/// own reasoning, applied to a label instead of a dropped row). See +/// docs/shapes.md S-164. +fn text_is_already_root_description(text: &str, result: &ParsedHelp) -> bool { + result + .description + .as_deref() + .is_some_and(|d| d.trim() == text.trim()) +} + /// True when the row at `lines[idx]` is flag-shaped and documents a real /// description, on its own line (a column gap) or on a deeper-indented /// line beneath it — tells a genuine option table (`mksquashfs`'s own @@ -1078,7 +1094,7 @@ fn emit_heading_block( // `split_shared_heading_rows`'s doc comment for why the BNF // fact is keyed on the row rather than the heading beside it. let heading_is_bnf = bnf_row_lines.contains(&flags_start); - let (end, entries, packed, argfile_entry, is_plus_sigil) = + let (end, entries, packed, argfile_entry, is_plus_sigil, is_alternation) = scan_flags_block(lines, flags_start, heading_is_bnf); i = end; if is_ignorable_heading(heading) { @@ -1093,14 +1109,22 @@ fn emit_heading_block( // as the group's label, and only there — every other block // still takes `meaningful_flag_group`'s answer unchanged. See // S-012. + // + // A label equal, verbatim, to the root description is refused + // (S-164): `grub-macbless`'s own `Mac-style bless on HFS or + // HFS+` is both its description and this block's own would-be + // heading, and a group repeating the description word for word + // names nothing new. let group = stanza_label .clone() - .or_else(|| meaningful_flag_group(heading.clone())); + .or_else(|| meaningful_flag_group(heading.clone())) + .filter(|g| !text_is_already_root_description(g, st.result)); let (seen, clean) = emit_flags_block( group, entries, packed, &is_plus_sigil, + &is_alternation, argfile_entry, st.result, ); @@ -1286,6 +1310,7 @@ fn emit_flush_heading( } else if !st.in_ignorable_section && heading_can_name_a_group(heading) && find_description_gap(h.line).is_none() + && !text_is_already_root_description(heading, st.result) && pending_label_names_a_real_table(lines, heading_idx + 1) { // A flush heading whose own rows sit at its own column rather @@ -1294,7 +1319,10 @@ fn emit_flush_heading( // group from the headingless flags-block shortcut. The gap // check guards the heading line itself: `nm`'s own `@FILE Read // options from FILE` row is not flag-shaped, so it reaches here - // looking like a heading, but it is a real row. See S-146. + // looking like a heading, but it is a real row. See S-146. A + // line already spent as the root description is refused here + // instead (S-164): Xvfb's own `use: X [:] [option]` is + // both its description and this shape's own heading candidate. st.pending_bare_label = Some(heading.clone()); } // Rewind to just past the original line and continue scanning @@ -1432,7 +1460,7 @@ fn scan_entries( // revisited as a heading — dcb and vdpa's `OPTIONS` row. // See S-042, noted as `bnf_row_lines`. let heading_is_bnf = bnf_row_lines.contains(&i); - let (end, entries, packed, argfile_entry, is_plus_sigil) = + let (end, entries, packed, argfile_entry, is_plus_sigil, is_alternation) = scan_flags_block(lines, i, heading_is_bnf); i = end; let (seen, clean) = emit_flags_block( @@ -1440,6 +1468,7 @@ fn scan_entries( entries, packed, &is_plus_sigil, + &is_alternation, argfile_entry, st.result, ); @@ -1696,7 +1725,23 @@ fn parse_body( // iteration made this function quadratic, found via the coverage // harness (spec §13.1) parsing a degenerate input in minutes instead // of milliseconds. - let description_bound = i.max(leading_prose_bound(&lines)); + let prose_bound = leading_prose_bound(&lines); + let mut description_bound = i.max(prose_bound); + // A headingless table with no blank line ahead of it and no + // recognized usage line (Xvfb's `use: X [:] [option]`, not + // `usage:`) reaches `leading_prose_bound`'s whole-document fallback + // untouched, so its option rows land in the description as well as + // being independently recovered by `scan_entries` below — the same + // text rendered twice. Consulted only in that narrow case (no blank + // line anywhere, no usage line), so an ordinary document's already- + // correct, cheap bound pays nothing extra. See docs/shapes.md S-165. + if usage_start.is_none() && prose_bound == lines.len() { + if let Some(flag_start) = + (i..lines.len()).find(|&j| starts_attested_headingless_flag_block(&lines, j)) + { + description_bound = description_bound.min(flag_start); + } + } if let Some(description) = extract_description(&lines, description_bound, usage_start, i) { result.description = Some(description); } diff --git a/xtask/src/coverage/mod.rs b/xtask/src/coverage/mod.rs index 5bc80212..b4fb95ee 100644 --- a/xtask/src/coverage/mod.rs +++ b/xtask/src/coverage/mod.rs @@ -10,10 +10,10 @@ mod aggregate; mod fingerprint; mod render_markdown; mod render_text; +mod round10; mod round7; mod round8; mod round9; -mod round10; mod score; use aggregate::compute_aggregate; diff --git a/xtask/src/coverage/round10.rs b/xtask/src/coverage/round10.rs index 34b4cdb8..177402b7 100644 --- a/xtask/src/coverage/round10.rs +++ b/xtask/src/coverage/round10.rs @@ -1,6 +1,8 @@ //! Round-10 family detectors. Atlas S-162: a leading option-rejection //! diagnostic line fused into the root description. S-163: a `+word` -//! option row. +//! option row, and a `+/-word`/`[+-]word` alternation-sigil row. S-164: a +//! root flag group that repeats the root description verbatim. S-165: a +//! headingless option table duplicated into the root description. use crate::detector::{Detector, ToolEvidence}; use mandible_core::CommandNode; @@ -14,6 +16,14 @@ pub(super) fn round10_family_counts( let leading_diagnostic = crate::detector::leading_diagnostic_line::LeadingDiagnosticLine.hits(&evidence); let plus_word = crate::detector::plus_word_option::PlusWordOption.hits(&evidence); + let plus_minus_alternation = + crate::detector::plus_minus_alternation_option::PlusMinusAlternationOption.hits(&evidence); + let description_reused_as_group = + crate::detector::description_reused_as_group_label::DescriptionReusedAsGroupLabel + .hits(&evidence); + let headingless_table_in_description = + crate::detector::headingless_table_in_root_description::HeadinglessTableInRootDescription + .hits(&evidence); vec![ ( "leading-diagnostic-line", @@ -25,5 +35,23 @@ pub(super) fn round10_family_counts( plus_word.len(), plus_word.into_iter().take(cap).collect(), ), + ( + "plus-minus-alternation-option", + plus_minus_alternation.len(), + plus_minus_alternation.into_iter().take(cap).collect(), + ), + ( + "description-reused-as-group-label", + description_reused_as_group.len(), + description_reused_as_group.into_iter().take(cap).collect(), + ), + ( + "headingless-table-in-root-description", + headingless_table_in_description.len(), + headingless_table_in_description + .into_iter() + .take(cap) + .collect(), + ), ] } diff --git a/xtask/src/detector/description_reused_as_group_label.rs b/xtask/src/detector/description_reused_as_group_label.rs new file mode 100644 index 00000000..b1d92d08 --- /dev/null +++ b/xtask/src/detector/description_reused_as_group_label.rs @@ -0,0 +1,108 @@ +//! `description-reused-as-group-label` (atlas S-164): a root flag's own +//! `group` equals, verbatim (trimmed), the node's own root `description` +//! — the same sentence shown twice, once as the description and once as +//! the label over a whole flag block (`fc-scan`, `grub-macbless`). Reads +//! only the parsed tree, since both fields already live on +//! `CommandNode`/`Entity`. + +use crate::detector::{Detector, Expect, Scope, SelfCheck, ToolEvidence}; +use mandible_core::{CommandNode, Entity, EntityKind, Provenance, Source, Text}; + +pub struct DescriptionReusedAsGroupLabel; + +impl Detector for DescriptionReusedAsGroupLabel { + fn name(&self) -> &'static str { + "description-reused-as-group-label" + } + + fn family(&self) -> Option<&'static str> { + None + } + + fn describes(&self) -> &'static str { + "a root flag's own group equals the node's root description verbatim, the same \ + sentence rendered twice" + } + + fn hits(&self, evidence: &ToolEvidence<'_>) -> Vec { + let Some(description) = evidence.root.description.as_ref().map(|t| t.as_str()) else { + return Vec::new(); + }; + let description = description.trim(); + if description.is_empty() { + return Vec::new(); + } + let mut groups_seen = std::collections::HashSet::new(); + let mut findings = Vec::new(); + for flag in evidence.root.flags() { + let Some(group) = flag.group.as_deref() else { + continue; + }; + if group.trim() == description && groups_seen.insert(group.to_string()) { + findings.push(format!( + "root flag group {group:?} repeats the root description verbatim" + )); + } + } + findings + } + + fn scope(&self) -> Scope { + Scope::full() + } + + fn self_checks(&self) -> Vec { + fn flag_with_group(spelling: &str, group: Option<&str>) -> Entity { + let mut e = Entity::new(EntityKind::Flag, Provenance::single(Source::HelpText)); + e.spellings.push(mandible_core::Spelling::long(spelling)); + e.group = group.map(str::to_string); + e + } + fn node(description: Option<&str>, flags: Vec) -> CommandNode { + let mut root = CommandNode::new("prog", Provenance::single(Source::HelpText)); + root.description = description.map(Text::sanitize); + root.set_entities_of(EntityKind::Flag, flags); + root + } + + let sentence = "Scan font files and directories, and print resulting pattern(s)"; + + vec![ + SelfCheck { + name: "fc-scan's own shape, the sentence doubles as the group", + why: "the defect itself: the same sentence is the description and a group", + expect: Expect::Fires(1), + raw: String::new(), + root: node( + Some(sentence), + vec![flag_with_group("brief", Some(sentence))], + ), + }, + SelfCheck { + name: "the same tree, the group cleared once the fix lands", + why: "once the group is refused, the same tree goes silent", + expect: Expect::Silent, + raw: String::new(), + root: node(Some(sentence), vec![flag_with_group("brief", None)]), + }, + SelfCheck { + name: "gcc-ranlib-13's own real label, distinct text", + why: "a group whose text differs from the description is a real label and must \ + never fire", + expect: Expect::Silent, + raw: String::new(), + root: node( + Some(sentence), + vec![flag_with_group("brief", Some("The options are"))], + ), + }, + SelfCheck { + name: "a node with no description at all", + why: "nothing was ever spent as a description, so no group can repeat it", + expect: Expect::Silent, + raw: String::new(), + root: node(None, vec![flag_with_group("brief", Some(sentence))]), + }, + ] + } +} diff --git a/xtask/src/detector/headingless_table_in_root_description.rs b/xtask/src/detector/headingless_table_in_root_description.rs new file mode 100644 index 00000000..997c55fa --- /dev/null +++ b/xtask/src/detector/headingless_table_in_root_description.rs @@ -0,0 +1,160 @@ +//! `headingless-table-in-root-description` (atlas S-165): the root's own +//! `description` still carries several of the tree's own flag spellings +//! as substrings — the option table rendered once as prose in the +//! description and a second time as the tree's real flags (`Xvfb`, before +//! the fix). Reads only the parsed tree: a description that happens to +//! repeat three or more of the node's own recovered spellings, each at a +//! real word boundary, is treated as the table leaking through rather +//! than a coincidental mention. + +use crate::detector::{Detector, Expect, Scope, SelfCheck, ToolEvidence}; +use mandible_core::{CommandNode, Entity, EntityKind, Provenance, Source, Text}; + +/// The fewest of the tree's own flag spellings a description must repeat +/// before this reads as the table leaking through rather than an +/// ordinary sentence that happens to mention one flag by name (a real, +/// common shape: `"see --verbose for more"`). +const MIN_REPEATED_SPELLINGS: usize = 3; + +/// True when `spelling` occurs in `text` at a real word boundary: not +/// preceded or followed by a letter, digit, `-` or `+` — so `-a` inside +/// `-audit` never counts as `-a` occurring. +fn occurs_at_word_boundary(text: &str, spelling: &str) -> bool { + if spelling.is_empty() { + return false; + } + let is_word_char = |c: char| c.is_ascii_alphanumeric() || c == '-' || c == '+'; + let mut start = 0; + while let Some(idx) = text[start..].find(spelling) { + let idx = start + idx; + let before_ok = text[..idx] + .chars() + .next_back() + .is_none_or(|c| !is_word_char(c)); + let after = idx + spelling.len(); + let after_ok = text[after..] + .chars() + .next() + .is_none_or(|c| !is_word_char(c)); + if before_ok && after_ok { + return true; + } + start = idx + 1; + } + false +} + +pub struct HeadinglessTableInRootDescription; + +impl Detector for HeadinglessTableInRootDescription { + fn name(&self) -> &'static str { + "headingless-table-in-root-description" + } + + fn family(&self) -> Option<&'static str> { + None + } + + fn describes(&self) -> &'static str { + "the root description still repeats several of the tree's own recovered flag \ + spellings, the option table rendered once as prose and again as real flags" + } + + fn hits(&self, evidence: &ToolEvidence<'_>) -> Vec { + let Some(description) = evidence.root.description.as_ref().map(|t| t.as_str()) else { + return Vec::new(); + }; + let mut repeated: Vec = Vec::new(); + for flag in evidence.root.flags() { + for spelling in &flag.spellings { + let rendered = spelling.typed(); + if occurs_at_word_boundary(description, &rendered) { + repeated.push(rendered); + break; + } + } + } + if repeated.len() >= MIN_REPEATED_SPELLINGS { + vec![format!( + "root description repeats {} of the tree's own flag spellings: {:?}", + repeated.len(), + repeated + )] + } else { + Vec::new() + } + } + + fn scope(&self) -> Scope { + Scope::full() + } + + fn self_checks(&self) -> Vec { + fn flag(spelling: mandible_core::Spelling) -> Entity { + let mut e = Entity::new(EntityKind::Flag, Provenance::single(Source::HelpText)); + e.spellings.push(spelling); + e + } + fn node(description: Option<&str>, flags: Vec) -> CommandNode { + let mut root = CommandNode::new("Xvfb", Provenance::single(Source::HelpText)); + root.description = description.map(Text::sanitize); + root.set_entities_of(EntityKind::Flag, flags); + root + } + + let table_prose = "use: X [:] [option] -a # default pointer acceleration \ + (factor) -ac disable access control restrictions -audit int set \ + audit trail level"; + + vec![ + SelfCheck { + name: "Xvfb's own shape, the table duplicated into the description", + why: "the defect itself: three of the tree's own real flags recur in the \ + description's own prose", + expect: Expect::Fires(1), + raw: String::new(), + root: node( + Some(table_prose), + vec![ + flag(mandible_core::Spelling::single_dash("a")), + flag(mandible_core::Spelling::single_dash("ac")), + flag(mandible_core::Spelling::single_dash("audit")), + ], + ), + }, + SelfCheck { + name: "the same tree, description trimmed to just the usage line", + why: "once the table no longer lands in the description, the same tree goes \ + silent", + expect: Expect::Silent, + raw: String::new(), + root: node( + Some("use: X [:] [option]"), + vec![ + flag(mandible_core::Spelling::single_dash("a")), + flag(mandible_core::Spelling::single_dash("ac")), + flag(mandible_core::Spelling::single_dash("audit")), + ], + ), + }, + SelfCheck { + name: "an ordinary description mentioning one real flag by name", + why: "one mention is common prose (\"see --verbose\"), never the whole table \ + leaking through, and must never fire", + expect: Expect::Silent, + raw: String::new(), + root: node( + Some("Run the build. See --verbose for more detail."), + vec![flag(mandible_core::Spelling::long("verbose"))], + ), + }, + SelfCheck { + name: "a node with no description at all", + why: "nothing to repeat a spelling in", + expect: Expect::Silent, + raw: String::new(), + root: node(None, vec![flag(mandible_core::Spelling::single_dash("a"))]), + }, + ] + } +} diff --git a/xtask/src/detector/leading_diagnostic_line.rs b/xtask/src/detector/leading_diagnostic_line.rs index 7174cab2..aecb2a02 100644 --- a/xtask/src/detector/leading_diagnostic_line.rs +++ b/xtask/src/detector/leading_diagnostic_line.rs @@ -22,15 +22,18 @@ fn looks_like_option_rejection(line: &str) -> bool { return false; } let body = match trimmed.split_once(": ") { - Some((prefix, rest)) if !prefix.is_empty() && !prefix.contains(char::is_whitespace) => { - rest - } + Some((prefix, rest)) if !prefix.is_empty() && !prefix.contains(char::is_whitespace) => rest, _ => trimmed, }; let lower = body.to_ascii_lowercase(); - ["invalid option", "unrecognized option", "unknown option", "illegal option"] - .iter() - .any(|kw| lower.starts_with(kw)) + [ + "invalid option", + "unrecognized option", + "unknown option", + "illegal option", + ] + .iter() + .any(|kw| lower.starts_with(kw)) } /// `raw`'s own first non-empty physical line, or `None` for an empty diff --git a/xtask/src/detector/mod.rs b/xtask/src/detector/mod.rs index 73b0d24c..3b73428b 100644 --- a/xtask/src/detector/mod.rs +++ b/xtask/src/detector/mod.rs @@ -49,19 +49,22 @@ pub(crate) mod choices_after_optional_placeholder; pub(crate) mod comma_glued_option_value; pub(crate) mod command_row_argument_placeholder; pub(crate) mod description_continuation_dash_flag; +pub(crate) mod description_reused_as_group_label; pub(crate) mod description_subcommands_list; pub(crate) mod examples_block_contaminates_last_flag; pub(crate) mod generic_option_placeholder_flag; pub(crate) mod glued_optional_group_spelling; pub(crate) mod glued_uppercase_shared_prefix; pub(crate) mod hash_in_spelling; +pub(crate) mod headingless_table_in_root_description; pub(crate) mod leading_diagnostic_line; -pub(crate) mod plus_word_option; pub(crate) mod multi_operand_usage_tail; pub(crate) mod nested_bracket_value; pub(crate) mod numbered_variadic_usage_tail; pub(crate) mod or_joined_alias_single_space_gap; pub(crate) mod or_joined_alias_with_values; +pub(crate) mod plus_minus_alternation_option; +pub(crate) mod plus_word_option; pub(crate) mod positional_description_block; pub(crate) mod single_dash_long_table; pub(crate) mod spaced_single_dash_long; @@ -752,6 +755,9 @@ pub fn registry() -> Vec> { Box::new(option_table_multiword_value_name::OptionTableMultiwordValueName), Box::new(leading_diagnostic_line::LeadingDiagnosticLine), Box::new(plus_word_option::PlusWordOption), + Box::new(plus_minus_alternation_option::PlusMinusAlternationOption), + Box::new(description_reused_as_group_label::DescriptionReusedAsGroupLabel), + Box::new(headingless_table_in_root_description::HeadinglessTableInRootDescription), ] } diff --git a/xtask/src/detector/plus_minus_alternation_option.rs b/xtask/src/detector/plus_minus_alternation_option.rs new file mode 100644 index 00000000..d5756dc5 --- /dev/null +++ b/xtask/src/detector/plus_minus_alternation_option.rs @@ -0,0 +1,182 @@ +//! `plus-minus-alternation-option` (atlas S-163): a `+/-word`, `-/+word`, +//! `[+-]word` or `[-+]word` alternation-sigil row names two flags, +//! `+word` and `-word`, on one row. Anchored to the row's own leading +//! token, the same evidence +//! `mandible_extract::help_text::sections::flag_rows::plus_minus_alternation_word` +//! requires, so `xxd`'s `-s [+][-]seek` (whose leading token is `-s`, not +//! one of these four sigils) is never in scope — checked independently +//! here since a detector reads only `raw`+`root`. + +use crate::detector::{Detector, Expect, Scope, SelfCheck, ToolEvidence}; +use mandible_core::{CommandNode, Provenance, Source}; + +/// [`crate::family_row::leading_token`] without its `trimmed == line` +/// indentation requirement — a headingless table (`Xvfb`'s own `+/-render` +/// and `[+-]accessx` rows, S-165) carries no indentation at all. Kept +/// local rather than widening the shared helper, since other detectors +/// read it. See `plus_word_option.rs`'s own copy of this same fix. +fn leading_token(line: &str) -> Option<(&str, &str)> { + let trimmed = line.trim_start(); + if trimmed.is_empty() { + return None; + } + let token = trimmed.split_whitespace().next()?; + let rest = &trimmed[token.len()..]; + Some((token, rest)) +} + +/// The base word right after a `+/-`/`-/+`/`[+-]`/`[-+]` alternation +/// sigil at the very start of `token`, when a letter-led run of +/// letters/digits/`-` follows immediately. `None` for every other token. +fn alternation_word(token: &str) -> Option<&str> { + let rest = token + .strip_prefix("+/-") + .or_else(|| token.strip_prefix("-/+")) + .or_else(|| token.strip_prefix("[+-]")) + .or_else(|| token.strip_prefix("[-+]"))?; + let word_end = rest + .char_indices() + .find(|(_, c)| !(c.is_ascii_alphanumeric() || *c == '-')) + .map_or(rest.len(), |(i, _)| i); + if word_end == 0 || !rest.starts_with(|c: char| c.is_ascii_alphabetic()) { + return None; + } + Some(&rest[..word_end]) +} + +/// `spelling` is compared against [`mandible_core::Spelling::typed`], not +/// the raw `.name` field: a `-word` spelling stores its dash separately +/// (`Dashes::Single`, `.name == "word"`), while a `+word` spelling stores +/// the sigil inside `.name` itself (`Dashes::None`, S-163's own +/// convention) — `.typed()` renders both the same way a user would type +/// them, which is the only form this detector's own `plus`/`minus` +/// spellings above are built to match. +fn tree_has_spelling(root: &CommandNode, spelling: &str) -> bool { + root.flags() + .any(|e| e.spellings.iter().any(|s| s.typed() == spelling)) +} + +pub struct PlusMinusAlternationOption; + +impl Detector for PlusMinusAlternationOption { + fn name(&self) -> &'static str { + "plus-minus-alternation-option" + } + + fn family(&self) -> Option<&'static str> { + None + } + + fn describes(&self) -> &'static str { + "a `+/-word`/`[+-]word` alternation-sigil row whose `+word` and `-word` pair does not \ + both reach the tree" + } + + fn hits(&self, evidence: &ToolEvidence<'_>) -> Vec { + let mut findings = Vec::new(); + for line in evidence.raw.lines() { + let Some((token, _)) = leading_token(line) else { + continue; + }; + let Some(word) = alternation_word(token) else { + continue; + }; + let plus = format!("+{word}"); + let minus = format!("-{word}"); + let plus_ok = tree_has_spelling(evidence.root, &plus); + let minus_ok = tree_has_spelling(evidence.root, &minus); + if !plus_ok || !minus_ok { + findings.push(format!( + "{token:?} did not expand to both {plus:?} and {minus:?} (have {plus}={plus_ok}, {minus}={minus_ok})" + )); + } + } + findings + } + + fn scope(&self) -> Scope { + Scope::full() + } + + fn self_checks(&self) -> Vec { + fn node_with_flags(name: &str, flags: Vec) -> CommandNode { + let mut root = CommandNode::new(name, Provenance::single(Source::HelpText)); + root.set_entities_of(mandible_core::EntityKind::Flag, flags); + root + } + fn flag(spelling: mandible_core::Spelling) -> mandible_core::Entity { + let mut e = mandible_core::Entity::new( + mandible_core::EntityKind::Flag, + Provenance::single(Source::HelpText), + ); + e.spellings.push(spelling); + e + } + fn plus(word: &str) -> mandible_core::Entity { + flag(mandible_core::Spelling::bare(format!("+{word}"))) + } + fn minus(word: &str) -> mandible_core::Entity { + flag(mandible_core::Spelling::single_dash(word)) + } + + let render_raw = + " +/-render\t\t turn on/off RENDER extension support(default on)\n".to_string(); + let accessx_raw = " [+-]accessx [ timeout [ ttb [ tpo [ ctrls ]]]] enable/disable \ + accessx\n" + .to_string(); + + vec![ + SelfCheck { + name: "Xvfb's own `+/-render` row, neither half recovered", + why: "the defect itself: the row names two flags and the tree has neither", + expect: Expect::Fires(1), + raw: render_raw.clone(), + root: node_with_flags("Xvfb", vec![]), + }, + SelfCheck { + name: "the same row, only `-render` recovered", + why: "one half missing is still a loss, not a silence", + expect: Expect::Fires(1), + raw: render_raw.clone(), + root: node_with_flags("Xvfb", vec![minus("render")]), + }, + SelfCheck { + name: "the same row, both `+render` and `-render` recovered", + why: "once both halves reach the tree, the row goes silent", + expect: Expect::Silent, + raw: render_raw, + root: node_with_flags("Xvfb", vec![plus("render"), minus("render")]), + }, + SelfCheck { + name: "Xvfb's own `[+-]accessx` row, neither half recovered", + why: "the bracketed sigil order names the same pair", + expect: Expect::Fires(1), + raw: accessx_raw.clone(), + root: node_with_flags("Xvfb", vec![]), + }, + SelfCheck { + name: "the same bracketed row, both halves recovered", + why: "once both halves reach the tree, the row goes silent", + expect: Expect::Silent, + raw: accessx_raw, + root: node_with_flags("Xvfb", vec![plus("accessx"), minus("accessx")]), + }, + SelfCheck { + name: "xxd's own `-s [+][-]seek` row, the named counter-case", + why: "the leading token is `-s`, not one of the four alternation sigils, and \ + must never fire — S-097 already refuses to fold this shape", + expect: Expect::Silent, + raw: " -s [+][-]seek seek offset (base is 16, +/- prefix optional)\n" + .to_string(), + root: node_with_flags("xxd", vec![]), + }, + SelfCheck { + name: "an ordinary `-word` row, no sigil at all", + why: "a plain dash-led row carries no alternation sigil and must never fire", + expect: Expect::Silent, + raw: " -render set render color alloc policy\n".to_string(), + root: node_with_flags("Xvfb", vec![minus("render")]), + }, + ] + } +} diff --git a/xtask/src/detector/plus_word_option.rs b/xtask/src/detector/plus_word_option.rs index 52c97bab..28a106fd 100644 --- a/xtask/src/detector/plus_word_option.rs +++ b/xtask/src/detector/plus_word_option.rs @@ -1,24 +1,32 @@ //! `plus-word-option` (atlas S-163): a `+word` option row (`+bs`, `+i`, //! `+s`) — a letter-led run after the sigil, distinct from S-095's own //! `plus-prefixed-option` (bare `+`/`+` only) — reaches no -//! entity in the tree. Requires indentation, the same evidence -//! `mandible_core::family_row::leading_token` and the parser's own -//! `scan_flags_block` both require, so an unindented specimen (`Xvfb`'s -//! own column-0 rows) is out of this family's reach until S-165's -//! headingless-table defect is fixed. -//! -//! A separate detector rather than a widening of `plus-prefixed-option`: -//! that family is already `REPAIRED` and ratchet-gated at zero -//! (docs/shapes.md S-095), so folding a still-open shape into it would -//! break the gate on tools this fix has not reached. Mirrors -//! `mandible_extract::help_text::sections::flag_rows::is_claimed_plus_token`'s -//! `+word` arm, checked independently here since a detector reads only -//! `raw`+`root`. +//! entity in the tree. Reads a row with [`leading_token_any_indent`], a +//! local copy of `family_row::leading_token` without its indentation +//! requirement: a headingless table (S-165) carries none at all. A +//! separate detector rather than a widening of `plus-prefixed-option`, +//! which is already `REPAIRED` and gated at zero (S-095). Mirrors +//! `mandible_extract`'s own `is_claimed_plus_token`, checked +//! independently since a detector reads only `raw`+`root`. use crate::detector::{Detector, Expect, Scope, SelfCheck, ToolEvidence}; -use crate::family_row::{leading_token, opens_description_column}; +use crate::family_row::opens_description_column; use mandible_core::{CommandNode, Provenance, Source}; +/// [`crate::family_row::leading_token`] without its `trimmed == line` +/// indentation requirement — see this module's own doc comment for why +/// this family needs that, unlike the four detectors that share the +/// original. +fn leading_token_any_indent(line: &str) -> Option<(&str, &str)> { + let trimmed = line.trim_start(); + if trimmed.is_empty() { + return None; + } + let token = trimmed.split_whitespace().next()?; + let rest = &trimmed[token.len()..]; + Some((token, rest)) +} + /// True when `token` is a `+word` spelling this family claims: `+` /// followed by a run opening with a letter, every later character /// alphanumeric or `-` — never a bare `+`, a bracketed placeholder @@ -29,9 +37,7 @@ fn is_plus_word_token(token: &str) -> bool { }; let mut chars = rest.chars(); match chars.next() { - Some(c) if c.is_ascii_alphabetic() => { - chars.all(|c| c.is_ascii_alphanumeric() || c == '-') - } + Some(c) if c.is_ascii_alphabetic() => chars.all(|c| c.is_ascii_alphanumeric() || c == '-'), _ => false, } } @@ -44,7 +50,7 @@ fn tree_has_spelling(root: &CommandNode, token: &str) -> bool { /// True when `line`'s own leading token is flag-shaped evidence: a real /// `-`-prefixed flag, or this same family's own `+word` claim. fn is_flag_shaped_neighbor(line: &str) -> bool { - let Some((token, _)) = leading_token(line) else { + let Some((token, _)) = leading_token_any_indent(line) else { return false; }; let token = token.trim_end_matches(','); @@ -87,7 +93,7 @@ impl Detector for PlusWordOption { let mut findings = Vec::new(); let lines: Vec<&str> = evidence.raw.lines().collect(); for (i, line) in lines.iter().enumerate() { - let Some((token, rest)) = leading_token(line) else { + let Some((token, rest)) = leading_token_any_indent(line) else { continue; }; let token = token.trim_end_matches(','); @@ -127,15 +133,19 @@ impl Detector for PlusWordOption { let fzf_raw = " -i Case-insensitive match (default: smart-case \ match)\n +i Case-sensitive match\n" .to_string(); - // `Xvfb`'s own row shape (multi-letter, `-word` neighbor) at its - // own indent (two spaces) — column 0, `Xvfb`'s real indentation, - // carries no evidence at all for `leading_token` (S-165's own - // headingless-table defect, not this family's to fix). + // The multi-letter shape at a real indent (two spaces). let indented_multiletter_raw = " -br create root window with black \ background\n +bs enable any backing \ store support\n -bs disable any \ backing store support\n" .to_string(); + // `Xvfb`'s own row shape verbatim: no indentation at all (S-165's + // headingless table), which `leading_token_any_indent` now reads. + let column_zero_raw = "-br create root window with black \ + background\n+bs enable any backing store \ + support\n-bs disable any backing store \ + support\n" + .to_string(); vec![ SelfCheck { @@ -167,6 +177,21 @@ impl Detector for PlusWordOption { raw: indented_multiletter_raw, root: node_with_flags("Xvfb", vec![plus_word_flag("bs")]), }, + SelfCheck { + name: "Xvfb's own column-0 row, no indentation at all", + why: "a headingless table (S-165) carries no leading whitespace, and this \ + family must still see it", + expect: Expect::Fires(1), + raw: column_zero_raw.clone(), + root: node_with_flags("Xvfb", vec![]), + }, + SelfCheck { + name: "the same column-0 row recovered as its own spelling", + why: "once recovered, the same raw row goes silent regardless of indentation", + expect: Expect::Silent, + raw: column_zero_raw, + root: node_with_flags("Xvfb", vec![plus_word_flag("bs")]), + }, SelfCheck { name: "a bare `+` row, S-095's own claim, not this family's", why: "a bare `+` has no letter run at all and must never be claimed here", From 9bc2986469d81cce98e11955d5c1ef7fd8c22705 Mon Sep 17 00:00:00 2001 From: Sadig Akhund Date: Sun, 13 Sep 2026 02:08:20 +0400 Subject: [PATCH 5/9] integration: cut the env-column detector comment to the contract Co-Authored-By: Claude Fable 5.1 --- xtask/src/coverage/round10.rs | 27 ++++++++----------- .../detector/header_declared_env_column.rs | 22 ++++----------- 2 files changed, 16 insertions(+), 33 deletions(-) diff --git a/xtask/src/coverage/round10.rs b/xtask/src/coverage/round10.rs index d6c05ab7..14a15f5e 100644 --- a/xtask/src/coverage/round10.rs +++ b/xtask/src/coverage/round10.rs @@ -2,14 +2,10 @@ //! diagnostic line fused into the root description. S-163: a `+word` //! option row, and a `+/-word`/`[+-]word` alternation-sigil row. S-164: a //! root flag group that repeats the root description verbatim. S-165: a -//! headingless option table duplicated into the root description. +//! headingless option table duplicated into the root description. S-166: a +//! header-declared three-column option table. Split into its own file for +//! the same line-count reason `round7.rs`, `round8.rs` and `round9.rs` are. -//! The round-10 family detectors. `header-declared-env-column` is atlas -//! S-166, W6's own header-declared three-column option table. Split into -//! its own file for the same line-count reason `round7.rs`, `round8.rs` -//! and `round9.rs` are. - -use super::score::FAMILY_DETECTOR_SAMPLES_PER_ROW; use crate::detector::{Detector, ToolEvidence}; use mandible_core::CommandNode; @@ -19,6 +15,9 @@ pub(super) fn round10_family_counts( ) -> Vec<(&'static str, usize, Vec)> { let cap = super::score::FAMILY_DETECTOR_SAMPLES_PER_ROW; let evidence = ToolEvidence { raw, root }; + let env_col = + crate::detector::header_declared_env_column::HeaderDeclaredEnvColumn.hits(&evidence); + let evidence = ToolEvidence { raw, root }; let leading_diagnostic = crate::detector::leading_diagnostic_line::LeadingDiagnosticLine.hits(&evidence); let plus_word = crate::detector::plus_word_option::PlusWordOption.hits(&evidence); @@ -59,14 +58,10 @@ pub(super) fn round10_family_counts( .take(cap) .collect(), ), + ( + "header-declared-env-column", + env_col.len(), + env_col.into_iter().take(cap).collect(), + ), ] - let cap = FAMILY_DETECTOR_SAMPLES_PER_ROW; - let evidence = ToolEvidence { raw, root }; - let env_col = - crate::detector::header_declared_env_column::HeaderDeclaredEnvColumn.hits(&evidence); - vec![( - "header-declared-env-column", - env_col.len(), - env_col.into_iter().take(cap).collect(), - )] } diff --git a/xtask/src/detector/header_declared_env_column.rs b/xtask/src/detector/header_declared_env_column.rs index e2becb02..b1510290 100644 --- a/xtask/src/detector/header_declared_env_column.rs +++ b/xtask/src/detector/header_declared_env_column.rs @@ -1,21 +1,9 @@ -//! `header-declared-env-column` (atlas S-166): a header-declared three- -//! column option table (`Argument`/`Env-variable`/`Description`, the -//! whole `qemu-*-static` fleet) whose own header row names its middle -//! column as an environment variable. The unfixed parser glues that -//! column onto the matching flag's description instead of reading it as -//! the flag's own [`Entity::env_var`] cross-reference (spec §4.5, §7 -//! Tier B rule 16). +//! `header-declared-env-column` (atlas S-166): a table whose own header row +//! names its columns `Argument`, `Env-variable` and `Description`, read at +//! one generic column gap so the variable glued onto the description. //! -//! Independent re-implementation of the header/row shape — no shared -//! code with `mandible_extract::help_text::sections`, so the detector -//! cannot agree with the parser by construction. -//! -//! The seed-7 audit labels `qemu-riscv64-static` "incomplete" and -//! describes this exact shape ("a triple column help text ... flags -//! being an alias of the env vars"), but no `DEFECT_FAMILIES` entry -//! names it yet and the entry carries no derived family, so -//! [`Detector::family`] returns `None` (spec §13.1e rule 6) rather than -//! forcing it onto an unrelated family. +//! Fixture: `corpus/qemu-riscv64-static/8.2.2`. 42 qemu binaries share one +//! help template. use crate::detector::{Detector, Expect, Scope, SelfCheck, ToolEvidence}; use mandible_core::CommandNode; From 543fd08b76d005236d52827f34f796652203ff7c Mon Sep 17 00:00:00 2001 From: Sadig Akhund Date: Sun, 13 Sep 2026 09:55:22 +0400 Subject: [PATCH 6/9] [S-163, S-168, S-145] fix Xvfb's +/-render collision, +extension/-extension value column, and its choice list +/-render's expansion no longer collides with the ordinary -render row; +word and -word now read the same value column and share one colon- introduced choice list (new S-168) instead of folding it into a description. Co-Authored-By: Claude Fable 5.1 --- CHANGELOG.md | 3 + corpus/Xvfb/audit-seed/expected.snap | 56 +++- corpus/Xvfb/audit-seed/meta.toml | 39 ++- docs/shapes.md | 93 ++++++- .../src/help_text/sections/flag_rows.rs | 133 +++++++++- .../src/help_text/sections/mod.rs | 9 + .../src/help_text/sections/repair.rs | 102 ++++++++ mandible-tui/src/render/detail_pane/entity.rs | 13 +- mandible-tui/src/render/detail_pane/layout.rs | 30 ++- .../detector/choice_list_under_placeholder.rs | 246 ++++++++++++++++++ xtask/src/detector/mod.rs | 5 + 11 files changed, 707 insertions(+), 22 deletions(-) create mode 100644 xtask/src/detector/choice_list_under_placeholder.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index af281dbb..c9dd82fb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -38,6 +38,9 @@ once it reaches a published 0.1.0 release. - [S-164] A root flag group that only repeated the node's own description verbatim is dropped now, so `mandible fc-scan`, `mandible fc-validate`, `mandible grub-macbless` and `mandible lto-dump` no longer show the same sentence twice. - [S-165] A headingless option table no longer duplicates into the root description, so `mandible Xvfb` shows its real description once and its whole flag table instead of the same text rendered twice. - [S-166] A header row that names its own columns as `Argument`, `Env-variable` and `Description` is now read at those exact offsets, so `mandible qemu-riscv64-static` and the rest of the `qemu-*-static` fleet show a clean description and each flag's own environment variable instead of the two glued together, keep `-cpu` and `-dfilter`'s own value names, and no longer invent an `-E` row from the prose paragraph below the table. +- [S-163] A `+/-name` alternation row's own expansion no longer collides with an ordinary row documenting the same spelling, and a `+word` row's value column is now borrowed onto its `-word` sibling when the ordinary repair can't recover a bare one, so `mandible Xvfb` keeps one `-render` with its four choices, one `+render`, and the same `name` value on both `+extension` and `-extension`; its `+word` spellings also render in the long column now, beside `-render`, instead of the short column at column 0. +- [S-168] A colon-introduced list of bare-name choices sitting directly under a placeholder row — or a `+word`/`-word` pair sharing one — now becomes that placeholder's own choices instead of folding into the row's own description, so `mandible Xvfb`'s `+extension`/`-extension` pair shows its 19 run-time-toggleable extension names as choices on both halves. +- [S-145] A single-dash-long table with no column padding at all now recovers a row's genuine bracketed value one space after its name, so `mandible Xvfb`'s `-render` and `-deferglyphs` keep their own choices instead of losing them outright. ## [0.7.0] - 2026-09-05 diff --git a/corpus/Xvfb/audit-seed/expected.snap b/corpus/Xvfb/audit-seed/expected.snap index e01e8c9e..ee172825 100644 --- a/corpus/Xvfb/audit-seed/expected.snap +++ b/corpus/Xvfb/audit-seed/expected.snap @@ -101,6 +101,8 @@ flags: - help-text - spellings: - -deferglyphs + value_name: '[none|all|16]' + value_kind: Optional description: defer loading of [no|all|16-bit] glyphs provenance: sources: @@ -239,6 +241,8 @@ flags: - help-text - spellings: - -render + value_name: '[default|mono|gray|color]' + value_kind: Optional description: set render color alloc policy provenance: sources: @@ -337,13 +341,55 @@ flags: - +extension value_name: name value_kind: Required + choices: + - Generic Event Extension + - MIT-SHM + - XTEST + - SECURITY + - XINERAMA + - XFIXES + - RENDER + - RANDR + - COMPOSITE + - DAMAGE + - MIT-SCREEN-SAVER + - DOUBLE-BUFFER + - RECORD + - DPMS + - X-Resource + - XVideo + - XVideo-MotionCompensation + - SELinux + - GLX description: Enable extension provenance: sources: - help-text - spellings: - -extension - description: 'Disable extension Only the following extensions can be run-time enabled/disabled: Generic Event Extension MIT-SHM XTEST SECURITY XINERAMA XFIXES RENDER RANDR COMPOSITE DAMAGE MIT-SCREEN-SAVER DOUBLE-BUFFER RECORD DPMS X-Resource XVideo XVideo-MotionCompensation SELinux GLX' + value_name: name + value_kind: Required + choices: + - Generic Event Extension + - MIT-SHM + - XTEST + - SECURITY + - XINERAMA + - XFIXES + - RENDER + - RANDR + - COMPOSITE + - DAMAGE + - MIT-SCREEN-SAVER + - DOUBLE-BUFFER + - RECORD + - DPMS + - X-Resource + - XVideo + - XVideo-MotionCompensation + - SELinux + - GLX + description: Disable extension provenance: sources: - help-text @@ -361,6 +407,8 @@ flags: - help-text - spellings: - -multicast + value_name: '[addr [hops]' + value_kind: Optional description: IPv6 multicast for XDMCP provenance: sources: @@ -452,12 +500,6 @@ flags: provenance: sources: - help-text -- spellings: - - -render - description: turn on/off RENDER extension support(default on) - provenance: - sources: - - help-text - spellings: - -linebias description: adjust thin line pixelization diff --git a/corpus/Xvfb/audit-seed/meta.toml b/corpus/Xvfb/audit-seed/meta.toml index a7622e88..3aa03683 100644 --- a/corpus/Xvfb/audit-seed/meta.toml +++ b/corpus/Xvfb/audit-seed/meta.toml @@ -3,6 +3,15 @@ # defects shared one specimen: the leading diagnostic (S-162), `+word` and # `+/-name`/`[+-]name` rows (S-163), and the headingless table landing in the # root description (S-165). All three are fixed. +# +# Round 11 repaired four further defects on this same specimen: the +# `+/-render` alternation row used to collide with the ordinary `-render` +# row, losing its four choices (S-163); `+extension`/`-extension` used to +# read different value columns for the same `name` placeholder (S-163); +# the colon-introduced list of run-time-toggleable extension names used to +# fold into `-extension`'s own description instead of becoming `name`'s +# choices, shared by both halves of the pair (S-168); and `+word` spellings +# used to render in the wrong column (mandible-tui, not this fixture). [bless] provenance = "agent" @@ -24,8 +33,9 @@ min_status = "ok" # The leading diagnostic never survives into the root description (S-162). must_not_describe_root = ["Unrecognized option", "disable access control restrictions"] # `+word` (S-163 rule 1) and the alternation-sigil rows (S-163 rules 2/3) -# all reach the tree, both halves each. -must_contain_flags = ["+bs", "-bs", "+accessx", "-accessx", "+render", "-render"] +# all reach the tree, both halves each. `+extension`/`-extension` reach it +# too, each exactly once (round 11's collision-avoidance rule). +must_contain_flags = ["+bs", "-bs", "+accessx", "-accessx", "+render", "-render", "+extension", "-extension"] # Neither the `use:` usage-label line nor `[+-]accessx`'s own wrapped # description survives as a fabricated group heading (S-164/S-165). must_not_contain_flag_group_prefixes = ["use: X", "Enable/disable accessx"] @@ -34,3 +44,28 @@ must_not_contain_flag_group_prefixes = ["use: X", "Enable/disable accessx"] # The headingless table no longer duplicates into the root description # (S-165): each flag keeps its own row's own description. "-a" = "default pointer acceleration" +# The `+/-render` alternation row's own expansion never overwrites the +# ordinary `-render` row's own description (round 11, S-163). +"-render" = "set render color alloc policy" +"+render" = "turn on/off RENDER extension support" +# The colon-introducer line is gone from `-extension`'s description, not +# folded in (S-168). +"-extension" = "Disable extension" +"+extension" = "Enable extension" + +[contract.must_value_name] +# The `+/-render` collision never costs `-render` its own four-choice +# bracket value (round 11, S-163). +"-render" = "default" +# `+extension` and `-extension` read the same value column (round 11, +# S-163): both name the placeholder `name`, never a garbled swallowed tail. +"-extension" = "name" +"+extension" = "name" + +[contract.must_attach_choices] +# The colon-introduced list of run-time-toggleable extensions (S-168) +# becomes `name`'s own choices, shared by both halves of the +# `+extension`/`-extension` pair — never folded into `-extension`'s +# description. +"-extension" = ["Generic Event Extension", "MIT-SHM", "GLX"] +"+extension" = ["Generic Event Extension", "MIT-SHM", "GLX"] diff --git a/docs/shapes.md b/docs/shapes.md index 0dead7bd..e9a99491 100644 --- a/docs/shapes.md +++ b/docs/shapes.md @@ -2682,7 +2682,7 @@ entry's `tools` field and nothing else. It does not get a new entry. -pf add list of pseudo file definitions from -Xhelp print compressor options for selected compressor -mem use physical memory for caches -- tools: mksquashfs, sqfstar +- tools: mksquashfs, sqfstar, Xvfb - handling: A table whose rows are column-0 `-word` spellings, tab- or column-gap separated from their descriptions, with at least two rows carrying an unambiguous, uniformly-lowercase multi-character name and no `--long` @@ -2699,7 +2699,14 @@ entry's `tools` field and nothing else. It does not get a new entry. unambiguous-evidence requirement exist because a single ambiguous row (a short flag glued to a capitalized description word) cannot tell a table from a coincidence on its own; a bundled-short-flag document - (digit- or case-mixed clusters) is excluded the same way. + (digit- or case-mixed clusters) is excluded the same way. Round 11: + `row_is_table_shaped`'s own gap test (tab or double-space) never fires + on a table with no column padding at all — Xvfb's own headingless shape + (S-165) — so a row's genuine `<...>`/`[...]` placeholder exactly one + space after the name is now admitted as the same evidence + (`-render [default|mono|gray|color]`, `-deferglyphs [none|all|16]`), + recovering both their bracket values without widening the repair to + bare, unbracketed words (still out of scope, S-117's own reasoning). - fleet: `single-dash-long-table` (`xtask/src/detector/single_dash_long_table.rs`) reads 14 tools/21 raw findings on a full-`PATH` sweep of 2323 tools, 2026-09-07, after the fix: unsquashfs, xkill, xev, setfont and others in the same @@ -2868,6 +2875,39 @@ entry's `tools` field and nothing else. It does not get a new entry. verbatim. (3) Neither rule fires where the sigil is not the row's own leading token: `xxd`'s `-s [+][-]seek` opens with `-s`, and stays refused, matching S-097's own counter-case. + Round 11 fixed three further defects on this same shape, all in + `help_text::sections::repair.rs`/`flag_rows.rs`. (4) An alternation + row's own expansion never duplicates or overwrites a spelling an + ordinary row elsewhere in the document already documents: + `+/-render`'s own `-render` half used to collide with the standalone + `-render [default|mono|gray|color]` row, and the collision cost the + ordinary row its own value and description + (`resolve_alternation_spelling_collisions`, run last, after every + other repair, so richness — an already-recovered value or choices — + decides which duplicate survives). (5) `+word` and its `-word` sibling + read the same value column: `parse_plus_sigil_spec`'s hand-rolled word + scan already reads a bare, unbracketed value correctly (`+extension + name`'s `name`), but the ordinary `-word` row goes through + `repair_single_dash_long_options` instead, which only recovers a + bracket/angle or `=`-glued value and silently drops a bare one — the + already-correct value is now borrowed onto the sibling + (`borrow_plus_word_value_for_dash_sibling`) rather than teaching the + ordinary repair to guess at bare words (still out of scope, S-117's own + reasoning). (6) A `+word` spelling longer than one character renders in + the wrong column: `mandible-tui`'s own column-choice test treated every + dashless spelling alike, so `+render`/`+extension` landed in the short + column at column 0 instead of beside `-render` in the long column; a + one-character dashless spelling (`+i`, `fzf`'s own row) still belongs in + the short column, and a modifier letter or environment variable name + (dashless for an unrelated reason) stays there regardless of length + (`mandible-tui/src/render/detail_pane/layout.rs::bare_spelling_column`, + spec §9.1a). The same distinction fixed the render *gap*: + `spelling_is_sigil` used to glue any dashless spelling's value with no + space (the argfile sigil's own `@` shape), which rendered + `+extension name` as `+extensionname`; now only a spelling that is + nothing but its own bare sigil character (`@`, the standalone `+`) + glues, never a `+word` that already carries its own word + (`mandible-tui/src/render/detail_pane/entity.rs::spelling_is_sigil`). - fleet: `plus-word-option` (`xtask/src/detector/plus_word_option.rs`, rule 1) and `plus-minus-alternation-option` (`xtask/src/detector/plus_minus_alternation_option.rs`, rules 2/3) are @@ -2973,3 +3013,52 @@ entry's `tools` field and nothing else. It does not get a new entry. `qemu-riscv64-static` and `qemu-arm64-static`; raw-shape grep over the seed's own captures reads 42 tools / 42 findings, the whole `qemu-*-static` set, 2026-09-12. + +### S-168: a colon-introduced choice list under a placeholder pair + +- id: S-168 +- looks like: | + +extension name Enable extension + -extension name Disable extension + Only the following extensions can be run-time enabled/disabled: + Generic Event Extension + MIT-SHM + XTEST +- tools: Xvfb; raw-shape grep also finds `chmem`'s "Supported zones:" + (`DMA`, `DMA32`, `Normal`, `Highmem`, `Movable`) under `-z, --zone ` +- handling: Fixed for Xvfb. A colon-terminated introducer sentence + (`looks_like_choice_list_introducer`) directly followed by at least two + bare-name item lines (`looks_like_choice_list_item` — a short run of + hyphenated words, refused the moment a genuine column gap appears, which + is what tells this apart from `as`'s own *described* sub-option rows, + S-015's territory) becomes the placeholder's own `choices` + (`mark_choice_list_rows`, `mandible-extract/src/help_text/sections/flag_rows.rs`). + Neither the introducer line nor the items are folded into the row's own + description. When the placeholder is shared by a `+word`/`-word` pair + (S-163), the same choices reach both halves + (`spec_word_after_sigil`'s pairing check), since the pair documents one + value column twice. Two further repairs on the same specimen ride along: + `+extension`/`-extension` used to read different value columns for the + same `name` placeholder — the `+word` grammar (`parse_plus_sigil_spec`) + already read a bare word correctly, and now that value is borrowed onto + the `-word` sibling when the ordinary single-dash-long repair could not + recover it on its own (`borrow_plus_word_value_for_dash_sibling`); and + the `+/-render` alternation row's own expansion used to collide with the + ordinary `-render` row, duplicating `-render` and losing its own + four-choice bracket value — the expansion now contributes only the + spelling an ordinary row does not already document + (`resolve_alternation_spelling_collisions`), both in + `mandible-extract/src/help_text/sections/repair.rs`. `-render`'s own + bracket value was lost separately: `row_is_table_shaped` required a + tab or double-space gap that a single-dash-long table with no column + padding at all (S-165's own headingless shape) never has; a genuine + `<...>`/`[...]` placeholder exactly one space after the name is now + admitted as the same evidence. +- fleet: `choice-list-under-placeholder` + (`xtask/src/detector/choice_list_under_placeholder.rs`) is family `None`, + so calibration reads NOT EVALUABLE; self-checks hold (4/4). Raw-shape + grep over `/home/ubuntu/projects/mandible/audit/queue-captures/`: 26 + tools / 26 findings, an upper bound on the raw shape alone, not the + tree-level count — most (`bash`, `perf`, `usbip`, `dmesg`, `bpftool`, and + others) are not audited here and their own parse is unexamined; `Xvfb` + is the one fixture fixed and corpus-pinned this round. 2026-09-13. diff --git a/mandible-extract/src/help_text/sections/flag_rows.rs b/mandible-extract/src/help_text/sections/flag_rows.rs index f6949c97..71024098 100644 --- a/mandible-extract/src/help_text/sections/flag_rows.rs +++ b/mandible-extract/src/help_text/sections/flag_rows.rs @@ -937,6 +937,101 @@ fn collect_flags_block_rows<'a>( (i, rows, argfile_entry) } +// S-168: a colon-introduced choice list under a placeholder row (Xvfb's +// `+extension`/`-extension` pair and its "run-time enabled/disabled" +// sentence) becomes that placeholder's own `choices`, shared by both +// halves of a `+word`/`-word` pair, never folded into a description. See +// docs/shapes.md S-168. + +/// True when a continuation line, once trimmed, is nothing but a +/// colon-terminated introducer sentence for a list nested directly below +/// it. Never itself flag-shaped, and never empty once the trailing colon +/// is stripped — a bare `:` alone introduces nothing. The caller +/// ([`mark_choice_list_rows`]) trusts this only once +/// [`MIN_NESTED_TABLE_ROWS`] further [`looks_like_choice_list_item`] rows +/// confirm a real list follows, so an ordinary description sentence that +/// happens to end in `:` with nothing list-shaped beneath it is never +/// mistaken for this. See docs/shapes.md S-168. +pub(super) fn looks_like_choice_list_introducer(text: &str) -> bool { + let trimmed = text.trim(); + let Some(body) = trimmed.strip_suffix(':') else { + return false; + }; + let body = body.trim(); + !body.is_empty() && !looks_like_flag_start(body) && !looks_like_bracket_flag_row(body) +} + +/// True when a continuation line is one list item in the shape +/// [`looks_like_choice_list_introducer`]'s own list is made of: a short +/// run (at most six words) of bare, hyphen-joined words, with none of the +/// sentence punctuation (`.`, `,`, `;`) an ordinary wrapped description +/// carries, and — critically — no genuine column gap +/// ([`find_multi_space_gap`]) anywhere in it. That gap is what tells this +/// bare-name list apart from `as`'s own described sub-option rows +/// (`c omit false conditionals`, S-015's territory, a name *and* a +/// description column, already handled by +/// [`choice_description_sub_row`]): a plain list item is nothing but its +/// own name. Matches `MIT-SHM`, `XVideo-MotionCompensation`, and the +/// three-word `Generic Event Extension` alike. See docs/shapes.md S-168. +pub(super) fn looks_like_choice_list_item(text: &str) -> bool { + let trimmed = text.trim(); + if trimmed.is_empty() || find_multi_space_gap(trimmed).is_some() { + return false; + } + let words: Vec<&str> = trimmed.split_whitespace().collect(); + words.len() <= 6 + && words + .iter() + .all(|w| !w.is_empty() && w.chars().all(|c| c.is_ascii_alphanumeric() || c == '-')) +} + +/// `rows` indices that belong to a colon-introduced choice list — the +/// introducer row itself, plus every item row beneath it — computed with +/// the whole block's rows in view so [`scan_flags_block`]'s own folding +/// loop needs no lookahead of its own. Requires at least +/// [`MIN_NESTED_TABLE_ROWS`] item rows after the introducer, the same +/// repetition floor every other nested-list recognizer in this file +/// uses: one line alone is cheap to produce by coincidence, a real run of +/// them is a list. See docs/shapes.md S-168. +fn mark_choice_list_rows(rows: &[FlagsBlockRow<'_>]) -> Vec { + let mut marks = vec![false; rows.len()]; + let mut i = 0; + while i < rows.len() { + if let FlagsBlockRow::Continuation(text) = rows[i] { + if looks_like_choice_list_introducer(text) { + let mut j = i + 1; + while let Some(FlagsBlockRow::Continuation(item)) = rows.get(j) { + if !looks_like_choice_list_item(item) { + break; + } + j += 1; + } + if j - (i + 1) >= MIN_NESTED_TABLE_ROWS { + for mark in &mut marks[i..j] { + *mark = true; + } + i = j; + continue; + } + } + } + i += 1; + } + marks +} + +/// The bare word right after a row's own leading `+`/`-` sigil — used only +/// to confirm a `+word`/`-word` pair share the same base word before a +/// colon-introduced choice list (S-168) is propagated from one half to +/// the other. `None` when the spec's first token carries neither sigil. +fn spec_word_after_sigil(spec: &str) -> Option<&str> { + let token = spec.split_whitespace().next()?; + let word = token + .strip_prefix('+') + .or_else(|| token.strip_prefix('-'))?; + (!word.is_empty()).then_some(word) +} + pub(super) fn scan_flags_block( lines: &[&str], start: usize, @@ -984,7 +1079,12 @@ pub(super) fn scan_flags_block( // `emit_flags_block` knows which single recovered entry to expand // into the row's own `+word`/`-word` pair. See docs/shapes.md S-163. let mut is_alternation: Vec = Vec::new(); - for row in rows { + // S-168: which `rows` indices belong to a colon-introduced choice + // list (the introducer line itself, plus every item beneath it) — + // computed once, with the whole block's rows in view, so the folding + // loop below can route each without its own lookahead. + let choice_list_rows = mark_choice_list_rows(&rows); + for (row_idx, row) in rows.into_iter().enumerate() { let plus_sigil_row = matches!(row, FlagsBlockRow::PlusSigil(_)); let alt_sigil_row = matches!(row, FlagsBlockRow::AlternationSigil(_)); let before = entries.len(); @@ -1077,7 +1177,36 @@ pub(super) fn scan_flags_block( } } FlagsBlockRow::Continuation(text) => { - if let Some(last) = entries.last_mut() { + if choice_list_rows[row_idx] { + // S-168: a colon-introduced choice list — the + // introducer line itself never becomes a choice or a + // description fragment; each item beneath it becomes a + // bare choice on this entry, and, when this entry is + // one half of a `+word`/`-word` pair (S-163), on its + // sibling too. See docs/shapes.md S-168. + if !looks_like_choice_list_introducer(text) { + let name = text.trim().to_string(); + if let Some(last) = entries.last_mut() { + if !last.2.iter().any(|(n, _)| n == &name) { + last.2.push((name.clone(), None)); + } + } + // The sibling half of a `+word`/`-word` pair sits + // exactly one entry back, tagged in `is_plus_sigil` + // (built incrementally, same loop) — never further + // back, and never when this entry isn't paired at + // all. + if entries.len() >= 2 { + let sibling_idx = entries.len() - 2; + let paired = is_plus_sigil.get(sibling_idx).copied().unwrap_or(false) + && spec_word_after_sigil(&entries[sibling_idx].0) + == spec_word_after_sigil(&entries[entries.len() - 1].0); + if paired && !entries[sibling_idx].2.iter().any(|(n, _)| n == &name) { + entries[sibling_idx].2.push((name, None)); + } + } + } + } else if let Some(last) = entries.last_mut() { // A continuation that completes an unclosed `<...>` // placeholder opened on the entry row above (jmod's // `--target-platform `) diff --git a/mandible-extract/src/help_text/sections/mod.rs b/mandible-extract/src/help_text/sections/mod.rs index 404b7a53..bcd11dfb 100644 --- a/mandible-extract/src/help_text/sections/mod.rs +++ b/mandible-extract/src/help_text/sections/mod.rs @@ -1836,6 +1836,15 @@ fn parse_body( // §7's row grammar) — `-help`'s row only qualifies once the repair // above has turned it into a single-dash spelling. See S-007. result.flags = recover_anchored_values(std::mem::take(&mut result.flags), raw); + // A `+word` row's own value column (S-163) is borrowed onto its + // `-word` sibling when the ordinary repair above could not recover a + // bare, unbracketed value (Xvfb's own `+extension name` / + // `-extension name`). See docs/shapes.md S-163. + result.flags = borrow_plus_word_value_for_dash_sibling(std::mem::take(&mut result.flags)); + // Last of all: an alternation-sigil row's own expansion (S-163) never + // duplicates or overwrites a spelling an ordinary row already + // documents (Xvfb's own `-render`). See docs/shapes.md S-163. + result.flags = resolve_alternation_spelling_collisions(std::mem::take(&mut result.flags), raw); result.confidence = compute_confidence(total_entries, clean_entries, !result.usage.is_empty()); result diff --git a/mandible-extract/src/help_text/sections/repair.rs b/mandible-extract/src/help_text/sections/repair.rs index 39e63fc3..b124ab7e 100644 --- a/mandible-extract/src/help_text/sections/repair.rs +++ b/mandible-extract/src/help_text/sections/repair.rs @@ -419,6 +419,17 @@ fn row_is_table_shaped(lines: &[&str], idx: usize) -> bool { if after.contains('\t') || after.contains(" ") { return true; } + // A genuine placeholder (`<...>`/`[...]`) exactly one space after the + // name is unambiguous evidence of a real value column even with no + // visual gap at all: Xvfb's own headingless table (S-165) never pads + // its columns, so `-render [default|mono|gray|color]` never earns the + // tab/double-space evidence above on its own row. See docs/shapes.md + // S-145. + if let Some(stripped) = after.strip_prefix(' ') { + if !stripped.starts_with(' ') && (stripped.starts_with('<') || stripped.starts_with('[')) { + return true; + } + } let indent = leading_whitespace(line); lines .get(idx + 1) @@ -585,6 +596,97 @@ pub(super) fn recover_anchored_values(mut flags: Vec, raw: &str) -> Vec< flags } +/// A `+word` row (`parse_plus_sigil_spec`) reads a bare, unbracketed value +/// correctly; its `-word` sibling goes through +/// [`repair_single_dash_long_options`] instead, which only recovers a +/// bracket/angle or `=`-glued value and drops a bare one. Borrows the +/// already-correct value from the `+word` side rather than teaching the +/// ordinary repair to guess at bare words. Never overwrites a `-word` row +/// that already carries its own value. See docs/shapes.md S-163. +pub(super) fn borrow_plus_word_value_for_dash_sibling(mut flags: Vec) -> Vec { + let borrowed: Vec<(String, String, ValueKind)> = flags + .iter() + .filter_map(|f| { + if f.spellings.len() != 1 || f.spellings[0].dashes != Dashes::None { + return None; + } + let name = &f.spellings[0].name; + let base = name.strip_prefix('+')?; + if base.is_empty() { + return None; + } + let value = f.value_name.as_ref()?; + Some((base.to_string(), value.clone(), f.value_kind)) + }) + .collect(); + for (base, value, kind) in borrowed { + if let Some(sibling) = flags.iter_mut().find(|f| { + f.spellings.len() == 1 + && f.spellings[0].dashes == Dashes::Single + && f.spellings[0].name == base + && f.value_name.is_none() + }) { + sibling.value_name = Some(value); + sibling.value_kind = kind; + } + } + flags +} + +/// When a `+/-name` alternation row (S-163) expands to a `-word` spelling +/// an *ordinary* row elsewhere already documents (Xvfb's own `-render`), +/// the expansion contributes only the spelling the existing row lacks, +/// never overwriting its value, choices or description. Run last, after +/// every other repair. Scoped to exactly the words the document's own +/// alternation rows name, never a general duplicate-spelling merge: `du`'s +/// `--time`/`--time=WORD` legitimately share one spelling for two forms +/// elsewhere in the fleet. See docs/shapes.md S-163. +pub(super) fn resolve_alternation_spelling_collisions( + mut flags: Vec, + raw: &str, +) -> Vec { + let alt_words: std::collections::HashSet = raw + .lines() + .filter_map(|line| { + let token = line.split_whitespace().next()?; + plus_minus_alternation_word(token).map(str::to_string) + }) + .collect(); + if alt_words.is_empty() { + return flags; + } + for word in alt_words { + let dupes: Vec = flags + .iter() + .enumerate() + .filter(|(_, f)| { + f.spellings.len() == 1 + && f.spellings[0].dashes == Dashes::Single + && f.spellings[0].name == word + }) + .map(|(i, _)| i) + .collect(); + if dupes.len() < 2 { + continue; + } + // Prefer the row that already carries a real value or choices — + // the pre-existing ordinary row's own value spec — over the + // alternation expansion's shared, valueless half. A tie (neither + // carries one) keeps the earliest, document-order entry. + let keep = dupes + .iter() + .copied() + .find(|&i| flags[i].value_name.is_some() || !flags[i].choices.is_empty()) + .unwrap_or(dupes[0]); + let mut drop: Vec = dupes.into_iter().filter(|&i| i != keep).collect(); + drop.sort_unstable_by(|a, b| b.cmp(a)); + for i in drop { + flags.remove(i); + } + } + flags +} + /// One [`recover_anchored_values`] run: same description, same table, each /// row one spelling. Finds the value this run's own well-parsed rows /// already agree the shared description takes (if any), then restores it diff --git a/mandible-tui/src/render/detail_pane/entity.rs b/mandible-tui/src/render/detail_pane/entity.rs index b71f7149..f8dcc570 100644 --- a/mandible-tui/src/render/detail_pane/entity.rs +++ b/mandible-tui/src/render/detail_pane/entity.rs @@ -94,10 +94,14 @@ pub(super) fn entity_value_text(flag: &Entity) -> Option { /// True when this entity's value placeholder glues directly onto its /// spelling with no space — the argfile sigil flag's row-verbatim shape, /// `@` (spec §4.5), rather than the ordinary `--output FILE` gap -/// (spec §9.3). Decided by shape (a single dashless spelling whose first -/// character is not alphanumeric), not by the literal `"@"`: a dashed -/// short option like `-?` must not match, since it does take a value -/// (`ffplay`'s `-? topic`) with the ordinary space. +/// (spec §9.3). Decided by shape (a single dashless spelling that is +/// nothing but its own leading sigil character, `@`, `+`), not by the +/// literal `"@"`: a dashed short option like `-?` must not match, since +/// it does take a value (`ffplay`'s `-? topic`) with the ordinary space — +/// and neither must a `+word` spelling that already carries its own word +/// (`+extension`, `+accessx`, S-163), whose value is a separate word in +/// the source and renders with the ordinary space the same way +/// `-extension`'s does, never glued into `+extensionname`. pub(super) fn spelling_is_sigil(flag: &Entity) -> bool { flag.spellings.len() == 1 && matches!(flag.spellings[0].dashes, Dashes::None) @@ -106,6 +110,7 @@ pub(super) fn spelling_is_sigil(flag: &Entity) -> bool { .chars() .next() .is_some_and(|c| !c.is_alphanumeric()) + && flag.spellings[0].name.chars().count() == 1 } /// True when a required value glues to its spelling by a literal diff --git a/mandible-tui/src/render/detail_pane/layout.rs b/mandible-tui/src/render/detail_pane/layout.rs index 1f427d10..893d5b77 100644 --- a/mandible-tui/src/render/detail_pane/layout.rs +++ b/mandible-tui/src/render/detail_pane/layout.rs @@ -60,16 +60,36 @@ pub(super) fn spelling_column(entity: &Entity, indent: usize) -> usize { indent + bare_spelling_column(entity) } +/// The word length a dashless spelling's own name carries once a single +/// leading sigil character (`+` in `"+render"`, `"+i"`) is stripped off — +/// zero for the bare sigil alone (`"+"`), and the name's own full length +/// for a spelling that never had one (a modifier letter, an environment +/// variable name). See [`bare_spelling_column`] and docs/shapes.md S-163. +fn dashless_word_len(name: &str) -> usize { + let mut chars = name.chars(); + match chars.next() { + Some(c) if !c.is_alphanumeric() => chars.count(), + _ => name.chars().count(), + } +} + /// [`spelling_column`] before the section's own indent is added. pub(super) fn bare_spelling_column(entity: &Entity) -> usize { if entity.spellings.len() > 2 { return SHORT_COLUMN; } - if entity - .spellings - .iter() - .any(|s| matches!(s.dashes, Dashes::None)) - { + // A dashless spelling reads as a short flag's own column when its word + // is at most one character (`+i`, the bare `+`), or when it isn't a + // `Flag` at all — a modifier letter or an environment variable name + // stays at the content edge whatever its length (spec §9.3, "MODIFIERS + // and ENVIRONMENT... stay laid out like FLAGS, against the content + // edge"). A longer dashless flag spelling (`+render`, `+extension`, + // S-163) is a single-dash-long-style spelling instead, and belongs + // beside `-render` in the long column. See docs/shapes.md S-163. + if entity.spellings.iter().any(|s| { + matches!(s.dashes, Dashes::None) + && (entity.kind != EntityKind::Flag || dashless_word_len(&s.name) <= 1) + }) { return SHORT_COLUMN; } if entity.short_spelling().is_some() { diff --git a/xtask/src/detector/choice_list_under_placeholder.rs b/xtask/src/detector/choice_list_under_placeholder.rs new file mode 100644 index 00000000..2e884300 --- /dev/null +++ b/xtask/src/detector/choice_list_under_placeholder.rs @@ -0,0 +1,246 @@ +//! `choice-list-under-placeholder` (atlas S-168): a colon-terminated +//! introducer sentence directly followed by a run of bare-name list +//! items, sitting under a placeholder row (or a `+word`/`-word` pair +//! sharing one), never folded into any row's own description. +//! +//! Fixture: `corpus/Xvfb/audit-seed`. `Xvfb --help`'s own +//! `+extension`/`-extension` pair documents the placeholder `name` this +//! way: a "the following extensions can be run-time enabled/disabled:" +//! sentence, then one extension name per line. + +use crate::detector::{Detector, Expect, Scope, SelfCheck, ToolEvidence}; +use mandible_core::CommandNode; + +/// The fewest item lines a colon introducer must be followed by before +/// this is trusted as a real list rather than an ordinary sentence that +/// happens to end in `:`. Mirrors +/// `mandible_extract`'s own `MIN_NESTED_TABLE_ROWS` floor — an +/// independent copy, since extract and xtask do not share code (S-163's +/// own precedent). +const MIN_LIST_ROWS: usize = 2; + +pub struct Finding { + pub introducer: String, + pub items: Vec, +} + +pub struct Report { + pub findings: Vec, +} + +/// True when `line`, once trimmed, is nothing but a colon-terminated +/// sentence — never itself a flag row. +fn looks_like_choice_list_introducer(line: &str) -> bool { + let trimmed = line.trim(); + let Some(body) = trimmed.strip_suffix(':') else { + return false; + }; + let body = body.trim(); + !body.is_empty() && !body.starts_with('-') && !body.starts_with('+') +} + +/// True when a genuine column gap (a tab, or two or more spaces past the +/// first non-blank character) exists in `line` — the same evidence +/// `choice_description_sub_row` uses to recognize a *described* row +/// (S-015), which this detector must never mistake for a bare list item. +fn has_real_column_gap(line: &str) -> bool { + let bytes = line.as_bytes(); + let mut seen_content = false; + let mut i = 0; + while i < bytes.len() { + let b = bytes[i]; + if b == b' ' || b == b'\t' { + let start = i; + let mut had_tab = false; + while i < bytes.len() && (bytes[i] == b' ' || bytes[i] == b'\t') { + had_tab |= bytes[i] == b'\t'; + i += 1; + } + if seen_content && (had_tab || i - start >= 2) { + return true; + } + } else { + seen_content = true; + i += 1; + } + } + false +} + +/// True when `line` is one bare-name list item: a short run of hyphenated +/// words, with no genuine column gap anywhere (that gap is what a +/// *described* row, S-015's territory, carries instead). +fn looks_like_choice_list_item(line: &str) -> bool { + let trimmed = line.trim(); + if trimmed.is_empty() || has_real_column_gap(trimmed) { + return false; + } + let words: Vec<&str> = trimmed.split_whitespace().collect(); + words.len() <= 6 + && words + .iter() + .all(|w| !w.is_empty() && w.chars().all(|c| c.is_ascii_alphanumeric() || c == '-')) +} + +/// True when some flag entity in `root` already carries every one of +/// `items` as its own `choices` — the fixed shape, regardless of which +/// entity (`+word` or `-word`) is checked first. +fn some_flag_carries_every_choice(root: &CommandNode, items: &[String]) -> bool { + root.flags().any(|f| { + items + .iter() + .all(|item| f.choices.iter().any(|c| &c.name == item)) + }) +} + +pub fn detect(raw: &str, root: &CommandNode) -> Report { + let mut findings = Vec::new(); + let lines: Vec<&str> = raw.lines().collect(); + let mut i = 0; + while i < lines.len() { + if looks_like_choice_list_introducer(lines[i]) { + let mut j = i + 1; + while j < lines.len() && looks_like_choice_list_item(lines[j]) { + j += 1; + } + if j - (i + 1) >= MIN_LIST_ROWS { + let items: Vec = lines[i + 1..j] + .iter() + .map(|l| l.trim().to_string()) + .collect(); + if !some_flag_carries_every_choice(root, &items) { + findings.push(Finding { + introducer: lines[i].trim().to_string(), + items, + }); + } + i = j; + continue; + } + } + i += 1; + } + Report { findings } +} + +pub struct ChoiceListUnderPlaceholder; + +impl Detector for ChoiceListUnderPlaceholder { + fn name(&self) -> &'static str { + "choice-list-under-placeholder" + } + + fn family(&self) -> Option<&'static str> { + None + } + + fn describes(&self) -> &'static str { + "a colon-introduced list of bare-name choices that never reached any flag's own \ + `choices` (S-168)" + } + + fn hits(&self, evidence: &ToolEvidence<'_>) -> Vec { + detect(evidence.raw, evidence.root) + .findings + .iter() + .map(|f| { + format!( + "{:?} introduces {} items never attached as choices", + f.introducer, + f.items.len() + ) + }) + .collect() + } + + fn scope(&self) -> Scope { + Scope::full() + } + + fn self_checks(&self) -> Vec { + self_checks() + } +} + +// ---------------------------------------------------------------------- +// Self-checks +// ---------------------------------------------------------------------- + +use mandible_core::{Choice, Entity, Provenance, Spelling}; + +/// Xvfb's own shape, byte-exact (`corpus/Xvfb/audit-seed/help.stderr.txt`). +const XVFB_EXTENSION_LIST: &str = "+extension name Enable extension\n\ + -extension name Disable extension\n\ + Only the following extensions can be run-time enabled/disabled:\n\ + \tGeneric Event Extension\n\ + \tMIT-SHM\n\ + \tXTEST\n"; + +fn flag_with_choices(spelling: Spelling, choices: Vec<&str>) -> Entity { + let mut e = Entity::new(mandible_core::EntityKind::Flag, Provenance::default()); + e.spellings = vec![spelling]; + e.choices = choices + .into_iter() + .map(|c| Choice { + name: c.to_string(), + description: None, + }) + .collect(); + e +} + +fn node_with(flags: Vec) -> CommandNode { + let mut root = CommandNode::new("Xvfb", Provenance::default()); + root.set_entities_of(mandible_core::EntityKind::Flag, flags); + root +} + +pub(crate) fn self_checks() -> Vec { + vec![ + SelfCheck { + name: "Xvfb's own extension list, never attached to either half of the pair", + why: "the defect itself: the colon-introduced list reaches no flag's own choices", + expect: Expect::Fires(1), + raw: XVFB_EXTENSION_LIST.to_string(), + root: node_with(vec![ + flag_with_choices(Spelling::bare("+extension"), vec![]), + flag_with_choices(Spelling::single_dash("extension"), vec![]), + ]), + }, + SelfCheck { + name: "Xvfb's own list, attached to the `-word` half", + why: "once either half of the pair carries every item as its own choices, the \ + list is accounted for", + expect: Expect::Silent, + raw: XVFB_EXTENSION_LIST.to_string(), + root: node_with(vec![ + flag_with_choices(Spelling::bare("+extension"), vec![]), + flag_with_choices( + Spelling::single_dash("extension"), + vec!["Generic Event Extension", "MIT-SHM", "XTEST"], + ), + ]), + }, + SelfCheck { + name: "an ordinary sentence ending in a colon with nothing list-shaped beneath it", + why: "one line, or none, is cheap to produce by coincidence — never enough evidence \ + of a real list on its own", + expect: Expect::Silent, + raw: "Notes:\nThis tool reads its configuration from the environment.\n".to_string(), + root: node_with(vec![]), + }, + SelfCheck { + name: "as's own described sub-option rows, each with a real column gap", + why: "a described row (S-015's own territory) is never mistaken for a bare list \ + item: the column gap between name and description is exactly what a plain \ + list item never carries", + expect: Expect::Silent, + raw: "Sub-options [default hls]:\n\ + \tc omit false conditionals\n\ + \td omit debugging directives\n\ + \tg include general info\n" + .to_string(), + root: node_with(vec![]), + }, + ] +} diff --git a/xtask/src/detector/mod.rs b/xtask/src/detector/mod.rs index e205d4fc..e7ab56ba 100644 --- a/xtask/src/detector/mod.rs +++ b/xtask/src/detector/mod.rs @@ -110,6 +110,10 @@ pub(crate) mod option_table_multiword_value_name; // Round-10 family detector (atlas S-166, W6's header-declared // three-column option table), same direct-`Detector`-impl shape. pub(crate) mod header_declared_env_column; +// Round-11 family detector (atlas S-168, Xvfb's colon-introduced choice +// list under a `+word`/`-word` placeholder pair), same direct-`Detector`- +// impl shape. +pub(crate) mod choice_list_under_placeholder; pub(crate) use calibration::*; pub(crate) use commands::*; @@ -763,6 +767,7 @@ pub fn registry() -> Vec> { Box::new(description_reused_as_group_label::DescriptionReusedAsGroupLabel), Box::new(headingless_table_in_root_description::HeadinglessTableInRootDescription), Box::new(header_declared_env_column::HeaderDeclaredEnvColumn), + Box::new(choice_list_under_placeholder::ChoiceListUnderPlaceholder), ] } From 23324afebbbd13810db82d9d2d8d0be82f5200ff Mon Sep 17 00:00:00 2001 From: Sadig Akhund Date: Sun, 13 Sep 2026 10:11:07 +0400 Subject: [PATCH 7/9] atlas: correct S-168's fleet claim and give S-145's widening its own count S-168 ships as a gated exception (1 tool at tree level, 26 raw-shape upper bound), not because it cleared the bar. S-145's placeholder-gap widening moves 6 tools on its own sweep evidence, named with the one partial recovery (llvm-lipo-18) and the -render duplicate's own fencing gap recorded. Co-Authored-By: Claude Fable 5.1 --- docs/shapes.md | 48 +++++++++++++++++++++++++++++++++++++----------- 1 file changed, 37 insertions(+), 11 deletions(-) diff --git a/docs/shapes.md b/docs/shapes.md index e9a99491..cf45693c 100644 --- a/docs/shapes.md +++ b/docs/shapes.md @@ -2682,7 +2682,8 @@ entry's `tools` field and nothing else. It does not get a new entry. -pf add list of pseudo file definitions from -Xhelp print compressor options for selected compressor -mem use physical memory for caches -- tools: mksquashfs, sqfstar, Xvfb +- tools: mksquashfs, sqfstar, Xvfb, jdb, jrunscript, llvm-libtool-darwin-18, + llvm-lipo-18, screen - handling: A table whose rows are column-0 `-word` spellings, tab- or column-gap separated from their descriptions, with at least two rows carrying an unambiguous, uniformly-lowercase multi-character name and no `--long` @@ -2704,9 +2705,18 @@ entry's `tools` field and nothing else. It does not get a new entry. on a table with no column padding at all — Xvfb's own headingless shape (S-165) — so a row's genuine `<...>`/`[...]` placeholder exactly one space after the name is now admitted as the same evidence - (`-render [default|mono|gray|color]`, `-deferglyphs [none|all|16]`), - recovering both their bracket values without widening the repair to - bare, unbracketed words (still out of scope, S-117's own reasoning). + (`-render [default|mono|gray|color]`, `-deferglyphs [none|all|16]`, + `-multicast [addr [hops]]`), recovering their bracket values without + widening the repair to bare, unbracketed words (still out of scope, + S-117's own reasoning). The same widening moved five further tools from + no value name to the tool's own literal text on a full-`PATH` sweep, + every one checked against its own `--help`: `jdb -dbgtrace [flags]`, + `jrunscript -encoding `, `llvm-libtool-darwin-18 -arch_only + `, `llvm-lipo-18 -arch `, `screen -wipe [match]`. All + five are gains, none a fabrication. One qualifier: `llvm-lipo-18`'s own + raw line is `-arch `, two values, and only the first is + recovered — a partial recovery, not a wrong one; the second value is + information the IR does not model. - fleet: `single-dash-long-table` (`xtask/src/detector/single_dash_long_table.rs`) reads 14 tools/21 raw findings on a full-`PATH` sweep of 2323 tools, 2026-09-07, after the fix: unsquashfs, xkill, xev, setfont and others in the same @@ -2719,7 +2729,11 @@ entry's `tools` field and nothing else. It does not get a new entry. jrunscript, perlbug, perlthanks, ckbcomp, containerd-shim-runc-v2, javax2jakarta, llvm-libtool-darwin-18, llvm-lipo-18 and winpr-makecert, 0 losses on a full-`PATH` sweep-diff of 2269 tools, 2026-09-07. All four - self-checks hold. No labelled member of this family exists in any audit + self-checks hold. Round 11's own placeholder-gap widening (above) clears + the five-tool bar on its own evidence: 6 tools gained a value name — + Xvfb, jdb, jrunscript, llvm-libtool-darwin-18, llvm-lipo-18 and screen — + 0 losses on a full-`PATH` sweep of 2323 tools, 2026-09-13. No labelled + member of this family exists in any audit seed; the self-checks are the only standing evidence. ### S-146: a flush heading or bare sub-label names no group @@ -3053,12 +3067,24 @@ entry's `tools` field and nothing else. It does not get a new entry. tab or double-space gap that a single-dash-long table with no column padding at all (S-165's own headingless shape) never has; a genuine `<...>`/`[...]` placeholder exactly one space after the name is now - admitted as the same evidence. + admitted as the same evidence. Fencing gap found on a break-it check: + disabling `resolve_alternation_spelling_collisions` still passes + `must_describe["-render"]` and `must_value_name["-render"]`, since a + duplicate entity does not stop the first matching one carrying the + right value — only `expected.snap`'s byte compare catches the + duplicate today. A `must_not_duplicate_spelling` contract field would + fence it directly; not added this round. - fleet: `choice-list-under-placeholder` (`xtask/src/detector/choice_list_under_placeholder.rs`) is family `None`, - so calibration reads NOT EVALUABLE; self-checks hold (4/4). Raw-shape + so calibration reads NOT EVALUABLE; self-checks hold (4/4). The + tree-level reach is **1 tool** — a full-`PATH` sweep of 2323 tools names + `choices changed` on Xvfb and nothing else, 2026-09-13 — well short of + the five-tool bar; this ships on the gated-exception route (§ common.md: + maintainer-audited, fixture promoted out of `[xfail]`, zero-loss sweep, + controls byte-identical), not because the rule cleared it. Raw-shape grep over `/home/ubuntu/projects/mandible/audit/queue-captures/`: 26 - tools / 26 findings, an upper bound on the raw shape alone, not the - tree-level count — most (`bash`, `perf`, `usbip`, `dmesg`, `bpftool`, and - others) are not audited here and their own parse is unexamined; `Xvfb` - is the one fixture fixed and corpus-pinned this round. 2026-09-13. + tools / 26 findings, an upper bound on the raw shape alone. The other 25 + (`bash`, `perf`, `usbip`, `dmesg`, `bpftool`, and others) are unaudited + and this rule does not reach them — nothing was fabricated on a tool + nobody read. `Xvfb` is the one fixture fixed and corpus-pinned this + round. 2026-09-13. From fbb62b2f338c2588f55216d4ab3468e7f5126fc2 Mon Sep 17 00:00:00 2001 From: Sadig Akhund Date: Sun, 13 Sep 2026 15:09:39 +0400 Subject: [PATCH 8/9] extract: bundle a flag block's row routing, and split the usage label tests Merging S-163's alternation rows beside S-155's argparse gate took emit_flags_block to eight arguments; RowRouting carries the three row facts. usage.rs reached 804 code lines, so its label and program-name predicates move to usage_label.rs. Xvfb gains 24 value names, fc-scan drops a group label that repeated its own description. Co-Authored-By: Claude Fable 5.1 --- corpus/Xvfb/audit-seed/expected.snap | 48 ++++ corpus/fc-scan/2.15.0/expected.snap | 5 - .../src/help_text/sections/emit.rs | 28 ++- .../src/help_text/sections/mod.rs | 25 +- .../src/help_text/sections/usage.rs | 221 +---------------- .../src/help_text/sections/usage_label.rs | 227 ++++++++++++++++++ 6 files changed, 315 insertions(+), 239 deletions(-) create mode 100644 mandible-extract/src/help_text/sections/usage_label.rs diff --git a/corpus/Xvfb/audit-seed/expected.snap b/corpus/Xvfb/audit-seed/expected.snap index ee172825..d9ecb405 100644 --- a/corpus/Xvfb/audit-seed/expected.snap +++ b/corpus/Xvfb/audit-seed/expected.snap @@ -17,12 +17,16 @@ flags: - help-text - spellings: - -audit + value_name: int + value_kind: Required description: set audit trail level provenance: sources: - help-text - spellings: - -auth + value_name: file + value_kind: Required description: select authorization file provenance: sources: @@ -83,12 +87,16 @@ flags: - help-text - spellings: - -displayfd + value_name: fd + value_kind: Required description: file descriptor to write display number to when ready to connect provenance: sources: - help-text - spellings: - -dpi + value_name: int + value_kind: Required description: screen resolution in dots per inch provenance: sources: @@ -123,6 +131,8 @@ flags: - help-text - spellings: - -fp + value_name: string + value_kind: Required description: default font path provenance: sources: @@ -153,18 +163,24 @@ flags: - help-text - spellings: - -ld + value_name: int + value_kind: Required description: limit data space to N Kb provenance: sources: - help-text - spellings: - -lf + value_name: int + value_kind: Required description: limit number of open files to N provenance: sources: - help-text - spellings: - -ls + value_name: int + value_kind: Required description: limit stack space to N Kb provenance: sources: @@ -177,18 +193,24 @@ flags: - help-text - spellings: - -maxclients + value_name: n + value_kind: Required description: set maximum number of clients (power of two) provenance: sources: - help-text - spellings: - -nolisten + value_name: string + value_kind: Required description: don't listen on protocol provenance: sources: - help-text - spellings: - -listen + value_name: string + value_kind: Required description: listen on protocol provenance: sources: @@ -263,6 +285,8 @@ flags: - help-text - spellings: - -seat + value_name: string + value_kind: Required description: seat to run on provenance: sources: @@ -327,6 +351,8 @@ flags: - help-text - spellings: - -schedInterval + value_name: int + value_kind: Required description: Set scheduler interval in msec provenance: sources: @@ -395,6 +421,8 @@ flags: - help-text - spellings: - -query + value_name: host-name + value_kind: Required description: contact named host for XDMCP provenance: sources: @@ -415,18 +443,24 @@ flags: - help-text - spellings: - -indirect + value_name: host-name + value_kind: Required description: contact named host for indirect XDMCP provenance: sources: - help-text - spellings: - -port + value_name: port-num + value_kind: Required description: UDP port number to send messages to provenance: sources: - help-text - spellings: - -from + value_name: local-address + value_kind: Required description: specify the local address to connect from provenance: sources: @@ -439,18 +473,24 @@ flags: - help-text - spellings: - -class + value_name: display-class + value_kind: Required description: specify display class to send in manage provenance: sources: - help-text - spellings: - -cookie + value_name: xdm-auth-bits + value_kind: Required description: specify the magic cookie for XDMCP provenance: sources: - help-text - spellings: - -displayID + value_name: display-id + value_kind: Required description: manufacturer display ID for request provenance: sources: @@ -502,24 +542,32 @@ flags: - help-text - spellings: - -linebias + value_name: n + value_kind: Required description: adjust thin line pixelization provenance: sources: - help-text - spellings: - -blackpixel + value_name: n + value_kind: Required description: pixel value for black provenance: sources: - help-text - spellings: - -whitepixel + value_name: n + value_kind: Required description: pixel value for white provenance: sources: - help-text - spellings: - -fbdir + value_name: directory + value_kind: Required description: put framebuffers in mmap'ed files in directory provenance: sources: diff --git a/corpus/fc-scan/2.15.0/expected.snap b/corpus/fc-scan/2.15.0/expected.snap index 5966ffa5..ce1ae914 100644 --- a/corpus/fc-scan/2.15.0/expected.snap +++ b/corpus/fc-scan/2.15.0/expected.snap @@ -13,7 +13,6 @@ flags: - spellings: - -b - --brief - group: Scan font files and directories, and print resulting pattern(s) description: display font pattern briefly provenance: sources: @@ -23,7 +22,6 @@ flags: - --format value_name: FORMAT value_kind: Required - group: Scan font files and directories, and print resulting pattern(s) description: use the given output format provenance: sources: @@ -33,7 +31,6 @@ flags: - --sysroot value_name: SYSROOT value_kind: Required - group: Scan font files and directories, and print resulting pattern(s) description: prepend SYSROOT to all paths for scanning provenance: sources: @@ -41,7 +38,6 @@ flags: - spellings: - -V - --version - group: Scan font files and directories, and print resulting pattern(s) description: display font config version and exit provenance: sources: @@ -49,7 +45,6 @@ flags: - spellings: - -h - --help - group: Scan font files and directories, and print resulting pattern(s) description: display this help and exit provenance: sources: diff --git a/mandible-extract/src/help_text/sections/emit.rs b/mandible-extract/src/help_text/sections/emit.rs index ca47ad89..6fa9b28f 100644 --- a/mandible-extract/src/help_text/sections/emit.rs +++ b/mandible-extract/src/help_text/sections/emit.rs @@ -42,16 +42,29 @@ fn partition_plus_sigil_entries( /// through, so neither call site repeats the packed/plus-sigil/argfile /// three-way split inline. Returns `(seen, clean)` for the caller's own /// running totals. +/// The per-row routing facts [`emit_flags_block`] needs beside the rows: +/// which entries take [`parse_plus_sigil_spec`]'s grammar (S-095), which +/// take the alternation-sigil expansion (S-163), and whether the document +/// is argparse's, which gates S-155's brace form. +pub(super) struct RowRouting<'a> { + pub is_plus_sigil: &'a [bool], + pub is_alternation: &'a [bool], + pub is_argparse: bool, +} + pub(super) fn emit_flags_block( group: Option, entries: Vec, packed: bool, - is_plus_sigil: &[bool], - is_alternation: &[bool], + routing: RowRouting<'_>, argfile_entry: Option, - is_argparse: bool, out: &mut ParsedHelp, ) -> (usize, usize) { + let RowRouting { + is_plus_sigil, + is_alternation, + is_argparse, + } = routing; let (mut seen, mut clean) = if packed { let (ordinary, plus_sigil, alternation) = partition_plus_sigil_entries(entries, is_plus_sigil, is_alternation); @@ -278,7 +291,14 @@ pub(super) fn emit_flags_with( } } } - push_flag_entity(spec, &description, &choice_names, group.clone(), is_argparse, out); + push_flag_entity( + spec, + &description, + &choice_names, + group.clone(), + is_argparse, + out, + ); } (seen, clean) } diff --git a/mandible-extract/src/help_text/sections/mod.rs b/mandible-extract/src/help_text/sections/mod.rs index 296afa46..5ff2e0d6 100644 --- a/mandible-extract/src/help_text/sections/mod.rs +++ b/mandible-extract/src/help_text/sections/mod.rs @@ -46,6 +46,7 @@ mod spelling; #[cfg(test)] mod test_support; mod usage; +mod usage_label; use backfill::*; use bullets::*; @@ -1244,14 +1245,16 @@ fn emit_heading_block( group, entries, packed, - &is_plus_sigil, - &is_alternation, + RowRouting { + is_plus_sigil: &is_plus_sigil, + is_alternation: &is_alternation, + // `argparse_subparser_quirk` is set only for + // `Framework::Argparse` (see profile.rs), so it doubles + // here as "this tool is argparse" for S-155's + // brace-alternation gate, with no new profile field. + is_argparse: profile.is_some_and(|p| p.argparse_subparser_quirk), + }, argfile_entry, - // `argparse_subparser_quirk` is set only for `Framework::Argparse` - // (see profile.rs), so it doubles here as "this tool is - // argparse" for S-155's brace-alternation gate, with no new - // profile field needed. - profile.is_some_and(|p| p.argparse_subparser_quirk), st.result, ); st.total_entries += seen; @@ -1614,10 +1617,12 @@ fn scan_entries( pending_group, entries, packed, - &is_plus_sigil, - &is_alternation, + RowRouting { + is_plus_sigil: &is_plus_sigil, + is_alternation: &is_alternation, + is_argparse: profile.is_some_and(|p| p.argparse_subparser_quirk), + }, argfile_entry, - profile.is_some_and(|p| p.argparse_subparser_quirk), st.result, ); st.total_entries += seen; diff --git a/mandible-extract/src/help_text/sections/usage.rs b/mandible-extract/src/help_text/sections/usage.rs index 2bf304f6..3c8521dc 100644 --- a/mandible-extract/src/help_text/sections/usage.rs +++ b/mandible-extract/src/help_text/sections/usage.rs @@ -2,228 +2,9 @@ //! mining the synopsis itself for positionals and for flags no option //! table documents. +pub use super::usage_label::*; use super::*; -/// True if `t` starts with `"usage:"`, case-insensitively. -/// -/// Compares raw bytes via `[u8]::get` rather than slicing the `str`, which -/// can panic when a multi-byte character (e.g. a box-drawing glyph) lands -/// off a UTF-8 boundary at the slice point. -pub fn starts_with_usage_prefix(t: &str) -> bool { - t.as_bytes() - .get(..6) - .map(|b| b.eq_ignore_ascii_case(b"usage:")) - .unwrap_or(false) -} - -/// True if `t` starts with `"or:"`, case-insensitively — GNU coreutils' -/// marker for a genuine *alternative* invocation form, distinct from a -/// wrapped continuation of the form above it. Without it, joining every -/// more-indented usage line onto its predecessor would swallow `or:`'s -/// alternative form too. See S-037 and corpus/du/9.4/help.txt. -pub fn starts_with_or_marker(t: &str) -> bool { - t.as_bytes() - .get(..3) - .map(|b| b.eq_ignore_ascii_case(b"or:")) - .unwrap_or(false) -} - -/// True if `t`'s only content, once trimmed, is the word `or` — any case — -/// with an optional trailing colon: `sg_luns`' bare second-form separator -/// (`corpus/sg_luns/1.45`), one whole physical line with nothing else on -/// it. Distinct from [`starts_with_or_marker`], which matches an `or:` -/// *prefix* even when real form content follows the colon on the same -/// line (`ip`'s `or: ip link ...`); a line this predicate matches carries -/// no such content and must contribute none to either usage form. -pub fn is_bare_or_form_separator(t: &str) -> bool { - t.trim().trim_end_matches(':').eq_ignore_ascii_case("or") -} - -/// True if `t`, trimmed, is a short label ending in the word `usage` -/// (case-insensitive), a colon, and nothing else — `perlthanks`'s -/// `Advanced usage:`, distinct from the literal `usage:` -/// [`starts_with_usage_prefix`] alone matches. At most three words, every -/// one plain ASCII alphabetic, so a sentence that merely ends near the -/// word `usage` never qualifies. See S-151, `corpus/perlthanks`. -pub fn starts_with_extended_usage_label(t: &str) -> bool { - let Some(head) = t.trim().strip_suffix(':') else { - return false; - }; - let words: Vec<&str> = head.split_whitespace().collect(); - if words.is_empty() || words.len() > 3 { - return false; - } - if !words - .last() - .expect("checked non-empty above") - .eq_ignore_ascii_case("usage") - { - return false; - } - words - .iter() - .all(|w| !w.is_empty() && w.chars().all(|c| c.is_ascii_alphabetic())) -} - -/// True if `t`, trimmed, is only a usage label with nothing after it on -/// the same line: the literal `usage:`/`or:` markers with an empty -/// remainder, or [`starts_with_extended_usage_label`]'s generalized form -/// (which by construction carries no remainder either). `fdisk`'s bare -/// `Usage:` line is this shape; the two real forms sit on the lines below -/// it. See S-150. -pub fn is_bare_usage_label(t: &str) -> bool { - let trimmed = t.trim(); - if starts_with_usage_prefix(trimmed) { - return trimmed - .get(6..) - .map(|rest| rest.trim().is_empty()) - .unwrap_or(true); - } - if starts_with_or_marker(trimmed) { - return trimmed - .get(3..) - .map(|rest| rest.trim().is_empty()) - .unwrap_or(true); - } - starts_with_extended_usage_label(trimmed) -} - -/// True when `t`, trimmed, opens with an alphabetic label of 2 to 20 -/// characters, a colon, and immediately (no space) the tool's own `name` -/// at a word boundary — `mksquashfs`'s `SYNTAX:mksquashfs source1 ...`. -/// Distinct from the two already-recognized markers (`usage:`, `or:`), -/// which are matched regardless of what follows; a label glued straight -/// to unrelated text, or to the name with a space, does not qualify. -/// Mirrors `xtask`'s `usage_label_glued_to_program_name` detector, kept as -/// an independent copy per this crate's convention of never depending on -/// `xtask`. Returns the byte offset in `t` where the tool's own name -/// begins, so the caller can drop the label. See S-142, issue #143. -pub fn label_glued_to_tool_name(t: &str, name: &str) -> Option { - if name.is_empty() { - return None; - } - let colon_idx = t.find(':')?; - let label = &t[..colon_idx]; - if label.len() < 2 || label.len() > 20 || !label.chars().all(|c| c.is_ascii_alphabetic()) { - return None; - } - if label.eq_ignore_ascii_case("usage") || label.eq_ignore_ascii_case("or") { - return None; - } - let after_idx = colon_idx + 1; - let after = t.get(after_idx..)?; - if after.is_empty() || after.starts_with(char::is_whitespace) { - return None; - } - // A URL scheme (`https://github.com/ajeetdsouza/zoxide`) glues its own - // colon straight to a `//` authority, and its path's own basename - // routinely coincides with the tool's own name (a homepage on the - // tool's own domain) — the false positive S-142's own atlas entry - // names. Never a usage label. - if after.starts_with("//") { - return None; - } - let first_token = after.split_whitespace().next().unwrap_or(after); - let basename = first_token.rsplit('/').next().unwrap_or(first_token); - let rest = basename.strip_prefix(name)?; - let boundary_ok = rest.is_empty() - || !rest - .chars() - .next() - .is_some_and(|c| c.is_alphanumeric() || c == '_'); - boundary_ok.then_some(after_idx) -} - -/// True if `t` (already trimmed of leading whitespace) begins with `name` -/// at a word boundary. Lets a tool that repeats its own name across lines -/// with no `or:`/`usage:` marker read as two entries rather than one -/// continuation swallowing the other. Word-boundary checked so `git` -/// doesn't also claim `gitk` or `git-foo`. See S-037. -pub fn starts_with_tool_name(t: &str, name: &str) -> bool { - if name.is_empty() { - return false; - } - match t.strip_prefix(name) { - Some(rest) => rest.is_empty() || rest.starts_with(char::is_whitespace), - None => false, - } -} - -/// True when `t`'s own first whitespace-delimited token spells the tool -/// under different notation than its resolved `name`: as a full path -/// (`/usr/bin/ar` against `ar`), or as the dotted stem a resolved name -/// itself extends (`vim` against `vim.basic`). Twin of -/// [`starts_with_tool_name`], kept as a separate, narrower predicate so -/// every other caller of that one stays exactly as strict as before — used -/// only at `sections/mod.rs`'s `is_own_name` site. See S-108 and -/// `corpus/ar/audit-seed2/help.txt`'s second usage line. -pub fn starts_with_tool_name_spelled_differently(t: &str, name: &str) -> bool { - if name.is_empty() { - return false; - } - let Some(first) = t.split_whitespace().next() else { - return false; - }; - let basename = first.rsplit('/').next().unwrap_or(first); - basename == name || basename == name.split('.').next().unwrap_or(name) -} - -/// True if `t` (already trimmed) is the C `fprintf(stderr, "%s: Usage: -/// ...", argv[0])` idiom's line: the tool's own name, a literal `": "`, -/// then `usage:` case-insensitively. [`starts_with_usage_prefix`] alone -/// misses this since it only tests the line's start. Kept tight: the -/// `usage:` must be preceded by *only* the name and `": "`, never scanned -/// for elsewhere in the line. See S-001 and corpus/nfsidmap/audit-seed/help.txt. -pub fn starts_with_name_prefixed_usage(t: &str, name: &str) -> bool { - if name.is_empty() { - return false; - } - t.strip_prefix(name) - .and_then(|rest| rest.strip_prefix(": ")) - .is_some_and(starts_with_usage_prefix) -} - -/// Drop the `: ` prefix [`starts_with_name_prefixed_usage`] -/// recognizes, wherever it sits in front of a usage label — not only the -/// document's first line. `nfsidmap`'s C `fprintf(stderr, "%s: Usage: -/// ...", argv[0])` idiom keeps that prefix glued to its own usage line -/// rendered; the diagnostic prefix is not the label, and a reader wants the -/// label. Returns `t` trimmed and unchanged when the prefix isn't present. -/// See docs/shapes.md S-162 and S-001. -pub(super) fn strip_name_prefixed_usage_label(t: &str, tool_name: Option<&str>) -> String { - let trimmed = t.trim(); - if let Some(name) = tool_name { - if starts_with_name_prefixed_usage(trimmed, name) { - // `starts_with_name_prefixed_usage` already confirmed `trimmed` - // opens with exactly `"{name}: "`. - return trimmed[name.len() + 2..].to_string(); - } - } - trimmed.to_string() -} - -/// True if `t` opens with `name` at a word boundary and its remainder -/// reads as usage-synopsis grammar rather than prose — the unlabelled -/// synopsis convention (`wpa_cli --help` opens `wpa_cli [-p] -/// [-i] ...` with no `Usage:` marker at all). A name match alone -/// is not evidence (`"tar is an archiving program..."` starts with `tar` -/// too), so both must hold: the remainder contains a docopt group -/// delimiter (spec §7 Tier B: `[`, `<`, `{`), and it does not read as an -/// English sentence ([`is_prose_sentence`]). See S-001 and wpa_cli's help. -pub fn looks_like_unlabeled_synopsis_line(t: &str, name: &str) -> bool { - let Some(rest) = t.strip_prefix(name) else { - return false; - }; - if !(rest.is_empty() || rest.starts_with(char::is_whitespace)) { - return false; - } - let rest = rest.trim_start(); - if rest.is_empty() { - return false; - } - rest.contains(['[', '<', '{']) && !is_prose_sentence(rest) -} - /// True if `lines[idx]` is a bare own-name invocation line — no bracket /// notation on the line itself — whose very next physical line is /// unambiguous flag-row evidence: [`looks_like_bracket_flag_row`] or diff --git a/mandible-extract/src/help_text/sections/usage_label.rs b/mandible-extract/src/help_text/sections/usage_label.rs new file mode 100644 index 00000000..99257193 --- /dev/null +++ b/mandible-extract/src/help_text/sections/usage_label.rs @@ -0,0 +1,227 @@ +//! Recognizing a usage label and the program name beside it: the label +//! vocabulary, the tool's own name at the head of a synopsis, and the +//! `: ` diagnostic prefix in front of a label. Split out of +//! `usage.rs` to keep that file under the size ceiling +//! (`scripts/shape_guard.sh`). + +use super::*; + +/// True if `t` starts with `"usage:"`, case-insensitively. +/// +/// Compares raw bytes via `[u8]::get` rather than slicing the `str`, which +/// can panic when a multi-byte character (e.g. a box-drawing glyph) lands +/// off a UTF-8 boundary at the slice point. +pub fn starts_with_usage_prefix(t: &str) -> bool { + t.as_bytes() + .get(..6) + .map(|b| b.eq_ignore_ascii_case(b"usage:")) + .unwrap_or(false) +} + +/// True if `t` starts with `"or:"`, case-insensitively — GNU coreutils' +/// marker for a genuine *alternative* invocation form, distinct from a +/// wrapped continuation of the form above it. Without it, joining every +/// more-indented usage line onto its predecessor would swallow `or:`'s +/// alternative form too. See S-037 and corpus/du/9.4/help.txt. +pub fn starts_with_or_marker(t: &str) -> bool { + t.as_bytes() + .get(..3) + .map(|b| b.eq_ignore_ascii_case(b"or:")) + .unwrap_or(false) +} + +/// True if `t`'s only content, once trimmed, is the word `or` — any case — +/// with an optional trailing colon: `sg_luns`' bare second-form separator +/// (`corpus/sg_luns/1.45`), one whole physical line with nothing else on +/// it. Distinct from [`starts_with_or_marker`], which matches an `or:` +/// *prefix* even when real form content follows the colon on the same +/// line (`ip`'s `or: ip link ...`); a line this predicate matches carries +/// no such content and must contribute none to either usage form. +pub fn is_bare_or_form_separator(t: &str) -> bool { + t.trim().trim_end_matches(':').eq_ignore_ascii_case("or") +} + +/// True if `t`, trimmed, is a short label ending in the word `usage` +/// (case-insensitive), a colon, and nothing else — `perlthanks`'s +/// `Advanced usage:`, distinct from the literal `usage:` +/// [`starts_with_usage_prefix`] alone matches. At most three words, every +/// one plain ASCII alphabetic, so a sentence that merely ends near the +/// word `usage` never qualifies. See S-151, `corpus/perlthanks`. +pub fn starts_with_extended_usage_label(t: &str) -> bool { + let Some(head) = t.trim().strip_suffix(':') else { + return false; + }; + let words: Vec<&str> = head.split_whitespace().collect(); + if words.is_empty() || words.len() > 3 { + return false; + } + if !words + .last() + .expect("checked non-empty above") + .eq_ignore_ascii_case("usage") + { + return false; + } + words + .iter() + .all(|w| !w.is_empty() && w.chars().all(|c| c.is_ascii_alphabetic())) +} + +/// True if `t`, trimmed, is only a usage label with nothing after it on +/// the same line: the literal `usage:`/`or:` markers with an empty +/// remainder, or [`starts_with_extended_usage_label`]'s generalized form +/// (which by construction carries no remainder either). `fdisk`'s bare +/// `Usage:` line is this shape; the two real forms sit on the lines below +/// it. See S-150. +pub fn is_bare_usage_label(t: &str) -> bool { + let trimmed = t.trim(); + if starts_with_usage_prefix(trimmed) { + return trimmed + .get(6..) + .map(|rest| rest.trim().is_empty()) + .unwrap_or(true); + } + if starts_with_or_marker(trimmed) { + return trimmed + .get(3..) + .map(|rest| rest.trim().is_empty()) + .unwrap_or(true); + } + starts_with_extended_usage_label(trimmed) +} + +/// True when `t`, trimmed, opens with an alphabetic label of 2 to 20 +/// characters, a colon, and immediately (no space) the tool's own `name` +/// at a word boundary — `mksquashfs`'s `SYNTAX:mksquashfs source1 ...`. +/// Distinct from the two already-recognized markers (`usage:`, `or:`), +/// which are matched regardless of what follows; a label glued straight +/// to unrelated text, or to the name with a space, does not qualify. +/// Mirrors `xtask`'s `usage_label_glued_to_program_name` detector, kept as +/// an independent copy per this crate's convention of never depending on +/// `xtask`. Returns the byte offset in `t` where the tool's own name +/// begins, so the caller can drop the label. See S-142, issue #143. +pub fn label_glued_to_tool_name(t: &str, name: &str) -> Option { + if name.is_empty() { + return None; + } + let colon_idx = t.find(':')?; + let label = &t[..colon_idx]; + if label.len() < 2 || label.len() > 20 || !label.chars().all(|c| c.is_ascii_alphabetic()) { + return None; + } + if label.eq_ignore_ascii_case("usage") || label.eq_ignore_ascii_case("or") { + return None; + } + let after_idx = colon_idx + 1; + let after = t.get(after_idx..)?; + if after.is_empty() || after.starts_with(char::is_whitespace) { + return None; + } + // A URL scheme (`https://github.com/ajeetdsouza/zoxide`) glues its own + // colon straight to a `//` authority, and its path's own basename + // routinely coincides with the tool's own name (a homepage on the + // tool's own domain) — the false positive S-142's own atlas entry + // names. Never a usage label. + if after.starts_with("//") { + return None; + } + let first_token = after.split_whitespace().next().unwrap_or(after); + let basename = first_token.rsplit('/').next().unwrap_or(first_token); + let rest = basename.strip_prefix(name)?; + let boundary_ok = rest.is_empty() + || !rest + .chars() + .next() + .is_some_and(|c| c.is_alphanumeric() || c == '_'); + boundary_ok.then_some(after_idx) +} + +/// True if `t` (already trimmed of leading whitespace) begins with `name` +/// at a word boundary. Lets a tool that repeats its own name across lines +/// with no `or:`/`usage:` marker read as two entries rather than one +/// continuation swallowing the other. Word-boundary checked so `git` +/// doesn't also claim `gitk` or `git-foo`. See S-037. +pub fn starts_with_tool_name(t: &str, name: &str) -> bool { + if name.is_empty() { + return false; + } + match t.strip_prefix(name) { + Some(rest) => rest.is_empty() || rest.starts_with(char::is_whitespace), + None => false, + } +} + +/// True when `t`'s own first whitespace-delimited token spells the tool +/// under different notation than its resolved `name`: as a full path +/// (`/usr/bin/ar` against `ar`), or as the dotted stem a resolved name +/// itself extends (`vim` against `vim.basic`). Twin of +/// [`starts_with_tool_name`], kept as a separate, narrower predicate so +/// every other caller of that one stays exactly as strict as before — used +/// only at `sections/mod.rs`'s `is_own_name` site. See S-108 and +/// `corpus/ar/audit-seed2/help.txt`'s second usage line. +pub fn starts_with_tool_name_spelled_differently(t: &str, name: &str) -> bool { + if name.is_empty() { + return false; + } + let Some(first) = t.split_whitespace().next() else { + return false; + }; + let basename = first.rsplit('/').next().unwrap_or(first); + basename == name || basename == name.split('.').next().unwrap_or(name) +} + +/// True if `t` (already trimmed) is the C `fprintf(stderr, "%s: Usage: +/// ...", argv[0])` idiom's line: the tool's own name, a literal `": "`, +/// then `usage:` case-insensitively. [`starts_with_usage_prefix`] alone +/// misses this since it only tests the line's start. Kept tight: the +/// `usage:` must be preceded by *only* the name and `": "`, never scanned +/// for elsewhere in the line. See S-001 and corpus/nfsidmap/audit-seed/help.txt. +pub fn starts_with_name_prefixed_usage(t: &str, name: &str) -> bool { + if name.is_empty() { + return false; + } + t.strip_prefix(name) + .and_then(|rest| rest.strip_prefix(": ")) + .is_some_and(starts_with_usage_prefix) +} + +/// Drop the `: ` prefix [`starts_with_name_prefixed_usage`] +/// recognizes, wherever it sits in front of a usage label — not only the +/// document's first line. `nfsidmap`'s C `fprintf(stderr, "%s: Usage: +/// ...", argv[0])` idiom keeps that prefix glued to its own usage line +/// rendered; the diagnostic prefix is not the label, and a reader wants the +/// label. Returns `t` trimmed and unchanged when the prefix isn't present. +/// See docs/shapes.md S-162 and S-001. +pub(super) fn strip_name_prefixed_usage_label(t: &str, tool_name: Option<&str>) -> String { + let trimmed = t.trim(); + if let Some(name) = tool_name { + if starts_with_name_prefixed_usage(trimmed, name) { + // `starts_with_name_prefixed_usage` already confirmed `trimmed` + // opens with exactly `"{name}: "`. + return trimmed[name.len() + 2..].to_string(); + } + } + trimmed.to_string() +} + +/// True if `t` opens with `name` at a word boundary and its remainder +/// reads as usage-synopsis grammar rather than prose — the unlabelled +/// synopsis convention (`wpa_cli --help` opens `wpa_cli [-p] +/// [-i] ...` with no `Usage:` marker at all). A name match alone +/// is not evidence (`"tar is an archiving program..."` starts with `tar` +/// too), so both must hold: the remainder contains a docopt group +/// delimiter (spec §7 Tier B: `[`, `<`, `{`), and it does not read as an +/// English sentence ([`is_prose_sentence`]). See S-001 and wpa_cli's help. +pub fn looks_like_unlabeled_synopsis_line(t: &str, name: &str) -> bool { + let Some(rest) = t.strip_prefix(name) else { + return false; + }; + if !(rest.is_empty() || rest.starts_with(char::is_whitespace)) { + return false; + } + let rest = rest.trim_start(); + if rest.is_empty() { + return false; + } + rest.contains(['[', '<', '{']) && !is_prose_sentence(rest) +} From 5d0b50e67a5636956b58220737a09941f49a8ab9 Mon Sep 17 00:00:00 2001 From: Sadig Akhund Date: Sun, 13 Sep 2026 15:14:41 +0400 Subject: [PATCH 9/9] extract: usage label predicates move beside the heading predicates The merged usage.rs reached 804 code lines, over the 800 ceiling. A usage label and the tool's own name at a synopsis head are the same question the heading predicates already answer, so they move to heading.rs rather than into a new file no ratio baseline covers. Co-Authored-By: Claude Fable 5.1 --- .../src/help_text/sections/heading.rs | 222 +++++++++++++++++ .../src/help_text/sections/mod.rs | 3 +- .../src/help_text/sections/usage.rs | 1 - .../src/help_text/sections/usage_label.rs | 227 ------------------ 4 files changed, 223 insertions(+), 230 deletions(-) delete mode 100644 mandible-extract/src/help_text/sections/usage_label.rs diff --git a/mandible-extract/src/help_text/sections/heading.rs b/mandible-extract/src/help_text/sections/heading.rs index a161f531..6aa72f16 100644 --- a/mandible-extract/src/help_text/sections/heading.rs +++ b/mandible-extract/src/help_text/sections/heading.rs @@ -1,6 +1,8 @@ //! Section headings: telling one from prose or a wrapped continuation, //! splitting a heading that shares its line with a row, recognizing a //! word grid or a man page, and naming the group a block's entries carry. +//! A usage label and the tool's own name at a synopsis head are the same +//! question asked of a labelled line, so their predicates live here too. use super::*; @@ -779,6 +781,226 @@ pub(super) fn meaningful_flag_group(heading: String) -> Option { } } +/// True if `t` starts with `"usage:"`, case-insensitively. +/// +/// Compares raw bytes via `[u8]::get` rather than slicing the `str`, which +/// can panic when a multi-byte character (e.g. a box-drawing glyph) lands +/// off a UTF-8 boundary at the slice point. +pub fn starts_with_usage_prefix(t: &str) -> bool { + t.as_bytes() + .get(..6) + .map(|b| b.eq_ignore_ascii_case(b"usage:")) + .unwrap_or(false) +} + +/// True if `t` starts with `"or:"`, case-insensitively — GNU coreutils' +/// marker for a genuine *alternative* invocation form, distinct from a +/// wrapped continuation of the form above it. Without it, joining every +/// more-indented usage line onto its predecessor would swallow `or:`'s +/// alternative form too. See S-037 and corpus/du/9.4/help.txt. +pub fn starts_with_or_marker(t: &str) -> bool { + t.as_bytes() + .get(..3) + .map(|b| b.eq_ignore_ascii_case(b"or:")) + .unwrap_or(false) +} + +/// True if `t`'s only content, once trimmed, is the word `or` — any case — +/// with an optional trailing colon: `sg_luns`' bare second-form separator +/// (`corpus/sg_luns/1.45`), one whole physical line with nothing else on +/// it. Distinct from [`starts_with_or_marker`], which matches an `or:` +/// *prefix* even when real form content follows the colon on the same +/// line (`ip`'s `or: ip link ...`); a line this predicate matches carries +/// no such content and must contribute none to either usage form. +pub fn is_bare_or_form_separator(t: &str) -> bool { + t.trim().trim_end_matches(':').eq_ignore_ascii_case("or") +} + +/// True if `t`, trimmed, is a short label ending in the word `usage` +/// (case-insensitive), a colon, and nothing else — `perlthanks`'s +/// `Advanced usage:`, distinct from the literal `usage:` +/// [`starts_with_usage_prefix`] alone matches. At most three words, every +/// one plain ASCII alphabetic, so a sentence that merely ends near the +/// word `usage` never qualifies. See S-151, `corpus/perlthanks`. +pub fn starts_with_extended_usage_label(t: &str) -> bool { + let Some(head) = t.trim().strip_suffix(':') else { + return false; + }; + let words: Vec<&str> = head.split_whitespace().collect(); + if words.is_empty() || words.len() > 3 { + return false; + } + if !words + .last() + .expect("checked non-empty above") + .eq_ignore_ascii_case("usage") + { + return false; + } + words + .iter() + .all(|w| !w.is_empty() && w.chars().all(|c| c.is_ascii_alphabetic())) +} + +/// True if `t`, trimmed, is only a usage label with nothing after it on +/// the same line: the literal `usage:`/`or:` markers with an empty +/// remainder, or [`starts_with_extended_usage_label`]'s generalized form +/// (which by construction carries no remainder either). `fdisk`'s bare +/// `Usage:` line is this shape; the two real forms sit on the lines below +/// it. See S-150. +pub fn is_bare_usage_label(t: &str) -> bool { + let trimmed = t.trim(); + if starts_with_usage_prefix(trimmed) { + return trimmed + .get(6..) + .map(|rest| rest.trim().is_empty()) + .unwrap_or(true); + } + if starts_with_or_marker(trimmed) { + return trimmed + .get(3..) + .map(|rest| rest.trim().is_empty()) + .unwrap_or(true); + } + starts_with_extended_usage_label(trimmed) +} + +/// True when `t`, trimmed, opens with an alphabetic label of 2 to 20 +/// characters, a colon, and immediately (no space) the tool's own `name` +/// at a word boundary — `mksquashfs`'s `SYNTAX:mksquashfs source1 ...`. +/// Distinct from the two already-recognized markers (`usage:`, `or:`), +/// which are matched regardless of what follows; a label glued straight +/// to unrelated text, or to the name with a space, does not qualify. +/// Mirrors `xtask`'s `usage_label_glued_to_program_name` detector, kept as +/// an independent copy per this crate's convention of never depending on +/// `xtask`. Returns the byte offset in `t` where the tool's own name +/// begins, so the caller can drop the label. See S-142, issue #143. +pub fn label_glued_to_tool_name(t: &str, name: &str) -> Option { + if name.is_empty() { + return None; + } + let colon_idx = t.find(':')?; + let label = &t[..colon_idx]; + if label.len() < 2 || label.len() > 20 || !label.chars().all(|c| c.is_ascii_alphabetic()) { + return None; + } + if label.eq_ignore_ascii_case("usage") || label.eq_ignore_ascii_case("or") { + return None; + } + let after_idx = colon_idx + 1; + let after = t.get(after_idx..)?; + if after.is_empty() || after.starts_with(char::is_whitespace) { + return None; + } + // A URL scheme (`https://github.com/ajeetdsouza/zoxide`) glues its own + // colon straight to a `//` authority, and its path's own basename + // routinely coincides with the tool's own name (a homepage on the + // tool's own domain) — the false positive S-142's own atlas entry + // names. Never a usage label. + if after.starts_with("//") { + return None; + } + let first_token = after.split_whitespace().next().unwrap_or(after); + let basename = first_token.rsplit('/').next().unwrap_or(first_token); + let rest = basename.strip_prefix(name)?; + let boundary_ok = rest.is_empty() + || !rest + .chars() + .next() + .is_some_and(|c| c.is_alphanumeric() || c == '_'); + boundary_ok.then_some(after_idx) +} + +/// True if `t` (already trimmed of leading whitespace) begins with `name` +/// at a word boundary. Lets a tool that repeats its own name across lines +/// with no `or:`/`usage:` marker read as two entries rather than one +/// continuation swallowing the other. Word-boundary checked so `git` +/// doesn't also claim `gitk` or `git-foo`. See S-037. +pub fn starts_with_tool_name(t: &str, name: &str) -> bool { + if name.is_empty() { + return false; + } + match t.strip_prefix(name) { + Some(rest) => rest.is_empty() || rest.starts_with(char::is_whitespace), + None => false, + } +} + +/// True when `t`'s own first whitespace-delimited token spells the tool +/// under different notation than its resolved `name`: as a full path +/// (`/usr/bin/ar` against `ar`), or as the dotted stem a resolved name +/// itself extends (`vim` against `vim.basic`). Twin of +/// [`starts_with_tool_name`], kept as a separate, narrower predicate so +/// every other caller of that one stays exactly as strict as before — used +/// only at `sections/mod.rs`'s `is_own_name` site. See S-108 and +/// `corpus/ar/audit-seed2/help.txt`'s second usage line. +pub fn starts_with_tool_name_spelled_differently(t: &str, name: &str) -> bool { + if name.is_empty() { + return false; + } + let Some(first) = t.split_whitespace().next() else { + return false; + }; + let basename = first.rsplit('/').next().unwrap_or(first); + basename == name || basename == name.split('.').next().unwrap_or(name) +} + +/// True if `t` (already trimmed) is the C `fprintf(stderr, "%s: Usage: +/// ...", argv[0])` idiom's line: the tool's own name, a literal `": "`, +/// then `usage:` case-insensitively. [`starts_with_usage_prefix`] alone +/// misses this since it only tests the line's start. Kept tight: the +/// `usage:` must be preceded by *only* the name and `": "`, never scanned +/// for elsewhere in the line. See S-001 and corpus/nfsidmap/audit-seed/help.txt. +pub fn starts_with_name_prefixed_usage(t: &str, name: &str) -> bool { + if name.is_empty() { + return false; + } + t.strip_prefix(name) + .and_then(|rest| rest.strip_prefix(": ")) + .is_some_and(starts_with_usage_prefix) +} + +/// Drop the `: ` prefix [`starts_with_name_prefixed_usage`] +/// recognizes, wherever it sits in front of a usage label — not only the +/// document's first line. `nfsidmap`'s C `fprintf(stderr, "%s: Usage: +/// ...", argv[0])` idiom keeps that prefix glued to its own usage line +/// rendered; the diagnostic prefix is not the label, and a reader wants the +/// label. Returns `t` trimmed and unchanged when the prefix isn't present. +/// See docs/shapes.md S-162 and S-001. +pub(super) fn strip_name_prefixed_usage_label(t: &str, tool_name: Option<&str>) -> String { + let trimmed = t.trim(); + if let Some(name) = tool_name { + if starts_with_name_prefixed_usage(trimmed, name) { + // `starts_with_name_prefixed_usage` already confirmed `trimmed` + // opens with exactly `"{name}: "`. + return trimmed[name.len() + 2..].to_string(); + } + } + trimmed.to_string() +} + +/// True if `t` opens with `name` at a word boundary and its remainder +/// reads as usage-synopsis grammar rather than prose — the unlabelled +/// synopsis convention (`wpa_cli --help` opens `wpa_cli [-p] +/// [-i] ...` with no `Usage:` marker at all). A name match alone +/// is not evidence (`"tar is an archiving program..."` starts with `tar` +/// too), so both must hold: the remainder contains a docopt group +/// delimiter (spec §7 Tier B: `[`, `<`, `{`), and it does not read as an +/// English sentence ([`is_prose_sentence`]). See S-001 and wpa_cli's help. +pub fn looks_like_unlabeled_synopsis_line(t: &str, name: &str) -> bool { + let Some(rest) = t.strip_prefix(name) else { + return false; + }; + if !(rest.is_empty() || rest.starts_with(char::is_whitespace)) { + return false; + } + let rest = rest.trim_start(); + if rest.is_empty() { + return false; + } + rest.contains(['[', '<', '{']) && !is_prose_sentence(rest) +} + #[cfg(test)] mod tests { use super::*; diff --git a/mandible-extract/src/help_text/sections/mod.rs b/mandible-extract/src/help_text/sections/mod.rs index 5ff2e0d6..acb5ab0d 100644 --- a/mandible-extract/src/help_text/sections/mod.rs +++ b/mandible-extract/src/help_text/sections/mod.rs @@ -46,7 +46,6 @@ mod spelling; #[cfg(test)] mod test_support; mod usage; -mod usage_label; use backfill::*; use bullets::*; @@ -62,7 +61,7 @@ use scan::*; use spelling::*; #[cfg(test)] use test_support::*; -pub use usage::*; +use usage::*; /// Hard cap on distinct entries (subcommands, flags, or choices) accepted /// from a single probe's output. Real `--help` output never remotely diff --git a/mandible-extract/src/help_text/sections/usage.rs b/mandible-extract/src/help_text/sections/usage.rs index 3c8521dc..26f8e1ef 100644 --- a/mandible-extract/src/help_text/sections/usage.rs +++ b/mandible-extract/src/help_text/sections/usage.rs @@ -2,7 +2,6 @@ //! mining the synopsis itself for positionals and for flags no option //! table documents. -pub use super::usage_label::*; use super::*; /// True if `lines[idx]` is a bare own-name invocation line — no bracket diff --git a/mandible-extract/src/help_text/sections/usage_label.rs b/mandible-extract/src/help_text/sections/usage_label.rs deleted file mode 100644 index 99257193..00000000 --- a/mandible-extract/src/help_text/sections/usage_label.rs +++ /dev/null @@ -1,227 +0,0 @@ -//! Recognizing a usage label and the program name beside it: the label -//! vocabulary, the tool's own name at the head of a synopsis, and the -//! `: ` diagnostic prefix in front of a label. Split out of -//! `usage.rs` to keep that file under the size ceiling -//! (`scripts/shape_guard.sh`). - -use super::*; - -/// True if `t` starts with `"usage:"`, case-insensitively. -/// -/// Compares raw bytes via `[u8]::get` rather than slicing the `str`, which -/// can panic when a multi-byte character (e.g. a box-drawing glyph) lands -/// off a UTF-8 boundary at the slice point. -pub fn starts_with_usage_prefix(t: &str) -> bool { - t.as_bytes() - .get(..6) - .map(|b| b.eq_ignore_ascii_case(b"usage:")) - .unwrap_or(false) -} - -/// True if `t` starts with `"or:"`, case-insensitively — GNU coreutils' -/// marker for a genuine *alternative* invocation form, distinct from a -/// wrapped continuation of the form above it. Without it, joining every -/// more-indented usage line onto its predecessor would swallow `or:`'s -/// alternative form too. See S-037 and corpus/du/9.4/help.txt. -pub fn starts_with_or_marker(t: &str) -> bool { - t.as_bytes() - .get(..3) - .map(|b| b.eq_ignore_ascii_case(b"or:")) - .unwrap_or(false) -} - -/// True if `t`'s only content, once trimmed, is the word `or` — any case — -/// with an optional trailing colon: `sg_luns`' bare second-form separator -/// (`corpus/sg_luns/1.45`), one whole physical line with nothing else on -/// it. Distinct from [`starts_with_or_marker`], which matches an `or:` -/// *prefix* even when real form content follows the colon on the same -/// line (`ip`'s `or: ip link ...`); a line this predicate matches carries -/// no such content and must contribute none to either usage form. -pub fn is_bare_or_form_separator(t: &str) -> bool { - t.trim().trim_end_matches(':').eq_ignore_ascii_case("or") -} - -/// True if `t`, trimmed, is a short label ending in the word `usage` -/// (case-insensitive), a colon, and nothing else — `perlthanks`'s -/// `Advanced usage:`, distinct from the literal `usage:` -/// [`starts_with_usage_prefix`] alone matches. At most three words, every -/// one plain ASCII alphabetic, so a sentence that merely ends near the -/// word `usage` never qualifies. See S-151, `corpus/perlthanks`. -pub fn starts_with_extended_usage_label(t: &str) -> bool { - let Some(head) = t.trim().strip_suffix(':') else { - return false; - }; - let words: Vec<&str> = head.split_whitespace().collect(); - if words.is_empty() || words.len() > 3 { - return false; - } - if !words - .last() - .expect("checked non-empty above") - .eq_ignore_ascii_case("usage") - { - return false; - } - words - .iter() - .all(|w| !w.is_empty() && w.chars().all(|c| c.is_ascii_alphabetic())) -} - -/// True if `t`, trimmed, is only a usage label with nothing after it on -/// the same line: the literal `usage:`/`or:` markers with an empty -/// remainder, or [`starts_with_extended_usage_label`]'s generalized form -/// (which by construction carries no remainder either). `fdisk`'s bare -/// `Usage:` line is this shape; the two real forms sit on the lines below -/// it. See S-150. -pub fn is_bare_usage_label(t: &str) -> bool { - let trimmed = t.trim(); - if starts_with_usage_prefix(trimmed) { - return trimmed - .get(6..) - .map(|rest| rest.trim().is_empty()) - .unwrap_or(true); - } - if starts_with_or_marker(trimmed) { - return trimmed - .get(3..) - .map(|rest| rest.trim().is_empty()) - .unwrap_or(true); - } - starts_with_extended_usage_label(trimmed) -} - -/// True when `t`, trimmed, opens with an alphabetic label of 2 to 20 -/// characters, a colon, and immediately (no space) the tool's own `name` -/// at a word boundary — `mksquashfs`'s `SYNTAX:mksquashfs source1 ...`. -/// Distinct from the two already-recognized markers (`usage:`, `or:`), -/// which are matched regardless of what follows; a label glued straight -/// to unrelated text, or to the name with a space, does not qualify. -/// Mirrors `xtask`'s `usage_label_glued_to_program_name` detector, kept as -/// an independent copy per this crate's convention of never depending on -/// `xtask`. Returns the byte offset in `t` where the tool's own name -/// begins, so the caller can drop the label. See S-142, issue #143. -pub fn label_glued_to_tool_name(t: &str, name: &str) -> Option { - if name.is_empty() { - return None; - } - let colon_idx = t.find(':')?; - let label = &t[..colon_idx]; - if label.len() < 2 || label.len() > 20 || !label.chars().all(|c| c.is_ascii_alphabetic()) { - return None; - } - if label.eq_ignore_ascii_case("usage") || label.eq_ignore_ascii_case("or") { - return None; - } - let after_idx = colon_idx + 1; - let after = t.get(after_idx..)?; - if after.is_empty() || after.starts_with(char::is_whitespace) { - return None; - } - // A URL scheme (`https://github.com/ajeetdsouza/zoxide`) glues its own - // colon straight to a `//` authority, and its path's own basename - // routinely coincides with the tool's own name (a homepage on the - // tool's own domain) — the false positive S-142's own atlas entry - // names. Never a usage label. - if after.starts_with("//") { - return None; - } - let first_token = after.split_whitespace().next().unwrap_or(after); - let basename = first_token.rsplit('/').next().unwrap_or(first_token); - let rest = basename.strip_prefix(name)?; - let boundary_ok = rest.is_empty() - || !rest - .chars() - .next() - .is_some_and(|c| c.is_alphanumeric() || c == '_'); - boundary_ok.then_some(after_idx) -} - -/// True if `t` (already trimmed of leading whitespace) begins with `name` -/// at a word boundary. Lets a tool that repeats its own name across lines -/// with no `or:`/`usage:` marker read as two entries rather than one -/// continuation swallowing the other. Word-boundary checked so `git` -/// doesn't also claim `gitk` or `git-foo`. See S-037. -pub fn starts_with_tool_name(t: &str, name: &str) -> bool { - if name.is_empty() { - return false; - } - match t.strip_prefix(name) { - Some(rest) => rest.is_empty() || rest.starts_with(char::is_whitespace), - None => false, - } -} - -/// True when `t`'s own first whitespace-delimited token spells the tool -/// under different notation than its resolved `name`: as a full path -/// (`/usr/bin/ar` against `ar`), or as the dotted stem a resolved name -/// itself extends (`vim` against `vim.basic`). Twin of -/// [`starts_with_tool_name`], kept as a separate, narrower predicate so -/// every other caller of that one stays exactly as strict as before — used -/// only at `sections/mod.rs`'s `is_own_name` site. See S-108 and -/// `corpus/ar/audit-seed2/help.txt`'s second usage line. -pub fn starts_with_tool_name_spelled_differently(t: &str, name: &str) -> bool { - if name.is_empty() { - return false; - } - let Some(first) = t.split_whitespace().next() else { - return false; - }; - let basename = first.rsplit('/').next().unwrap_or(first); - basename == name || basename == name.split('.').next().unwrap_or(name) -} - -/// True if `t` (already trimmed) is the C `fprintf(stderr, "%s: Usage: -/// ...", argv[0])` idiom's line: the tool's own name, a literal `": "`, -/// then `usage:` case-insensitively. [`starts_with_usage_prefix`] alone -/// misses this since it only tests the line's start. Kept tight: the -/// `usage:` must be preceded by *only* the name and `": "`, never scanned -/// for elsewhere in the line. See S-001 and corpus/nfsidmap/audit-seed/help.txt. -pub fn starts_with_name_prefixed_usage(t: &str, name: &str) -> bool { - if name.is_empty() { - return false; - } - t.strip_prefix(name) - .and_then(|rest| rest.strip_prefix(": ")) - .is_some_and(starts_with_usage_prefix) -} - -/// Drop the `: ` prefix [`starts_with_name_prefixed_usage`] -/// recognizes, wherever it sits in front of a usage label — not only the -/// document's first line. `nfsidmap`'s C `fprintf(stderr, "%s: Usage: -/// ...", argv[0])` idiom keeps that prefix glued to its own usage line -/// rendered; the diagnostic prefix is not the label, and a reader wants the -/// label. Returns `t` trimmed and unchanged when the prefix isn't present. -/// See docs/shapes.md S-162 and S-001. -pub(super) fn strip_name_prefixed_usage_label(t: &str, tool_name: Option<&str>) -> String { - let trimmed = t.trim(); - if let Some(name) = tool_name { - if starts_with_name_prefixed_usage(trimmed, name) { - // `starts_with_name_prefixed_usage` already confirmed `trimmed` - // opens with exactly `"{name}: "`. - return trimmed[name.len() + 2..].to_string(); - } - } - trimmed.to_string() -} - -/// True if `t` opens with `name` at a word boundary and its remainder -/// reads as usage-synopsis grammar rather than prose — the unlabelled -/// synopsis convention (`wpa_cli --help` opens `wpa_cli [-p] -/// [-i] ...` with no `Usage:` marker at all). A name match alone -/// is not evidence (`"tar is an archiving program..."` starts with `tar` -/// too), so both must hold: the remainder contains a docopt group -/// delimiter (spec §7 Tier B: `[`, `<`, `{`), and it does not read as an -/// English sentence ([`is_prose_sentence`]). See S-001 and wpa_cli's help. -pub fn looks_like_unlabeled_synopsis_line(t: &str, name: &str) -> bool { - let Some(rest) = t.strip_prefix(name) else { - return false; - }; - if !(rest.is_empty() || rest.starts_with(char::is_whitespace)) { - return false; - } - let rest = rest.trim_start(); - if rest.is_empty() { - return false; - } - rest.contains(['[', '<', '{']) && !is_prose_sentence(rest) -}