diff --git a/src/audits/behavioral/destructive_ops.rs b/src/audits/behavioral/destructive_ops.rs index dcc54d1..30ab408 100644 --- a/src/audits/behavioral/destructive_ops.rs +++ b/src/audits/behavioral/destructive_ops.rs @@ -8,8 +8,9 @@ use crate::runner::HelpOutput; /// Subcommand names that imply destructive intent — irreversible writes -/// targeted at the agent-managed resource. Case-insensitive substring on -/// the subcommand name (so `delete-key` and `force-delete` both match). +/// targeted at the agent-managed resource. Matched case-insensitively at +/// the start of the name or of a `-`/`_` segment, so `delete-key`, +/// `force-delete` and `dropdb` match while `format` and `confirm` do not. const DESTRUCTIVE_VERBS: &[&str] = &[ "delete", "remove", "rm", "destroy", "purge", "wipe", "reset", "drop", "clean", "force-", ]; @@ -35,7 +36,16 @@ pub(crate) fn destructive_subcommands(help: &HelpOutput) -> Vec<&String> { pub(crate) fn is_destructive(name: &str) -> bool { let lower = name.to_lowercase(); - DESTRUCTIVE_VERBS.iter().any(|v| lower.contains(v)) + segment_starts(&lower).any(|segment| DESTRUCTIVE_VERBS.iter().any(|v| segment.starts_with(v))) +} + +/// Every suffix of `name` that begins at the name's start or immediately +/// after a `-` or `_` delimiter: `reset-keys` yields `reset-keys` and `keys`. +fn segment_starts(name: &str) -> impl Iterator { + std::iter::once(name).chain( + name.match_indices(['-', '_']) + .map(|(i, delimiter)| &name[i + delimiter.len()..]), + ) } pub(crate) fn is_read_verb(name: &str) -> bool { @@ -77,11 +87,33 @@ mod tests { } #[test] - fn substring_match_for_compound_names() { - // `delete-all` should match — substring is correct here because - // the destructive intent of `delete` carries over. - assert!(is_destructive("delete-all")); - assert!(is_destructive("dropdb")); + fn segment_prefix_keeps_compound_names_destructive() { + for name in &[ + "delete-all", + "dropdb", + "rmdir", + "cleanup", + "purgeall", + "reset-keys", + "force-push", + "remove_all", + ] { + assert!(is_destructive(name), "{name} should be destructive"); + } + } + + #[test] + fn verb_after_a_delimiter_is_destructive() { + for name in &[ + "queue-purge", + "config-reset", + "session_wipe", + "db-drop", + "cache_clean", + ] { + assert!(is_destructive(name), "{name} should be destructive"); + } + assert!(!is_destructive("config-firmware")); } #[test] @@ -91,6 +123,13 @@ mod tests { } } + #[test] + fn verb_inside_a_word_is_not_destructive() { + for name in &["format", "transform", "perform", "confirm", "firmware"] { + assert!(!is_destructive(name), "{name} should not be destructive"); + } + } + #[test] fn read_verbs_match_exactly() { assert!(is_read_verb("list")); diff --git a/src/audits/behavioral/json_output.rs b/src/audits/behavioral/json_output.rs index ce0dc14..7a5fe7e 100644 --- a/src/audits/behavioral/json_output.rs +++ b/src/audits/behavioral/json_output.rs @@ -1,6 +1,7 @@ use crate::audit::Audit; +use crate::audits::behavioral::subcommand_help::should_skip; use crate::project::Project; -use crate::runner::{BinaryRunner, RunStatus}; +use crate::runner::{BinaryRunner, HelpOutput, RunStatus}; use crate::types::{AuditGroup, AuditLayer, AuditResult, AuditStatus, Confidence}; pub struct JsonOutputAudit; @@ -47,7 +48,7 @@ impl Audit for JsonOutputAudit { } else { // Flag not in top-level help. Probe subcommands, since most CLIs // (gh, kubectl, cargo) put --output on subcommands, not top-level. - probe_subcommands(runner, &output) + probe_subcommands(runner, project.help_output()) } } _ => AuditStatus::Skip("could not run --help to detect output flags".into()), @@ -65,12 +66,13 @@ impl Audit for JsonOutputAudit { } } -/// Parse subcommand names from --help output and audit each for --output/--format. +/// Probe each top-level subcommand from the shared help parse for --output/--format. /// /// Most CLI frameworks (clap, cobra, argparse) list subcommands under a "Commands:" -/// or "Subcommands:" section. We parse those names and probe each one. -fn probe_subcommands(runner: &BinaryRunner, help_output: &str) -> AuditStatus { - let subcommands = parse_subcommand_names(help_output); +/// or "Subcommands:" section, and hand-written help under `... commands:`; the +/// shared parser reads all of them. +fn probe_subcommands(runner: &BinaryRunner, help: Option<&HelpOutput>) -> AuditStatus { + let subcommands = subcommands_to_probe(help); if subcommands.is_empty() { return AuditStatus::OptOut( "no --output/--format flag detected — tool does not ship structured output. \ @@ -103,47 +105,17 @@ fn probe_subcommands(runner: &BinaryRunner, help_output: &str) -> AuditStatus { ) } -/// Extract subcommand names from CLI --help output. -/// -/// Looks for a "Commands:" or "Subcommands:" section and parses the first word -/// of each indented line. Stops at the next section header or blank line gap. -fn parse_subcommand_names(help_output: &str) -> Vec { - let mut names = Vec::new(); - let mut in_commands_section = false; - - for line in help_output.lines() { - let trimmed = line.trim(); - - // Detect the start of a commands section - if trimmed.eq_ignore_ascii_case("commands:") - || trimmed.eq_ignore_ascii_case("subcommands:") - || trimmed.starts_with("Commands:") - || trimmed.starts_with("Subcommands:") - { - in_commands_section = true; - continue; - } - - if in_commands_section { - // End of section: non-indented non-empty line (next section header) - if !trimmed.is_empty() && !line.starts_with(' ') && !line.starts_with('\t') { - break; - } - // Skip empty lines within the section - if trimmed.is_empty() { - continue; - } - // Extract the first word as the subcommand name - if let Some(name) = trimmed.split_whitespace().next() { - // Skip "help" subcommand (meta, not a real command) - if name != "help" { - names.push(name.to_string()); - } - } - } - } - - names +/// Top-level subcommand names worth probing for an output flag: the shared +/// parser's names minus the built-ins (`help`, shell completions) that the +/// subcommand-help helper skips. `help` in particular echoes top-level help, +/// so probing it could move this row on a tool that has no output flag. +fn subcommands_to_probe(help: Option<&HelpOutput>) -> Vec<&str> { + help.map(HelpOutput::subcommands) + .unwrap_or_default() + .iter() + .map(String::as_str) + .filter(|name| !should_skip(name)) + .collect() } /// Try safe subcommands with the detected flag to validate actual JSON output. @@ -376,23 +348,81 @@ esac } #[test] - fn parse_subcommand_names_clap_format() { - let help = "Usage: mycli [COMMAND]\n\nCommands:\n audit Run audits\n list List items\n help Print help\n\nOptions:\n -h, --help Print help\n"; - let names = parse_subcommand_names(help); - assert_eq!(names, vec!["audit", "list"]); + fn json_output_probes_hand_written_command_block() { + // herdr-shaped help: a `Common commands:` block whose entries all lead + // with the tool name. The `count` subcommand carries the output flag. + let script = r#" +case "$*" in + *count*--output*json*|*count*--output=json*) + echo '{"count":3}';; + *count*--help*) + echo "Usage: test count [--output FORMAT]";; + *--help*) + echo "Usage: test [options] + +Common commands: + test Launch interactively + test count Count things + test list List things";; + *) + echo "hello";; +esac +"#; + let project = test_project_with_sh_script(script); + let result = JsonOutputAudit.run(&project).expect("audit should run"); + assert_eq!(result.status, AuditStatus::Pass, "got {:?}", result.status); + } + + #[test] + fn json_output_does_not_probe_the_help_subcommand() { + // `help --help` advertises --output and would validate as JSON, but + // `help` is a built-in the audit never probes, so the tool opts out. + let script = r#" +case "$*" in + *help*--output*json*|*help*--output=json*) + echo '{"help":true}';; + help*--help*) + echo "Usage: test help [--output FORMAT]";; + *--help*) + echo "Usage: test [COMMAND] + +Commands: + audit Run audits + help Print help";; + *) + echo "hello";; +esac +"#; + let project = test_project_with_sh_script(script); + let result = JsonOutputAudit.run(&project).expect("audit should run"); + assert!( + matches!(result.status, AuditStatus::OptOut(_)), + "expected OptOut, got {:?}", + result.status + ); + } + + #[test] + fn subcommands_to_probe_clap_format_drops_help() { + let help = HelpOutput::from_raw( + "Usage: mycli [COMMAND]\n\nCommands:\n audit Run audits\n list List items\n help Print help\n\nOptions:\n -h, --help Print help\n", + ); + assert_eq!(subcommands_to_probe(Some(&help)), ["audit", "list"]); } #[test] - fn parse_subcommand_names_empty() { - let help = "Usage: mycli [OPTIONS]\n\nOptions:\n -h, --help Print help\n"; - let names = parse_subcommand_names(help); - assert!(names.is_empty()); + fn subcommands_to_probe_empty_without_block_or_probe() { + let help = + HelpOutput::from_raw("Usage: mycli [OPTIONS]\n\nOptions:\n -h, --help Print help\n"); + assert!(subcommands_to_probe(Some(&help)).is_empty()); + assert!(subcommands_to_probe(None).is_empty()); } #[test] - fn parse_subcommand_names_subcommands_header() { - let help = "Subcommands:\n run Execute\n build Compile\n"; - let names = parse_subcommand_names(help); - assert_eq!(names, vec!["run", "build"]); + fn subcommands_to_probe_hand_written_block() { + let help = HelpOutput::from_raw( + "Usage: tool [options]\n\nCommon commands:\n tool Launch\n tool run Execute\n tool build Compile\n tool completion zsh Shell completions\n", + ); + assert_eq!(subcommands_to_probe(Some(&help)), ["run", "build"]); } } diff --git a/src/audits/behavioral/standard_names.rs b/src/audits/behavioral/standard_names.rs index 21447b3..94e5448 100644 --- a/src/audits/behavioral/standard_names.rs +++ b/src/audits/behavioral/standard_names.rs @@ -208,9 +208,10 @@ impl Audit for StandardNamesAudit { } } -/// Core unit for tests. Returns Skip when no subcommands are present (the -/// "if CLI uses subcommands" applicability is vacuously satisfied), Pass when -/// at least the threshold fraction matches the allow-list, Warn otherwise. +/// Core unit for tests. Returns Skip when no subcommand names were parsed +/// from `--help` (the "if CLI uses subcommands" applicability is vacuously +/// satisfied), Pass when at least the threshold fraction matches the +/// allow-list, Warn otherwise. /// /// `domain_verbs` extends the built-in [`STANDARD_VERBS`] list with per-CLI /// platform vocabulary (typically loaded from `.anc.toml`). Recognition is @@ -232,7 +233,7 @@ pub(crate) fn audit_standard_names( if subs.is_empty() { return StandardNamesResult { - status: AuditStatus::Skip("no subcommands parsed from --help".into()), + status: AuditStatus::Skip(help.missing_subcommands_reason().into()), mitigation: None, }; } diff --git a/src/audits/behavioral/subcommand_examples.rs b/src/audits/behavioral/subcommand_examples.rs index d7ed3fb..a77e8b1 100644 --- a/src/audits/behavioral/subcommand_examples.rs +++ b/src/audits/behavioral/subcommand_examples.rs @@ -14,7 +14,7 @@ //! - Contains the literal `Examples:` / `EXAMPLES` section header. //! //! Fail when any non-skipped subcommand misses an example. Vacuous Skip -//! when the binary has no subcommands. +//! when no subcommand names were parsed from the top-level `--help`. use crate::audit::Audit; use crate::audits::behavioral::subcommand_help::probe_subcommands; @@ -52,10 +52,10 @@ impl Audit for SubcommandExamplesAudit { fn run(&self, project: &Project) -> anyhow::Result { let status = match project.help_output() { None => AuditStatus::Skip("could not probe --help".into()), - Some(top_help) if top_help.subcommands().is_empty() => AuditStatus::Skip( - "binary has no subcommands; MUST applies conditionally to CLIs that use them." - .into(), - ), + Some(top_help) if top_help.subcommands().is_empty() => AuditStatus::Skip(format!( + "{}; the MUST applies conditionally to CLIs that use subcommands.", + top_help.missing_subcommands_reason() + )), Some(top_help) => { let runner = project.runner_ref(); let subhelp = probe_subcommands(runner, top_help); diff --git a/src/audits/behavioral/subcommand_help.rs b/src/audits/behavioral/subcommand_help.rs index eadf247..3e4faa0 100644 --- a/src/audits/behavioral/subcommand_help.rs +++ b/src/audits/behavioral/subcommand_help.rs @@ -54,7 +54,8 @@ pub(crate) fn probe_subcommands( out } -fn should_skip(name: &str) -> bool { +/// Whether `name` is a built-in the subcommand probes leave alone. +pub(crate) fn should_skip(name: &str) -> bool { SKIP_SUBCOMMANDS .iter() .any(|s| name.eq_ignore_ascii_case(s)) diff --git a/src/runner/help_probe/mod.rs b/src/runner/help_probe/mod.rs index 898ec51..ae9ccb7 100644 --- a/src/runner/help_probe/mod.rs +++ b/src/runner/help_probe/mod.rs @@ -1,15 +1,17 @@ //! Shared `--help` probe + lazy parsers. //! //! The runner spawns ` --help` exactly once per target. The captured -//! text is parsed on demand into three views — flags, env hints, subcommands. -//! Behavioral audits that need to inspect the help surface share the same -//! `HelpOutput` for a given target so none of them re-spawn the binary. +//! text is parsed on demand into views: flags, env hints, command blocks, and +//! subcommands. Behavioral audits that need to inspect the help surface share +//! the same `HelpOutput` for a given target so none of them re-spawn the +//! binary. //! //! Parsers are English-only by convention: we match on clap's output shape -//! (`Commands:`, `[env: FOO]`, leading-whitespace flag lines). Localized help -//! is a named exception in `docs/coverage-matrix.md` — audits that consume -//! these parsers should Skip, not Warn, when the raw text lacks an English -//! help surface. +//! (`Commands:`, `[env: FOO]`, leading-whitespace flag lines) and on the +//! hand-written command-block shape (`Common commands:` headers whose entries +//! lead with the tool's own name). Localized help is a named exception in +//! `docs/coverage-matrix.md` — audits that consume these parsers should Skip, +//! not Warn, when the raw text lacks an English help surface. //! //! `parse_env_hints` uses two complementary patterns: //! - **Pattern 1 (clap-style)**: `[env: FOO]` annotations inside the flag @@ -91,28 +93,69 @@ pub struct EnvHint { pub source: EnvHintSource, } +/// A `... commands:` block from the help surface, kept alongside the names +/// read from it so audits can report what was read, not only what survived. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct CommandBlock { + /// The header line as written, e.g. `Commands:` or `Common commands:`. + pub header: String, + /// Entry lines as written, without their indentation. + pub entries: Vec, + /// The tool-name token every entry led with. Names are read from what + /// follows it; `None` when the block is not prefixed. + pub prefix: Option, +} + +impl CommandBlock { + /// The invocation part of `entry`: the text before the two-space (or + /// tab) description gap, minus the shared prefix when the block is + /// prefixed. Empty for the bare-invocation entry, whose invocation is + /// the tool name alone; that entry may also carry its description after + /// a single space (`tool Launch the app`), so in a prefixed block a + /// gapless entry whose remainder starts with an uppercase letter is read + /// as description only: commands are lowercase by convention, sentences + /// are capitalized. + pub fn command_text<'a>(&self, entry: &'a str) -> &'a str { + let invocation = before_description_gap(entry.split('\t').next().unwrap_or(entry)); + if self.prefix.is_none() { + return invocation; + } + let rest = invocation + .split_once(char::is_whitespace) + .map_or("", |(_, rest)| rest.trim_start()); + let has_description_gap = invocation.len() < entry.len(); + if !has_description_gap && rest.starts_with(char::is_uppercase) { + return ""; + } + rest + } +} + /// Shared, lazily-parsed view over ` --help`. Construct via /// [`HelpOutput::probe`] in runner code, or [`HelpOutput::from_raw`] in tests. pub struct HelpOutput { raw: String, + /// File stem of the binary the probe spawned, `None` for text built + /// via [`HelpOutput::from_raw`]. Joins the `Usage:` line's tool name as + /// a prefix candidate when command blocks are parsed. + binary_stem: Option, flags: OnceLock>, env_hints: OnceLock>, - /// Reserved for P3/P6 subcommand-structure audits. Parsed lazily like - /// the other views; no current behavioral audit consumes it, so the - /// compiler would flag it as dead code without this allow. - #[allow(dead_code)] + command_blocks: OnceLock>, subcommands: OnceLock>, } impl HelpOutput { /// Build a `HelpOutput` from captured help text. The primary seam for /// unit tests — pass a fixture string and exercise the parsers without - /// spawning a binary. + /// spawning a binary. Does no parsing; every view is built on first use. pub fn from_raw(raw: impl Into) -> Self { Self { raw: raw.into(), + binary_stem: None, flags: OnceLock::new(), env_hints: OnceLock::new(), + command_blocks: OnceLock::new(), subcommands: OnceLock::new(), } } @@ -137,7 +180,9 @@ impl HelpOutput { let mut raw = String::with_capacity(help.stdout.len() + help.stderr.len()); raw.push_str(&help.stdout); raw.push_str(&help.stderr); - Ok(Self::from_raw(raw)) + let mut parsed = Self::from_raw(raw); + parsed.binary_stem = runner.binary_stem().map(str::to_string); + Ok(parsed) } } } @@ -157,12 +202,30 @@ impl HelpOutput { self.env_hints.get_or_init(|| parse_env_hints(&self.raw)) } - /// Subcommand names parsed out of the help surface. Lazy + cached. - /// Reserved for P3/P6 audits; no behavioral audit consumes this yet. - #[allow(dead_code)] + /// Every command block found in the help surface, in order. Lazy + cached. + pub fn command_blocks(&self) -> &[CommandBlock] { + self.command_blocks.get_or_init(|| { + let candidates = prefix_candidates(&self.raw, self.binary_stem.as_deref()); + parse_command_blocks(&self.raw, &candidates) + }) + } + + /// Single-token top-level subcommand names read from every command + /// block, deduplicated in order of first appearance. Lazy + cached. pub fn subcommands(&self) -> &[String] { self.subcommands - .get_or_init(|| parse_subcommands(&self.raw)) + .get_or_init(|| subcommand_names(self.command_blocks())) + } + + /// Why [`HelpOutput::subcommands`] is empty, worded as what the parser + /// observed so an audit never asserts the tool has no subcommands when it + /// means none were parsed. + pub fn missing_subcommands_reason(&self) -> &'static str { + if self.command_blocks().is_empty() { + "no command block found in --help output, so no subcommands were parsed" + } else { + "a command block was found in --help output but no subcommand names could be parsed from it" + } } } @@ -187,10 +250,7 @@ fn parse_flags(raw: &str) -> Vec { if trimmed.starts_with("---") { continue; } - // Header = everything before clap's two-space description gap. When - // there's no description on the same line the whole remainder is the - // header. - let header = trimmed.split(" ").next().unwrap_or(trimmed); + let header = before_description_gap(trimmed); let mut short: Option = None; let mut long: Option = None; @@ -212,6 +272,13 @@ fn parse_flags(raw: &str) -> Vec { flags } +/// The text before clap's two-space description gap: the flag header of a +/// flag line, the invocation of a command entry. The whole line when there +/// is no description on it. +fn before_description_gap(line: &str) -> &str { + line.split(" ").next().unwrap_or(line) +} + /// Extract a `--long` flag name from a token like `--long`, `--long=`, /// `--long[=]`, or `--long `. Returns `None` when `candidate` is /// not a long flag. @@ -308,36 +375,127 @@ fn is_env_var_name(s: &str) -> bool { .all(|c| c.is_ascii_uppercase() || c.is_ascii_digit() || c == '_') } -/// Parse the `Commands:` / `Subcommands:` block. We collect the first -/// whitespace-separated token on each line until the block terminates -/// (empty line, or a new non-indented section header). -#[allow(dead_code)] -fn parse_subcommands(raw: &str) -> Vec { - let mut out = Vec::new(); - let mut in_section = false; - for line in raw.lines() { - let trimmed = line.trim(); - let is_header = matches!(trimmed, "Commands:" | "Subcommands:" | "SUBCOMMANDS:"); - if is_header { - in_section = true; +/// The tool's own name as the `Usage:` line states it: the first token after +/// `Usage:` on the same line, or on the next non-empty line when the header +/// stands alone (cobra, clap v3). A path is reduced to its final component. +fn usage_tool_name(raw: &str) -> Option { + let mut lines = raw.lines(); + while let Some(line) = lines.next() { + let Some((label, rest)) = line.trim().split_once(':') else { continue; - } - if !in_section { + }; + if !label.eq_ignore_ascii_case("usage") { continue; } - if trimmed.is_empty() { - // Blank line ends the block. - in_section = false; + let candidate = rest.split_whitespace().next().or_else(|| { + lines + .find(|l| !l.trim().is_empty()) + .and_then(|l| l.split_whitespace().next()) + })?; + let name = candidate.rsplit(['/', '\\']).next().unwrap_or(candidate); + return name + .chars() + .next() + .is_some_and(|c| c.is_ascii_alphanumeric() || c == '_' || c == '.') + .then(|| name.to_string()); + } + None +} + +/// A block header is any line whose last word is `commands:` or +/// `subcommands:`, case-insensitively: clap's `Commands:`, cobra's +/// `Available Commands:`, and hand-written `Common commands:` alike. +fn is_command_header(trimmed: &str) -> bool { + trimmed.split_whitespace().last().is_some_and(|last| { + last.eq_ignore_ascii_case("commands:") || last.eq_ignore_ascii_case("subcommands:") + }) +} + +/// The tokens a prefixed block's entries may lead with: the `Usage:` line's +/// tool name and the probe-supplied binary stem, deduplicated. Both are +/// needed because a usage line can lead with a launcher (`npx tool`, +/// `python -m tool`) or a suffixed file name (`tool.exe`) while the entries +/// lead with the bare tool name. +fn prefix_candidates(raw: &str, binary_stem: Option<&str>) -> Vec { + let mut candidates: Vec = usage_tool_name(raw).into_iter().collect(); + if let Some(stem) = binary_stem + && !candidates.iter().any(|c| c == stem) + { + candidates.push(stem.to_string()); + } + candidates +} + +/// Collect every command block: a header, then the indented lines that +/// follow it until a blank line or a non-indented line closes the block. +/// The block's first entry sets its entry indentation; a later line +/// indented more deeply than that continues the previous entry (a wrapped +/// description, a nested subcommand) and is not recorded. Headers with no +/// entries are dropped. +fn parse_command_blocks(raw: &str, candidates: &[String]) -> Vec { + let mut blocks: Vec = Vec::new(); + let mut entry_indent: Option = None; + let mut in_block = false; + for line in raw.lines() { + let trimmed = line.trim(); + if is_command_header(trimmed) { + blocks.push(CommandBlock { + header: trimmed.to_string(), + entries: Vec::new(), + prefix: None, + }); + entry_indent = None; + in_block = true; + } else if !in_block { continue; + } else if trimmed.is_empty() || !line.starts_with(char::is_whitespace) { + in_block = false; + } else if let Some(block) = blocks.last_mut() { + let indent = line.chars().take_while(|c| c.is_whitespace()).count(); + let first = *entry_indent.get_or_insert(indent); + if indent <= first { + block.entries.push(trimmed.to_string()); + } } - if !line.starts_with(' ') { - // A new top-level section header ended the commands block. - break; - } - if let Some(name) = trimmed.split_whitespace().next() - && is_subcommand_name(name) - { - out.push(name.to_string()); + } + blocks.retain(|block| !block.entries.is_empty()); + for block in &mut blocks { + block.prefix = shared_tool_prefix(&block.entries, candidates); + } + blocks +} + +/// The candidate every entry in the block leads with, the hand-written shape +/// `herdr status` / `herdr server stop`. Agreement has to be unanimous so a +/// command that merely shares the tool's name is not eaten, and it is judged +/// per block so a one-entry block strips on its own. Candidates are tried in +/// order, so the `Usage:` line's name wins over the binary stem when both +/// would match. +fn shared_tool_prefix(entries: &[String], candidates: &[String]) -> Option { + candidates + .iter() + .find(|candidate| { + entries + .iter() + .all(|entry| entry.split_whitespace().next() == Some(candidate.as_str())) + }) + .cloned() +} + +/// Single-token top-level names: the first token of each entry after any +/// shared prefix, validated by [`is_subcommand_name`] and deduplicated in +/// order of first appearance. A nested entry (`server stop`) contributes +/// `server`; an entry the prefix strip leaves empty contributes nothing. +fn subcommand_names(blocks: &[CommandBlock]) -> Vec { + let mut out: Vec = Vec::new(); + for block in blocks { + for entry in &block.entries { + let Some(name) = block.command_text(entry).split_whitespace().next() else { + continue; + }; + if is_subcommand_name(name) && !out.iter().any(|seen| seen == name) { + out.push(name.to_string()); + } } } out @@ -345,7 +503,6 @@ fn parse_subcommands(raw: &str) -> Vec { /// Subcommand names are kebab-case/snake_case identifiers. Anything else — /// `[options]`, ``, punctuation — is not a subcommand. -#[allow(dead_code)] fn is_subcommand_name(s: &str) -> bool { !s.is_empty() && s.chars() @@ -402,6 +559,36 @@ Usage: xurl-rs URL // `runner/help_probe/env_hints_bash.rs`. Parent keeps only Pattern 1 // + shared-fixture tests. + // Hand-written help modeled on herdr: a `Usage:` line naming the tool, + // two `... commands:` blocks, and every entry led by the tool name. + const HERDR_HELP: &str = r#"herdr — terminal workspace manager for AI coding agents + +Usage: herdr [options] + herdr --session [options] + herdr server stop + +Common commands: + herdr Launch or attach to the persistent session + herdr status [server|client] Show local client and running server status + herdr update Download and install the latest version + herdr completion zsh Generate shell completions for zsh + herdr server stop Stop the running server via the API socket + herdr channel set Choose the stable or preview update channel + herdr server reload-config Reload config.toml in the running server + herdr config reset-keys Back up config.toml and remove custom keybindings + herdr channel Manage the stable or preview update channel + herdr machine Manage saved SSH machines + herdr api Inspect socket API metadata and live runtime state + +Advanced commands: + herdr server Run as headless server + +Options: + --session Use or create a named persistent session + --version, -V Print version and exit + --help, -h Show this help +"#; + // Localized help — ensures parsers degrade to empty without panicking. const NON_ENGLISH_HELP: &str = r#"用法: outil [选项] @@ -491,16 +678,211 @@ Usage: xurl-rs URL #[test] fn parse_subcommands_reads_commands_block() { - let subs = parse_subcommands(CLAP_HELP); - assert!(subs.iter().any(|s| s == "audit")); - assert!(subs.iter().any(|s| s == "generate")); - assert!(subs.iter().any(|s| s == "completions")); + let help = HelpOutput::from_raw(CLAP_HELP); + assert_eq!( + help.subcommands(), + ["audit", "completions", "generate", "help"] + ); + let [block] = help.command_blocks() else { + panic!("expected one block, got {:?}", help.command_blocks()); + }; + assert_eq!(block.header, "Commands:"); + assert_eq!(block.prefix, None); + } + + #[test] + fn single_entry_clap_block_is_not_read_as_prefixed() { + // Unanimity is trivial in a one-entry block; the tool name is what + // distinguishes `audit Run audits` from `herdr status`. + let help = + HelpOutput::from_raw("Usage: tool \n\nCommands:\n audit Run audits\n"); + assert_eq!(help.subcommands(), ["audit"]); + assert_eq!(help.command_blocks()[0].prefix, None); + } + + #[test] + fn hand_written_block_records_prefix_and_entries() { + let help = HelpOutput::from_raw(HERDR_HELP); + let blocks = help.command_blocks(); + assert_eq!(blocks.len(), 2); + assert_eq!(blocks[0].header, "Common commands:"); + assert_eq!(blocks[0].prefix.as_deref(), Some("herdr")); + assert_eq!(blocks[0].entries.len(), 11); + assert_eq!( + blocks[0].entries[0], + "herdr Launch or attach to the persistent session" + ); + assert_eq!(blocks[0].command_text(&blocks[0].entries[0]), ""); + assert_eq!( + blocks[0].command_text(&blocks[0].entries[1]), + "status [server|client]" + ); + assert_eq!(blocks[1].header, "Advanced commands:"); + assert_eq!(blocks[1].prefix.as_deref(), Some("herdr")); + assert_eq!( + blocks[1].entries, + ["herdr server Run as headless server"] + ); + } + + #[test] + fn nested_and_placeholder_entries_contribute_top_level_name_only() { + let help = HelpOutput::from_raw(HERDR_HELP); + let subs = help.subcommands(); + assert!( + subs.iter().all(|s| !s.contains(['<', '[', '>', ']'])), + "{subs:?}" + ); + assert!( + !subs + .iter() + .any(|s| s == "stop" || s == "set" || s == "herdr"), + "{subs:?}" + ); + } + + #[test] + fn block_without_a_shared_prefix_is_read_as_written() { + let help = HelpOutput::from_raw( + "Usage: tool \n\nCommon commands:\n status Show status\n tool-up Bring the tool up\n", + ); + assert_eq!(help.subcommands(), ["status", "tool-up"]); + assert_eq!(help.command_blocks()[0].prefix, None); + } + + #[test] + fn command_that_shares_the_tool_name_breaks_unanimity() { + let help = HelpOutput::from_raw( + "Usage: tool \n\nCommands:\n tool Run the tool\n status Show status\n", + ); + assert_eq!(help.subcommands(), ["tool", "status"]); + assert_eq!(help.command_blocks()[0].prefix, None); + } + + #[test] + fn cobra_style_header_and_usage_on_next_line() { + let help = HelpOutput::from_raw( + "Usage:\n kubectl [command]\n\nAvailable Commands:\n apply Apply a configuration\n get Display resources\n\nFlags:\n -h, --help help\n", + ); + assert_eq!(help.subcommands(), ["apply", "get"]); + assert_eq!(help.command_blocks()[0].header, "Available Commands:"); + } + + #[test] + fn usage_tool_name_reads_the_usage_line() { + assert_eq!(usage_tool_name(HERDR_HELP).as_deref(), Some("herdr")); + assert_eq!(usage_tool_name(CLAP_HELP).as_deref(), Some("anc")); + assert_eq!( + usage_tool_name("USAGE:\n anc \n").as_deref(), + Some("anc") + ); + assert_eq!( + usage_tool_name("Usage: ./target/debug/anc \n").as_deref(), + Some("anc") + ); + assert_eq!(usage_tool_name("Usage: [OPTIONS]\n"), None); + assert_eq!(usage_tool_name(NON_ENGLISH_HELP), None); + } + + #[test] + fn probe_falls_back_to_the_binary_name_without_a_usage_line() { + use crate::audits::behavioral::tests::test_project_with_sh_script; + let project = test_project_with_sh_script( + "echo 'Common commands:'; echo ' test status Show status'; echo ' test server Run the server'", + ); + let help = project.help_output().expect("probe succeeds"); + assert_eq!(help.subcommands(), ["status", "server"]); + } + + #[test] + fn missing_subcommands_reason_distinguishes_no_block_from_unparsed_block() { + let none = HelpOutput::from_raw(BARE_HELP); + assert!(none.subcommands().is_empty()); + assert!( + none.missing_subcommands_reason() + .contains("no command block") + ); + + let unparsed = HelpOutput::from_raw("Usage: tool\n\nCommands:\n A placeholder\n"); + assert!(unparsed.subcommands().is_empty()); + assert!( + unparsed + .missing_subcommands_reason() + .contains("but no subcommand names") + ); + } + + #[test] + fn hand_written_block_yields_top_level_names_without_the_tool_prefix() { + let help = HelpOutput::from_raw(HERDR_HELP); + assert_eq!( + help.subcommands(), + [ + "status", + "update", + "completion", + "server", + "channel", + "config", + "machine", + "api", + ] + ); + } + + #[test] + fn wrapped_description_continuation_keeps_the_prefixed_block_intact() { + let wrapped = HERDR_HELP.replace( + " herdr status [server|client] Show local client and running server status\n", + " herdr status [server|client] Show local client and running server\n status, including uptime\n", + ); + assert_ne!(wrapped, HERDR_HELP, "fixture line was rewritten"); + let help = HelpOutput::from_raw(&wrapped); + assert_eq!( + help.subcommands(), + HelpOutput::from_raw(HERDR_HELP).subcommands() + ); + assert!(!help.subcommands().iter().any(|s| s == "herdr")); + assert_eq!(help.command_blocks()[0].entries.len(), 11); + } + + #[test] + fn deeper_indented_nested_lines_are_continuations_of_their_parent_entry() { + let help = HelpOutput::from_raw( + "Usage: tool \n\nCommands:\n server Manage\n start Start it\n stop Stop it\n", + ); + assert_eq!(help.subcommands(), ["server"]); + assert_eq!(help.command_blocks()[0].entries, ["server Manage"]); + } + + #[test] + fn probe_strips_the_binary_stem_when_the_usage_line_leads_with_a_launcher() { + use crate::audits::behavioral::tests::test_project_with_sh_script; + let project = test_project_with_sh_script( + "echo 'Usage: npx test [options]'; echo ''; echo 'Commands:'; echo ' test status Show status'; echo ' test server Run the server'", + ); + let help = project.help_output().expect("probe succeeds"); + assert_eq!(help.subcommands(), ["status", "server"]); + assert_eq!(help.command_blocks()[0].prefix.as_deref(), Some("test")); + } + + #[test] + fn single_space_bare_entry_in_a_prefixed_block_is_description_only() { + let help = HelpOutput::from_raw( + "Usage: tool [options]\n\nCommands:\n tool Launch the app\n tool status Show status\n", + ); + assert_eq!(help.subcommands(), ["status"]); + let block = &help.command_blocks()[0]; + assert_eq!(block.prefix.as_deref(), Some("tool")); + assert_eq!(block.command_text(&block.entries[0]), ""); + assert_eq!(block.command_text(&block.entries[1]), "status"); } #[test] fn parse_subcommands_empty_without_block() { - let subs = parse_subcommands(BARE_HELP); - assert!(subs.is_empty()); + let help = HelpOutput::from_raw(BARE_HELP); + assert!(help.subcommands().is_empty()); + assert!(help.command_blocks().is_empty()); } #[test] @@ -516,7 +898,11 @@ Usage: xurl-rs URL assert!(f.short.is_some() || f.long.is_some()); } assert!(parse_env_hints(NON_ENGLISH_HELP).is_empty()); - assert!(parse_subcommands(NON_ENGLISH_HELP).is_empty()); + assert!( + HelpOutput::from_raw(NON_ENGLISH_HELP) + .subcommands() + .is_empty() + ); } #[test] diff --git a/src/runner/mod.rs b/src/runner/mod.rs index 91d5498..5841de3 100644 --- a/src/runner/mod.rs +++ b/src/runner/mod.rs @@ -80,6 +80,14 @@ impl BinaryRunner { }) } + /// File stem of the binary this runner spawns: `anc` for `target/debug/anc` + /// and for `anc.exe`. The stem rather than the file name because the help + /// probe compares it against the bare token on a `Usage:` line, which + /// never carries an extension. + pub fn binary_stem(&self) -> Option<&str> { + self.binary.file_stem().and_then(|stem| stem.to_str()) + } + /// Run the binary with the given args and env overrides. /// /// Results are cached by (args, env_overrides). `NO_COLOR=1` is always set. diff --git a/tests/fixtures/handwritten-help/tally b/tests/fixtures/handwritten-help/tally new file mode 100755 index 0000000..30970e0 --- /dev/null +++ b/tests/fixtures/handwritten-help/tally @@ -0,0 +1,46 @@ +#!/bin/sh +# Hand-written help in the herdr shape: a `Usage:` line naming the tool, two +# `... commands:` blocks, and every entry led by the tool's own name. +case "$1" in + --help|-h) + cat <<'EOF' +tally — count things from the terminal + +Usage: tally [options] + tally count + tally report ... + +Common commands: + tally Launch the interactive counter + tally count Count entries under a path + tally list List saved tallies + tally show Show one saved tally + tally format