diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 32bd73d9..075de734 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -38,6 +38,7 @@ jobs: cli: - '.github/workflows/ci.yml' - 'acdc-cli/**' + - 'acdc-execute/**' - 'acdc-parser/**' - 'converters/**' - 'Cargo.toml' @@ -82,7 +83,9 @@ jobs: version: 0.16.0 - uses: Swatinem/rust-cache@v2 - name: Run clippy - run: cargo clippy --all-targets --all-features -- --deny clippy::pedantic + run: cargo clippy --all-targets --all-features -- --deny clippy::pedantic --deny clippy::todo + - name: Run clippy (execute library) + run: cargo clippy -p acdc-execute --all-targets --all-features -- --deny clippy::pedantic --deny clippy::todo - name: Install wasm32 target run: rustup target add wasm32-unknown-unknown - name: Run clippy (WASM editor) @@ -153,8 +156,12 @@ jobs: version: 0.16.0 - uses: Swatinem/rust-cache@v2 - uses: taiki-e/install-action@nextest - - name: Run CLI tests - run: cargo nextest run -p acdc-cli --all-features --verbose --no-tests warn --test-threads 1 # we've added this until https://github.com/nextest-rs/nextest/pull/3553 gets merged + - name: Run CLI and execute library tests + run: cargo nextest run -p acdc-cli -p acdc-execute --all-features --verbose --no-tests warn --test-threads 1 # we've added this until https://github.com/nextest-rs/nextest/pull/3553 gets merged + - name: Run execute-only tests + run: cargo nextest run -p acdc-cli -p acdc-execute --no-default-features --features execute --verbose --test-threads 1 + - name: Test execute with parser-only substitution features + run: cargo nextest run -p acdc-execute --no-default-features --features acdc-parser/pre-spec-subs -E 'test(substitution_follows_the_parser_feature_even_when_features_are_unified)' --verbose - name: Smoke-test backend-only CLI builds shell: bash run: | @@ -164,6 +171,7 @@ jobs: - name: Smoke-test tool-only CLI builds shell: bash run: | + cargo run -p acdc-cli --no-default-features --features execute -- execute --help cargo run -p acdc-cli --no-default-features --features inspect -- inspect --help cargo run -p acdc-cli --no-default-features --features lint -- lint --help cargo run -p acdc-cli --no-default-features --features tck -- tck --help @@ -220,3 +228,5 @@ jobs: - uses: Swatinem/rust-cache@v2 - name: Run doc tests run: cargo test --doc --all-features --verbose + - name: Run execute library doc tests + run: cargo test --doc -p acdc-execute --all-features --verbose diff --git a/Cargo.lock b/Cargo.lock index 968e59b1..4ee6b996 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -12,6 +12,7 @@ dependencies = [ "acdc-converters-markdown", "acdc-converters-pdf", "acdc-converters-terminal", + "acdc-execute", "acdc-lint", "acdc-parser", "chrono", @@ -20,6 +21,7 @@ dependencies = [ "miette", "open", "rayon", + "regex", "serde", "serde_json", "tempfile", @@ -155,6 +157,18 @@ dependencies = [ "web-sys", ] +[[package]] +name = "acdc-execute" +version = "0.1.0" +dependencies = [ + "acdc-converters-core", + "acdc-parser", + "petgraph", + "rstest", + "tempfile", + "thiserror", +] + [[package]] name = "acdc-lint" version = "0.1.0" @@ -1526,6 +1540,12 @@ version = "0.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" +[[package]] +name = "fixedbitset" +version = "0.5.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1d674e81391d1e1ab681a28d99df07927c6d4aa5b027d7da16ba32d1d21ecd99" + [[package]] name = "flate2" version = "1.1.9" @@ -1563,6 +1583,12 @@ version = "1.0.7" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3f9eec918d3f24069decb9af1554cad7c880e2da24a9afd88aca000531ab82c1" +[[package]] +name = "foldhash" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d9c4f5dac5e15c24eb999c26181a6ca40b39fe946cbe4c263c7209467bc83af2" + [[package]] name = "foldhash" version = "0.2.0" @@ -1778,7 +1804,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d99fc21d493812643aae86d53b7bbd02f376434a90317e8a790bc209fdd6605e" dependencies = [ "bytemuck", - "foldhash", + "foldhash 0.2.0", "hashbrown 0.17.1", "log", "peniko", @@ -1861,13 +1887,22 @@ version = "0.14.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e5274423e17b7c9fc20b6e7e208532f9b19825d82dfd615708b70edd83df41f1" +[[package]] +name = "hashbrown" +version = "0.15.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1" +dependencies = [ + "foldhash 0.1.5", +] + [[package]] name = "hashbrown" version = "0.17.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" dependencies = [ - "foldhash", + "foldhash 0.2.0", ] [[package]] @@ -3223,6 +3258,17 @@ version = "2.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" +[[package]] +name = "petgraph" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8701b58ea97060d5e5b155d383a69952a60943f0e6dfe30b04c287beb0b27455" +dependencies = [ + "fixedbitset", + "hashbrown 0.15.5", + "indexmap", +] + [[package]] name = "phf" version = "0.13.1" diff --git a/Cargo.toml b/Cargo.toml index 6c946909..e3281539 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -2,6 +2,7 @@ members = [ "acdc-cli", "acdc-editor-wasm", + "acdc-execute", "acdc-lint", "acdc-lsp", "acdc-parser", @@ -43,6 +44,8 @@ acdc-converters-manpage = { path = "converters/manpage", default-features = fals acdc-converters-markdown = { path = "converters/markdown" } acdc-converters-pdf = { path = "converters/pdf", default-features = false } acdc-converters-terminal = { path = "converters/terminal", default-features = false } +acdc-editor-wasm = { path = "acdc-editor-wasm" } +acdc-execute = { path = "acdc-execute", default-features = false } acdc-lint = { path = "acdc-lint", default-features = false } acdc-parser = { path = "acdc-parser", default-features = false } base64 = "0.23" @@ -54,6 +57,7 @@ image = { version = "0.25", default-features = false, features = ["png", "jpeg", libghostty-vt = { git = "https://github.com/Uzaaft/libghostty-rs", rev = "5988a0b78b4aa804d1c12e66bbfe662bd97d81c0" } lopdf = "0.45" mockito = "1" +petgraph = { version = "0.8", default-features = false, features = ["std"] } pretty_assertions = "1.4" rayon = "1.11" relative-path = "2.0.1" diff --git a/acdc-cli/CHANGELOG.md b/acdc-cli/CHANGELOG.md index 1c51173d..d3ec42f4 100644 --- a/acdc-cli/CHANGELOG.md +++ b/acdc-cli/CHANGELOG.md @@ -7,8 +7,17 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Changed + +- `execute --list` now prints one bullet per command with labeled ids, + interpreters, enclosing section titles, direct dependencies, and optional + descriptions. Commands outside sections omit the section title. Quoted + values escape control characters so each command stays on one line. + ### Fixed +- Warning batches read each referenced source file once, reducing repeated I/O + for documents with many warnings, including commands that use `subs=attributes`. - `lint -D one-sentence-per-line` now rejects description-list values with multiple sentences on one line or a sentence split across lines, including values written after the term's delimiter. @@ -98,6 +107,19 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 page sizes can use portrait or landscape output, `pdf-page-size` accepts custom dimensions, `pdf-page-margin` sets per-document margins, and `--strict` makes unresolved PDF images or logos fail instead of falling back with a warning. +- Builds with the `execute` feature run listing and source blocks marked with + the `command` role. Use `--id` or `--id-regex` to select commands and their + dependencies, `--dry-run` to inspect scripts, and `--list` to see descriptions. + `interpreter=` overrides the source language; `--cwd` and repeated `--env` + options configure child processes. +- Command execution preserves literal script text after includes and conditionals; + `subs=attributes` or `subs=+attributes` expands document attributes from each + command's source position without shell quoting or removing callouts. Missing + enabled attributes and incomplete command input fail before any process starts. + Dry-run shows the prepared script. Builds without `pre-spec-subs` reject explicit + `subs` settings. Failed commands block dependents while independent commands + continue; `--exit-on-failure` + stops sooner. Source diagnostics identify commands in included files. - The `terminal-emulator` build feature renders `[terminal]` session blocks through `libghostty-vt` on the `--backend terminal` path. Requires a Zig toolchain to build the bundled library, which is statically linked so the diff --git a/acdc-cli/Cargo.toml b/acdc-cli/Cargo.toml index 1a38295a..9c87d05d 100644 --- a/acdc-cli/Cargo.toml +++ b/acdc-cli/Cargo.toml @@ -17,6 +17,7 @@ doc = false [dependencies] acdc-converters-core.workspace = true +acdc-execute = { workspace = true, optional = true } acdc-converters-html = { workspace = true, optional = true } acdc-lint = { workspace = true, optional = true } acdc-parser.workspace = true @@ -30,6 +31,7 @@ crossterm.workspace = true miette = { version = "7", features = ["derive", "fancy"] } open = "5" rayon.workspace = true +regex = { version = "1", optional = true } serde.workspace = true serde_json.workspace = true thiserror.workspace = true @@ -52,6 +54,7 @@ pre-spec-subs = [ "acdc-converters-manpage?/pre-spec-subs", "acdc-converters-pdf?/pre-spec-subs", "acdc-converters-terminal?/pre-spec-subs", + "acdc-execute?/pre-spec-subs", "acdc-lint?/pre-spec-subs", ] @@ -81,16 +84,17 @@ terminal-emulator = [ ] # Development tools +execute = ["dep:acdc-execute", "dep:regex"] # Run command blocks from AsciiDoc documents inspect = [] # AST inspection tool for debugging parser output tck = [] # Technology Compatibility Kit for spec compliance testing # Compatibility features -setext = ["acdc-parser/setext", "acdc-lint?/setext"] # Enable Setext-style (underlined) header parsing -network = ["acdc-parser/network", "acdc-lint?/network", "acdc-converters-pdf?/network"] # Enable remote include parsing and remote PDF images +setext = ["acdc-parser/setext", "acdc-execute?/setext", "acdc-lint?/setext"] # Enable Setext-style (underlined) header parsing +network = ["acdc-parser/network", "acdc-execute?/network", "acdc-lint?/network", "acdc-converters-pdf?/network"] # Enable remote include parsing and remote PDF images # Feature groups for convenience all-backends = ["html", "manpage", "markdown", "pdf", "terminal"] # Enable all output converters -dev-tools = ["inspect"] # Enable development utilities +dev-tools = ["execute", "inspect"] # Enable development utilities test-tools = ["tck"] # Enable testing and compliance tools [lints.rust] diff --git a/acdc-cli/README.adoc b/acdc-cli/README.adoc index a5ea3ee3..95c62355 100644 --- a/acdc-cli/README.adoc +++ b/acdc-cli/README.adoc @@ -1,7 +1,7 @@ = `acdc` command-line interface `acdc` is the command-line entry point for the acdc AsciiDoc toolchain. -It converts documents, runs project lints, displays a structural parser outline, and provides the AsciiDoc TCK adapter. +It converts documents, runs project lints and command blocks, displays a structural parser outline, and provides the AsciiDoc TCK adapter. Use `acdc --help` for the exhaustive option reference. This guide focuses on common workflows and build-time features. @@ -161,6 +161,109 @@ cargo run -p acdc-cli --features inspect -- inspect document.adoc --show-locatio `--max-depth` limits displayed nesting. Color is used only on interactive stdout and respects `NO_COLOR`. +=== `execute` + +Build with `execute` to run *command blocks* defined in an AsciiDoc document. +A command block is a listing or source block carrying the `command` role and an explicit `id`. +Ids contain only ASCII letters, digits, `_`, and `-`; they must be nonempty and cannot start with `-`. +The optional `deps` attribute names prerequisites, separated by commas. +The optional `description` attribute supplies a summary for `--list`. + +[source,asciidoc] +.... +== Build + +[.command, id=build, description="Build the project"] +[source, bash] +---- +cargo build +---- + +== Test + +[.command, id=test, deps="build", description="Run the tests"] +[source, bash] +---- +cargo nextest run +---- +.... + +[source,console] +.... +acdc execute --dry-run README.adoc +acdc execute --list README.adoc +acdc execute --id build README.adoc +acdc execute --id-regex '^test-' --exit-on-failure README.adoc +acdc execute --id test --cwd project --env MODE=ci README.adoc +.... + +Commands are discovered in document order, including inside nested containers, AsciiDoc table cells, and included files, then executed in dependency order. +Exact `--id` and `--id-regex` selectors form a union, and selected commands always run together with their transitive dependencies; each command runs at most once. +Without selectors, every command is selected. +An unknown id or a regex that matches no commands is an error. +`--dry-run` prints the selected commands and scripts; `--list` prints a bulleted list with labeled ids, interpreters, enclosing section titles, direct dependencies, and optional descriptions. +Both include dependencies and run no commands; the two options cannot be combined. + +For the example above, `--list` prints: + +[source,text] +.... +- id=build, interpreter="bash", section="Build", description="Build the project" +- id=test, interpreter="bash", section="Test", deps="build", description="Run the tests" +.... + +Dependency ids are sorted alphabetically. +The section is the nearest enclosing section's title as plain text; commands outside sections omit it. +Interpreters, section titles, and descriptions are quoted, with control characters escaped to keep each command on one line. + +The block attribute `interpreter` overrides the source language: `[source,python,role=command,id=check,interpreter=python3]` uses Python highlighting and runs `python3`. +Without an override, the source language selects the interpreter; a block without either uses `sh`. +The interpreter is one executable name or path, passed directly to the operating system without shell argument splitting or an allowlist. +Relative interpreter paths are resolved from the caller's working directory, including when `--cwd` is set; bare executable names use the operating system's search rules. + +Includes and conditionals are processed before execution. +Preprocessing normalizes line endings and removes trailing whitespace from lines. +Script bodies are literal by default, including attribute references and callout markers. +Enable attribute expansion with `subs=attributes`, `subs=+attributes`, or `subs=attributes+` on a command block. +The `normal` group also enables attributes; `subs=none` and `subs=-attributes` keep them literal. +Only enabled attribute references are expanded; other rendering substitutions do not change the script, and callouts remain literal. +Use comments in the script language for annotations. + +[source,asciidoc] +.... +:target: build-output + +[.command,id=show-target,subs=+attributes] +---- +printf '%s\n' '{target}' +---- +.... + +Attribute values are inserted as text without shell quoting; write the quoting required by the selected interpreter. +When attribute expansion is enabled, `\{name}` becomes literal `{name}` without looking up the attribute. +An empty attribute value is valid; a missing unescaped attribute is an error. +Each command uses the attributes active at its source position. +Attributes defined inside an AsciiDoc table cell stay in that cell; values inherited by the cell cannot be reassigned there. +All scripts are prepared before selection or execution, so a missing attribute in an unselected command also stops the invocation. +Dry-run displays the same prepared script that execution passes to the interpreter. +Malformed command blocks, incomplete structural recovery, and missing required includes are errors before any command starts. + +A command that exits with a nonzero status blocks its dependent commands; independent commands still run. +`--exit-on-failure` stops at the first failed command. +A signal termination or an error creating the script or starting its interpreter always stops the run. +Any failed command makes the invocation fail, including when independent commands finish successfully. + +Each command inherits the caller's working directory, environment, and standard streams. +`--cwd PATH` changes the child working directory; relative paths are resolved from the caller's directory. +It does not change how the input document or its includes are resolved. +Repeat `--env NAME=VALUE` to override variables for the children; the last value for a name wins. +An empty value is allowed, and only the first `=` separates the name from its value. +These options leave the calling process unchanged. +`-S`/`--safe-mode` limits what the document may include while parsing; it does not sandbox the executed commands. +A build with the `execute` feature only is available via `cargo build -p acdc-cli --no-default-features --features execute`. +Add `pre-spec-subs` to that feature list to enable command attribute substitutions. +Without this feature, documents that request any `subs` setting are rejected before execution. + === `tck` The `tck` feature provides the JSON stdin/stdout adapter used by the AsciiDoc Test Compatibility Kit. @@ -213,12 +316,16 @@ echo '{"contents":"= Hello","path":"test.adoc","type":"block"}' \ | no | Remote includes and remote PDF assets when PDF is selected. +| `execute` +| no +| `acdc execute` command: run command blocks defined in AsciiDoc documents. + | `inspect`, `tck` | no | Developer/specification commands. |=== -Convenience groups are `all-backends`, `dev-tools` (`inspect`), and `test-tools` (`tck`). +Convenience groups are `all-backends`, `dev-tools` (`execute`, `inspect`), and `test-tools` (`tck`). A build with no command-enabling features exits with a diagnostic that lists the available command features. == Diagnostics and exit status diff --git a/acdc-cli/src/error.rs b/acdc-cli/src/error.rs index eb322611..07433afe 100644 --- a/acdc-cli/src/error.rs +++ b/acdc-cli/src/error.rs @@ -4,8 +4,19 @@ // https://github.com/zkat/miette/pull/459 for more details. #![allow(unused_assignments)] -use std::path::Path; - +use std::{ + collections::HashMap, + path::{Path, PathBuf}, + sync::Arc, +}; + +#[cfg(any( + feature = "html", + feature = "manpage", + feature = "markdown", + feature = "pdf", + feature = "terminal" +))] use acdc_converters_core::Warning as ConverterWarning; #[cfg(feature = "lint")] use acdc_lint::{LintDiagnostic, LintLevel}; @@ -14,7 +25,14 @@ use miette::{Diagnostic, NamedSource, Report, SourceSpan}; #[cfg(feature = "lint")] use miette::{LabeledSpan, Severity}; -/// Rich error wrapper for beautiful miette display with source code +/// An error with its source snippet. +#[cfg(any( + feature = "html", + feature = "manpage", + feature = "markdown", + feature = "pdf", + feature = "terminal" +))] #[derive(Debug, Diagnostic, thiserror::Error)] #[error("{message}")] #[diagnostic()] @@ -32,6 +50,115 @@ pub(crate) struct RichError { position_advice: String, } +#[cfg(feature = "execute")] +#[derive(Debug, Diagnostic, thiserror::Error)] +#[error("{error}")] +pub(crate) struct LocatedError { + #[source] + error: E, + #[related] + locations: Vec, +} + +#[cfg(feature = "execute")] +impl LocatedError { + pub(crate) fn new(error: E, locations: impl IntoIterator) -> Self { + Self { + error, + locations: locations + .into_iter() + .map(|location| SourceContext::new(&location)) + .collect(), + } + } + + pub(crate) fn with_cached_source( + error: E, + location: &SourceLocation, + sources: &mut SourceCache, + ) -> Self { + let content = location.file.as_deref().and_then(|path| sources.load(path)); + Self { + error, + locations: vec![SourceContext::with_content(location, content)], + } + } + + pub(crate) fn without_source(error: E, location: &SourceLocation) -> Self { + Self { + error, + locations: vec![SourceContext::with_content(location, None)], + } + } +} + +#[derive(Debug, Default)] +pub(crate) struct SourceCache { + sources: HashMap>>, +} + +impl SourceCache { + fn load(&mut self, path: &Path) -> Option> { + if let Some(source) = self.sources.get(path) { + return source.clone(); + } + let source = std::fs::read_to_string(path).ok().map(Arc::new); + self.sources.insert(path.to_owned(), source.clone()); + source + } +} + +#[cfg(feature = "execute")] +#[derive(Debug, Diagnostic, thiserror::Error)] +#[error("{description}")] +struct SourceContext { + description: String, + #[source_code] + src: Option>>, + #[label("command source")] + span: Option, +} + +#[cfg(feature = "execute")] +impl SourceContext { + fn new(location: &SourceLocation) -> Self { + let content = location + .file + .as_deref() + .and_then(|path| std::fs::read_to_string(path).ok()) + .map(Arc::new); + Self::with_content(location, content) + } + + fn with_content(location: &SourceLocation, content: Option>) -> Self { + let (line, column) = source_location_line_column(location); + let mut description = match &location.file { + Some(path) => format!("{}:{line}:{column}", path.display()), + None => format!("line {line}, column {column}"), + }; + if let Some(chain) = location + .location + .start + .file + .as_deref() + .filter(|chain| !chain.is_empty()) + { + description.push_str(" (include chain: "); + description.push_str(&chain.join(" -> ")); + description.push(')'); + } + let span = content + .as_deref() + .map(|source| source_span_from_source_location(location, source)); + let src = content.map(|source| NamedSource::new(description.clone(), source)); + Self { + description, + src, + span, + } + } +} + /// Rich warning wrapper: same shape as `RichError` but with `severity = Warning` so /// miette renders it in the warning palette (yellow, `⚠`) rather than the error palette /// (red, `✗`). @@ -45,7 +172,7 @@ pub(crate) struct RichWarning { advice: Option, #[source_code] - src: NamedSource, + src: NamedSource>, #[label("{position_advice}")] span: SourceSpan, @@ -64,16 +191,24 @@ pub(crate) struct PlainWarning { advice: Option, } -#[derive(Debug, Clone, Copy)] +#[derive(Debug, Default)] pub(crate) struct WarningReportContext<'a> { file: Option<&'a Path>, + sources: SourceCache, } impl<'a> WarningReportContext<'a> { - pub(crate) const fn new() -> Self { - Self { file: None } + pub(crate) fn new() -> Self { + Self::default() } + #[cfg(any( + feature = "html", + feature = "manpage", + feature = "markdown", + feature = "pdf", + feature = "terminal" + ))] pub(crate) const fn with_optional_file(mut self, file: Option<&'a Path>) -> Self { self.file = file; self @@ -81,7 +216,7 @@ impl<'a> WarningReportContext<'a> { } pub(crate) trait WarningReport { - fn to_report(&self, context: WarningReportContext<'_>) -> Report; + fn to_report(&self, context: &mut WarningReportContext<'_>) -> Report; } #[cfg(feature = "lint")] @@ -238,22 +373,17 @@ fn offset_from_position(source: &str, line: u32, column: u32) -> usize { source.len() } -/// Build a miette `Report` from extracted warning fields. -/// -/// Tries to load the referenced source file and produce a `RichWarning` with a -/// span/snippet; falls back to a source-less `PlainWarning` when the warning has -/// no location, no file path, or the file can't be read. `fallback_file` -/// (typically the file being processed) is used when the warning's own -/// `SourceLocation` has no path; pass `None` for stdin input. +/// Attach a source snippet when the warning's file or the context's fallback is readable. +/// Warnings in one context share source text; missing files remain source-less. fn build_warning_report( message: String, advice: Option, location: Option<&SourceLocation>, - fallback_file: Option<&Path>, + context: &mut WarningReportContext<'_>, ) -> Report { let rich = location.and_then(|loc| { - let path = loc.file.as_deref().or(fallback_file)?; - let source_str = std::fs::read_to_string(path).ok()?; + let path = loc.file.as_deref().or(context.file)?; + let source_str = context.sources.load(path)?; let span = source_span_from_source_location(loc, &source_str); let (line, column) = source_location_line_column(loc); Some(RichWarning { @@ -272,23 +402,30 @@ fn build_warning_report( } impl WarningReport for ParserWarning { - fn to_report(&self, context: WarningReportContext<'_>) -> Report { + fn to_report(&self, context: &mut WarningReportContext<'_>) -> Report { build_warning_report( self.kind.to_string(), self.advice().map(str::to_string), self.source_location(), - context.file, + context, ) } } +#[cfg(any( + feature = "html", + feature = "manpage", + feature = "markdown", + feature = "pdf", + feature = "terminal" +))] impl WarningReport for ConverterWarning { - fn to_report(&self, context: WarningReportContext<'_>) -> Report { + fn to_report(&self, context: &mut WarningReportContext<'_>) -> Report { build_warning_report( self.to_string(), self.advice().map(str::to_string), self.source_location(), - context.file, + context, ) } } @@ -355,6 +492,13 @@ impl LintDiagnosticReport for LintDiagnostic { } } +#[cfg(any( + feature = "html", + feature = "manpage", + feature = "markdown", + feature = "pdf", + feature = "terminal" +))] pub(crate) fn display(e: &E) -> Report { if let Some(parser_error) = acdc_converters_core::find_parser_error(e) && let Some(source_location) = parser_error.source_location() @@ -394,6 +538,41 @@ mod tests { use super::*; + #[test] + fn warning_reports_reuse_the_first_source_snapshot() -> Result<(), Box> { + let file = tempfile::NamedTempFile::new()?; + std::fs::write(file.path(), "original source\n")?; + let location = + SourceLocation::at_position(Some(file.path().to_owned()), Position::new(1, 1)); + let mut context = WarningReportContext::new(); + let first = build_warning_report("first".into(), None, Some(&location), &mut context); + std::fs::write(file.path(), "changed source\n")?; + let second = build_warning_report("second".into(), None, Some(&location), &mut context); + let first = first + .downcast_ref::() + .ok_or("missing first source")?; + let second = second + .downcast_ref::() + .ok_or("missing second source")?; + assert_eq!(second.src.inner().as_str(), "original source\n"); + assert!(Arc::ptr_eq(first.src.inner(), second.src.inner())); + Ok(()) + } + + #[test] + fn warning_reports_cache_unreadable_sources() -> Result<(), Box> { + let directory = tempfile::tempdir()?; + let path = directory.path().join("missing.adoc"); + let location = SourceLocation::at_position(Some(path.clone()), Position::new(1, 1)); + let mut context = WarningReportContext::new(); + let first = build_warning_report("first".into(), None, Some(&location), &mut context); + std::fs::write(path, "created later\n")?; + let second = build_warning_report("second".into(), None, Some(&location), &mut context); + assert!(first.downcast_ref::().is_some()); + assert!(second.downcast_ref::().is_some()); + Ok(()) + } + #[test] fn resolves_unicode_columns_to_byte_offsets() { assert_eq!(offset_from_position("éx\nnext", 1, 1), 0); @@ -443,4 +622,78 @@ mod tests { assert_eq!(span.offset(), 6); assert_eq!(span.len(), 3); } + + #[cfg(feature = "execute")] + #[test] + fn unresolved_command_sources_keep_the_include_chain() { + let mut position = Position::new(7, 3); + position.file = Some(Arc::new(vec![ + "outer.adoc".into(), + "nested/inner.adoc".into(), + ])); + let context = SourceContext::new(&SourceLocation::at_position(None, position)); + + assert_eq!( + context.description, + "line 7, column 3 (include chain: outer.adoc -> nested/inner.adoc)" + ); + assert!(context.src.is_none()); + assert!(context.span.is_none()); + } + + #[cfg(feature = "execute")] + #[test] + fn skipped_command_diagnostics_do_not_load_source_text() + -> Result<(), Box> { + let file = tempfile::NamedTempFile::new()?; + std::fs::write(file.path(), "source must not be retained")?; + let location = + SourceLocation::at_position(Some(file.path().to_owned()), Position::new(1, 1)); + let report = LocatedError::without_source(std::io::Error::other("skipped"), &location); + let context = report + .locations + .first() + .ok_or("missing source coordinates")?; + + assert!( + context + .description + .contains(file.path().to_string_lossy().as_ref()) + ); + assert!(context.src.is_none()); + assert!(context.span.is_none()); + Ok(()) + } + + #[cfg(feature = "execute")] + #[test] + fn execution_failures_share_one_source_allocation_per_file() + -> Result<(), Box> { + let file = tempfile::NamedTempFile::new()?; + std::fs::write(file.path(), "first\nsecond\n")?; + let location = + SourceLocation::at_position(Some(file.path().to_owned()), Position::new(1, 1)); + let mut cache = SourceCache::default(); + let first = + LocatedError::with_cached_source(std::io::Error::other("first"), &location, &mut cache); + let second = LocatedError::with_cached_source( + std::io::Error::other("second"), + &location, + &mut cache, + ); + let first = first + .locations + .first() + .and_then(|context| context.src.as_ref()) + .ok_or("missing first source")?; + let second = second + .locations + .first() + .and_then(|context| context.src.as_ref()) + .ok_or("missing second source")?; + + assert!(Arc::ptr_eq(first.inner(), second.inner())); + assert_eq!(cache.sources.len(), 1); + Ok(()) + } } diff --git a/acdc-cli/src/main.rs b/acdc-cli/src/main.rs index 52624876..fc1bc0a5 100644 --- a/acdc-cli/src/main.rs +++ b/acdc-cli/src/main.rs @@ -4,6 +4,7 @@ feature = "markdown", feature = "pdf", feature = "terminal", + feature = "execute", feature = "inspect", feature = "lint", feature = "tck", @@ -16,6 +17,7 @@ use clap::{CommandFactory, FromArgMatches, Parser, Subcommand}; feature = "markdown", feature = "pdf", feature = "terminal", + feature = "execute", feature = "lint" ))] mod error; @@ -35,6 +37,7 @@ mod timing; feature = "markdown", feature = "pdf", feature = "terminal", + feature = "execute", feature = "inspect", feature = "lint", feature = "tck", @@ -53,6 +56,7 @@ struct Cli { feature = "markdown", feature = "pdf", feature = "terminal", + feature = "execute", feature = "inspect", feature = "lint", feature = "tck", @@ -69,6 +73,10 @@ enum Commands { /// Convert `AsciiDoc` documents to various output formats Convert(subcommands::convert::Args), + #[cfg(feature = "execute")] + /// Execute command blocks defined in `AsciiDoc` documents + Execute(subcommands::execute::Args), + #[cfg(feature = "inspect")] /// Show a structural outline of an `AsciiDoc` document Inspect(subcommands::inspect::Args), @@ -104,6 +112,7 @@ fn setup_logging() { feature = "markdown", feature = "pdf", feature = "terminal", + feature = "execute", feature = "inspect", feature = "lint", feature = "tck", @@ -127,6 +136,8 @@ fn main() { feature = "terminal" ))] Commands::Convert(_) => true, + #[cfg(feature = "execute")] + Commands::Execute(_) => true, #[cfg(feature = "inspect")] Commands::Inspect(_) => true, #[cfg(feature = "tck")] @@ -142,6 +153,9 @@ fn main() { ))] Commands::Convert(args) => subcommands::convert::run(&args), + #[cfg(feature = "execute")] + Commands::Execute(args) => subcommands::execute::run(&args), + #[cfg(feature = "inspect")] Commands::Inspect(args) => { subcommands::inspect::run(&args).map_err(|e| miette::miette!("Inspect failed: {e}")) @@ -188,6 +202,7 @@ fn main() { feature = "markdown", feature = "pdf", feature = "terminal", + feature = "execute", feature = "inspect", feature = "lint", feature = "tck", @@ -196,7 +211,7 @@ fn main() { setup_logging(); eprintln!( "acdc was built without any subcommand features. Enable at least \ - one of: html, manpage, markdown, pdf, terminal, inspect, lint, tck." + one of: html, manpage, markdown, pdf, terminal, execute, inspect, lint, tck." ); std::process::exit(2); } diff --git a/acdc-cli/src/subcommands/convert.rs b/acdc-cli/src/subcommands/convert.rs index 33fa1fce..e7ab3d8a 100644 --- a/acdc-cli/src/subcommands/convert.rs +++ b/acdc-cli/src/subcommands/convert.rs @@ -690,18 +690,18 @@ trait WarningRenderer { impl WarningRenderer for [Warning] { fn render(&self, context: WarningRenderContext<'_>) { - let context = WarningReportContext::new().with_optional_file(context.file); + let mut context = WarningReportContext::new().with_optional_file(context.file); for warning in self { - eprintln!("{:?}", warning.to_report(context)); + eprintln!("{:?}", warning.to_report(&mut context)); } } } impl WarningRenderer for [acdc_converters_core::Warning] { fn render(&self, context: WarningRenderContext<'_>) { - let context = WarningReportContext::new().with_optional_file(context.file); + let mut context = WarningReportContext::new().with_optional_file(context.file); for warning in self { - eprintln!("{:?}", warning.to_report(context)); + eprintln!("{:?}", warning.to_report(&mut context)); } } } @@ -734,6 +734,8 @@ mod tests { let cli = crate::Cli::try_parse_from(raw).map_err(|error| miette::miette!(error))?; match cli.command { Commands::Convert(args) => Ok(args), + #[cfg(feature = "execute")] + Commands::Execute(_) => Err(miette::miette!("test command selected execute")), #[cfg(feature = "inspect")] Commands::Inspect(_) => Err(miette::miette!("test command selected inspect")), #[cfg(feature = "lint")] diff --git a/acdc-cli/src/subcommands/execute.rs b/acdc-cli/src/subcommands/execute.rs new file mode 100644 index 00000000..39f3bf4c --- /dev/null +++ b/acdc-cli/src/subcommands/execute.rs @@ -0,0 +1,644 @@ +//! Execute command blocks from `AsciiDoc` files. +//! +//! Commands are listing or source blocks set with the `command` role. + +use std::{ + ffi::OsString, + io::{self, Write}, + path::PathBuf, +}; + +use acdc_execute::{ + CommandGraph, CommandId, CommandState, ExecError, ExecutionOptions, ExecutionPlan, + ExecutionReport, ProcessOptions, SkipReason, +}; +use acdc_parser::{ParseResult, SafeMode}; +use clap::{ArgAction, Args as ClapArgs}; +use miette::{Diagnostic, IntoDiagnostic as _}; +use regex::Regex; + +use crate::error::{LocatedError, SourceCache, WarningReport, WarningReportContext}; + +/// Execute command blocks defined in an `AsciiDoc` file +#[derive(ClapArgs, Debug)] +pub struct Args { + /// Input `AsciiDoc` file + pub file: PathBuf, + + /// Select command blocks whose id exactly matches this value + #[arg(long = "id", value_name = "ID", action = ArgAction::Append)] + pub ids: Vec, + + /// Select command blocks whose id matches this regex + #[arg(long = "id-regex", value_name = "REGEX", action = ArgAction::Append)] + pub id_regexes: Vec, + + /// Print selected commands and scripts instead of running them + #[arg(long, conflicts_with = "list")] + pub dry_run: bool, + + /// List command ids, interpreters, sections, dependencies, and descriptions without running them + #[arg(long)] + pub list: bool, + + /// Working directory for child processes (defaults to the caller's directory) + #[arg(long, value_name = "DIRECTORY")] + pub cwd: Option, + + /// Override a child environment variable; repeat to set more values + #[arg(long = "env", value_name = "NAME=VALUE", value_parser = parse_environment, action = ArgAction::Append)] + pub environment: Vec<(OsString, OsString)>, + + /// Stop all commands at the first unsuccessful child exit + #[arg(long)] + pub exit_on_failure: bool, + + /// Safe mode to use while parsing the document + /// + /// This limits document reads and includes; it does not sandbox commands. + #[arg(short = 'S', long, value_parser = clap::value_parser!(SafeMode), default_value = "safe")] + pub safe_mode: SafeMode, +} + +fn parse_environment(value: &str) -> Result<(OsString, OsString), String> { + let (name, value) = value.split_once('=').ok_or("expected NAME=VALUE")?; + if name.is_empty() || name.contains('\0') || value.contains('\0') { + return Err( + "environment names must be nonempty and names and values must not contain NUL".into(), + ); + } + Ok((name.into(), value.into())) +} + +pub fn run(args: &Args) -> miette::Result<()> { + let parser_options = acdc_parser::Options::builder() + .with_safe_mode(args.safe_mode) + .build() + .map_err(parser_report)?; + let parsed = acdc_parser::parse_file(&args.file, &parser_options).map_err(parser_report)?; + let graph = CommandGraph::try_from(&parsed).map_err(|error| { + let locations = error + .source_location() + .into_iter() + .chain(error.related_location()) + .cloned() + .collect::>(); + miette::Report::new(LocatedError::new(error, locations)) + })?; + report_warnings(&parsed)?; + drop(parsed); + let selected = select(&graph, &args.ids, &args.id_regexes)?; + + if args.dry_run || args.list { + let mut stdout = io::stdout().lock(); + if args.list { + write_list(&mut stdout, &graph, &selected).into_diagnostic()?; + } else { + write_plan(&mut stdout, &selected).into_diagnostic()?; + } + stdout.flush().into_diagnostic()?; + } else { + let report = selected.execute(&ExecutionOptions { + exit_on_failure: args.exit_on_failure, + process: ProcessOptions { + current_dir: args.cwd.clone(), + env: args.environment.clone(), + }, + }); + report_execution(report)?; + } + Ok(()) +} + +fn parser_report(error: acdc_parser::Error) -> miette::Report { + let location = error.source_location().cloned(); + miette::Report::new(LocatedError::new(error, location)) +} + +fn report_warnings(parsed: &ParseResult) -> miette::Result<()> { + let mut context = WarningReportContext::new(); + let mut stderr = io::stderr().lock(); + for warning in parsed.warnings() { + writeln!(stderr, "{:?}", warning.to_report(&mut context)).into_diagnostic()?; + } + Ok(()) +} + +/// Selectors form a union; each must match, and selected commands include their prerequisites. +fn select<'graph>( + graph: &'graph CommandGraph, + ids: &[String], + id_regexes: &[Regex], +) -> miette::Result> { + if ids.is_empty() && id_regexes.is_empty() { + return Ok(graph.plan_all()); + } + let mut selected = Vec::new(); + for id in ids { + let id = CommandId::new(id) + .map_err(|error| miette::miette!("invalid --id value {id:?}: {error}"))?; + if !graph.contains(id.as_str()) { + return Err(miette::miette!("unknown command id: {id}")); + } + selected.push(id); + } + for regex in id_regexes { + let previous_count = selected.len(); + selected.extend( + graph + .ids() + .filter(|id| regex.is_match(id.as_str())) + .cloned(), + ); + if selected.len() == previous_count { + return Err(miette::miette!( + "--id-regex `{}` matched no commands", + regex.as_str() + )); + } + } + graph.plan_for(&selected).into_diagnostic() +} + +fn write_plan(output: &mut impl Write, plan: &ExecutionPlan<'_>) -> io::Result<()> { + for block in plan.commands() { + writeln!( + output, + "{} ({})", + block.metadata.id, block.metadata.interpreter + )?; + for line in block.script.split_inclusive('\n') { + write!(output, " {line}")?; + } + if !block.script.ends_with('\n') { + writeln!(output)?; + } + } + Ok(()) +} + +fn write_list( + output: &mut impl Write, + graph: &CommandGraph, + plan: &ExecutionPlan<'_>, +) -> io::Result<()> { + for block in plan.commands() { + write!( + output, + "- id={}, interpreter={:?}", + block.metadata.id, block.metadata.interpreter + )?; + if let Some(section) = &block.metadata.section_title { + write!(output, ", section={section:?}")?; + } + let mut dependencies = graph + .dependencies(block.metadata.id.as_str()) + .into_iter() + .flatten() + .map(CommandId::as_str) + .collect::>(); + dependencies.sort_unstable(); + if !dependencies.is_empty() { + write!(output, ", deps=\"")?; + for (index, dependency) in dependencies.iter().enumerate() { + if index > 0 { + write!(output, ",")?; + } + write!(output, "{dependency}")?; + } + write!(output, "\"")?; + } + if let Some(description) = &block.metadata.description { + write!(output, ", description={description:?}")?; + } + writeln!(output)?; + } + Ok(()) +} + +#[derive(Debug, thiserror::Error)] +enum CommandIssue { + #[error("command `{id}` {error}")] + Failed { + id: CommandId, + #[source] + error: ExecError, + }, + #[error("command `{id}` skipped because prerequisite `{dependency}` did not succeed")] + Dependency { + id: CommandId, + dependency: CommandId, + }, + #[error("command `{id}` skipped after failure of `{command}`")] + Stopped { id: CommandId, command: CommandId }, +} + +#[derive(Debug, Diagnostic, thiserror::Error)] +#[error("command execution failed ({failed} failed, {skipped} skipped)")] +struct ExecutionFailed { + failed: usize, + skipped: usize, + #[related] + issues: Vec>, +} + +fn report_execution(report: ExecutionReport<'_>) -> miette::Result<()> { + if report.is_success() { + return Ok(()); + } + let mut failed = 0; + let mut skipped = 0; + let mut issues = Vec::new(); + let mut sources = SourceCache::default(); + for outcome in report.into_outcomes() { + let diagnostic = match outcome.state { + CommandState::Succeeded => continue, + CommandState::Failed(error) => { + failed += 1; + LocatedError::with_cached_source( + CommandIssue::Failed { + id: outcome.command.metadata.id.clone(), + error, + }, + &outcome.command.location, + &mut sources, + ) + } + CommandState::Skipped(reason) => { + skipped += 1; + let id = outcome.command.metadata.id.clone(); + let issue = match reason { + SkipReason::DependencyFailed { dependency } => CommandIssue::Dependency { + id, + dependency: dependency.clone(), + }, + SkipReason::StoppedAfterFailure { command } => CommandIssue::Stopped { + id, + command: command.clone(), + }, + }; + LocatedError::without_source(issue, &outcome.command.location) + } + }; + issues.push(diagnostic); + } + Err(miette::Report::new(ExecutionFailed { + failed, + skipped, + issues, + })) +} + +#[cfg(test)] +#[allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] +mod tests { + use std::path::PathBuf; + + use acdc_parser::{Block, DelimitedBlockType, Options, SafeMode}; + use clap::Parser; + use regex::Regex; + + use super::{Args, select, write_list, write_plan}; + use acdc_execute::{CommandBlock, CommandGraph, CommandGraphBuilder}; + + #[derive(Parser)] + struct TestCli { + #[command(flatten)] + args: Args, + } + + fn parse_graph(src: &str) -> CommandGraph { + let parsed = acdc_parser::parse(src, &Options::default()) + .unwrap_or_else(|error| panic!("parse failed: {error}")); + CommandGraph::try_from(&parsed) + .unwrap_or_else(|error| panic!("graph should build: {error}")) + } + + fn select_ids(graph: &CommandGraph, ids: &[&str], regexes: &[&str]) -> Vec { + let regexes: Vec = regexes + .iter() + .map(|r| Regex::new(r).unwrap_or_else(|e| panic!("test regex {r:?}: {e}"))) + .collect(); + let ids: Vec = ids.iter().map(ToString::to_string).collect(); + select(graph, &ids, ®exes) + .unwrap_or_else(|error| panic!("selection should succeed: {error}")) + .commands() + .map(|block| block.metadata.id.as_str().to_string()) + .collect() + } + + fn cmd(id: &str, deps: Option<&str>, body: &str) -> String { + let deps = deps.map(|d| format!(", deps=\"{d}\"")).unwrap_or_default(); + format!("[.command, id={id}{deps}]\n----\n{body}\n----\n") + } + + #[test] + fn parses_execute_flags() { + let cli = TestCli::parse_from([ + "test", + "README.adoc", + "--id", + "build", + "--id", + "test", + "--id-regex", + "^deploy-", + "--dry-run", + "--exit-on-failure", + ]); + + assert_eq!(cli.args.file, PathBuf::from("README.adoc")); + assert_eq!(cli.args.ids, ["build", "test"]); + assert_eq!(cli.args.id_regexes.len(), 1); + assert_eq!( + cli.args.id_regexes.first().map(regex::Regex::as_str), + Some("^deploy-") + ); + assert!(cli.args.dry_run); + assert!(cli.args.exit_on_failure); + assert_eq!(cli.args.safe_mode, SafeMode::Safe); + } + + #[test] + fn rejects_invalid_id_regex() { + let err = TestCli::try_parse_from(["test", "README.adoc", "--id-regex", "["]); + assert!(err.is_err()); + } + + #[test] + fn dry_run_preserves_trailing_blank_lines() { + let block = CommandBlock::new( + "build".parse().unwrap_or_else(|e| panic!("{e}")), + "echo hello\n\n".into(), + None, + acdc_parser::SourceLocation::at_location(None, acdc_parser::Location::default()), + ); + let mut builder = CommandGraphBuilder::new(); + builder.add(block, Vec::new()); + let graph = builder.build().unwrap(); + let mut output = Vec::new(); + write_plan(&mut output, &graph.plan_all()).unwrap(); + assert_eq!(output, b"build (sh)\n echo hello\n \n"); + } + + #[test] + fn parses_the_command_block_shape() -> miette::Result<()> { + let input = "[.command, id=build]\n----\necho hello\n----\n"; + let parsed = acdc_parser::parse(input, &Options::default()) + .map_err(|error| miette::miette!(error.to_string()))?; + let Some(Block::DelimitedBlock(block)) = parsed.document().blocks.first() else { + return Err(miette::miette!( + "expected command markup to parse as a delimited block" + )); + }; + + assert!(matches!( + block.inner, + DelimitedBlockType::DelimitedListing(_) + )); + assert_eq!(block.metadata.roles, ["command"]); + assert_eq!( + block.metadata.id.as_ref().map(|anchor| anchor.id), + Some("build") + ); + + Ok(()) + } + + // ------------------------------------------------------------------ + // Selection + // ------------------------------------------------------------------ + + #[test] + fn no_selectors_selects_all_commands_in_execution_order() { + let src = format!( + "{}\n{}\n{}", + cmd("c", Some("b"), "echo c"), + cmd("a", None, "echo a"), + cmd("b", Some("a"), "echo b"), + ); + let graph = parse_graph(&src); + assert_eq!(select_ids(&graph, &[], &[]), ["a", "b", "c"]); + } + + #[test] + fn exact_id_selects_one_command() { + let src = format!("{}\n{}", cmd("a", None, "echo a"), cmd("b", None, "echo b")); + let graph = parse_graph(&src); + assert_eq!(select_ids(&graph, &["b"], &[]), ["b"]); + } + + #[test] + fn regex_selects_matching_commands() { + let src = format!( + "{}\n{}\n{}", + cmd("test-unit", None, "echo 1"), + cmd("test-integration", None, "echo 2"), + cmd("build", None, "echo 3"), + ); + let graph = parse_graph(&src); + assert_eq!( + select_ids(&graph, &[], &["^test-"]), + ["test-unit", "test-integration"] + ); + } + + #[test] + fn exact_and_regex_selectors_form_a_union() { + let src = format!( + "{}\n{}\n{}\n{}", + cmd("build", None, "echo build"), + cmd("deploy-staging", Some("build"), "echo staging"), + cmd("deploy-prod", Some("build"), "echo prod"), + cmd("docs", None, "echo docs"), + ); + let graph = parse_graph(&src); + let mut ids = select_ids(&graph, &["docs"], &["^deploy-"]); + ids.sort(); + // `build` is pulled in as a transitive dependency of both deploy targets. + assert_eq!(ids, ["build", "deploy-prod", "deploy-staging", "docs"]); + } + + #[test] + fn selection_includes_transitive_dependencies() { + let src = format!( + "{}\n{}\n{}", + cmd("gen", None, "echo gen"), + cmd("build", Some("gen"), "echo build"), + cmd("other", None, "echo other"), + ); + let graph = parse_graph(&src); + assert_eq!(select_ids(&graph, &["build"], &[]), ["gen", "build"]); + } + + #[test] + fn each_command_is_selected_only_once() { + let src = format!( + "{}\n{}", + cmd("deploy-x", None, "echo x"), + cmd("build", None, "echo b") + ); + let graph = parse_graph(&src); + let ids = select_ids(&graph, &["deploy-x"], &["^deploy-"]); + assert_eq!(ids, ["deploy-x"]); + } + + #[test] + fn unknown_exact_id_is_an_error() { + let graph = parse_graph(&cmd("a", None, "echo a")); + let regexes = Vec::new(); + let ids = vec!["missing".to_string()]; + let error = select(&graph, &ids, ®exes).expect_err("should fail"); + assert!(error.to_string().contains("unknown command id: missing")); + } + + #[test] + fn regex_matching_no_commands_is_an_error() { + let graph = parse_graph(&cmd("a", None, "echo a")); + let regexes = vec![Regex::new("^nope-").unwrap()]; + let error = select(&graph, &[], ®exes).expect_err("should fail"); + assert!(error.to_string().contains("matched no commands")); + } + + #[test] + fn invalid_exact_id_value_is_an_error() { + let graph = parse_graph(&cmd("a", None, "echo a")); + let regexes = Vec::new(); + let ids = vec!["bad id".to_string()]; + let error = select(&graph, &ids, ®exes).expect_err("should fail"); + assert!(error.to_string().contains("invalid --id value")); + } + + #[test] + fn list_and_dry_run_conflict() { + let error = TestCli::try_parse_from(["test", "commands.adoc", "--list", "--dry-run"]); + assert!(error.is_err()); + } + + #[test] + fn parses_ordered_environment_overrides_and_cwd() { + let cli = TestCli::parse_from([ + "test", + "commands.adoc", + "--cwd", + "work", + "--env", + "VALUE=first", + "--env", + "VALUE=last=kept", + "--env", + "EMPTY=", + ]); + assert_eq!(cli.args.cwd, Some(PathBuf::from("work"))); + assert_eq!( + cli.args.environment, + vec![ + ("VALUE".into(), "first".into()), + ("VALUE".into(), "last=kept".into()), + ("EMPTY".into(), "".into()), + ] + ); + } + + #[test] + fn rejects_missing_environment_assignment_or_name() { + for value in ["NAME", "=value"] { + assert!(TestCli::try_parse_from(["test", "commands.adoc", "--env", value]).is_err()); + } + } + + #[test] + fn list_includes_descriptions_and_prerequisites_without_scripts() { + let graph = parse_graph(concat!( + "= Commands\n\n", + "[.command,id=build,description=Compile]\n----\necho build\n----\n\n", + "== Quality\n\n", + "[source,bash,role=command,id=lint,deps=build]\n----\necho lint\n----\n\n", + "=== *Unit* tests\n\n", + "[source,python,role=command,id=test,deps=\"lint,build,lint\",interpreter=python3]\n", + "----\nprint('test')\n----\n", + )); + let plan = select(&graph, &["test".into()], &[]).unwrap(); + let mut output = Vec::new(); + write_list(&mut output, &graph, &plan).unwrap(); + assert_eq!( + String::from_utf8(output).unwrap(), + concat!( + "- id=build, interpreter=\"sh\", description=\"Compile\"\n", + "- id=lint, interpreter=\"bash\", section=\"Quality\", deps=\"build\"\n", + "- id=test, interpreter=\"python3\", section=\"Unit tests\", deps=\"build,lint\"\n", + ) + ); + } + + #[test] + fn list_quotes_paths_and_escapes_control_characters() { + let mut builder = CommandGraphBuilder::new(); + let mut command = CommandBlock::new( + "multiline".parse().unwrap(), + String::new(), + Some("tools/my \"shell\"\n\u{1b}".into()), + acdc_parser::SourceLocation::at_location(None, acdc_parser::Location::default()), + ) + .with_description(Some("first\r\nsecond\nthird\rfourth\t\u{1b}".into())); + command.metadata.section_title = Some("Setup \"tools\"\n\u{1b}".into()); + builder.add(command, Vec::new()); + let graph = builder.build().unwrap(); + let mut output = Vec::new(); + write_list(&mut output, &graph, &graph.plan_all()).unwrap(); + assert_eq!( + String::from_utf8(output).unwrap(), + concat!( + r#"- id=multiline, interpreter="tools/my \"shell\"\n\u{1b}""#, + r#", section="Setup \"tools\"\n\u{1b}""#, + r#", description="first\r\nsecond\nthird\rfourth\t\u{1b}""#, + "\n", + ) + ); + } + + #[test] + fn dry_run_adds_only_a_display_newline() { + let mut builder = CommandGraphBuilder::new(); + builder.add( + CommandBlock::new( + "literal".parse().unwrap(), + "last line".into(), + None, + acdc_parser::SourceLocation::at_location(None, acdc_parser::Location::default()), + ), + Vec::new(), + ); + let graph = builder.build().unwrap(); + let plan = graph.plan_all(); + let mut output = Vec::new(); + write_plan(&mut output, &plan).unwrap(); + assert_eq!(output, b"literal (sh)\n last line\n"); + assert_eq!(plan.commands().next().unwrap().script, "last line"); + } + + #[test] + fn output_write_errors_are_returned() { + struct BrokenOutput; + impl std::io::Write for BrokenOutput { + fn write(&mut self, _bytes: &[u8]) -> std::io::Result { + Err(std::io::ErrorKind::BrokenPipe.into()) + } + fn flush(&mut self) -> std::io::Result<()> { + Ok(()) + } + } + let graph = parse_graph(&cmd("build", None, "true")); + let plan = graph.plan_all(); + assert_eq!( + write_plan(&mut BrokenOutput, &plan).unwrap_err().kind(), + std::io::ErrorKind::BrokenPipe + ); + assert_eq!( + write_list(&mut BrokenOutput, &graph, &plan) + .unwrap_err() + .kind(), + std::io::ErrorKind::BrokenPipe + ); + } +} diff --git a/acdc-cli/src/subcommands/mod.rs b/acdc-cli/src/subcommands/mod.rs index 188c9f61..f45a9808 100644 --- a/acdc-cli/src/subcommands/mod.rs +++ b/acdc-cli/src/subcommands/mod.rs @@ -7,6 +7,9 @@ ))] pub mod convert; +#[cfg(feature = "execute")] +pub mod execute; + #[cfg(feature = "inspect")] pub mod inspect; diff --git a/acdc-cli/tests/cli.rs b/acdc-cli/tests/cli.rs index 6c0ab565..e094a125 100644 --- a/acdc-cli/tests/cli.rs +++ b/acdc-cli/tests/cli.rs @@ -6,7 +6,12 @@ use std::{ #[cfg(any(feature = "html", feature = "terminal", feature = "inspect"))] use tempfile::tempdir; -#[cfg(any(feature = "html", feature = "terminal", feature = "inspect"))] +#[cfg(any( + feature = "html", + feature = "terminal", + feature = "inspect", + feature = "execute" +))] use std::fs; fn run_acdc(args: &[&str], input: Option<&str>) -> io::Result { @@ -39,6 +44,7 @@ fn output_text(bytes: &[u8]) -> String { feature = "markdown", feature = "pdf", feature = "terminal", + feature = "execute", feature = "inspect", feature = "lint", feature = "tck", @@ -567,3 +573,459 @@ fn inspect_resolves_includes_and_omits_ansi_when_piped() -> Result<(), Box Result<(), Box> { + let temp = tempfile::tempdir()?; + let document = temp.path().join("commands.adoc"); + fs::write( + &document, + "[.command, id=test, deps=\"build\"]\n----\necho testing\n----\n\n\ + [.command, id=build]\n[source, bash]\n----\necho building\n----\n", + )?; + let document_arg = document.to_string_lossy(); + + let output = run_acdc(&["execute", "--dry-run", document_arg.as_ref()], None)?; + let stdout = output_text(&output.stdout); + + assert!(output.status.success()); + let build = stdout.find("build (bash)").ok_or("build plan missing")?; + let test = stdout.find("test (sh)").ok_or("test plan missing")?; + assert!(build < test, "dependency must print first: {stdout}"); + assert!(stdout.contains(" echo testing")); + Ok(()) +} + +#[cfg(all(feature = "execute", unix))] +#[test] +fn execute_runs_selected_commands() -> Result<(), Box> { + let temp = tempfile::tempdir()?; + let marker = temp.path().join("ran"); + let document = temp.path().join("commands.adoc"); + fs::write( + &document, + format!( + "[.command, id=touch-marker]\n----\necho run >> {}\n----\n", + marker.display() + ), + )?; + let document_arg = document.to_string_lossy(); + + let output = run_acdc( + &["execute", "--id", "touch-marker", document_arg.as_ref()], + None, + )?; + + assert!(output.status.success()); + assert!(marker.exists()); + Ok(()) +} + +#[cfg(all(feature = "execute", unix))] +#[test] +fn execute_failure_returns_a_failing_exit() -> Result<(), Box> { + let temp = tempfile::tempdir()?; + let document = temp.path().join("commands.adoc"); + fs::write(&document, "[.command, id=fail]\n----\nexit 7\n----\n")?; + let document_arg = document.to_string_lossy(); + + let output = run_acdc(&["execute", document_arg.as_ref()], None)?; + let stderr = output_text(&output.stderr); + + assert_eq!(output.status.code(), Some(1)); + assert!(stderr.contains("`fail` exited with")); + Ok(()) +} + +#[cfg(all(feature = "execute", unix))] +mod execution { + use std::path::Path; + + use super::*; + + fn run_document( + directory: &Path, + source: &str, + flags: &[&str], + ) -> Result> { + let document = directory.join("commands.adoc"); + fs::write(&document, source)?; + Ok(Command::new(env!("CARGO_BIN_EXE_acdc")) + .arg("execute") + .arg(&document) + .args(flags) + .current_dir(directory) + .env("ACDC_EXECUTE_INHERITED", "inherited") + .output()?) + } + + #[test] + fn execute_blocks_transitive_dependents_but_runs_independent_commands() + -> Result<(), Box> { + let directory = tempfile::tempdir()?; + let output = run_document( + directory.path(), + concat!( + "[.command,id=build]\n----\nexit 7\n----\n\n", + "[.command,id=test,deps=build]\n----\ntouch test\n----\n\n", + "[.command,id=deploy,deps=test]\n----\ntouch deploy\n----\n\n", + "[.command,id=independent]\n----\ntouch independent\n----\n", + ), + &[], + )?; + assert_eq!(output.status.code(), Some(1)); + let stderr = output_text(&output.stderr); + assert!(stderr.contains("command `build` exited with"), "{stderr}"); + assert!( + stderr.contains("prerequisite `build` did not succeed"), + "{stderr}" + ); + assert!( + stderr.contains("prerequisite `test` did not succeed"), + "{stderr}" + ); + assert!(!directory.path().join("test").exists()); + assert!(!directory.path().join("deploy").exists()); + assert!(directory.path().join("independent").exists()); + Ok(()) + } + + #[test] + fn execute_exit_on_failure_stops_independent_commands() -> Result<(), Box> { + let directory = tempfile::tempdir()?; + let output = run_document( + directory.path(), + concat!( + "[.command,id=bad]\n----\nexit 7\n----\n\n", + "[.command,id=after]\n----\ntouch after\n----\n", + ), + &["--exit-on-failure"], + )?; + assert_eq!(output.status.code(), Some(1)); + assert!(!directory.path().join("after").exists()); + assert!(output_text(&output.stderr).contains("skipped after failure of `bad`")); + Ok(()) + } + + #[test] + fn execute_cwd_and_env_override_only_child_settings() -> Result<(), Box> { + let directory = tempfile::tempdir()?; + fs::create_dir(directory.path().join("child"))?; + let output = run_document( + directory.path(), + concat!( + "[.command,id=child]\n----\n", + "printf '%s|%s' \"$ACDC_EXECUTE_INHERITED\" \"$ACDC_EXECUTE_VALUE\" > value\n----\n", + ), + &[ + "--cwd", + "child", + "--env", + "ACDC_EXECUTE_VALUE=first", + "--env", + "ACDC_EXECUTE_VALUE=last=kept", + ], + )?; + assert!(output.status.success(), "{}", output_text(&output.stderr)); + assert_eq!( + fs::read_to_string(directory.path().join("child/value"))?, + "inherited|last=kept" + ); + assert!(!directory.path().join("value").exists()); + Ok(()) + } + + #[test] + fn execute_defaults_inherit_caller_directory_and_environment() -> Result<(), Box> { + let directory = tempfile::tempdir()?; + let output = run_document( + directory.path(), + "[.command,id=defaults]\n----\nprintf '%s' \"$ACDC_EXECUTE_INHERITED\" > inherited\n----\n", + &[], + )?; + assert!(output.status.success(), "{}", output_text(&output.stderr)); + assert_eq!( + fs::read_to_string(directory.path().join("inherited"))?, + "inherited" + ); + Ok(()) + } + + #[test] + fn execute_list_includes_prerequisites_and_descriptions_without_running() + -> Result<(), Box> { + let directory = tempfile::tempdir()?; + let output = run_document( + directory.path(), + concat!( + "= Commands\n\n", + "[.command,id=build,description=Compile]\n----\ntouch build\n----\n\n", + "== Tests\n\n", + "[source,bash,role=command,id=test,deps=build,interpreter=sh]\n", + "----\ntouch test\n----\n", + ), + &["--list", "--id", "test"], + )?; + assert!(output.status.success(), "{}", output_text(&output.stderr)); + assert_eq!( + output_text(&output.stdout), + concat!( + "- id=build, interpreter=\"sh\", description=\"Compile\"\n", + "- id=test, interpreter=\"sh\", section=\"Tests\", deps=\"build\"\n", + ) + ); + assert!(!directory.path().join("build").exists()); + assert!(!directory.path().join("test").exists()); + Ok(()) + } + + #[test] + fn execute_rejects_recovered_source_before_any_child_starts() -> Result<(), Box> { + for source in [ + "[.command,id=incomplete]\n----\ntouch marker\n", + "[.command,id=missing]\n----\ninclude::missing.sh[]\ntouch marker\n----\n", + ] { + let directory = tempfile::tempdir()?; + let output = run_document(directory.path(), source, &[])?; + assert_eq!( + output.status.code(), + Some(1), + "{}", + output_text(&output.stderr) + ); + assert!(!directory.path().join("marker").exists()); + } + Ok(()) + } + + #[test] + fn execute_preserves_callout_text_in_heredocs() -> Result<(), Box> { + let directory = tempfile::tempdir()?; + let output = run_document( + directory.path(), + "[.command,id=literal]\n----\ncat <<'EOF'\n<1>\nEOF\n----\n", + &[], + )?; + assert!(output.status.success(), "{}", output_text(&output.stderr)); + assert_eq!(output.stdout, b"<1>\n"); + Ok(()) + } + + #[test] + fn execute_keeps_attribute_references_literal_by_default() -> Result<(), Box> { + let directory = tempfile::tempdir()?; + let source = concat!( + "= Commands\n:value: expanded\n\n", + "[.command,id=literal]\n----\n", + "printf '%s' '{value}|\\{value}|<1>'\n----\n", + ); + let dry_run = run_document(directory.path(), source, &["--dry-run"])?; + assert!(dry_run.status.success(), "{}", output_text(&dry_run.stderr)); + assert_eq!( + output_text(&dry_run.stdout), + "literal (sh)\n printf '%s' '{value}|\\{value}|<1>'\n" + ); + let executed = run_document(directory.path(), source, &[])?; + assert!( + executed.status.success(), + "{}", + output_text(&executed.stderr) + ); + assert_eq!(executed.stdout, b"{value}|\\{value}|<1>"); + Ok(()) + } + + #[cfg(feature = "pre-spec-subs")] + #[test] + fn execute_dry_run_and_interpreter_use_the_same_attribute_substitutions() + -> Result<(), Box> { + for substitutions in ["attributes", "+attributes", "attributes+", "normal"] { + let directory = tempfile::tempdir()?; + let source = format!( + "= Commands\n:value: hello <1>\n:empty:\n\n\ + [.command,id=expanded,subs=\"{substitutions}\"]\n----\n\ + printf '%s|%s|%s|%s' '{{value}}' '\\{{missing}}' '{{empty}}' '<2>'\n----\n" + ); + let dry_run = run_document(directory.path(), &source, &["--dry-run"])?; + assert!( + dry_run.status.success(), + "{substitutions}: {}", + output_text(&dry_run.stderr) + ); + assert_eq!( + output_text(&dry_run.stdout), + "expanded (sh)\n printf '%s|%s|%s|%s' 'hello <1>' '{missing}' '' '<2>'\n", + "{substitutions}" + ); + let executed = run_document(directory.path(), &source, &[])?; + assert!( + executed.status.success(), + "{substitutions}: {}", + output_text(&executed.stderr) + ); + assert_eq!( + executed.stdout, b"hello <1>|{missing}||<2>", + "{substitutions}" + ); + } + Ok(()) + } + + #[cfg(feature = "pre-spec-subs")] + #[test] + fn execute_keeps_disabled_attribute_substitutions_literal() -> Result<(), Box> { + for substitutions in ["none", "-attributes"] { + let directory = tempfile::tempdir()?; + let source = format!( + "= Commands\n:value: expanded\n\n\ + [.command,id=literal,subs=\"{substitutions}\"]\n----\n\ + printf '%s' '{{value}}|\\{{value}}|<1>'\n----\n" + ); + let output = run_document(directory.path(), &source, &[])?; + assert!( + output.status.success(), + "{substitutions}: {}", + output_text(&output.stderr) + ); + assert_eq!(output.stdout, b"{value}|\\{value}|<1>", "{substitutions}"); + } + Ok(()) + } + + #[cfg(feature = "pre-spec-subs")] + #[test] + fn execute_rejects_missing_attributes_in_unselected_commands_before_starting() + -> Result<(), Box> { + for flags in [vec!["--id", "first"], vec!["--id", "first", "--dry-run"]] { + let directory = tempfile::tempdir()?; + let output = run_document( + directory.path(), + concat!( + "[.command,id=first]\n----\nprintf ran > marker\n----\n\n", + "[.command,id=broken,subs=attributes]\n----\nprintf '%s' '{missing}'\n----\n", + ), + &flags, + )?; + assert_eq!(output.status.code(), Some(1)); + let stderr = output_text(&output.stderr); + assert!( + stderr.contains("command `broken` references missing document attribute `missing`"), + "{stderr}" + ); + assert!(stderr.contains("commands.adoc"), "{stderr}"); + assert!(!directory.path().join("marker").exists()); + } + Ok(()) + } + + #[cfg(feature = "pre-spec-subs")] + #[test] + fn execute_attribute_substitutions_follow_source_order() -> Result<(), Box> { + let directory = tempfile::tempdir()?; + let source = concat!( + "= Commands\n:value: first\n\n", + "[.command,id=first,subs=attributes]\n----\nprintf '%s\\n' '{value}'\n----\n\n", + ":value: second\n\n", + "[.command,id=second,subs=attributes]\n----\nprintf '%s\\n' '{value}'\n----\n", + ); + let dry_run = run_document(directory.path(), source, &["--dry-run"])?; + assert!(dry_run.status.success(), "{}", output_text(&dry_run.stderr)); + assert_eq!( + output_text(&dry_run.stdout), + concat!( + "first (sh)\n printf '%s\\n' 'first'\n", + "second (sh)\n printf '%s\\n' 'second'\n", + ) + ); + let output = run_document(directory.path(), source, &[])?; + assert!(output.status.success(), "{}", output_text(&output.stderr)); + assert_eq!(output.stdout, b"first\nsecond\n"); + Ok(()) + } + + #[cfg(not(feature = "pre-spec-subs"))] + #[test] + fn execute_without_substitution_support_rejects_explicit_subs_before_starting() + -> Result<(), Box> { + for substitutions in ["attributes", "none"] { + let directory = tempfile::tempdir()?; + let source = format!( + "[.command,id=first]\n----\nprintf ran > marker\n----\n\n\ + [.command,id=unsupported,subs={substitutions}]\n----\nprintf '%s' '{{value}}'\n----\n" + ); + let output = run_document(directory.path(), &source, &["--id", "first"])?; + assert_eq!(output.status.code(), Some(1)); + let stderr = output_text(&output.stderr); + assert!(stderr.contains("subs"), "{stderr}"); + assert!(stderr.contains("commands.adoc"), "{stderr}"); + assert!(!directory.path().join("marker").exists()); + } + Ok(()) + } + + #[test] + fn execute_reports_the_resolved_included_file_for_invalid_commands() + -> Result<(), Box> { + let directory = tempfile::tempdir()?; + fs::create_dir(directory.path().join("nested"))?; + let included = directory.path().join("nested/broken.adoc"); + fs::write(&included, "[.command]\n----\ntrue\n----\n")?; + let output = run_document(directory.path(), "include::nested/broken.adoc[]\n", &[])?; + assert_eq!(output.status.code(), Some(1)); + let stderr = output_text(&output.stderr); + assert!(stderr.contains("missing an id"), "{stderr}"); + assert!( + stderr.contains(included.to_string_lossy().as_ref()), + "{stderr}" + ); + assert!(stderr.contains("[.command]"), "{stderr}"); + Ok(()) + } + + #[test] + fn execute_keeps_both_duplicate_declarations_in_diagnostics() -> Result<(), Box> { + let directory = tempfile::tempdir()?; + fs::write( + directory.path().join("first.adoc"), + "[.command,id=duplicate]\n----\ntrue\n----\n", + )?; + fs::write( + directory.path().join("second.adoc"), + "[.command,id=duplicate]\n----\ntrue\n----\n", + )?; + let output = run_document( + directory.path(), + "include::first.adoc[]\n\ninclude::second.adoc[]\n", + &[], + )?; + assert_eq!(output.status.code(), Some(1)); + let stderr = output_text(&output.stderr); + assert!(stderr.contains("duplicate command id"), "{stderr}"); + assert!(stderr.contains("first.adoc"), "{stderr}"); + assert!(stderr.contains("second.adoc"), "{stderr}"); + Ok(()) + } +} + +#[cfg(feature = "execute")] +#[test] +fn execute_reports_unknown_and_duplicate_selectors_as_diagnostics() +-> Result<(), Box> { + let temp = tempfile::tempdir()?; + let document = temp.path().join("commands.adoc"); + fs::write(&document, "[.command, id=build]\n----\ntrue\n----\n")?; + let document_arg = document.to_string_lossy(); + + let unknown = run_acdc(&["execute", "--id", "missing", document_arg.as_ref()], None)?; + assert_eq!(unknown.status.code(), Some(1)); + assert!(output_text(&unknown.stderr).contains("unknown command id: missing")); + + let no_match = run_acdc( + &["execute", "--id-regex", "^zzz", document_arg.as_ref()], + None, + )?; + assert_eq!(no_match.status.code(), Some(1)); + assert!(output_text(&no_match.stderr).contains("matched no commands")); + Ok(()) +} diff --git a/acdc-execute/CHANGELOG.md b/acdc-execute/CHANGELOG.md new file mode 100644 index 00000000..c82b8762 --- /dev/null +++ b/acdc-execute/CHANGELOG.md @@ -0,0 +1,36 @@ +# Changelog + +All notable changes to `acdc-execute` will be documented in this file. + +The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), +and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). + +## [Unreleased] + +### Changed + +- Library callers now import command-discovery errors as + `acdc_execute::DiscoveryError`. +- Building `acdc-execute` no longer enables unused graph implementations in its + dependencies. + +### Added + +- Library callers can inspect a command's direct prerequisites and nearest + enclosing section title. +- Creating execution plans uses less temporary memory. +- Discover and run listing and source blocks marked with the `command` role, + including commands in nested containers, AsciiDoc table cells, and included files. + Scripts retain their literal text after includes and conditionals; enable + textual document-attribute expansion with `subs=attributes` or `subs=+attributes`. + Expansion uses attributes at the command's source position and preserves + callouts. Missing enabled attributes fail before execution. Builds without + `pre-spec-subs` reject explicit `subs` settings. The `interpreter` attribute + overrides the source language. +- Select commands with their dependencies and inspect execution outcomes. + A failed prerequisite blocks its dependents; callers can stop all commands at + the first failure. Infrastructure errors and signal termination always stop + execution. Invalid or incomplete command input fails before execution. +- Configure child working directories and environment variables while preserving + inherited defaults. Source diagnostics identify malformed commands and + dependency errors, including their locations in included files. diff --git a/acdc-execute/Cargo.toml b/acdc-execute/Cargo.toml new file mode 100644 index 00000000..89caded9 --- /dev/null +++ b/acdc-execute/Cargo.toml @@ -0,0 +1,29 @@ +[package] +name = "acdc-execute" +version = "0.1.0" +edition.workspace = true +description = "Discover and run command blocks defined in AsciiDoc documents." +license.workspace = true +repository = "https://github.com/nlopes/acdc" +keywords = ["asciidoc", "command-runner", "task-runner"] +categories = ["command-line-utilities", "development-tools"] +publish = false + +[lints] +workspace = true + +[features] +default = ["pre-spec-subs", "setext"] +pre-spec-subs = ["acdc-parser/pre-spec-subs", "acdc-converters-core/pre-spec-subs"] +setext = ["acdc-parser/setext"] +network = ["acdc-parser/network"] + +[dependencies] +acdc-converters-core.workspace = true +acdc-parser.workspace = true +petgraph.workspace = true +tempfile.workspace = true +thiserror.workspace = true + +[dev-dependencies] +rstest.workspace = true diff --git a/acdc-execute/README.adoc b/acdc-execute/README.adoc new file mode 100644 index 00000000..47e067af --- /dev/null +++ b/acdc-execute/README.adoc @@ -0,0 +1,37 @@ += `acdc-execute` + +Run named commands from AsciiDoc documents, with dependency ordering and execution reports. +For command-line use, see the link:../acdc-cli/README.adoc[CLI guide]. + +== Example + +Save a command in `tasks.adoc`: + +[source,asciidoc] +.... +[source,sh,role=command,id=hello] +---- +printf 'Hello\n' +---- +.... + +Parse it, build a command graph, and run its execution plan: + +[source,rust] +---- +use acdc_execute::{CommandGraph, ExecutionOptions}; +use acdc_parser::{Options, parse_file}; + +let parsed = parse_file("tasks.adoc", &Options::default())?; +let graph = CommandGraph::try_from(&parsed)?; +let report = graph.plan_all().execute(&ExecutionOptions::default()); + +println!("All commands succeeded: {}", report.is_success()); +---- + +The report records which commands succeeded, failed, or were skipped. +A failed command blocks its dependents; independent commands continue by default. + +Commands run with the caller's permissions, so only execute documents you trust. + +See the link:../acdc-cli/README.adoc[CLI guide] for command syntax and substitution rules, and the link:src/command/mod.rs[library API] for selection, process options, and result handling. diff --git a/acdc-execute/src/command/error.rs b/acdc-execute/src/command/error.rs new file mode 100644 index 00000000..6ef7bf7c --- /dev/null +++ b/acdc-execute/src/command/error.rs @@ -0,0 +1,113 @@ +//! Errors reported while constructing, selecting, and executing commands. + +use acdc_parser::SourceLocation; + +use super::CommandId; + +/// Error returned when a string is not a valid [`CommandId`]. +#[derive(Debug, thiserror::Error)] +pub enum InvalidCommandId { + /// The id was empty. + #[error("command id must not be empty")] + Empty, + /// The id started with `-`, which would collide with command-line flag parsing. + #[error("command id {id:?} must not start with '-'")] + LeadingDash { + /// The rejected id. + id: String, + }, + /// The id contained a character outside the allowed set. + #[error("command id {id:?} contains illegal character {ch:?}")] + IllegalChar { + /// The rejected id. + id: String, + /// The first offending character. + ch: char, + }, +} + +/// Error returned when a [`CommandBlock`](super::CommandBlock) fails to execute. +#[derive(Debug, thiserror::Error)] +pub enum ExecError { + /// The temporary script file could not be created or written. + #[error("could not write temporary script file: {0}")] + TempFile(#[source] std::io::Error), + /// Starting or waiting for the interpreter failed. + #[error("could not run interpreter: {0}")] + Process(#[source] std::io::Error), + /// The child exited unsuccessfully, including termination by a signal. + #[error("exited with {0}")] + Failed(std::process::ExitStatus), +} + +/// Error returned when building a [`CommandGraph`](super::CommandGraph) from a +/// [`CommandGraphBuilder`](super::CommandGraphBuilder) fails. +#[derive(Debug, thiserror::Error)] +pub enum BuildError { + /// Two commands share the same id. + #[error("duplicate command id: {id}")] + DuplicateId { + /// The repeated id. + id: CommandId, + /// The first declaration. + first_location: Box, + /// The duplicate declaration. + duplicate_location: Box, + }, + /// A declared dependency does not name any added command. + #[error("command `{command}` declares unknown dependency: {dep}")] + UnknownDep { + /// The command declaring the dependency. + command: CommandId, + /// The dependency that could not be resolved. + dep: CommandId, + /// The dependent command's declaration. + location: Box, + }, + /// A dependency edge would introduce a cycle. + /// + /// The reported edge runs from `dep` (the prerequisite) to `command` (the dependent) and is + /// one edge in the cycle. A self-dependency has `dep == command`. + #[error("dependency cycle includes: {dep} -> {command}")] + Cycle { + /// The dependent command. + command: CommandId, + /// The prerequisite that closes the cycle. + dep: CommandId, + /// The dependent command's declaration. + command_location: Box, + /// The prerequisite's declaration. + dep_location: Box, + }, +} + +impl BuildError { + /// The declaration that caused graph validation to fail. + #[must_use] + pub fn source_location(&self) -> &SourceLocation { + match self { + Self::DuplicateId { + duplicate_location, .. + } => duplicate_location, + Self::UnknownDep { location, .. } => location, + Self::Cycle { + command_location, .. + } => command_location, + } + } + + /// The other declaration involved in a duplicate id or dependency cycle. + #[must_use] + pub fn related_location(&self) -> Option<&SourceLocation> { + match self { + Self::DuplicateId { first_location, .. } => Some(first_location), + Self::Cycle { dep_location, .. } => Some(dep_location), + Self::UnknownDep { .. } => None, + } + } +} + +/// Error returned when a command id is not found in a [`CommandGraph`](super::CommandGraph). +#[derive(Debug, thiserror::Error)] +#[error("unknown command: {0}")] +pub struct UnknownCommand(pub CommandId); diff --git a/acdc-execute/src/command/mod.rs b/acdc-execute/src/command/mod.rs new file mode 100644 index 00000000..662688e5 --- /dev/null +++ b/acdc-execute/src/command/mod.rs @@ -0,0 +1,1362 @@ +//! Command construction and execution. + +mod error; + +pub use error::{BuildError, ExecError, InvalidCommandId, UnknownCommand}; + +use std::{ + borrow::{Borrow, Cow}, + cmp::Reverse, + collections::{BinaryHeap, HashMap, HashSet}, + ffi::OsString, + fmt, + io::Write as _, + path::{Path, PathBuf}, + process::Command, +}; + +use acdc_parser::SourceLocation; +use petgraph::{ + graph::{DiGraph, NodeIndex}, + prelude::Direction, + visit::{Dfs, EdgeRef, Reversed}, +}; + +const DEFAULT_INTERPRETER: &str = "sh"; + +/// A unique identifier for a command. +/// +/// IDs contain only ASCII alphanumerics, `-`, and `_`, and must not start with `-`. +/// Construct via [`CommandId::new`] or [`str::parse`]. +#[derive(Clone, Debug, Hash, Eq, PartialEq)] +pub struct CommandId(String); + +impl CommandId { + /// Validate `id` and wrap it. + /// + /// # Errors + /// + /// Returns [`InvalidCommandId`] if `id` is empty, starts with `-`, or contains + /// a character outside ASCII alphanumerics, `-`, and `_`. + pub fn new(id: impl Into) -> Result { + let id: String = id.into(); + if id.is_empty() { + return Err(InvalidCommandId::Empty); + } + if id.starts_with('-') { + return Err(InvalidCommandId::LeadingDash { id }); + } + if let Some(ch) = id.chars().find(|&c| !Self::is_legal(c)) { + return Err(InvalidCommandId::IllegalChar { id, ch }); + } + Ok(Self(id)) + } + + /// The id as a string slice. + #[must_use] + pub fn as_str(&self) -> &str { + &self.0 + } + + fn is_legal(c: char) -> bool { + c.is_ascii_alphanumeric() || matches!(c, '-' | '_') + } +} + +impl std::str::FromStr for CommandId { + type Err = InvalidCommandId; + + fn from_str(s: &str) -> Result { + Self::new(s) + } +} + +impl Borrow for CommandId { + fn borrow(&self) -> &str { + &self.0 + } +} + +impl fmt::Display for CommandId { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(&self.0) + } +} + +/// Metadata for a command block. +#[derive(Clone, Debug)] +pub struct CommandMetadata { + /// The identifier for the command, e.g. `build` or `test`. + pub id: CommandId, + /// The interpreter name or path to use to execute the command. + /// + /// This value is what might appear *last* in the *shebang* of a script, e.g. `python3` for + /// `#!/usr/bin/env python3`, or `bash` for `#!/usr/bin/env bash`. It is passed directly to + /// [`std::process::Command`] without an allowlist or shell-string parsing, so a document can + /// select any executable available to the calling process. Executing a document is not a + /// sandbox boundary. + pub interpreter: String, + /// A short, human-readable summary of what the command does, shown when listing commands. + pub description: Option, + /// Plain-text title of the nearest enclosing section, or `None` outside a section. + pub section_title: Option, +} + +/// A single command with its metadata. +#[derive(Clone, Debug)] +pub struct CommandBlock { + /// The metadata associated with this command block. + pub metadata: CommandMetadata, + /// The script body. + pub script: String, + /// The resolved source file and original document location of this command. + pub location: SourceLocation, +} + +impl CommandBlock { + /// Preserve `script` unchanged and default `interpreter` to `"sh"` when absent. + #[must_use] + pub fn new( + id: CommandId, + script: String, + interpreter: Option, + location: SourceLocation, + ) -> Self { + Self { + metadata: CommandMetadata { + id, + interpreter: interpreter.unwrap_or_else(|| DEFAULT_INTERPRETER.to_string()), + description: None, + section_title: None, + }, + script, + location, + } + } + + /// Attach a description, returning the updated block. + #[must_use] + pub fn with_description(mut self, description: Option) -> Self { + self.metadata.description = description; + self + } + + /// Execute the script with inherited working directory, environment, and stdio. + /// + /// # Errors + /// + /// Returns [`ExecError`] if preparing, running, or waiting for the script fails. + pub fn execute(&self) -> Result<(), ExecError> { + self.execute_with(&ProcessOptions::default()) + } + + /// Execute the exact script bytes with options applied only to the child process. + /// + /// The interpreter receives a temporary script file as one argument. Stdio is inherited, + /// and the temporary file is removed after the child exits. Scripts are not sandboxed. + /// Explicit relative interpreter paths are resolved from the caller's directory, including + /// when a child working directory is set. Bare interpreter names use the operating system's + /// executable search. + /// + /// # Errors + /// + /// Returns [`ExecError::TempFile`] if the script cannot be written, [`ExecError::Process`] + /// if starting or waiting for the interpreter fails, or [`ExecError::Failed`] if the + /// child exits unsuccessfully. + pub fn execute_with(&self, options: &ProcessOptions) -> Result<(), ExecError> { + let mut tmp = tempfile::NamedTempFile::new().map_err(ExecError::TempFile)?; + tmp.write_all(self.script.as_bytes()) + .map_err(ExecError::TempFile)?; + tmp.flush().map_err(ExecError::TempFile)?; + + let interpreter = Path::new(&self.metadata.interpreter); + let resolved_interpreter = if options.current_dir.is_some() + && interpreter.is_relative() + && self.metadata.interpreter.contains(std::path::is_separator) + { + Some( + std::env::current_dir() + .map_err(ExecError::Process)? + .join(interpreter), + ) + } else { + None + }; + let mut child = Command::new(resolved_interpreter.as_deref().unwrap_or(interpreter)); + child + .arg(tmp.path()) + .envs(options.env.iter().map(|(name, value)| (name, value))); + if let Some(directory) = &options.current_dir { + child.current_dir(directory); + } + let status = child.status().map_err(ExecError::Process)?; + + drop(tmp); + + if status.success() { + Ok(()) + } else { + Err(ExecError::Failed(status)) + } + } +} + +/// Child-process settings. Defaults inherit the caller's working directory and environment. +#[derive(Clone, Debug, Default)] +pub struct ProcessOptions { + /// Working directory for each child; relative paths are resolved from the caller's directory. + pub current_dir: Option, + /// Environment overrides, applied in order so the last value for a name wins. + pub env: Vec<(OsString, OsString)>, +} + +/// Policy and child-process settings for an execution plan. +#[derive(Clone, Debug, Default)] +pub struct ExecutionOptions { + /// Stop all remaining commands after the first unsuccessful child exit. + /// + /// Infrastructure errors and signal termination always stop the plan. + pub exit_on_failure: bool, + /// Working directory and environment applied to each child. + pub process: ProcessOptions, +} + +/// A directed acyclic graph of commands, typically pulled from a single `AsciiDoc` file. +/// +/// Edges run from prerequisite to dependent: an edge `A → B` means "A must run before B." +#[derive(Debug)] +pub struct CommandGraph { + graph: DiGraph, + index: HashMap, + /// All nodes in topological order; the single source of iteration order. + order: Vec, +} + +impl CommandGraph { + /// The id of every command, in topological execution order. + pub fn ids(&self) -> impl Iterator { + let graph = &self.graph; + self.order.iter().map(move |idx| &graph[*idx].metadata.id) + } + + /// Whether `id` names a command in this graph. + #[must_use] + pub fn contains(&self, id: &str) -> bool { + self.index.contains_key(id) + } + + /// Direct prerequisites of `id`, each returned once in unspecified order. + /// + /// Returns `None` if `id` does not name a command in this graph. + #[must_use] + pub fn dependencies<'graph>( + &'graph self, + id: &str, + ) -> Option + use<'graph>> { + let &node = self.index.get(id)?; + Some( + self.graph + .neighbors_directed(node, Direction::Incoming) + .map(|dependency| &self.graph[dependency].metadata.id), + ) + } + + /// Select every command in execution order without copying script bodies. + #[must_use] + pub fn plan_all(&self) -> ExecutionPlan<'_> { + ExecutionPlan { + graph: self, + order: Cow::Borrowed(&self.order), + } + } + + /// Select the given commands and their transitive dependencies in execution order. + /// + /// Each selected command appears once. The plan borrows script bodies from this graph. + /// + /// # Errors + /// + /// Returns [`UnknownCommand`] if any id in `ids` is not in the graph. + pub fn plan_for(&self, ids: &[CommandId]) -> Result, UnknownCommand> { + let rev = Reversed(&self.graph); + let mut dfs = Dfs::empty(rev); + let mut relevant = vec![false; self.graph.node_count()]; + + // Reuse one DFS so shared ancestors are visited once; `move_to` resets the stack while + // retaining the visitor's visited set. + for id in ids { + let &target = self + .index + .get(id) + .ok_or_else(|| UnknownCommand(id.clone()))?; + + dfs.move_to(target); + while let Some(node) = dfs.next(rev) { + if let Some(selected) = relevant.get_mut(node.index()) { + *selected = true; + } + } + } + + let order = self + .order + .iter() + .copied() + .filter(|node| relevant.get(node.index()) == Some(&true)) + .collect(); + Ok(ExecutionPlan { + graph: self, + order: Cow::Owned(order), + }) + } +} + +/// Builds a [`CommandGraph`] from commands added in any order. +/// +/// Commands may declare dependencies on ids that have not been added yet; all ids are resolved +/// once, at [`build`](CommandGraphBuilder::build). This lifts the ordering constraint that direct +/// insertion would impose, at the cost of deferring every validation error to build time. +#[derive(Debug, Default)] +pub struct CommandGraphBuilder { + pending: Vec<(CommandBlock, Vec)>, +} + +impl CommandGraphBuilder { + /// Create an empty builder. + #[must_use] + pub fn new() -> Self { + Self::default() + } + + /// Register a command and the ids of its prerequisites. + /// + /// Ids are not validated here; duplicates, unknown deps, and cycles are all reported by + /// [`build`](CommandGraphBuilder::build). + pub fn add(&mut self, block: CommandBlock, deps: Vec) { + self.pending.push((block, deps)); + } + + /// Resolve all commands into a [`CommandGraph`]. + /// + /// # Errors + /// + /// Returns [`BuildError::DuplicateId`] if two commands share an id, [`BuildError::UnknownDep`] + /// if a dep names no added command, or [`BuildError::Cycle`] if the dependencies are cyclic. + pub fn build(self) -> Result { + let mut graph = CommandGraph { + graph: DiGraph::new(), + index: HashMap::new(), + order: Vec::new(), + }; + + // Pass 1: add every node and index its id, so deps can be resolved in any order. + let mut edges: Vec<(NodeIndex, Vec)> = Vec::with_capacity(self.pending.len()); + for (block, deps) in self.pending { + let id = block.metadata.id.clone(); + if let Some(&first) = graph.index.get::(id.borrow()) { + return Err(BuildError::DuplicateId { + id, + first_location: Box::new(graph.graph[first].location.clone()), + duplicate_location: Box::new(block.location), + }); + } + let node = graph.graph.add_node(block); + graph.index.insert(id, node); + edges.push((node, deps)); + } + + // Pass 2: wire edges. + for (node, deps) in edges { + let mut seen: HashSet = HashSet::new(); + for dep in deps { + let dep_node = *graph.index.get::(dep.borrow()).ok_or_else(|| { + BuildError::UnknownDep { + command: graph.graph[node].metadata.id.clone(), + dep: dep.clone(), + location: Box::new(graph.graph[node].location.clone()), + } + })?; + if dep_node == node { + return Err(BuildError::Cycle { + command: graph.graph[node].metadata.id.clone(), + dep, + command_location: Box::new(graph.graph[node].location.clone()), + dep_location: Box::new(graph.graph[node].location.clone()), + }); + } + // A dep listed twice would add a parallel edge; skip the repeat. + if !seen.insert(dep_node) { + continue; + } + graph.graph.add_edge(dep_node, node, ()); + } + } + + // Pass 3: Kahn's algorithm, smallest insertion index first, so commands + // without dependencies keep document order. Remaining prerequisite edges + // after sorting contain a cycle. + let count = graph.graph.node_count(); + let mut indegree = vec![0_usize; count]; + for edge in graph.graph.edge_references() { + if let Some(degree) = indegree.get_mut(edge.target().index()) { + *degree += 1; + } + } + let mut ready: BinaryHeap> = indegree + .iter() + .enumerate() + .filter(|(_, degree)| **degree == 0) + .map(|(index, _)| Reverse(index)) + .collect(); + graph.order = Vec::with_capacity(count); + while let Some(Reverse(i)) = ready.pop() { + let node = NodeIndex::new(i); + for successor in graph.graph.neighbors(node) { + if let Some(degree) = indegree.get_mut(successor.index()) { + *degree -= 1; + if *degree == 0 { + ready.push(Reverse(successor.index())); + } + } + } + graph.order.push(node); + } + if graph.order.len() != count { + let (command, dep) = cycle_edge(&graph.graph, &indegree); + return Err(BuildError::Cycle { + command: graph.graph[command].metadata.id.clone(), + dep: graph.graph[dep].metadata.id.clone(), + command_location: Box::new(graph.graph[command].location.clone()), + dep_location: Box::new(graph.graph[dep].location.clone()), + }); + } + + Ok(graph) + } +} + +fn cycle_edge( + graph: &DiGraph, + remaining_indegree: &[usize], +) -> (NodeIndex, NodeIndex) { + let start = remaining_indegree + .iter() + .position(|degree| *degree > 0) + .map_or_else(|| NodeIndex::new(0), NodeIndex::new); + let mut visited = HashSet::new(); + let mut command = start; + + loop { + visited.insert(command); + let dep = graph + .neighbors_directed(command, Direction::Incoming) + .filter(|node| { + remaining_indegree + .get(node.index()) + .is_some_and(|degree| *degree > 0) + }) + .min() + .unwrap_or(command); + if visited.contains(&dep) { + return (command, dep); + } + command = dep; + } +} + +/// Selected commands and their prerequisites, borrowing their scripts from a validated graph. +#[derive(Debug)] +pub struct ExecutionPlan<'graph> { + graph: &'graph CommandGraph, + order: Cow<'graph, [NodeIndex]>, +} + +impl<'graph> ExecutionPlan<'graph> { + /// Selected commands in execution order, including their prerequisites. + #[must_use] + pub fn commands(&self) -> impl ExactSizeIterator + '_ { + self.order.iter().map(|node| &self.graph.graph[*node]) + } + + /// Number of selected commands, including prerequisites. + #[must_use] + pub fn len(&self) -> usize { + self.order.len() + } + + /// Whether no commands were selected. + #[must_use] + pub fn is_empty(&self) -> bool { + self.order.is_empty() + } + + /// Run commands synchronously and return an outcome for every selected command. + /// + /// Commands with unsuccessful prerequisites are skipped. Independent commands continue + /// after ordinary nonzero exits unless [`ExecutionOptions::exit_on_failure`] is set. + /// Infrastructure errors and signal termination always stop all remaining commands. + #[must_use] + pub fn execute(&self, options: &ExecutionOptions) -> ExecutionReport<'graph> { + let mut outcomes = Vec::with_capacity(self.len()); + let mut succeeded = vec![false; self.graph.graph.node_count()]; + let mut stopped_after = None; + + for &node in self.order.iter() { + let command = &self.graph.graph[node]; + let state = if let Some(command) = stopped_after { + CommandState::Skipped(SkipReason::StoppedAfterFailure { command }) + } else if let Some(dependency) = self + .graph + .graph + .neighbors_directed(node, Direction::Incoming) + .filter(|dependency| succeeded.get(dependency.index()) != Some(&true)) + .min() + { + CommandState::Skipped(SkipReason::DependencyFailed { + dependency: &self.graph.graph[dependency].metadata.id, + }) + } else { + match command.execute_with(&options.process) { + Ok(()) => { + if let Some(success) = succeeded.get_mut(node.index()) { + *success = true; + } + CommandState::Succeeded + } + Err(error) => { + let must_stop = match &error { + ExecError::Failed(status) => { + options.exit_on_failure || status.code().is_none() + } + ExecError::TempFile(_) | ExecError::Process(_) => true, + }; + if must_stop { + stopped_after = Some(&command.metadata.id); + } + CommandState::Failed(error) + } + } + }; + outcomes.push(CommandOutcome { command, state }); + } + + ExecutionReport { outcomes } + } +} + +/// Results in plan order, including commands that were skipped. +#[derive(Debug)] +pub struct ExecutionReport<'graph> { + outcomes: Vec>, +} + +impl<'graph> ExecutionReport<'graph> { + /// Results in the order commands appeared in the plan. + #[must_use] + pub fn outcomes(&self) -> &[CommandOutcome<'graph>] { + &self.outcomes + } + + /// Whether every selected command succeeded. An empty plan succeeds. + #[must_use] + pub fn is_success(&self) -> bool { + self.outcomes + .iter() + .all(|outcome| matches!(outcome.state, CommandState::Succeeded)) + } + + /// Take the outcomes, preserving structured errors for the caller. + #[must_use] + pub fn into_outcomes(self) -> Vec> { + self.outcomes + } +} + +/// A command's outcome and its source metadata. +#[derive(Debug)] +pub struct CommandOutcome<'graph> { + /// The command that ran or was skipped. + pub command: &'graph CommandBlock, + /// The result of executing this command. + pub state: CommandState<'graph>, +} + +/// Execution result for one selected command. +#[derive(Debug)] +pub enum CommandState<'graph> { + /// The child exited successfully. + Succeeded, + /// Preparing or running the child failed. + Failed(ExecError), + /// The command was not started. + Skipped(SkipReason<'graph>), +} + +/// Why a selected command was not started. +#[derive(Clone, Copy, Debug)] +pub enum SkipReason<'graph> { + /// A prerequisite failed or was itself skipped. + DependencyFailed { + /// The first unsuccessful prerequisite in document order. + dependency: &'graph CommandId, + }, + /// Execution stopped after an earlier command failed. + StoppedAfterFailure { + /// The command whose failure stopped execution. + command: &'graph CommandId, + }, +} + +#[cfg(test)] +#[allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] +mod tests { + //! Unit tests for the command model, graph construction, and execution ordering. + + use std::collections::HashSet; + + use rstest::rstest; + + use super::*; + + // -------------------------------------------------------------------------- + // Helpers + // -------------------------------------------------------------------------- + + fn id(s: &str) -> CommandId { + CommandId::new(s).unwrap_or_else(|e| panic!("test id {s:?} should be valid: {e}")) + } + + fn location() -> SourceLocation { + SourceLocation::at_location(None, acdc_parser::Location::default()) + } + + fn block_at(name: &str) -> CommandBlock { + CommandBlock::new(id(name), format!("echo \"{name}\""), None, location()) + } + + /// Build a graph from `(command, [deps])` specs, expecting success. + fn graph(specs: &[(&str, &[&str])]) -> CommandGraph { + build(specs).unwrap_or_else(|e| panic!("graph should build: {e}")) + } + + /// Attempt to build a graph from `(command, [deps])` specs. + fn build(specs: &[(&str, &[&str])]) -> Result { + let mut builder = CommandGraphBuilder::new(); + for (name, deps) in specs { + builder.add(block_at(name), deps.iter().map(|d| id(d)).collect()); + } + builder.build() + } + + /// Collect selected command ids in execution order. + fn order(plan: &ExecutionPlan<'_>) -> Vec { + plan.commands() + .map(|b| b.metadata.id.as_str().to_string()) + .collect() + } + + /// Index of `name` within an ordered id list. + fn pos(ids: &[String], name: &str) -> usize { + ids.iter() + .position(|s| s == name) + .unwrap_or_else(|| panic!("{name:?} missing from {ids:?}")) + } + + // -------------------------------------------------------------------------- + // `CommandId` validation + // -------------------------------------------------------------------------- + + #[rstest] + #[case("foo")] + #[case("some-command")] + #[case("cmd_123")] + #[case("a")] + #[case("A")] + #[case("_")] + #[case("123")] + #[case("a-b_c-1")] + fn new_accepts_valid_ids(#[case] raw: &str) { + let id = CommandId::new(raw).unwrap_or_else(|e| panic!("{raw:?} should be valid: {e}")); + assert_eq!(id.as_str(), raw); + } + + #[test] + fn new_rejects_empty() { + assert!(matches!(CommandId::new(""), Err(InvalidCommandId::Empty))); + } + + #[rstest] + #[case("-foo")] + #[case("-")] + #[case("--bar")] + fn new_rejects_leading_dash(#[case] raw: &str) { + match CommandId::new(raw) { + Err(InvalidCommandId::LeadingDash { id }) => assert_eq!(id, raw), + other => panic!("expected LeadingDash, got {other:?}"), + } + } + + #[rstest] + #[case(" foo", ' ')] + #[case("foo bar", ' ')] + #[case("foo.bar", '.')] + #[case("foo/bar", '/')] + #[case("héllo", 'é')] + fn new_rejects_illegal_char(#[case] raw: &str, #[case] expected: char) { + match CommandId::new(raw) { + Err(InvalidCommandId::IllegalChar { id, ch }) => { + assert_eq!(id, raw); + assert_eq!(ch, expected); + } + other => panic!("expected IllegalChar, got {other:?}"), + } + } + + #[test] + fn from_str_matches_new() { + let parsed: CommandId = "build".parse().unwrap_or_else(|e| panic!("{e}")); + assert_eq!(parsed, id("build")); + } + + #[test] + fn borrow_enables_str_lookup() { + let mut set = HashSet::new(); + set.insert(id("build")); + assert!(set.contains("build")); + } + + #[test] + fn display_renders_inner_string() { + assert_eq!(id("deploy").to_string(), "deploy"); + } + + // -------------------------------------------------------------------------- + // `CommandBlock` + // -------------------------------------------------------------------------- + + #[test] + fn new_stores_id_script_and_default_interpreter() { + let block = CommandBlock::new(id("build"), "cargo build".into(), None, location()); + assert_eq!(block.metadata.id, id("build")); + assert_eq!(block.metadata.interpreter, "sh"); + assert_eq!(block.script, "cargo build"); + } + + #[test] + fn new_stores_explicit_interpreter() { + let block = CommandBlock::new(id("test"), String::new(), Some("bash".into()), location()); + assert_eq!(block.metadata.interpreter, "bash"); + } + + #[test] + fn new_preserves_multiline_script() { + let script = "set -e\ncargo build\necho done\n".to_string(); + let block = CommandBlock::new(id("build"), script.clone(), None, location()); + assert_eq!(block.script, script); + } + + #[test] + fn clone_is_independent() { + let block = CommandBlock::new( + id("deploy"), + "echo deploy".into(), + Some("bash".into()), + location(), + ); + let mut cloned = block.clone(); + cloned.script.push_str("\necho done\n"); + assert_eq!(block.script, "echo deploy"); + assert_eq!(cloned.script, "echo deploy\necho done\n"); + } + + // -------------------------------------------------------------------------- + // `CommandGraphBuilder::build` — success + // -------------------------------------------------------------------------- + + #[test] + fn build_empty_yields_empty_graph() { + let built = CommandGraphBuilder::new() + .build() + .unwrap_or_else(|e| panic!("{e}")); + assert_eq!(built.plan_all().len(), 0); + } + + #[test] + fn build_single_command() { + assert_eq!(order(&graph(&[("build", &[])]).plan_all()), ["build"]); + } + + #[test] + fn build_independent_commands_keep_document_order() { + // Kahn's algorithm with the smallest insertion index first preserves the + // order commands were added in when no dependencies force another order. + let built = graph(&[("c", &[]), ("a", &[]), ("b", &[])]); + assert_eq!(order(&built.plan_all()), ["c", "a", "b"]); + } + + #[test] + fn build_resolves_forward_referenced_dep() { + // `build` depends on `gen`, which is added *after* it. + let built = graph(&[("build", &["gen"]), ("gen", &[])]); + let ids = order(&built.plan_all()); + assert!(pos(&ids, "gen") < pos(&ids, "build")); + } + + #[test] + fn build_orders_chain_dependencies() { + let built = graph(&[("c", &["b"]), ("b", &["a"]), ("a", &[])]); + assert_eq!(order(&built.plan_all()), ["a", "b", "c"]); + } + + #[test] + fn build_orders_diamond_dependencies() { + let built = graph(&[("d", &["b", "c"]), ("b", &["a"]), ("c", &["a"]), ("a", &[])]); + let ids = order(&built.plan_all()); + assert!(pos(&ids, "a") < pos(&ids, "b")); + assert!(pos(&ids, "a") < pos(&ids, "c")); + assert!(pos(&ids, "b") < pos(&ids, "d")); + assert!(pos(&ids, "c") < pos(&ids, "d")); + } + + #[test] + fn build_dedupes_duplicate_dep() { + let built = graph(&[("b", &["a", "a"]), ("a", &[])]); + assert_eq!(order(&built.plan_all()), ["a", "b"]); + } + + // -------------------------------------------------------------------------- + // `CommandGraphBuilder::build` — errors + // -------------------------------------------------------------------------- + + #[test] + fn build_rejects_duplicate_id() { + match build(&[("build", &[]), ("build", &[])]) { + Err(BuildError::DuplicateId { id: got, .. }) => assert_eq!(got, id("build")), + other => panic!("expected DuplicateId, got {other:?}"), + } + } + + #[test] + fn build_rejects_unknown_dep() { + match build(&[("build", &["missing"])]) { + Err(BuildError::UnknownDep { dep: got, .. }) => assert_eq!(got, id("missing")), + other => panic!("expected UnknownDep, got {other:?}"), + } + } + + #[test] + fn build_rejects_self_dependency() { + match build(&[("a", &["a"])]) { + Err(BuildError::Cycle { command, dep, .. }) => { + assert_eq!(command, id("a")); + assert_eq!(dep, id("a")); + } + other => panic!("expected Cycle, got {other:?}"), + } + } + + #[test] + fn build_rejects_two_node_cycle() { + match build(&[("a", &["b"]), ("b", &["a"])]) { + Err(BuildError::Cycle { command, dep, .. }) => { + assert!( + (command == id("a") && dep == id("b")) + || (command == id("b") && dep == id("a")), + "cycle edge must name two distinct cycle members, got {command} -> {dep}" + ); + } + other => panic!("expected Cycle, got {other:?}"), + } + } + + #[test] + fn build_rejects_longer_cycle() { + match build(&[("a", &["c"]), ("b", &["a"]), ("c", &["b"])]) { + Err(BuildError::Cycle { command, dep, .. }) => { + assert_ne!(command, dep); + assert!( + (command == id("a") && dep == id("c")) + || (command == id("b") && dep == id("a")) + || (command == id("c") && dep == id("b")), + "reported edge must belong to the cycle: {dep} -> {command}" + ); + } + other => panic!("expected Cycle, got {other:?}"), + } + } + + #[test] + fn build_error_display() { + let duplicate = build(&[("build", &[]), ("build", &[])]).unwrap_err(); + assert_eq!(duplicate.to_string(), "duplicate command id: build"); + let missing = build(&[("build", &["gen"])]).unwrap_err(); + assert_eq!( + missing.to_string(), + "command `build` declares unknown dependency: gen" + ); + let cycle = build(&[("a", &["a"])]).unwrap_err(); + assert_eq!(cycle.to_string(), "dependency cycle includes: a -> a"); + } + + #[test] + fn invalid_id_display() { + assert_eq!( + InvalidCommandId::Empty.to_string(), + "command id must not be empty" + ); + assert_eq!(UnknownCommand(id("x")).to_string(), "unknown command: x"); + } + + // -------------------------------------------------------------------------- + // `CommandGraph::plan_for` + // -------------------------------------------------------------------------- + + #[test] + fn plan_for_unknown_id_errors() { + let built = graph(&[("a", &[])]); + match built.plan_for(&[id("nope")]) { + Err(UnknownCommand(got)) => assert_eq!(got, id("nope")), + Ok(_) => panic!("expected UnknownCommand"), + } + } + + #[test] + fn plan_for_empty_ids_is_empty() { + let built = graph(&[("a", &[]), ("b", &[])]); + let queue = built.plan_for(&[]).unwrap_or_else(|e| panic!("{e}")); + assert_eq!(queue.len(), 0); + } + + #[test] + fn plan_for_single_command_without_deps() { + let built = graph(&[("a", &[]), ("b", &[])]); + let queue = built.plan_for(&[id("a")]).unwrap_or_else(|e| panic!("{e}")); + assert_eq!(order(&queue), ["a"]); + } + + #[test] + fn plan_for_includes_transitive_deps_in_order() { + let built = graph(&[("c", &["b"]), ("b", &["a"]), ("a", &[]), ("unused", &[])]); + let queue = built.plan_for(&[id("c")]).unwrap_or_else(|e| panic!("{e}")); + assert_eq!(order(&queue), ["a", "b", "c"]); + } + + #[test] + fn plan_for_diamond_dedupes_shared_ancestor() { + let built = graph(&[("d", &["b", "c"]), ("b", &["a"]), ("c", &["a"]), ("a", &[])]); + let queue = built.plan_for(&[id("d")]).unwrap_or_else(|e| panic!("{e}")); + let ids = order(&queue); + assert_eq!( + ids.len(), + 4, + "shared ancestor `a` must appear once: {ids:?}" + ); + assert!(pos(&ids, "a") < pos(&ids, "b")); + assert!(pos(&ids, "a") < pos(&ids, "c")); + assert!(pos(&ids, "b") < pos(&ids, "d")); + assert!(pos(&ids, "c") < pos(&ids, "d")); + } + + #[test] + fn plan_for_duplicate_input_ids_dedupes() { + let built = graph(&[("a", &[]), ("b", &["a"])]); + let queue = built + .plan_for(&[id("b"), id("b")]) + .unwrap_or_else(|e| panic!("{e}")); + assert_eq!(order(&queue), ["a", "b"]); + } + + #[test] + fn plan_for_multiple_targets_dedupes() { + // `b` and `c` are independent targets that both depend on `a`. + let built = graph(&[("a", &[]), ("b", &["a"]), ("c", &["a"])]); + let queue = built + .plan_for(&[id("b"), id("c")]) + .unwrap_or_else(|e| panic!("{e}")); + let mut ids = order(&queue); + ids.sort(); + assert_eq!(ids, ["a", "b", "c"]); + } + + #[test] + fn plan_for_target_that_is_ancestor_of_another_target() { + // `a` is itself a prerequisite of `c`; requesting both must not duplicate `a`. + let built = graph(&[("a", &[]), ("b", &["a"]), ("c", &["b"])]); + let queue = built + .plan_for(&[id("c"), id("a")]) + .unwrap_or_else(|e| panic!("{e}")); + assert_eq!(order(&queue), ["a", "b", "c"]); + } + + #[test] + fn plan_commands_report_exact_len() { + let built = graph(&[("a", &[]), ("b", &["a"]), ("c", &["b"])]); + let plan = built.plan_for(&[id("c")]).unwrap(); + assert_eq!(plan.len(), 3); + let mut commands = plan.commands(); + assert_eq!(commands.len(), 3); + commands.next(); + assert_eq!(commands.len(), 2); + } + + #[test] + fn ids_iterates_in_topological_order() { + let built = graph(&[("c", &["b"]), ("b", &["a"]), ("a", &[])]); + assert_eq!( + built.ids().map(CommandId::as_str).collect::>(), + ["a", "b", "c"] + ); + assert!(built.contains("b")); + assert!(!built.contains("z")); + } + + #[test] + fn dependencies_return_only_unique_direct_prerequisites() { + let built = graph(&[ + ("deploy", &["test", "build", "test"]), + ("test", &["build"]), + ("build", &["prepare"]), + ("prepare", &[]), + ]); + let dependencies = { + let id = String::from("deploy"); + built.dependencies(&id).unwrap() + }; + let mut dependencies = dependencies.map(CommandId::as_str).collect::>(); + dependencies.sort_unstable(); + assert_eq!(dependencies, ["build", "test"]); + assert_eq!(built.dependencies("prepare").unwrap().count(), 0); + assert!(built.dependencies("missing").is_none()); + } + + // -------------------------------------------------------------------------- + // `CommandGraph::plan_all` + // -------------------------------------------------------------------------- + + #[test] + fn plan_all_yields_every_command_topologically() { + let built = graph(&[("a", &[]), ("b", &["a"]), ("c", &[]), ("d", &["b"])]); + let ids = order(&built.plan_all()); + assert_eq!(ids.len(), 4); + assert!(pos(&ids, "a") < pos(&ids, "b")); + assert!(pos(&ids, "b") < pos(&ids, "d")); + } + + // -------------------------------------------------------------------------- + // Execution + // -------------------------------------------------------------------------- + + #[cfg(unix)] + #[test] + fn execute_succeeds_on_zero_exit() { + let block = CommandBlock::new(id("ok"), "true".into(), None, location()); + block.execute().unwrap_or_else(|e| panic!("{e}")); + } + + #[cfg(unix)] + #[test] + fn execute_fails_on_nonzero_exit() { + let block = CommandBlock::new(id("bad"), "exit 3".into(), None, location()); + match block.execute() { + Err(ExecError::Failed(status)) => { + assert_eq!(status.code(), Some(3)); + } + Err(e) => panic!("expected Failed, got {e:?}"), + Ok(()) => panic!("expected failure"), + } + } + + #[test] + fn execute_reports_missing_interpreter() { + let block = CommandBlock::new( + id("nope"), + "true".into(), + Some("acdc-execute-nonexistent-interpreter".into()), + location(), + ); + assert!(matches!(block.execute(), Err(ExecError::Process(_)))); + } + + #[rstest] + #[case("")] + #[case("printf hello")] + #[case("printf hello\r\n\r\n")] + #[case("printf hello\n\n ")] + fn constructor_preserves_script_bytes(#[case] script: &str) { + let block = CommandBlock::new(id("exact"), script.to_owned(), None, location()); + assert_eq!(block.script.as_bytes(), script.as_bytes()); + } + + #[test] + fn plans_borrow_the_graphs_command_storage() { + let built = graph(&[("first", &[]), ("last", &["first"])]); + let first = built.plan_all(); + let second = built.plan_for(&[id("last")]).unwrap(); + for (original, selected) in first.commands().zip(second.commands()) { + assert!(std::ptr::eq(original, selected)); + } + } + + #[test] + fn graph_errors_keep_resolved_source_locations() { + let first = SourceLocation::at_position( + Some("first.adoc".into()), + acdc_parser::Position::new(3, 1), + ); + let second = SourceLocation::at_position( + Some("included/second.adoc".into()), + acdc_parser::Position::new(7, 1), + ); + let command = |location| CommandBlock::new(id("build"), String::new(), None, location); + let mut builder = CommandGraphBuilder::new(); + builder.add(command(first.clone()), Vec::new()); + builder.add(command(second.clone()), Vec::new()); + let error = builder.build().unwrap_err(); + assert_eq!(error.source_location(), &second); + assert_eq!(error.related_location(), Some(&first)); + + let mut builder = CommandGraphBuilder::new(); + builder.add(command(second.clone()), vec![id("missing")]); + let error = builder.build().unwrap_err(); + assert_eq!(error.source_location(), &second); + assert!(matches!(error, BuildError::UnknownDep { command, dep, .. } + if command == id("build") && dep == id("missing"))); + } + + #[cfg(unix)] + mod process_tests { + use std::{fs, os::unix::fs::PermissionsExt}; + + use super::*; + + fn script(name: &str, body: &str) -> CommandBlock { + CommandBlock::new(id(name), body.to_owned(), None, location()) + } + + fn execution_graph(specs: &[(&str, &[&str], &str)]) -> CommandGraph { + let mut builder = CommandGraphBuilder::new(); + for (name, dependencies, body) in specs { + builder.add( + script(name, body), + dependencies + .iter() + .map(|dependency| id(dependency)) + .collect(), + ); + } + builder.build().unwrap() + } + + fn options(directory: &Path) -> ExecutionOptions { + ExecutionOptions { + process: ProcessOptions { + current_dir: Some(directory.to_owned()), + env: Vec::new(), + }, + ..ExecutionOptions::default() + } + } + + #[test] + fn failed_prerequisite_skips_transitive_dependents_and_runs_independent_commands() { + let directory = tempfile::tempdir().unwrap(); + let graph = execution_graph(&[ + ("build", &[], "exit 7"), + ("test", &["build"], "touch test"), + ("deploy", &["test"], "touch deploy"), + ("independent", &[], "touch independent"), + ]); + let report = graph.plan_all().execute(&options(directory.path())); + assert!(!report.is_success()); + assert!(matches!(report.outcomes(), [ + CommandOutcome { state: CommandState::Failed(ExecError::Failed(status)), .. }, + CommandOutcome { state: CommandState::Skipped(SkipReason::DependencyFailed { dependency: first }), .. }, + CommandOutcome { state: CommandState::Skipped(SkipReason::DependencyFailed { dependency: second }), .. }, + CommandOutcome { state: CommandState::Succeeded, .. }, + ] if status.code() == Some(7) && first.as_str() == "build" && second.as_str() == "test")); + assert!(!directory.path().join("test").exists()); + assert!(!directory.path().join("deploy").exists()); + assert!(directory.path().join("independent").exists()); + } + + #[test] + fn shared_prerequisite_runs_once_before_both_branches() { + let directory = tempfile::tempdir().unwrap(); + let graph = execution_graph(&[ + ("join", &["left", "right"], "printf join >> order"), + ("left", &["first"], "printf 'left ' >> order"), + ("right", &["first"], "printf 'right ' >> order"), + ("first", &[], "printf 'first ' >> order"), + ]); + let report = graph + .plan_for(&[id("join")]) + .unwrap() + .execute(&options(directory.path())); + assert!(report.is_success()); + assert_eq!( + fs::read_to_string(directory.path().join("order")).unwrap(), + "first left right join" + ); + } + + #[test] + fn diamond_failure_skips_both_branches_and_their_join() { + let directory = tempfile::tempdir().unwrap(); + let graph = execution_graph(&[ + ("first", &[], "exit 3"), + ("left", &["first"], "touch left"), + ("right", &["first"], "touch right"), + ("join", &["right", "left"], "touch join"), + ]); + let report = graph.plan_all().execute(&options(directory.path())); + assert!(matches!(report.outcomes().last(), Some(CommandOutcome { + state: CommandState::Skipped(SkipReason::DependencyFailed { dependency }), .. + }) if dependency.as_str() == "left")); + assert_eq!(fs::read_dir(directory.path()).unwrap().count(), 0); + } + + #[test] + fn exit_on_failure_skips_independent_commands() { + let directory = tempfile::tempdir().unwrap(); + let graph = execution_graph(&[("bad", &[], "exit 1"), ("after", &[], "touch after")]); + let mut options = options(directory.path()); + options.exit_on_failure = true; + let report = graph.plan_all().execute(&options); + assert!(matches!(report.outcomes().last(), Some(CommandOutcome { + state: CommandState::Skipped(SkipReason::StoppedAfterFailure { command }), .. + }) if command.as_str() == "bad")); + assert!(!directory.path().join("after").exists()); + } + + #[test] + fn missing_interpreter_stops_all_commands() { + let directory = tempfile::tempdir().unwrap(); + let mut builder = CommandGraphBuilder::new(); + let mut bad = script("bad", "true"); + bad.metadata.interpreter = directory + .path() + .join("missing-interpreter") + .display() + .to_string(); + builder.add(bad, Vec::new()); + builder.add(script("dependent", "touch dependent"), vec![id("bad")]); + builder.add(script("independent", "touch independent"), Vec::new()); + let graph = builder.build().unwrap(); + let report = graph.plan_all().execute(&options(directory.path())); + assert!(matches!( + report.outcomes().first(), + Some(CommandOutcome { + state: CommandState::Failed(ExecError::Process(_)), + .. + }) + )); + assert!(report.outcomes().iter().skip(1).all(|outcome| matches!( + outcome.state, + CommandState::Skipped(SkipReason::StoppedAfterFailure { .. }) + ))); + assert_eq!(fs::read_dir(directory.path()).unwrap().count(), 0); + } + + #[test] + fn signal_termination_stops_independent_commands() { + let directory = tempfile::tempdir().unwrap(); + let graph = execution_graph(&[ + ("signal", &[], "kill -TERM $$"), + ("after", &[], "touch after"), + ]); + let report = graph.plan_all().execute(&options(directory.path())); + assert!(matches!(report.outcomes().first(), Some(CommandOutcome { + state: CommandState::Failed(ExecError::Failed(status)), .. + }) if status.code().is_none())); + assert!(matches!( + report.outcomes().last(), + Some(CommandOutcome { + state: CommandState::Skipped(SkipReason::StoppedAfterFailure { .. }), + .. + }) + )); + assert!(!directory.path().join("after").exists()); + } + + #[test] + fn child_options_do_not_change_the_parent() { + let directory = tempfile::tempdir().unwrap(); + let parent_directory = std::env::current_dir().unwrap(); + let parent_value = std::env::var_os("ACDC_EXECUTE_TEST_VALUE"); + let options = ProcessOptions { + current_dir: Some(directory.path().to_owned()), + env: vec![ + ("ACDC_EXECUTE_TEST_VALUE".into(), "first".into()), + ("ACDC_EXECUTE_TEST_VALUE".into(), "last=kept".into()), + ], + }; + script("child", "printf '%s' \"$ACDC_EXECUTE_TEST_VALUE\" > value") + .execute_with(&options) + .unwrap(); + assert_eq!( + fs::read_to_string(directory.path().join("value")).unwrap(), + "last=kept" + ); + assert_eq!(std::env::current_dir().unwrap(), parent_directory); + assert_eq!(std::env::var_os("ACDC_EXECUTE_TEST_VALUE"), parent_value); + } + + #[test] + fn interpreter_receives_exact_script_bytes() { + let directory = tempfile::tempdir().unwrap(); + let interpreter = directory.path().join("copy script"); + fs::write( + &interpreter, + "#!/bin/sh\ncat \"$1\" > \"$ACDC_SCRIPT_COPY\"\n", + ) + .unwrap(); + fs::set_permissions(&interpreter, fs::Permissions::from_mode(0o700)).unwrap(); + let options = ProcessOptions { + env: vec![( + "ACDC_SCRIPT_COPY".into(), + directory.path().join("copy").into_os_string(), + )], + ..ProcessOptions::default() + }; + for source in ["", "printf hello", "one\r\ntwo\r\n\r\n", "trailing\n\n "] { + let command = CommandBlock::new( + id("exact"), + source.into(), + Some(interpreter.display().to_string()), + location(), + ); + command.execute_with(&options).unwrap(); + assert_eq!( + fs::read(directory.path().join("copy")).unwrap(), + source.as_bytes() + ); + } + } + + #[test] + fn explicit_relative_interpreter_uses_the_callers_directory() { + let caller = std::env::current_dir().unwrap(); + let directory = tempfile::tempdir_in(&caller).unwrap(); + let interpreter = directory.path().join("interpreter"); + fs::write(&interpreter, "#!/bin/sh\n/bin/sh \"$1\"\n").unwrap(); + fs::set_permissions(&interpreter, fs::Permissions::from_mode(0o700)).unwrap(); + let child_directory = directory.path().join("child"); + fs::create_dir(&child_directory).unwrap(); + let relative = interpreter.strip_prefix(&caller).unwrap(); + let command = CommandBlock::new( + id("relative"), + "touch marker".into(), + Some(relative.display().to_string()), + location(), + ); + command + .execute_with(&ProcessOptions { + current_dir: Some(child_directory.clone()), + env: Vec::new(), + }) + .unwrap(); + assert!(child_directory.join("marker").exists()); + assert_eq!(std::env::current_dir().unwrap(), caller); + } + } +} diff --git a/acdc-execute/src/discovery/error.rs b/acdc-execute/src/discovery/error.rs new file mode 100644 index 00000000..513536f7 --- /dev/null +++ b/acdc-execute/src/discovery/error.rs @@ -0,0 +1,104 @@ +//! Errors reported while discovering executable commands. + +use acdc_parser::{SourceLocation, WarningKind}; + +use crate::command::{BuildError, InvalidCommandId}; + +/// A document could not provide a complete, valid command graph. +#[derive(Debug, thiserror::Error)] +pub enum DiscoveryError { + /// The parser could not preserve source content, structure, or requested substitutions. + #[error("cannot execute recovered source: {source}")] + RecoveredSource { + /// The parser's typed recovery diagnostic. + #[source] + source: WarningKind, + /// The original source position, when available. + location: Option>, + }, + /// A command or dependency has an invalid identifier. + #[error("command `{command}`: {source}")] + InvalidId { + /// The command declaring the invalid identifier or dependency. + command: String, + /// The identifier validation error. + source: InvalidCommandId, + /// Where the command is declared. + location: Box, + }, + /// Command dependencies could not form a valid graph. + #[error(transparent)] + Build(#[from] BuildError), + /// A command block has no identifier. + #[error("command block is missing an id")] + MissingId { + /// Where the block is declared. + location: Box, + }, + /// A marked block is not a listing or source paragraph. + #[error("command `{id}` is not a listing or source block")] + NotAScript { + /// The declared identifier. + id: String, + /// Where the block is declared. + location: Box, + }, + /// A script block has no retained source text. + #[error("command `{id}` has no retained script source")] + MissingSource { + /// The declared identifier. + id: String, + /// Where the block is declared. + location: Box, + }, + /// An interpreter must be a nonempty executable name or path. + #[error("command `{id}` requires a nonempty interpreter name or path without NUL bytes")] + InvalidInterpreter { + /// The declared identifier. + id: String, + /// Where the block is declared. + location: Box, + }, + /// Attribute substitution referenced an unset or unknown document attribute. + #[error("command `{id}` references missing document attribute `{name}`")] + MissingAttribute { + /// The command containing the reference. + id: String, + /// The first unresolved attribute name in the script. + name: String, + /// Where the command is declared. + location: Box, + }, +} + +impl DiscoveryError { + /// The original file and position associated with this error, when known. + #[must_use] + pub fn source_location(&self) -> Option<&SourceLocation> { + match self { + Self::RecoveredSource { location, .. } => location.as_deref(), + Self::Build(error) => Some(error.source_location()), + Self::InvalidId { location, .. } + | Self::MissingId { location } + | Self::NotAScript { location, .. } + | Self::MissingSource { location, .. } + | Self::InvalidInterpreter { location, .. } + | Self::MissingAttribute { location, .. } => Some(location), + } + } + + /// A second source position involved in a duplicate identifier or cycle. + #[must_use] + pub fn related_location(&self) -> Option<&SourceLocation> { + match self { + Self::Build(error) => error.related_location(), + Self::RecoveredSource { .. } + | Self::InvalidId { .. } + | Self::MissingId { .. } + | Self::NotAScript { .. } + | Self::MissingSource { .. } + | Self::InvalidInterpreter { .. } + | Self::MissingAttribute { .. } => None, + } + } +} diff --git a/acdc-execute/src/discovery/mod.rs b/acdc-execute/src/discovery/mod.rs new file mode 100644 index 00000000..40c6da70 --- /dev/null +++ b/acdc-execute/src/discovery/mod.rs @@ -0,0 +1,1173 @@ +//! Discover executable commands from parsed `AsciiDoc` source. + +mod error; + +pub use error::DiscoveryError; + +use acdc_converters_core::{InlineTextTransform, TraversalContext, visitor::Visitor}; +use acdc_parser::{ + AttributeValue, Block, BlockMetadata, DelimitedBlockType, InlineNode, ParseResult, Section, + SourceLocation, Substitution, VERBATIM, substitute_attributes, +}; + +use crate::command::{CommandBlock, CommandGraph, CommandGraphBuilder, CommandId}; + +impl TryFrom<&ParseResult> for CommandGraph { + type Error = DiscoveryError; + + fn try_from(parsed: &ParseResult) -> Result { + validate_source(parsed)?; + let document = parsed.document(); + let mut traversal = TraversalContext::new(&document.attributes); + let mut collector = CommandCollector { + parsed, + builder: CommandGraphBuilder::new(), + section_title: None, + }; + collector.visit_document(&mut traversal, document)?; + Ok(collector.builder.build()?) + } +} + +fn validate_source(parsed: &ParseResult) -> Result<(), DiscoveryError> { + if let Some(warning) = parsed.source_recovery() { + return Err(DiscoveryError::RecoveredSource { + source: warning.kind.clone(), + location: warning.source_location().cloned().map(Box::new), + }); + } + Ok(()) +} + +struct CommandCollector<'doc> { + parsed: &'doc ParseResult, + builder: CommandGraphBuilder, + section_title: Option<&'doc [InlineNode<'doc>]>, +} + +impl<'doc> Visitor<'doc> for CommandCollector<'doc> { + type Error = DiscoveryError; + + fn before_block( + &mut self, + traversal: &mut TraversalContext<'doc>, + block: &'doc Block<'doc>, + ) -> Result<(), Self::Error> { + if let Some(metadata) = block.metadata() + && metadata.roles.contains(&"command") + { + let (mut command, dependencies) = parse_command_block( + block, + metadata, + self.parsed.source_location(block.location()), + traversal, + )?; + command.metadata.section_title = self.section_title.map(|title| { + InlineTextTransform::default() + .decode_char_refs(true) + .references(&self.parsed.document().references) + .to_string(title) + }); + self.builder.add(command, dependencies); + } + Ok(()) + } + + fn visit_section( + &mut self, + traversal: &mut TraversalContext<'doc>, + section: &'doc Section<'doc>, + ) -> Result<(), Self::Error> { + let parent = self.section_title.replace(§ion.title); + let result = traversal.visit_blocks(self, §ion.content); + self.section_title = parent; + result + } + + fn visit_inline_nodes( + &mut self, + _traversal: &mut TraversalContext<'doc>, + _nodes: &[InlineNode<'_>], + ) -> Result<(), Self::Error> { + Ok(()) + } +} + +fn parse_command_block( + block: &Block<'_>, + metadata: &BlockMetadata<'_>, + location: SourceLocation, + traversal: &TraversalContext<'_>, +) -> Result<(CommandBlock, Vec), DiscoveryError> { + let anchor = metadata + .id + .as_ref() + .ok_or_else(|| DiscoveryError::MissingId { + location: Box::new(location.clone()), + })?; + #[expect( + clippy::wildcard_enum_match_arm, + reason = "Only script block variants are accepted" + )] + let source_text = match block { + Block::DelimitedBlock(block) + if matches!(block.inner, DelimitedBlockType::DelimitedListing(_)) => + { + block.source_text() + } + Block::Paragraph(paragraph) if matches!(metadata.style, Some("source" | "listing")) => { + paragraph.source_text() + } + _ => { + return Err(DiscoveryError::NotAScript { + id: anchor.id.to_owned(), + location: Box::new(location), + }); + } + } + .ok_or_else(|| DiscoveryError::MissingSource { + id: anchor.id.to_owned(), + location: Box::new(location.clone()), + })?; + + let invalid_id = |source| DiscoveryError::InvalidId { + command: anchor.id.to_owned(), + source, + location: Box::new(location.clone()), + }; + let id = anchor.id.parse().map_err(invalid_id)?; + let dependencies = metadata + .attributes + .get_string("deps") + .map(|dependencies| { + dependencies + .split(',') + .map(str::trim) + .filter(|dependency| !dependency.is_empty()) + .map(str::parse) + .collect::, _>>() + .map_err(invalid_id) + }) + .transpose()? + .unwrap_or_default(); + let interpreter = + source_interpreter(metadata).map_err(|()| DiscoveryError::InvalidInterpreter { + id: anchor.id.to_owned(), + location: Box::new(location.clone()), + })?; + let description = metadata + .attributes + .get_string("description") + .map(std::borrow::Cow::into_owned); + let script = if metadata.uses_substitution(&Substitution::Attributes, VERBATIM) { + prepare_script(source_text, traversal, anchor.id, &location)? + } else { + source_text.to_owned() + }; + Ok(( + CommandBlock::new(id, script, interpreter, location).with_description(description), + dependencies, + )) +} + +fn prepare_script( + source: &str, + traversal: &TraversalContext<'_>, + id: &str, + location: &SourceLocation, +) -> Result { + let mut missing = None; + let script = substitute_attributes(source, |name| { + let value = traversal.get(name); + if value.is_none() && missing.is_none() { + missing = Some(name.to_owned()); + } + value + }); + if let Some(name) = missing { + return Err(DiscoveryError::MissingAttribute { + id: id.to_owned(), + name, + location: Box::new(location.clone()), + }); + } + Ok(script.into_owned()) +} + +fn source_interpreter(metadata: &BlockMetadata<'_>) -> Result, ()> { + let interpreter = match metadata.attributes.get("interpreter") { + Some(AttributeValue::String(value)) => Some(value.to_string()), + Some(_) => return Err(()), + None if metadata.style == Some("source") => metadata + .attributes + .get_string("language") + .map(std::borrow::Cow::into_owned), + None => None, + }; + if interpreter + .as_ref() + .is_some_and(|value| value.trim().is_empty() || value.contains('\0')) + { + return Err(()); + } + Ok(interpreter) +} + +#[cfg(test)] +#[allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] +mod tests { + //! Unit tests for discovery: building a [`CommandGraph`] from a parsed `AsciiDoc` document. + + use acdc_parser::{Options, ParseResult, SafeMode}; + use rstest::rstest; + + use crate::{ + DiscoveryError, + command::{BuildError, CommandBlock, CommandGraph}, + }; + + // -------------------------------------------------------------------------- + // Helpers + // -------------------------------------------------------------------------- + + fn parse(src: &str) -> ParseResult { + acdc_parser::parse(src, &Options::default()).unwrap_or_else(|e| panic!("parse failed: {e}")) + } + + fn try_graph(src: &str) -> Result { + CommandGraph::try_from(&parse(src)) + } + + fn graph(src: &str) -> CommandGraph { + try_graph(src).unwrap_or_else(|e| panic!("graph should build: {e}")) + } + + fn err(src: &str) -> DiscoveryError { + try_graph(src).expect_err("graph build should fail") + } + + /// Command ids in execution (topological) order. + fn ids(built: &CommandGraph) -> Vec { + built + .plan_all() + .commands() + .map(|b| b.metadata.id.as_str().to_string()) + .collect() + } + + /// The single command block with `id`, failing the test if absent. + fn find(built: &CommandGraph, id: &str) -> CommandBlock { + built + .plan_all() + .commands() + .find(|b| b.metadata.id.as_str() == id) + .cloned() + .unwrap_or_else(|| panic!("command {id:?} not found")) + } + + /// Index of `name` within an ordered id list. + fn pos(ids: &[String], name: &str) -> usize { + ids.iter() + .position(|s| s == name) + .unwrap_or_else(|| panic!("{name:?} missing from {ids:?}")) + } + + /// Render a single command block. `deps` and `lang` are optional; `body` is the script. + fn cmd(id: &str, deps: Option<&str>, lang: Option<&str>, body: &str) -> String { + let deps = deps.map(|d| format!(", deps=\"{d}\"")).unwrap_or_default(); + let lang = lang.map(|l| format!("[source, {l}]\n")).unwrap_or_default(); + format!("[.command, id={id}{deps}]\n{lang}----\n{body}\n----\n") + } + + /// Wrap `body` in a level-1 section titled `name`. + fn section(name: &str, body: &str) -> String { + format!("== {name}\n\n{body}") + } + + // -------------------------------------------------------------------------- + // Discovery: where command blocks are found + // -------------------------------------------------------------------------- + + #[test] + fn empty_document_yields_empty_graph() { + assert_eq!(ids(&graph("= Doc\n")), Vec::::new()); + } + + #[test] + fn document_with_no_commands_yields_empty_graph() { + let src = "= Doc\n\nSome prose.\n\n[source, bash]\n----\necho not-a-command\n----\n"; + assert_eq!(ids(&graph(src)), Vec::::new()); + } + + #[rstest] + #[case::top_level(cmd("build", None, None, "echo hi"))] + #[case::open_block(format!("--\n{}--\n", cmd("build", None, None, "echo hi")))] + #[case::source_paragraph("[source,bash,role=command,id=build]\necho hi".into())] + #[case::listing_paragraph("[listing,role=command,id=build]\necho hi".into())] + #[case::nested_in_section(format!( + "= Doc\n\n{}", + section("Build", &cmd("build", None, None, "echo hi")) +))] + #[case::nested_subsection(format!( + "= Doc\n\n== Top\n\n{}", + section("Build", &cmd("build", None, None, "echo hi")) +))] + #[case::example_block(format!("= Doc\n\n====\n{}====\n", cmd("build", None, None, "echo hi")))] + #[case::sidebar_block(format!("= Doc\n\n****\n{}****\n", cmd("build", None, None, "echo hi")))] + #[case::quote_block(format!("= Doc\n\n____\n{}____\n", cmd("build", None, None, "echo hi")))] + #[case::admonition_block(format!( + "= Doc\n\n[NOTE]\n====\n{}====\n", + cmd("build", None, None, "echo hi") +))] + #[case::ordered_list_item(format!("= Doc\n\n. Build it\n+\n{}", cmd("build", None, None, "echo hi")))] + #[case::unordered_list_item(format!("= Doc\n\n* Build it\n+\n{}", cmd("build", None, None, "echo hi")))] + #[case::description_list_item(format!( + "= Doc\n\nBuild:: Compile the project\n+\n{}", + cmd("build", None, None, "echo hi") +))] + fn single_command_is_discovered(#[case] src: String) { + assert_eq!(ids(&graph(&src)), ["build"]); + } + + #[test] + fn commands_across_multiple_sections_are_all_found() { + let src = format!( + "= Doc\n\n{}\n{}", + section("Build", &cmd("build", None, None, "echo build")), + section("Test", &cmd("test", Some("build"), None, "echo test")), + ); + let ids = ids(&graph(&src)); + assert_eq!(ids.len(), 2); + assert!(pos(&ids, "build") < pos(&ids, "test")); + } + + #[test] + fn commands_keep_the_nearest_section_title_as_plain_text() { + let built = graph(&format!( + concat!( + "= Commands\n:project: acdc\n\n{}\n", + "== *Build* {{project}}{{apos}}s tools\n\n{}\n", + "=== _Unit_ tests\n\n--\n{}--\n\n", + "== Deploy\n\n{}\n", + "[discrete]\n=== A separate heading\n\n{}\n", + ), + cmd("preamble", None, None, "true"), + cmd("build", None, None, "true"), + cmd("test", None, None, "true"), + cmd("deploy", None, None, "true"), + cmd("after-discrete", None, None, "true"), + )); + assert_eq!(find(&built, "preamble").metadata.section_title, None); + assert_eq!( + find(&built, "build").metadata.section_title.as_deref(), + Some("Build acdc's tools") + ); + assert_eq!( + find(&built, "test").metadata.section_title.as_deref(), + Some("Unit tests") + ); + for id in ["deploy", "after-discrete"] { + assert_eq!( + find(&built, id).metadata.section_title.as_deref(), + Some("Deploy") + ); + } + } + + #[rstest] + #[case(None)] + #[case(Some("Outer section"))] + fn table_cell_sections_restore_the_enclosing_section(#[case] parent: Option<&str>) { + let body = format!( + "[cols=a]\n|===\na|== Cell section\n\n{}\n|===\n\n{}", + cmd("cell", None, None, "true"), + cmd("after", None, None, "true"), + ); + let source = parent.map_or_else(|| body.clone(), |title| section(title, &body)); + let built = graph(&format!("= Commands\n\n{source}")); + assert_eq!( + find(&built, "cell").metadata.section_title.as_deref(), + Some("Cell section") + ); + assert_eq!( + find(&built, "after").metadata.section_title.as_deref(), + parent + ); + } + + #[test] + fn commands_across_distinct_list_items_are_all_found() { + let src = format!( + "= Doc\n\n. First\n+\n{}\n. Second\n+\n{}", + cmd("build", None, None, "echo build"), + cmd("test", Some("build"), None, "echo test"), + ); + let ids = ids(&graph(&src)); + assert_eq!(ids.len(), 2); + assert!(pos(&ids, "build") < pos(&ids, "test")); + } + + // -------------------------------------------------------------------------- + // Non-commands are ignored + // -------------------------------------------------------------------------- + + #[test] + fn listing_without_command_role_is_ignored() { + let src = "[source, bash]\n----\necho hi\n----\n"; + assert!(ids(&graph(src)).is_empty()); + } + + #[test] + fn command_role_without_id_errors() { + // A `command` block with no id is a typo, not a no-op: surface it. + let src = "[.command]\n----\necho hi\n----\n"; + assert!(matches!(err(src), DiscoveryError::MissingId { .. })); + } + + #[test] + fn command_role_on_non_listing_block_errors() { + // An example block carrying the command role is not a script; it cannot be run. + let src = "[.command, id=x]\n====\nsome content\n====\n"; + assert!(matches!(err(src), DiscoveryError::NotAScript { ref id, .. } if id == "x")); + } + + #[test] + fn non_script_paragraph_with_command_role_is_an_error() { + assert!(matches!( + err("[.command,id=prose]\nThis is a paragraph.\n"), + DiscoveryError::NotAScript { .. } + )); + assert!(matches!( + err("[.command]\nThis is a paragraph.\n"), + DiscoveryError::MissingId { .. } + )); + } + + // -------------------------------------------------------------------------- + // Discovery across included files + // -------------------------------------------------------------------------- + + #[test] + fn command_in_included_file_is_discovered() { + let dir = tempfile::tempdir().unwrap_or_else(|e| panic!("{e}")); + let main = dir.path().join("main.adoc"); + let included = dir.path().join("commands.adoc"); + std::fs::write(&main, "= Doc\n\n== Build\n\ninclude::commands.adoc[]\n") + .unwrap_or_else(|e| panic!("{e}")); + std::fs::write(&included, cmd("build", None, None, "echo hi")) + .unwrap_or_else(|e| panic!("{e}")); + + let parsed = acdc_parser::parse_file(&main, &Options::default()) + .unwrap_or_else(|e| panic!("parse_file failed: {e}")); + let built = CommandGraph::try_from(&parsed).unwrap_or_else(|e| panic!("{e}")); + + assert_eq!(ids(&built), ["build"]); + assert_eq!( + find(&built, "build").metadata.section_title.as_deref(), + Some("Build") + ); + } + + // -------------------------------------------------------------------------- + // Shell / language + // -------------------------------------------------------------------------- + + #[test] + fn default_interpreter_is_sh_when_no_language() { + let block = find(&graph(&cmd("build", None, None, "echo hi")), "build"); + assert_eq!(block.metadata.interpreter, "sh"); + } + + #[test] + fn source_block_with_no_language_defaults_to_sh() { + // `[source]` with no language: style=="source" but no None-valued attribute key. + // Different code path from having no [source,...] annotation at all. + let src = "[.command, id=build]\n[source]\n----\necho hi\n----\n"; + let block = find(&graph(src), "build"); + assert_eq!(block.metadata.interpreter, "sh"); + } + + #[test] + fn source_interpreter_ignores_block_options() { + let src = "[.command, id=build]\n[source, bash, %linenums]\n----\necho hi\n----\n"; + let block = find(&graph(src), "build"); + assert_eq!(block.metadata.interpreter, "bash"); + } + + #[rstest] + #[case("bash")] + #[case("zsh")] + #[case("python3")] + #[case("fish")] + fn source_interpreter_sets_interpreter(#[case] lang: &str) { + let block = find(&graph(&cmd("build", None, Some(lang), "echo hi")), "build"); + assert_eq!(block.metadata.interpreter, lang); + } + + // -------------------------------------------------------------------------- + // Script body + // -------------------------------------------------------------------------- + + #[rstest] + #[case::single_line("cargo build", "cargo build\n")] + #[case::multiline("set -e\ncargo build", "set -e\ncargo build\n")] + fn script_body_is_captured(#[case] body: &str, #[case] expected: &str) { + let block = find(&graph(&cmd("build", None, None, body)), "build"); + assert_eq!(block.script, expected); + } + + #[test] + fn script_location_reports_the_block_line() { + // The command block opens on line 3 (after the title and a blank line). + let src = "= Doc\n\n[.command, id=build]\n----\necho hi\n----\n"; + let block = find(&graph(src), "build"); + assert_eq!(block.location.location.start.line, 3); + } + + // -------------------------------------------------------------------------- + // Dependencies + // -------------------------------------------------------------------------- + + #[test] + fn dependency_orders_prerequisite_first() { + let src = format!( + "{}\n{}", + cmd("build", None, None, "echo build"), + cmd("test", Some("build"), None, "echo test"), + ); + assert_eq!(ids(&graph(&src)), ["build", "test"]); + } + + #[test] + fn forward_referenced_dependency_resolves() { + // `test` is declared *before* the `build` it depends on. + let src = format!( + "{}\n{}", + cmd("test", Some("build"), None, "echo test"), + cmd("build", None, None, "echo build"), + ); + assert_eq!(ids(&graph(&src)), ["build", "test"]); + } + + #[test] + fn multiple_dependencies_all_precede_dependent() { + let src = format!( + "{}\n{}\n{}", + cmd("a", None, None, "echo a"), + cmd("b", None, None, "echo b"), + cmd("c", Some("a, b"), None, "echo c"), + ); + let ids = ids(&graph(&src)); + assert!(pos(&ids, "a") < pos(&ids, "c")); + assert!(pos(&ids, "b") < pos(&ids, "c")); + } + + #[rstest] + #[case("build", &["build"])] + #[case("a,b", &["a", "b"])] + #[case("a, b", &["a", "b"])] + #[case(" a , b ", &["a", "b"])] + #[case("a, b, c", &["a", "b", "c"])] + fn deps_attribute_is_split_and_trimmed(#[case] deps: &str, #[case] expected: &[&str]) { + // Provide every named prerequisite so the graph resolves, then check ordering. + let mut src = String::new(); + for dep in expected { + src.push_str(&cmd(dep, None, None, "echo dep")); + src.push('\n'); + } + src.push_str(&cmd("target", Some(deps), None, "echo target")); + + let ids = ids(&graph(&src)); + for dep in expected { + assert!( + pos(&ids, dep) < pos(&ids, "target"), + "{dep} should precede target" + ); + } + } + + #[test] + fn trailing_and_empty_dep_segments_are_ignored() { + let src = format!( + "{}\n{}", + cmd("build", None, None, "echo build"), + cmd("test", Some("build, ,"), None, "echo test"), + ); + assert_eq!(ids(&graph(&src)), ["build", "test"]); + } + + #[test] + fn empty_deps_attribute_yields_no_dependencies() { + // deps="" splits to [""], filtered to nothing — command has no prerequisites. + let src = cmd("build", Some(""), None, "echo build"); + assert_eq!(ids(&graph(&src)), ["build"]); + } + + // -------------------------------------------------------------------------- + // Errors + // -------------------------------------------------------------------------- + + #[test] + fn invalid_dependency_id_errors() { + let src = format!( + "{}\n{}", + cmd("build", None, None, "echo build"), + cmd("test", Some("bad id"), None, "echo test"), + ); + assert!(matches!(err(&src), DiscoveryError::InvalidId { .. })); + } + + #[test] + fn invalid_command_id_errors() { + let src = "[.command, id=bad.id]\n----\necho hi\n----\n"; + assert!(matches!(err(src), DiscoveryError::InvalidId { .. })); + } + + #[test] + fn duplicate_command_id_errors() { + let src = format!( + "{}\n{}", + cmd("build", None, None, "echo one"), + cmd("build", None, None, "echo two"), + ); + assert!(matches!( + err(&src), + DiscoveryError::Build(BuildError::DuplicateId { .. }) + )); + } + + #[test] + fn unknown_dependency_errors() { + let src = cmd("test", Some("missing"), None, "echo test"); + assert!(matches!( + err(&src), + DiscoveryError::Build(BuildError::UnknownDep { .. }) + )); + } + + #[test] + fn self_dependency_errors() { + let src = cmd("loop", Some("loop"), None, "echo loop"); + assert!(matches!( + err(&src), + DiscoveryError::Build(BuildError::Cycle { .. }) + )); + } + + #[test] + fn dependency_cycle_errors() { + let src = format!( + "{}\n{}", + cmd("a", Some("b"), None, "echo a"), + cmd("b", Some("a"), None, "echo b"), + ); + assert!(matches!( + err(&src), + DiscoveryError::Build(BuildError::Cycle { .. }) + )); + } + + #[test] + fn missing_id_error_reports_the_block_line() { + let src = "= Doc\n\n[.command]\n----\necho hi\n----\n"; + let DiscoveryError::MissingId { location } = err(src) else { + panic!("expected MissingId") + }; + assert_eq!(location.location.start.line, 3); + } + + #[test] + fn not_a_script_error_reports_the_block_line() { + let src = "= Doc\n\n[.command, id=x]\n====\nsome content\n====\n"; + let DiscoveryError::NotAScript { location, .. } = err(src) else { + panic!("expected NotAScript") + }; + assert_eq!(location.location.start.line, 3); + } + + // -------------------------------------------------------------------------- + // Safe mode is a document-read policy, not a command sandbox + // -------------------------------------------------------------------------- + + #[test] + fn safe_mode_still_discovers_commands() { + // Commands are discovered and validated identically under SECURE safe mode; + // safe mode only limits what the document may include. + let options = Options::builder() + .with_safe_mode(SafeMode::Secure) + .build() + .unwrap_or_else(|e| panic!("{e}")); + let src = "[.command, id=build]\n----\necho hi\n----\n"; + let parsed = acdc_parser::parse(src, &options).unwrap_or_else(|e| panic!("{e}")); + let built = CommandGraph::try_from(&parsed).unwrap_or_else(|e| panic!("{e}")); + assert_eq!(ids(&built), ["build"]); + } + + // -------------------------------------------------------------------------- + // End-to-end example + // -------------------------------------------------------------------------- + + #[test] + fn readme_example_builds_expected_graph() { + let src = "= My Project\n\n== Build\n\n[.command, id=build]\n[source, bash]\n----\ncargo xtask build\n----\n\n== Tests\n\n[.command, id=test, deps=\"build\"]\n[source, bash]\n----\ncargo nextest run\n----\n"; + let built = graph(src); + let plan = built.plan_all(); + let blocks: Vec<&CommandBlock> = plan.commands().collect(); + + let names: Vec<&str> = blocks.iter().map(|b| b.metadata.id.as_str()).collect(); + assert_eq!(names, ["build", "test"]); + let first = blocks.first().unwrap_or_else(|| panic!("build missing")); + let second = blocks.get(1).unwrap_or_else(|| panic!("test missing")); + assert_eq!(first.metadata.interpreter, "bash"); + assert_eq!(first.script, "cargo xtask build\n"); + assert_eq!(second.script, "cargo nextest run\n"); + } + + #[rstest] + #[case("cat <<'EOF'\n<1>\nEOF\n")] + #[case("printf '%s' \\<1>\n")] + #[case("\n<.>\n")] + #[case("echo '{value}' *bold*\n\n")] + fn scripts_preserve_literal_body(#[case] body: &str) { + let source = format!( + ":value: replaced\n\n[source,bash,role=command,id=literal]\n----\n{body}----\n" + ); + assert_eq!(find(&graph(&source), "literal").script, body); + } + + #[test] + fn substitution_follows_the_parser_feature_even_when_features_are_unified() { + let mut parsed = parse( + ":value: expanded\n\n[.command,id=run,subs=attributes]\n----\nprintf '{value}'\n----\n", + ); + let unsupported = parsed.source_recovery().is_some(); + parsed.take_warnings(); + let result = CommandGraph::try_from(&parsed); + if unsupported { + assert!(matches!( + result, + Err(DiscoveryError::RecoveredSource { .. }) + )); + } else { + assert_eq!(find(&result.unwrap(), "run").script, "printf 'expanded'\n"); + } + } + + #[cfg(feature = "pre-spec-subs")] + mod attribute_substitution { + use super::*; + + fn command(id: &str, subs: &str, body: &str) -> String { + format!("[source,sh,role=command,id={id},subs=\"{subs}\"]\n----\n{body}----\n") + } + + #[rstest] + #[case("attributes", "expanded")] + #[case("+attributes", "expanded")] + #[case("attributes+", "expanded")] + #[case("normal", "expanded")] + #[case("none", "{value}")] + #[case("-attributes", "{value}")] + #[case("verbatim", "{value}")] + #[case("-attributes,+attributes", "expanded")] + #[case("+attributes,-attributes", "{value}")] + fn resolves_substitutions_against_verbatim_defaults( + #[case] subs: &str, + #[case] expected: &str, + ) { + let source = format!(":value: expanded\n\n{}", command("run", subs, "{value}\n")); + assert_eq!(find(&graph(&source), "run").script, format!("{expected}\n")); + } + + #[rstest] + #[case("source,sh")] + #[case("listing")] + fn source_paragraphs_expand_attributes_without_adding_a_newline(#[case] style: &str) { + let source = format!( + ":value: expanded\n\n[{style},role=command,id=run,subs=attributes]\nprintf '{{value}}'" + ); + assert_eq!(find(&graph(&source), "run").script, "printf 'expanded'"); + } + + #[test] + fn enabled_attributes_leave_other_code_and_callouts_unchanged() { + let body = "printf '*bold* _italic_ <1> -- ... {value}'\n\n \n"; + let source = format!(":value: expanded\n\n{}", command("run", "normal", body)); + assert_eq!( + find(&graph(&source), "run").script, + "printf '*bold* _italic_ <1> -- ... expanded'\n\n\n" + ); + } + + #[test] + fn escaped_references_are_literal_and_do_not_require_a_value() { + let source = format!( + ":value: expanded\n\n{}", + command("run", "attributes", "\\{value} \\{missing} {value}\n") + ); + assert_eq!( + find(&graph(&source), "run").script, + "{value} {missing} expanded\n" + ); + } + + #[test] + fn malformed_references_remain_literal() { + let source = command("run", "attributes", "{} {not a name} {missing\n"); + assert_eq!( + find(&graph(&source), "run").script, + "{} {not a name} {missing\n" + ); + } + + #[test] + fn first_missing_attribute_is_a_located_preflight_error() { + let source = command("run", "attributes", "{first} {second}\n"); + let error = err(&source); + assert!( + matches!(&error, DiscoveryError::MissingAttribute { id, name, .. } if id == "run" && name == "first") + ); + assert!(error.source_location().is_some()); + assert!(error.related_location().is_none()); + assert_eq!( + error.to_string(), + "command `run` references missing document attribute `first`" + ); + } + + #[test] + fn presence_attributes_expand_to_empty_text() { + let source = format!( + ":empty:\n\n{}", + command("run", "attributes", "before{empty}after\n") + ); + assert_eq!(find(&graph(&source), "run").script, "beforeafter\n"); + } + + #[test] + fn unset_attributes_mask_header_values() { + let source = format!( + ":value: header\n\nBody.\n\n:value!:\n\n{}", + command("run", "attributes", "{value}\n") + ); + assert!( + matches!(err(&source), DiscoveryError::MissingAttribute { name, .. } if name == "value") + ); + } + + #[test] + fn disabled_attributes_do_not_reject_missing_references() { + let source = command("run", "none", "{missing} \\{also-missing}\n"); + assert_eq!( + find(&graph(&source), "run").script, + "{missing} \\{also-missing}\n" + ); + } + + #[test] + fn substituted_values_are_not_expanded_again() { + let options = Options::with_attributes([("value", "{missing}")]).unwrap(); + let source = command("run", "attributes", "{value}\n"); + let parsed = acdc_parser::parse(&source, &options).unwrap(); + assert_eq!( + find(&CommandGraph::try_from(&parsed).unwrap(), "run").script, + "{missing}\n" + ); + } + + #[test] + fn source_order_attributes_are_frozen_before_dependency_ordering() { + let source = concat!( + "= Commands\n:value: header\n\n", + "[.command,id=first,deps=second,subs=attributes]\n----\n{value}\n----\n\n", + ":value: later\n\n", + "[.command,id=second,subs=attributes]\n----\n{value}\n----\n", + ); + let graph = graph(source); + assert_eq!(ids(&graph), ["second", "first"]); + assert_eq!(find(&graph, "first").script, "header\n"); + assert_eq!(find(&graph, "second").script, "later\n"); + } + + #[rstest] + #[case::inherited_value_is_locked("value", "outer\n")] + #[case::local_value_is_accepted("local", "cell\n")] + fn table_cell_attributes_obey_inheritance_and_scope( + #[case] name: &str, + #[case] expected: &str, + ) { + let source = format!( + concat!( + "= Commands\n:value: outer\n\n[cols=\"a,a\"]\n|===\n", + "|\nCell one.\n\n:{name}: cell\n\n", + "[.command,id=cell,subs=attributes]\n----\n{{{name}}}\n----\n", + "|\n[.command,id=sibling,subs=attributes]\n----\n{{value}}\n----\n", + "|===\n\n", + "[.command,id=after,subs=attributes]\n----\n{{value}}\n----\n", + ), + name = name + ); + let graph = graph(&source); + assert_eq!(find(&graph, "cell").script, expected); + assert_eq!(find(&graph, "sibling").script, "outer\n"); + assert_eq!(find(&graph, "after").script, "outer\n"); + } + + #[rstest] + #[case::sibling(true)] + #[case::parent(false)] + fn local_table_attributes_are_missing_outside_their_cell(#[case] sibling: bool) { + let cell = command("inside", "attributes", "{local}\n"); + let outside = command("outside", "attributes", "{local}\n"); + let mut source = format!("[cols=a]\n|===\n|\n:local: cell\n\n{cell}\n"); + source.push_str(if sibling { "|\n" } else { "|===\n\n" }); + source.push_str(&outside); + if sibling { + source.push_str("|===\n"); + } + assert!( + matches!(err(&source), DiscoveryError::MissingAttribute { id, name, .. } + if id == "outside" && name == "local") + ); + } + + #[test] + fn table_cells_use_their_initial_document_attributes() { + let source = concat!( + "= Commands\n:doctype: book\n\n[cols=a]\n|===\n|\n", + "[.command,id=cell,subs=attributes]\n----\n{doctype}\n----\n|===\n\n", + "[.command,id=after,subs=attributes]\n----\n{doctype}\n----\n", + ); + let graph = graph(source); + assert_eq!(find(&graph, "cell").script, "article\n"); + assert_eq!(find(&graph, "after").script, "book\n"); + } + + #[test] + fn included_attribute_events_apply_in_source_order() { + let directory = tempfile::tempdir().unwrap(); + let included = directory.path().join("included.adoc"); + std::fs::write( + &included, + format!( + ":value: included\n\n{}", + command("included", "attributes", "{value}\n") + ), + ) + .unwrap(); + let main = directory.path().join("main.adoc"); + std::fs::write( + &main, + concat!( + "= Commands\n:value: initial\n\nBody.\n\ninclude::included.adoc[]\n\n", + "[.command,id=after,subs=attributes]\n----\n{value}\n----\n", + ), + ) + .unwrap(); + let parsed = acdc_parser::parse_file(&main, &Options::default()).unwrap(); + let graph = CommandGraph::try_from(&parsed).unwrap(); + let included_command = find(&graph, "included"); + assert_eq!(included_command.script, "included\n"); + assert_eq!( + included_command.location.file.as_deref(), + Some(included.as_path()) + ); + assert_eq!(find(&graph, "after").script, "included\n"); + } + + #[test] + fn drained_warnings_do_not_change_missing_attribute_validation() { + let mut parsed = parse(&command("run", "attributes", "{missing}\n")); + parsed.take_warnings(); + assert!( + matches!(CommandGraph::try_from(&parsed), Err(DiscoveryError::MissingAttribute { name, .. }) if name == "missing") + ); + } + } + + #[test] + fn source_paragraph_preserves_missing_final_newline() { + let source = "[source,bash,role=command,id=literal]\nprintf '%s' '<1>'"; + assert_eq!(find(&graph(source), "literal").script, "printf '%s' '<1>'"); + } + + #[test] + fn table_commands_respect_header_cell_semantics() { + let source = format!( + "[cols=a,options=\"header,footer\"]\n|===\na|{}\na|{}\na|{}\n|===\n", + cmd("header", None, None, "echo header"), + cmd("body", None, None, "echo body"), + cmd("footer", None, None, "echo footer"), + ); + // Header cells use inline content even with an explicit AsciiDoc cell style. + assert_eq!(ids(&graph(&source)), ["body", "footer"]); + } + + #[test] + fn marked_media_is_an_error() { + assert!(matches!( + err("[.command,id=image]\nimage::image.png[]"), + DiscoveryError::NotAScript { .. } + )); + } + + #[rstest] + #[case("[.command,id=broken]\n----\necho hi")] + #[case("[cols=\"1,1\"]\n|===\n|only one cell\n|===")] + #[case("|===\n|unterminated")] + #[case("ifdef::missing[]\nremaining content")] + fn recovered_documents_cannot_build_command_graphs(#[case] broken: &str) { + let source = format!("{}\n{broken}", cmd("valid", None, None, "echo valid")); + assert!(matches!( + err(&source), + DiscoveryError::RecoveredSource { .. } + )); + } + + #[test] + fn presentation_warnings_do_not_block_commands() { + let source = format!( + "See <>.\n\n{}", + cmd("valid", None, None, "echo valid") + ); + let mut parsed = parse(&source); + assert!(parsed.warnings().iter().any(|warning| matches!( + warning.kind, + acdc_parser::WarningKind::UnresolvedReference { .. } + ))); + assert!(!parsed.take_warnings().is_empty()); + assert_eq!(ids(&CommandGraph::try_from(&parsed).unwrap()), ["valid"]); + } + + #[test] + fn routing_warnings_does_not_permit_recovered_commands() { + let mut parsed = parse("[.command,id=broken]\n----\necho hi"); + assert!(!parsed.take_warnings().is_empty()); + assert!(parsed.warnings().is_empty()); + assert!(matches!( + CommandGraph::try_from(&parsed), + Err(DiscoveryError::RecoveredSource { .. }) + )); + } + + #[test] + fn unclosed_included_conditionals_keep_the_original_opening_location() { + let directory = tempfile::tempdir().unwrap(); + let main = directory.path().join("main.adoc"); + let included = directory.path().join("conditional.adoc"); + std::fs::write(&main, "include::conditional.adoc[lines=3..4]\n").unwrap(); + std::fs::write(&included, "ignored\n\nifdef::missing[]\nremaining\n").unwrap(); + let parsed = acdc_parser::parse_file(&main, &Options::default()).unwrap(); + let error = CommandGraph::try_from(&parsed).unwrap_err(); + let location = error.source_location().unwrap(); + assert_eq!(location.file.as_deref(), Some(included.as_path())); + assert_eq!(location.location.start.line, 3); + } + + #[rstest] + #[case("interpreter=\"\"")] + #[case("interpreter=\" \"")] + fn empty_interpreter_overrides_are_errors(#[case] attribute: &str) { + let source = format!("[source,bash,role=command,id=run,{attribute}]\n----\necho hi\n----"); + assert!(matches!( + err(&source), + DiscoveryError::InvalidInterpreter { .. } + )); + } + + #[test] + fn interpreter_override_is_a_single_executable_path() { + let source = "[source,python,role=command,id=run,interpreter=\"/a path/python3\"]\n----\nprint('hi')\n----"; + assert_eq!( + find(&graph(source), "run").metadata.interpreter, + "/a path/python3" + ); + } + + #[test] + fn non_string_and_nul_interpreters_are_rejected() { + for value in [ + acdc_parser::AttributeValue::Bool(true), + acdc_parser::AttributeValue::String("bad\0name".into()), + ] { + let mut metadata = acdc_parser::BlockMetadata::default(); + metadata.attributes.insert("interpreter".into(), value); + assert!(super::source_interpreter(&metadata).is_err()); + } + } + + #[test] + fn required_includes_block_graphs_but_absent_optional_includes_do_not() { + let directory = tempfile::tempdir().unwrap(); + let options = Options::builder() + .with_base_dir(directory.path()) + .build() + .unwrap(); + for (attributes, blocked) in [("", true), ("opts=optional", false)] { + let source = format!( + "include::missing.adoc[{attributes}]\n\n{}", + cmd("run", None, None, "echo hi") + ); + let parsed = acdc_parser::parse(&source, &options).unwrap(); + let result = CommandGraph::try_from(&parsed); + if blocked { + assert!(matches!( + result, + Err(DiscoveryError::RecoveredSource { .. }) + )); + } else { + assert_eq!(ids(&result.unwrap()), ["run"]); + } + } + } + + #[test] + fn nested_selected_include_retains_script_and_original_location() { + let directory = tempfile::tempdir().unwrap(); + let chapters = directory.path().join("chapters"); + std::fs::create_dir(&chapters).unwrap(); + let main = directory.path().join("main.adoc"); + let outer = chapters.join("outer.adoc"); + let scripts = directory.path().join("scripts.adoc"); + std::fs::write(&main, "include::chapters/outer.adoc[]").unwrap(); + std::fs::write(&outer, "include::../scripts.adoc[tag=command]").unwrap(); + std::fs::write( + &scripts, + format!( + "ignored\n// tag::command[]\n{}// end::command[]\n", + cmd("run", None, None, "cat <<'EOF'\n<1>\nEOF") + ), + ) + .unwrap(); + let parsed = acdc_parser::parse_file( + &main, + &Options::builder() + .with_safe_mode(SafeMode::Safe) + .build() + .unwrap(), + ) + .unwrap(); + let command = find(&CommandGraph::try_from(&parsed).unwrap(), "run"); + assert_eq!(command.script, "cat <<'EOF'\n<1>\nEOF\n"); + assert_eq!(command.location.file.as_deref(), Some(scripts.as_path())); + assert_eq!(command.location.location.start.line, 3); + } + + #[rstest] + #[case("[.command]\n----\necho hi\n----", false)] + #[case("[.command,id=bad.id]\n----\necho hi\n----", false)] + #[case("[.command,id=run,deps=missing]\n----\necho hi\n----", true)] + fn included_errors_keep_the_original_file(#[case] source: &str, #[case] graph_error: bool) { + let directory = tempfile::tempdir().unwrap(); + let main = directory.path().join("main.adoc"); + let included = directory.path().join("included.adoc"); + std::fs::write(&main, "include::included.adoc[]").unwrap(); + std::fs::write(&included, format!("intro\n\n{source}")).unwrap(); + let parsed = acdc_parser::parse_file(&main, &Options::default()).unwrap(); + let error = CommandGraph::try_from(&parsed).unwrap_err(); + let location = error.source_location().unwrap(); + assert_eq!(location.file.as_deref(), Some(included.as_path())); + assert_eq!(location.location.start.line, 3); + assert_eq!(matches!(error, DiscoveryError::Build(_)), graph_error); + } +} diff --git a/acdc-execute/src/lib.rs b/acdc-execute/src/lib.rs new file mode 100644 index 00000000..37fd779f --- /dev/null +++ b/acdc-execute/src/lib.rs @@ -0,0 +1,11 @@ +//! Discover and run command blocks defined in `AsciiDoc` documents. + +pub mod command; +pub mod discovery; + +pub use command::{ + BuildError, CommandBlock, CommandGraph, CommandGraphBuilder, CommandId, CommandMetadata, + CommandOutcome, CommandState, ExecError, ExecutionOptions, ExecutionPlan, ExecutionReport, + InvalidCommandId, ProcessOptions, SkipReason, UnknownCommand, +}; +pub use discovery::DiscoveryError; diff --git a/acdc-parser/CHANGELOG.md b/acdc-parser/CHANGELOG.md index 4a50b73a..15b69304 100644 --- a/acdc-parser/CHANGELOG.md +++ b/acdc-parser/CHANGELOG.md @@ -7,8 +7,23 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Added + +- Applications can inspect block metadata and source locations without matching + each block variant. +- Applications can read block bodies after preprocessing and before inline + substitutions, with normalized line endings, and resolve source locations to + the original included files. Serialized output and semantic equality are unchanged. +- Applications can distinguish omitted or recovered source content from + presentation warnings before acting on a parsed document, even after routing + the warning list elsewhere. Rendering can continue with the existing recovery + behavior. Includes disabled by a zero depth limit, secure mode, or URI access + policy now produce located warnings. + ### Changed +- Parsing file-based conditionals and checking enabled block substitutions use + fewer temporary allocations. - Repeated automatic cross-references use less memory when document attributes stay unchanged. Caption and section-signifier changes still apply in source order. @@ -66,6 +81,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - Escaped concealed index shorthand retains its outer parentheses and processes the inner visible term, matching Asciidoctor. Escaped visible terms remain literal. +- Textual attribute substitution preserves escaped references without looking + them up, accepts empty values, and leaves non-reference braces unchanged. + Builds without `pre-spec-subs` now classify ignored `subs` requests as source + recovery, so applications can reject incomplete substitution behavior even + after routing parser warnings elsewhere. + - Description-list values now report accurate source locations after extra whitespace or a line break following the term. A newline between formatted spans no longer extends its source location into the next line. diff --git a/acdc-parser/README.adoc b/acdc-parser/README.adoc index 92062ac7..c9fe82ea 100644 --- a/acdc-parser/README.adoc +++ b/acdc-parser/README.adoc @@ -141,6 +141,28 @@ assert_eq!(depth.text(), Some("064")); `with_defaults` replaces earlier defaults. Use `false` or `()` to unset an attribute, including one supplied by defaults. +== Source text and diagnostics + +`Paragraph::source_text()` and `DelimitedBlock::source_text()` expose the block body after document preprocessing and before inline substitutions. +Preprocessing normalizes line endings, removes trailing whitespace from lines, and processes includes and conditionals. +Delimited block text excludes metadata and delimiters, but retains the newline before the closing delimiter. +These methods return `None` for programmatically constructed or synthetic blocks without retained source text. +Retained text does not change ASG serialization or semantic equality. + +Consumers can use `BlockMetadata::uses_substitution(&Substitution::Attributes, VERBATIM)` to check whether a block enables attribute substitution with verbatim defaults. +The `substitute_attributes` helper performs one textual pass: valid unescaped references use the supplied lookup, and `\{name}` becomes literal `{name}` without a lookup. +Missing references remain in the text so the caller can report them; present empty values replace the reference with empty text. +The helper does not apply shell quoting or other inline substitutions. + +Use `ParseResult::source_location(&location)` to resolve an AST location to the file that supplied it, including nested and partial includes. +A location that crosses files is anchored in the file containing its start. +Unknown include chains have no resolved file. +`WarningKind::ContentRecovery` identifies omitted, replaced, or recovered source content separately from presentation warnings. +`ParseResult::source_recovery()` returns the first warning that affects source content or structure, including unmatched block boundaries and incomplete tables. +It excludes presentation warnings and remains available after `take_warnings()` moves the warning list elsewhere. +In builds without `pre-spec-subs`, an explicit `subs` setting records source recovery rather than silently permitting execution with different substitution behavior. +Rendering can continue after recovery; applications that require complete input must check `source_recovery()` before using recovered content. + == Intrinsic document attributes acdc initializes the intrinsic backend, input, time, safe-mode, and environment attributes before preprocessing. @@ -215,11 +237,11 @@ let options = acdc_parser::Options::builder() ---- The entry document does not count toward the limit; each currently open included file counts as one level. -A value of `0` disables built-in include processing and leaves each directive as literal content without a diagnostic. +A value of `0` disables built-in include processing and leaves each directive as literal content with a located content-recovery warning. A string value can have surrounding Unicode whitespace, but the complete trimmed value must be a non-negative ASCII decimal integer. Malformed, empty, decimal-fraction, and negative values return an `InvalidDocumentAttribute` configuration error when options are built. The original spelling of a valid value remains visible as the document attribute, and very large positive values saturate safely without overflow. -At a positive limit, the blocked directive is preserved, a located diagnostic is added to `ParseResult::warnings()`, and parsing continues. +At a positive limit, the blocked directive is preserved, a located content-recovery warning is added to `ParseResult::warnings()`, and parsing continues. Only an exact `include::` directive is processed; block macro names that merely begin with `include` remain ordinary content without include diagnostics. Declarations in document content are consumed but cannot change or unset the trusted value and do not appear as `Block::DocumentAttribute` nodes in the AST. diff --git a/acdc-parser/benches/conditional_bench.rs b/acdc-parser/benches/conditional_bench.rs index 99db33c5..59d586a9 100644 --- a/acdc-parser/benches/conditional_bench.rs +++ b/acdc-parser/benches/conditional_bench.rs @@ -12,6 +12,8 @@ //! `slow_path_control` forces the ordinary preprocessor rebuild without using a //! conditional. Keep the controls when comparing revisions so uniform machine //! or codegen shifts are distinguishable from conditional-path changes. +//! `conditional_files` repeats empty conditionals through `parse_file` to measure +//! file-path overhead separately from parsing their contents. //! //! For an acceptance comparison, put this same benchmark in the old and new //! worktrees, then run @@ -19,9 +21,9 @@ //! The runner performs seven alternating pairs and fails unless both active and //! inactive cases improve by at least 2% after adjustment by `plain_control`. -use std::{fmt::Write as _, hint::black_box}; +use std::{fmt::Write as _, fs, hint::black_box}; -use acdc_parser::{Options, parse}; +use acdc_parser::{Options, parse, parse_file}; use criterion::{BenchmarkId, Criterion, criterion_group, criterion_main}; const LINE_COUNTS: [usize; 2] = [1_000, 10_000]; @@ -104,6 +106,25 @@ fn conditional_benchmark(c: &mut Criterion) { } group.finish(); + + let mut group = c.benchmark_group("conditional_files"); + let path = std::env::temp_dir().join(format!( + "acdc-conditional-benchmark-{}.adoc", + std::process::id() + )); + drop(fs::File::create_new(&path).expect("create benchmark file")); + for count in LINE_COUNTS { + let input = "ifdef::bench-active[]\nendif::bench-active[]\n".repeat(count); + fs::write(&path, input).expect("write file conditionals"); + for (name, options) in [("active", &active_options), ("inactive", &inactive_options)] { + assert!(parse_file(&path, options).is_ok()); + group.bench_with_input(BenchmarkId::new(name, count), options, |b, options| { + b.iter(|| black_box(parse_file(black_box(&path), options))); + }); + } + } + fs::remove_file(path).expect("remove benchmark file"); + group.finish(); } criterion_group!(benches, conditional_benchmark); diff --git a/acdc-parser/benches/parser_bench.rs b/acdc-parser/benches/parser_bench.rs index 3fb70d49..fb240b08 100644 --- a/acdc-parser/benches/parser_bench.rs +++ b/acdc-parser/benches/parser_bench.rs @@ -7,6 +7,14 @@ use criterion::{BenchmarkId, Criterion, criterion_group, criterion_main}; fn parse_benchmark(c: &mut Criterion) { let mut group = c.benchmark_group("parser"); + group.bench_function("parse_inline/control", |b| { + b.iter(|| { + black_box( + Parser::new(black_box("Plain words with *strong* and _emphasis_.")).parse_inline(), + ) + }); + }); + let fixture_files_without_ext = vec![ "basic_header", "stem_blocks", diff --git a/acdc-parser/src/grammar/document.rs b/acdc-parser/src/grammar/document.rs index be578461..753826d6 100644 --- a/acdc-parser/src/grammar/document.rs +++ b/acdc-parser/src/grammar/document.rs @@ -36,8 +36,8 @@ use crate::{ table::parse_table_cell, }, model::{ - Caption, CaptionKind, LeveloffsetRange, ListLevel, Locateable, PositionalAttribute, - SectionKind, SectionLevel, Substitution, caption, section, strip_quotes, substitute, + Caption, CaptionKind, LeveloffsetRange, ListLevel, PositionalAttribute, SectionKind, + SectionLevel, Substitution, caption, section, strip_quotes, substitute, substitution::{HEADER, SubstitutionPlan, VERBATIM}, }, }; @@ -419,6 +419,7 @@ struct DelimitedParams<'input> { /// Language captured after a Markdown ```` ``` ```` fence, if any. lang: Option<&'input str>, content: &'input str, + source_text: &'input str, open_start: usize, start: usize, content_start: usize, @@ -475,7 +476,7 @@ fn build_delimited_block<'input>( Ok(assemble_delimited( metadata, - p.open_delim, + p, inner, block_metadata.title.clone(), location, @@ -487,7 +488,7 @@ fn build_delimited_block<'input>( /// Assemble the common `Block::DelimitedBlock` shell shared by every kind. fn assemble_delimited<'input>( metadata: BlockMetadata<'input>, - open_delim: &'input str, + p: &DelimitedParams<'input>, inner: DelimitedBlockType<'input>, title: Title<'input>, location: Location, @@ -496,12 +497,13 @@ fn assemble_delimited<'input>( ) -> Block<'input> { Block::DelimitedBlock(DelimitedBlock { metadata, - delimiter: open_delim, + delimiter: p.open_delim, inner, title, location, open_delimiter_location: Some(open_delimiter_location), close_delimiter_location, + source_text: Some(p.source_text), }) } @@ -537,7 +539,7 @@ fn build_example_block<'input>( } Ok(assemble_delimited( metadata, - p.open_delim, + p, DelimitedBlockType::DelimitedExample(blocks), block_metadata.title.clone(), location, @@ -1160,10 +1162,12 @@ fn store_named_block_attribute<'input>( metadata.substitutions = Some(parse_subs_attribute(value)); } #[cfg(not(feature = "pre-spec-subs"))] - state.add_generic_warning_at( - "The subs= attribute is not honoured in this build (the `pre-spec-subs` feature is disabled). The draft AsciiDoc spec drops the substitution model in favour of an inline parsing grammar; this attribute will be silently ignored.".to_string(), - location.clone().unwrap_or_default(), - ); + state.add_warning(Warning::new( + WarningKind::ContentRecovery { + message: "The subs= attribute is not honoured in this build (the `pre-spec-subs` feature is disabled). Requested content substitutions were ignored.".into(), + }, + Some(state.create_error_source_location(location.clone().unwrap_or_default())), + )); } "attribution" => { metadata.attribution = Some(Attribution::new(plain_attribute_value(value, location))); @@ -2211,6 +2215,7 @@ fn get_literal_paragraph<'input>( #[cfg(not(feature = "pre-spec-subs"))] let _ = content_start; Ok(Block::Paragraph(Paragraph { + source_text: Some(content), content: inlines, metadata, title: block_metadata.title.clone(), @@ -2811,6 +2816,7 @@ fn parse_table_block_impl<'input>( .with_presentation(presentation); Ok(Block::DelimitedBlock(DelimitedBlock { + source_text: None, metadata: metadata.clone(), delimiter: state.intern_str(open_delim), inner: DelimitedBlockType::DelimitedTable(table), @@ -4321,16 +4327,23 @@ peg::parser! { // recovery; `build_delimited_block` emits the unterminated warning). rule generic_delimited_block(start: usize, offset: usize, block_metadata: &BlockParsingMetadata<'input>) -> Result, Error> = open_start:position!() open:block_open() (eol() / ![_]) - content_start:position!() content:until_block_close(open.1) content_end:position!() - close:(eol() close_start:position!() close_delim:block_close_delim(open.1) { (close_start, close_delim) })? + content_start:position!() + source_text:$(until_block_close(open.1) (eol() &block_close_delim(open.1))?) body_end:position!() + close:(close_start:position!() close_delim:block_close_delim(open.1) { (close_start, close_delim) })? { + let content = if close.is_some() { + source_text.strip_suffix('\n').unwrap_or(source_text) + } else { + source_text + }; + let content_end = body_end - (source_text.len() - content.len()); let kind = match (open.0, block_metadata.metadata.style) { (DelimitedKind::Open, Some("source" | "listing")) => DelimitedKind::Listing, (DelimitedKind::Open, Some("literal")) => DelimitedKind::Literal, (kind, _) => kind, }; build_delimited_block(state, block_metadata, &DelimitedParams { - kind, open_delim: open.1, lang: open.2, content, + kind, open_delim: open.1, lang: open.2, content, source_text, open_start, start, content_start, content_end, end: span_end, offset, close, }) } @@ -4370,11 +4383,18 @@ peg::parser! { // entry point reuses the shared skeleton and builder. rule comment_block(start: usize, offset: usize, block_metadata: &BlockParsingMetadata<'input>) -> Result, Error> = open_start:position!() open_delim:comment_delimiter() (eol() / ![_]) - content_start:position!() content:until_block_close(open_delim) content_end:position!() - close:(eol() close_start:position!() close_delim:block_close_delim(open_delim) { (close_start, close_delim) })? + content_start:position!() + source_text:$(until_block_close(open_delim) (eol() &block_close_delim(open_delim))?) body_end:position!() + close:(close_start:position!() close_delim:block_close_delim(open_delim) { (close_start, close_delim) })? { + let content = if close.is_some() { + source_text.strip_suffix('\n').unwrap_or(source_text) + } else { + source_text + }; + let content_end = body_end - (source_text.len() - content.len()); build_delimited_block(state, block_metadata, &DelimitedParams { - kind: DelimitedKind::Comment, open_delim, lang: None, content, + kind: DelimitedKind::Comment, open_delim, lang: None, content, source_text, open_start, start, content_start, content_end, end: span_end, offset, close, }) } @@ -5957,6 +5977,12 @@ peg::parser! { match content { Ok(blocks) => description.extend(blocks), Err(e) => { + state.add_warning(Warning::new( + WarningKind::ContentRecovery { + message: format!("discarded attached content: {e}").into(), + }, + e.source_location().cloned(), + )); tracing::error!(?e, "Error processing attached content"); } } @@ -6137,6 +6163,7 @@ peg::parser! { } Ok(Block::DelimitedBlock(DelimitedBlock { + source_text: None, metadata, delimiter: "\"", inner: DelimitedBlockType::DelimitedQuote(blocks), @@ -6219,6 +6246,7 @@ peg::parser! { }; Ok(Block::DelimitedBlock(DelimitedBlock { + source_text: None, metadata, delimiter: ">", inner: DelimitedBlockType::DelimitedQuote(blocks), @@ -6309,6 +6337,7 @@ peg::parser! { return get_literal_paragraph(state, content, start, content_start, span_end, offset, block_metadata); } + let source_text = content; let content = if is_styled_verbatim { let content_location = state.create_block_location(content_start, span_end, offset); @@ -6376,6 +6405,7 @@ peg::parser! { metadata: block_metadata.metadata.clone(), title, blocks: vec![Block::Paragraph(Paragraph { + source_text: Some(source_text), content, metadata: block_metadata.metadata.clone(), title: Title::default(), @@ -6391,6 +6421,7 @@ peg::parser! { tracing::debug!(?content, "found paragraph block"); Ok(Block::Paragraph(Paragraph { + source_text: Some(source_text), content, metadata, title, diff --git a/acdc-parser/src/grammar/inline_processing.rs b/acdc-parser/src/grammar/inline_processing.rs index 843d33ce..d73fe183 100644 --- a/acdc-parser/src/grammar/inline_processing.rs +++ b/acdc-parser/src/grammar/inline_processing.rs @@ -76,7 +76,7 @@ pub(crate) fn adjust_peg_error_position( /// Helper for error recovery when parsing from a substring /// -/// Adjusts error positions to the original document and logs the error +/// Records a content-recovery warning at the original source position. pub(crate) fn adjust_and_log_parse_error( err: &peg::error::ParseError, parsed_text: &str, @@ -85,6 +85,12 @@ pub(crate) fn adjust_and_log_parse_error( context: &str, ) { let adjusted_error = adjust_peg_error_position(err, parsed_text, doc_start_offset, state); + state.add_warning(crate::Warning::new( + crate::WarningKind::ContentRecovery { + message: format!("{context}: {adjusted_error}").into(), + }, + adjusted_error.source_location().cloned(), + )); tracing::error!(?adjusted_error, ?context, "Parsing error occurred"); } diff --git a/acdc-parser/src/lib.rs b/acdc-parser/src/lib.rs index dccc184d..9e5bcfdc 100644 --- a/acdc-parser/src/lib.rs +++ b/acdc-parser/src/lib.rs @@ -484,7 +484,8 @@ fn parse_input( // unwraps it. let warnings_for_state = Rc::clone(&warnings_handle); - ParseResult::try_new(owner, warnings_handle, move |owner| { + let source_files = parsed::SourceFiles::new(file_path.clone(), &source_ranges); + ParseResult::try_new(owner, warnings_handle, source_files, move |owner| { let mut state = grammar::ParserState::new(&owner.source, &owner.arena); state.document_attributes = Rc::new(options_owned.document_attributes.clone()); state.options = Rc::new(options_owned); diff --git a/acdc-parser/src/model/inlines/mod.rs b/acdc-parser/src/model/inlines/mod.rs index 4b9e466b..a13d1d19 100644 --- a/acdc-parser/src/model/inlines/mod.rs +++ b/acdc-parser/src/model/inlines/mod.rs @@ -9,7 +9,7 @@ mod text; pub use macros::*; pub use text::*; -use crate::{Anchor, ElementAttributes, Image, Location, Source, model::Locateable}; +use crate::{Anchor, ElementAttributes, Image, Location, Source}; /// An `InlineNode` represents an inline node in a document. /// @@ -44,12 +44,6 @@ impl InlineNode<'_> { /// Returns the source location of this inline node. #[must_use] pub fn location(&self) -> &Location { - ::location(self) - } -} - -impl Locateable for InlineNode<'_> { - fn location(&self) -> &Location { match self { InlineNode::PlainText(t) => &t.location, InlineNode::RawText(t) => &t.location, @@ -96,30 +90,8 @@ impl InlineNode<'_> { } } -impl Locateable for InlineMacro<'_> { - fn location(&self) -> &Location { - match self { - Self::Footnote(f) => &f.location, - Self::Icon(i) => &i.location, - Self::Image(img) => &img.location, - Self::Keyboard(k) => &k.location, - Self::Button(b) => &b.location, - Self::Menu(m) => &m.location, - Self::Url(u) => &u.location, - Self::Mailto(m) => &m.location, - Self::Link(l) => &l.location, - Self::Autolink(a) => &a.location, - Self::CrossReference(x) => &x.location, - Self::Pass(p) => &p.location, - Self::Stem(s) => &s.location, - Self::IndexTerm(i) => &i.location, - } - } -} - impl InlineMacro<'_> { - /// Mutable access to this macro's own location. Counterpart to - /// [`Locateable::location`] for [`InlineMacro`]. + /// Mutable access to this macro's own location. Counterpart to [`Self::location`]. pub(crate) fn location_mut(&mut self) -> &mut Location { match self { Self::Footnote(f) => &mut f.location, @@ -214,7 +186,22 @@ impl InlineMacro<'_> { /// Returns the source location of this inline macro. #[must_use] pub fn location(&self) -> &Location { - ::location(self) + match self { + Self::Footnote(f) => &f.location, + Self::Icon(i) => &i.location, + Self::Image(img) => &img.location, + Self::Keyboard(k) => &k.location, + Self::Button(b) => &b.location, + Self::Menu(m) => &m.location, + Self::Url(u) => &u.location, + Self::Mailto(m) => &m.location, + Self::Link(l) => &l.location, + Self::Autolink(a) => &a.location, + Self::CrossReference(x) => &x.location, + Self::Pass(p) => &p.location, + Self::Stem(s) => &s.location, + Self::IndexTerm(i) => &i.location, + } } } diff --git a/acdc-parser/src/model/location.rs b/acdc-parser/src/model/location.rs index c8162525..8c1f8c25 100644 --- a/acdc-parser/src/model/location.rs +++ b/acdc-parser/src/model/location.rs @@ -130,11 +130,6 @@ impl SourceRange { } } -pub(crate) trait Locateable { - /// Get a reference to the location. - fn location(&self) -> &Location; -} - /// A `Location` represents a location in a document. /// /// After parsing completes, a `Location` is **original-source-relative**: its diff --git a/acdc-parser/src/model/metadata.rs b/acdc-parser/src/model/metadata.rs index 9dc4e690..312c52df 100644 --- a/acdc-parser/src/model/metadata.rs +++ b/acdc-parser/src/model/metadata.rs @@ -11,6 +11,7 @@ use super::{ attribution::{Attribution, CiteTitle}, caption::Caption, location::Location, + substitution::Substitution, }; pub type Role<'a> = &'a str; @@ -133,6 +134,23 @@ impl<'a> BlockMetadata<'a> { Self::default() } + /// Whether a substitution is enabled after applying this block's `subs` setting. + /// + /// `defaults` contains individual substitutions, such as [`crate::VERBATIM`] + /// for source blocks. Without `pre-spec-subs`, only these defaults apply. + #[must_use] + pub fn uses_substitution( + &self, + substitution: &Substitution, + defaults: &[Substitution], + ) -> bool { + #[cfg(feature = "pre-spec-subs")] + if let Some(spec) = &self.substitutions { + return spec.contains(substitution, defaults); + } + defaults.contains(substitution) + } + /// The anchor that defines this block's id: the explicit `id` (`[#id]`), /// otherwise the first `[[id]]` anchor. `None` when the block has no id. pub(crate) fn id_anchor(&self) -> Option<&Anchor<'a>> { diff --git a/acdc-parser/src/model/mod.rs b/acdc-parser/src/model/mod.rs index 3babd583..40aabbff 100644 --- a/acdc-parser/src/model/mod.rs +++ b/acdc-parser/src/model/mod.rs @@ -346,8 +346,33 @@ pub enum Block<'a> { } impl<'a> Block<'a> { + /// The source location of this block. + #[must_use] + pub fn location(&self) -> &Location { + match self { + Block::Section(s) => &s.location, + Block::Paragraph(p) => &p.location, + Block::UnorderedList(l) => &l.location, + Block::OrderedList(l) => &l.location, + Block::DescriptionList(l) => &l.location, + Block::CalloutList(l) => &l.location, + Block::DelimitedBlock(d) => &d.location, + Block::Admonition(a) => &a.location, + Block::TableOfContents(t) => &t.location, + Block::DiscreteHeader(h) => &h.location, + Block::DocumentAttribute(a) => &a.location, + Block::ThematicBreak(tb) => &tb.location, + Block::PageBreak(pb) => &pb.location, + Block::Image(i) => &i.location, + Block::Audio(a) => &a.location, + Block::Video(v) => &v.location, + Block::Comment(c) => &c.location, + } + } + /// This block's metadata, for the blocks that carry any. - pub(crate) fn metadata(&self) -> Option<&BlockMetadata<'a>> { + #[must_use] + pub fn metadata(&self) -> Option<&BlockMetadata<'a>> { match self { Block::Section(block) => Some(&block.metadata), Block::DelimitedBlock(block) => Some(&block.metadata), @@ -437,33 +462,9 @@ impl<'a> Block<'a> { } } -impl Locateable for Block<'_> { - fn location(&self) -> &Location { - match self { - Block::Section(s) => &s.location, - Block::Paragraph(p) => &p.location, - Block::UnorderedList(l) => &l.location, - Block::OrderedList(l) => &l.location, - Block::DescriptionList(l) => &l.location, - Block::CalloutList(l) => &l.location, - Block::DelimitedBlock(d) => &d.location, - Block::Admonition(a) => &a.location, - Block::TableOfContents(t) => &t.location, - Block::DiscreteHeader(h) => &h.location, - Block::DocumentAttribute(a) => &a.location, - Block::ThematicBreak(tb) => &tb.location, - Block::PageBreak(pb) => &pb.location, - Block::Image(i) => &i.location, - Block::Audio(a) => &a.location, - Block::Video(v) => &v.location, - Block::Comment(c) => &c.location, - } - } -} - impl Block<'_> { /// Mutable access to this block's own location (the post-parse source remap - /// pass rewrites it). Counterpart to [`Locateable::location`]. + /// pass rewrites it). Counterpart to [`Self::location`]. pub(crate) fn location_mut(&mut self) -> &mut Location { match self { Block::Section(s) => &mut s.location, @@ -673,13 +674,23 @@ impl Serialize for TableOfContents<'_> { } /// A `Paragraph` represents a paragraph in a document. -#[derive(Clone, Debug, PartialEq)] +#[derive(Clone, Debug)] #[non_exhaustive] pub struct Paragraph<'a> { pub metadata: BlockMetadata<'a>, pub title: Title<'a>, pub content: Vec>, pub location: Location, + pub(crate) source_text: Option<&'a str>, +} + +impl PartialEq for Paragraph<'_> { + fn eq(&self, other: &Self) -> bool { + self.metadata == other.metadata + && self.title == other.title + && self.content == other.content + && self.location == other.location + } } impl<'a> Paragraph<'a> { @@ -691,9 +702,19 @@ impl<'a> Paragraph<'a> { title: Title::default(), content, location, + source_text: None, } } + /// The paragraph body after document preprocessing, before inline substitutions. + /// + /// Absent for programmatically constructed paragraphs. This text is not part + /// of ASG serialization or semantic equality. + #[must_use] + pub fn source_text(&self) -> Option<&'a str> { + self.source_text + } + /// Set the metadata. #[must_use] pub fn with_metadata(mut self, metadata: BlockMetadata<'a>) -> Self { @@ -710,7 +731,7 @@ impl<'a> Paragraph<'a> { } /// A `DelimitedBlock` represents a delimited block in a document. -#[derive(Clone, Debug, PartialEq)] +#[derive(Clone, Debug)] #[non_exhaustive] pub struct DelimitedBlock<'a> { pub metadata: BlockMetadata<'a>, @@ -720,6 +741,19 @@ pub struct DelimitedBlock<'a> { pub location: Location, pub open_delimiter_location: Option, pub close_delimiter_location: Option, + pub(crate) source_text: Option<&'a str>, +} + +impl PartialEq for DelimitedBlock<'_> { + fn eq(&self, other: &Self) -> bool { + self.metadata == other.metadata + && self.inner == other.inner + && self.delimiter == other.delimiter + && self.title == other.title + && self.location == other.location + && self.open_delimiter_location == other.open_delimiter_location + && self.close_delimiter_location == other.close_delimiter_location + } } impl<'a> DelimitedBlock<'a> { @@ -734,9 +768,20 @@ impl<'a> DelimitedBlock<'a> { location, open_delimiter_location: None, close_delimiter_location: None, + source_text: None, } } + /// The body after document preprocessing, before inline substitutions. + /// + /// Delimiters and metadata are excluded; a newline before the closing delimiter + /// is retained. Absent for programmatically constructed or synthetic blocks. + /// This text is not part of ASG serialization or semantic equality. + #[must_use] + pub fn source_text(&self) -> Option<&'a str> { + self.source_text + } + /// Set the metadata. #[must_use] pub fn with_metadata(mut self, metadata: BlockMetadata<'a>) -> Self { diff --git a/acdc-parser/src/model/substitution.rs b/acdc-parser/src/model/substitution.rs index 114e5c52..751365a7 100644 --- a/acdc-parser/src/model/substitution.rs +++ b/acdc-parser/src/model/substitution.rs @@ -330,6 +330,28 @@ impl Serialize for SubstitutionSpec { #[cfg(feature = "pre-spec-subs")] impl SubstitutionSpec { + pub(crate) fn contains(&self, substitution: &Substitution, defaults: &[Substitution]) -> bool { + match self { + Self::Explicit(substitutions) => substitutions.contains(substitution), + // The last operation affecting a substitution determines its membership. + Self::Modifiers(operations) => operations + .iter() + .rev() + .find_map(|operation| { + let (affected, enabled) = match operation { + SubstitutionOp::Append(affected) | SubstitutionOp::Prepend(affected) => { + (affected, true) + } + SubstitutionOp::Remove(affected) => (affected, false), + }; + substitution_members(affected, GroupContext::Block) + .contains(substitution) + .then_some(enabled) + }) + .unwrap_or_else(|| defaults.contains(substitution)), + } + } + /// Apply modifier operations to a default substitution list. /// /// This is used by converters to resolve modifiers with the appropriate baseline. @@ -631,6 +653,37 @@ mod tests { use super::*; use crate::AttributeValue; + proptest::proptest! { + #[test] + fn membership_matches_resolved_modifier_order( + operations in proptest::collection::vec((0_usize..9, 0_u8..3), 0..20), + ) { + let substitutions = [ + Substitution::SpecialChars, Substitution::Attributes, + Substitution::Replacements, Substitution::Macros, + Substitution::PostReplacements, Substitution::Quotes, + Substitution::Callouts, Substitution::Normal, Substitution::Verbatim, + ]; + let operations = operations.into_iter().filter_map(|(index, operation)| { + substitutions.get(index).map(|substitution| match operation { + 0 => SubstitutionOp::Append(substitution.clone()), + 1 => SubstitutionOp::Prepend(substitution.clone()), + _ => SubstitutionOp::Remove(substitution.clone()), + }) + }).collect(); + let spec = SubstitutionSpec::Modifiers(operations); + for defaults in [NORMAL, VERBATIM, &[]] { + let resolved = spec.resolve(defaults); + for substitution in &substitutions { + proptest::prop_assert_eq!( + spec.contains(substitution, defaults), + resolved.contains(substitution), + ); + } + } + } + } + // Helper to extract explicit list from SubstitutionSpec #[allow(clippy::panic)] fn explicit(spec: &SubstitutionSpec) -> &Vec { @@ -1195,9 +1248,14 @@ mod tests { } } -/// Expand known attribute references, leaving unresolved references unchanged. +/// Expand known attribute references once, leaving unresolved references unchanged. +/// +/// Names contain one or more ASCII letters, digits, underscores, or hyphens. +/// A backslash before a valid reference is removed and protects that reference +/// from lookup. Only valid, unescaped references call `lookup`; inserted values +/// are not scanned again. Other text and line endings are preserved. /// -/// The returned text borrows the input when no reference is replaced. +/// The result borrows the input when no reference is replaced or escape removed. #[must_use] pub fn substitute_attributes<'text, 'value>( text: &'text str, @@ -1206,14 +1264,30 @@ pub fn substitute_attributes<'text, 'value>( let mut remaining = text; let mut unwritten = text; let mut output: Option = None; - while let Some((_, candidate)) = remaining.split_once('{') { - let Some((name, rest)) = candidate.split_once('}') else { - break; + while let Some((prefix, candidate)) = remaining.split_once('{') { + let name_len = candidate + .bytes() + .take_while(|byte| byte.is_ascii_alphanumeric() || matches!(*byte, b'_' | b'-')) + .count(); + let (name, after_name) = candidate.split_at(name_len); + let Some(rest) = after_name.strip_prefix('}') else { + remaining = after_name; + continue; }; remaining = rest; - if let Some(value) = lookup(name) { + if name.is_empty() { + continue; + } + let prefix_len = unwritten.len() - candidate.len() - 1; + if prefix.ends_with('\\') { + let output = output.get_or_insert_with(|| String::with_capacity(text.len())); + output.push_str(unwritten.split_at(prefix_len - 1).0); + output.push('{'); + output.push_str(name); + output.push('}'); + unwritten = rest; + } else if let Some(value) = lookup(name) { let output = output.get_or_insert_with(|| String::with_capacity(text.len())); - let prefix_len = unwritten.len() - candidate.len() - 1; output.push_str(unwritten.split_at(prefix_len).0); let _ = value.write_text(output); unwritten = rest; @@ -1227,3 +1301,124 @@ pub fn substitute_attributes<'text, 'value>( None => Cow::Borrowed(text), } } + +#[cfg(test)] +mod attribute_substitution_tests { + use std::borrow::Cow; + + use crate::{DocumentAttributeValue, Options}; + + use super::substitute_attributes; + + #[rstest::rstest] + #[case(r"\{shell}", "{shell}")] + #[case(r"\\{shell}", r"\{shell}")] + #[case(r"\\\{missing}", r"\\{missing}")] + fn escaped_references_do_not_call_lookup(#[case] input: &str, #[case] expected: &str) { + let value = DocumentAttributeValue::from("unexpected"); + let mut names = Vec::new(); + let output = substitute_attributes(input, |name| { + names.push(name.to_owned()); + Some(&value) + }); + assert_eq!(output, expected); + assert!(names.is_empty()); + } + + #[rstest::rstest] + #[case("λ plain\r\ntext\n", &[])] + #[case("{missing}", &["missing"])] + #[case("{{missing}}", &["missing"])] + #[case("{} {bad name} {name.part} {café} {counter:n}", &[])] + #[case(r"\{bad name} {shell\} ${shell:-fallback}", &[])] + #[case("{unfinished", &[])] + fn unchanged_text_stays_borrowed(#[case] input: &str, #[case] expected_names: &[&str]) { + let mut names = Vec::new(); + let output = substitute_attributes(input, |name| { + names.push(name.to_owned()); + None + }); + assert!(matches!(output, Cow::Borrowed(value) if std::ptr::eq(value, input))); + assert_eq!(names, expected_names); + } + + #[test] + fn adjacent_escaped_known_and_unknown_references_are_distinct() { + let value = DocumentAttributeValue::from("zsh"); + let mut names = Vec::new(); + let output = substitute_attributes(r"{shell}\{shell}{missing}{shell}", |name| { + names.push(name.to_owned()); + (name == "shell").then_some(&value) + }); + assert_eq!(output, "zsh{shell}{missing}zsh"); + assert_eq!(names, ["shell", "missing", "shell"]); + } + + #[rstest::rstest] + #[case("{{shell}}", "{zsh}", 1)] + #[case("{outer{shell}}", "{outerzsh}", 1)] + #[case(r"\{{shell}}", r"\{zsh}", 1)] + #[case(r"{\{shell}}", "{{shell}}", 0)] + #[case(r"\{outer{shell}}", r"\{outerzsh}", 1)] + #[case("{é{shell}}", "{ézsh}", 1)] + fn valid_inner_references_expand_once( + #[case] input: &str, + #[case] expected: &str, + #[case] calls: usize, + ) { + let value = DocumentAttributeValue::from("zsh"); + let mut names = Vec::new(); + let output = substitute_attributes(input, |name| { + names.push(name.to_owned()); + Some(&value) + }); + assert_eq!(output, expected); + assert_eq!(names, vec!["shell"; calls]); + } + + #[test] + fn names_follow_the_parser_reference_grammar() { + let value = DocumentAttributeValue::from("x"); + let mut names = Vec::new(); + let output = substitute_attributes("{_}{-}{0}{A-b_2}", |name| { + names.push(name.to_owned()); + Some(&value) + }); + assert_eq!(output, "xxxx"); + assert_eq!(names, ["_", "-", "0", "A-b_2"]); + } + + #[test] + fn replacement_text_is_not_scanned_again() { + let first = DocumentAttributeValue::from("{second}\\{third}\r\n"); + let second = DocumentAttributeValue::from("done"); + let mut names = Vec::new(); + let output = substitute_attributes("{first}/{second}", |name| { + names.push(name.to_owned()); + match name { + "first" => Some(&first), + "second" => Some(&second), + _ => None, + } + }); + assert_eq!(output, "{second}\\{third}\r\n/done"); + assert_eq!(names, ["first", "second"]); + } + + #[test] + fn values_keep_lexical_spelling_and_empty_values_are_present() -> Result<(), crate::Error> { + let options = Options::builder() + .with_attribute("max-include-depth", "003") + .with_attribute("quoted", "\"word\"") + .with_attribute("empty", "") + .with_attribute("present", true) + .build()?; + let attributes = options.document_attributes(); + let output = substitute_attributes( + "\r\n{max-include-depth}/{quoted}/{empty}/{present}\n", + |name| attributes.get(name), + ); + assert_eq!(output, "\r\n003/\"word\"//\n"); + Ok(()) + } +} diff --git a/acdc-parser/src/parsed.rs b/acdc-parser/src/parsed.rs index db4f1fd2..f2636eea 100644 --- a/acdc-parser/src/parsed.rs +++ b/acdc-parser/src/parsed.rs @@ -15,11 +15,47 @@ //! `OwnedSource` covers the parse-failure case (we have the text but no //! AST) without paying for an empty arena. -use std::{cell::RefCell, rc::Rc}; +use std::{cell::RefCell, collections::HashMap, path::PathBuf, rc::Rc}; use bumpalo::Bump; -use crate::{Document, InlineNode, Warning}; +use crate::{ + Document, InlineNode, Location, SourceLocation, Warning, WarningKind, model::SourceRange, +}; + +#[derive(Debug, Default)] +pub(crate) struct SourceFiles { + primary: Option, + included: HashMap, PathBuf>, +} + +impl SourceFiles { + pub(crate) fn new(primary: Option, ranges: &[SourceRange]) -> Self { + let mut included = HashMap::new(); + for range in ranges { + if !range.file_chain.is_empty() + && let Some(file) = &range.file + && !included.contains_key(range.file_chain.as_slice()) + { + included.insert(range.file_chain.clone(), file.clone()); + } + } + Self { primary, included } + } + + fn location(&self, location: &Location) -> SourceLocation { + let file = match location + .start + .file + .as_deref() + .filter(|chain| !chain.is_empty()) + { + Some(chain) => self.included.get(chain.as_slice()), + None => self.primary.as_ref(), + }; + SourceLocation::at_location(file.cloned(), location.clone()) + } +} /// Owner-side of the self-referential parse cell: holds the preprocessed /// source text and the arena that parser-allocated strings live in. The @@ -80,22 +116,40 @@ self_cell::self_cell! { pub struct ParseResult { cell: ParsedDocumentCell, warnings: Vec, + source_recovery: Option>, + source_files: SourceFiles, } impl ParseResult { - /// Internal constructor used by the `parse_*` entry points. Takes the - /// owner, a shared warnings handle (the `ParserState` holds its own - /// clone of this `Rc`; we recover the collected warnings after the - /// builder returns), and a fallible dependent builder. + /// Build the document with its source map and collect warnings after construction. pub(crate) fn try_new( owner: OwnedInput, warnings_handle: Rc>>, + source_files: SourceFiles, builder: impl for<'a> FnOnce(&'a OwnedInput) -> Result, E>, ) -> Result { let cell = ParsedDocumentCell::try_new(owner, builder)?; + let warnings = recover_warnings(warnings_handle); + let source_recovery = warnings + .iter() + .find(|warning| { + matches!( + warning.kind, + WarningKind::ContentRecovery { .. } + | WarningKind::UnterminatedDelimitedBlock { .. } + | WarningKind::UnterminatedTable { .. } + | WarningKind::TableUnknownFormat { .. } + | WarningKind::TableIncompleteRow + | WarningKind::TableCellOverflow { .. } + | WarningKind::TableColumnCount { .. } + ) + }) + .map(|warning| Box::new(Warning::new(warning.kind.clone(), warning.location.clone()))); Ok(Self { cell, - warnings: recover_warnings(warnings_handle), + warnings, + source_recovery, + source_files, }) } @@ -115,12 +169,32 @@ impl ParseResult { &self.cell.borrow_owner().source } + /// Resolve a location from this result's AST to its original source file and position. + /// + /// Includes use the file resolved during preprocessing, even when selection + /// or indentation changed the parsed text. A span crossing files is anchored + /// in the file containing its start. Unknown include chains have no file. + #[must_use] + pub fn source_location(&self, location: &Location) -> SourceLocation { + self.source_files.location(location) + } + /// Borrow the collected warnings. #[must_use] pub fn warnings(&self) -> &[Warning] { &self.warnings } + /// The first warning for recovered source content or structure, when present. + /// + /// This includes omitted content, ignored substitutions, and unmatched block + /// boundaries, but excludes presentation warnings. It remains available after + /// [`Self::take_warnings`]. + #[must_use] + pub fn source_recovery(&self) -> Option<&Warning> { + self.source_recovery.as_deref() + } + /// Take the warnings out of this result, leaving an empty warnings /// slice behind. Useful when the caller wants to route warnings /// independently of the AST (e.g. attach them to an LSP diagnostic diff --git a/acdc-parser/src/preprocessor/include.rs b/acdc-parser/src/preprocessor/include.rs index 83c18db6..2143ef11 100644 --- a/acdc-parser/src/preprocessor/include.rs +++ b/acdc-parser/src/preprocessor/include.rs @@ -870,13 +870,13 @@ impl<'a> Include<'a> { if target.starts_with(base_dir) { return Ok(target); } - self.warn_unlocated("include file is outside of jail; recovering automatically"); + self.warn_located("include file is outside of jail; recovering automatically"); return Ok(Self::rebase_absolute_target(base_dir, &target)); } let mut resolved = absolute_normalized(current_parent)?; if !resolved.starts_with(base_dir) { - self.warn_unlocated("include file is outside of jail; recovering automatically"); + self.warn_located("include file is outside of jail; recovering automatically"); return Ok(Self::rebase_absolute_target(base_dir, target)); } @@ -896,7 +896,7 @@ impl<'a> Include<'a> { } } if recovered { - self.warn_unlocated( + self.warn_located( "include file has illegal reference to ancestor of jail; recovering automatically", ); } @@ -1031,6 +1031,9 @@ impl<'a> Include<'a> { attribute_list_as_written: &str, ) -> Result { if !self.context.allows_uri_read { + self.warn_located(format!( + "include uri not read because URI access is disabled: {url}" + )); return Ok(UrlIncludeOutcome::Fallback(IncludeResult::link_fallback( self.target_as_written(), self.options.document_attributes.contains_key("compat-mode"), @@ -1088,6 +1091,10 @@ impl<'a> Include<'a> { pub(crate) fn process(&self, attribute_list_as_written: &str) -> Result { if self.options.safe_mode == SafeMode::Secure { + self.warn_located(format!( + "include not read in secure mode: {}", + self.target_as_written() + )); return Ok(IncludeResult::link_fallback( self.target_as_written(), self.options.document_attributes.contains_key("compat-mode"), @@ -1117,7 +1124,10 @@ impl<'a> Include<'a> { (memory_base.as_path(), memory_base.as_path()) } SourceOrigin::Uri(_) => { - tracing::error!(?target, "local include target has a URI source origin"); + self.warn_located(format!( + "local include target has a URI source origin: {}", + target.display() + )); return Ok(IncludeResult::empty()); } }; @@ -1165,6 +1175,9 @@ impl<'a> Include<'a> { } Target::UnsupportedUri(uri) => { if !self.context.allows_uri_read { + self.warn_located(format!( + "include uri not read because URI access is disabled: {uri}" + )); return Ok(IncludeResult::link_fallback( self.target_as_written(), self.options.document_attributes.contains_key("compat-mode"), @@ -1211,22 +1224,15 @@ impl<'a> Include<'a> { location: Location::point(crate::Position::from_line_col(self.line_number, 1)), }; let warning = Warning::new( - crate::WarningKind::Other(message.into()), + crate::WarningKind::ContentRecovery { + message: message.into(), + }, Some(source_location), ); tracing::warn!("{warning}"); self.warnings.borrow_mut().push(warning); } - /// Push a warning with no source location. The remaining callers are - /// Safe/Server jail-recovery conditions whose location contract is tracked - /// separately from directive-specific read failures. - fn warn_unlocated(&self, message: impl Into>) { - let warning = Warning::new(crate::WarningKind::Other(message.into()), None); - tracing::warn!("{warning}"); - self.warnings.borrow_mut().push(warning); - } - fn resolve_end_line(end: isize, max_size: usize) -> Option { match end { n if n < 0 => max_size.checked_sub(1), diff --git a/acdc-parser/src/preprocessor/mod.rs b/acdc-parser/src/preprocessor/mod.rs index 1aca43da..c7ed3aca 100644 --- a/acdc-parser/src/preprocessor/mod.rs +++ b/acdc-parser/src/preprocessor/mod.rs @@ -340,6 +340,7 @@ pub(super) fn absolute_normalized(path: &Path) -> Result { struct ConditionalFrame<'input> { conditional: conditional::Conditional<'input>, active: bool, + opening_line: usize, } /// Mutable state accumulated during preprocessing. @@ -673,6 +674,17 @@ impl Preprocessor { tracing::warn!(?warning); self.warnings.borrow_mut().push(warning); } + + fn recover_content_at(&self, message: impl Into>, location: SourceLocation) { + let warning = Warning::new( + WarningKind::ContentRecovery { + message: message.into(), + }, + Some(location), + ); + tracing::warn!(?warning); + self.warnings.borrow_mut().push(warning); + } } impl Preprocessor { @@ -886,7 +898,10 @@ impl Preprocessor { .map_or("", |(_, attributes)| attributes); return Ok(Some(include.process(attribute_list_as_written)?)); } - tracing::error!(%line, "source origin is missing - include directive cannot be processed"); + self.recover_content_at( + format!("source origin is missing; include directive cannot be processed: {line}"), + Self::create_source_location(line_number, None), + ); Ok(None) } @@ -1117,6 +1132,7 @@ impl Preprocessor { stack.push(ConditionalFrame { conditional, active, + opening_line: ctx.source_line, }); } return Ok(true); @@ -1158,15 +1174,16 @@ impl Preprocessor { out.note_source_line(ctx.input_line); out.push_line(Cow::Borrowed(&line[1..])); } else if has_include_directive_shape(line) { - // A limit of 0 disables built-in includes silently; exceeding a - // positive limit preserves the directive and warns. if let Some(limit) = self.include_context.blocked_limit(options.safe_mode) { - if limit != 0 { - self.add_warning_at( - format!("maximum include depth of {limit} exceeded"), - Self::create_source_location(ctx.source_line, ctx.current_file()), - ); - } + let message = if limit == 0 { + "include not read because maximum include depth is zero".to_owned() + } else { + format!("maximum include depth of {limit} exceeded") + }; + self.recover_content_at( + message, + Self::create_source_location(ctx.source_line, ctx.current_file()), + ); out.push_source_line(line, ctx.input_line); return Ok(()); } @@ -1420,6 +1437,13 @@ impl Preprocessor { line_number += 1; } + let file = source_origin.and_then(SourceOrigin::as_path); + for frame in conditional_stack { + self.recover_content_at( + "conditional directive has no matching endif", + Self::create_source_location(frame.opening_line, file), + ); + } out.flush_run(); out.flush_borrowed_run(); diff --git a/acdc-parser/src/proptests/invariants.rs b/acdc-parser/src/proptests/invariants.rs index ef7715b5..daa7b5cd 100644 --- a/acdc-parser/src/proptests/invariants.rs +++ b/acdc-parser/src/proptests/invariants.rs @@ -10,8 +10,8 @@ use proptest::prelude::*; use std::rc::Rc; use crate::{ - Block, DelimitedBlock, DelimitedBlockType, Document, InlineNode, Location, Options, - model::Locateable, parse, parse_inline, + Block, DelimitedBlock, DelimitedBlockType, Document, InlineNode, Location, Options, parse, + parse_inline, }; use super::generators::*; diff --git a/acdc-parser/src/warning.rs b/acdc-parser/src/warning.rs index 864c8451..c4712df0 100644 --- a/acdc-parser/src/warning.rs +++ b/acdc-parser/src/warning.rs @@ -80,7 +80,7 @@ impl Warning { WarningKind::InvalidDocumentAttribute { .. } => { Some("Use a value in the attribute's documented domain") } - WarningKind::Other(_) => None, + WarningKind::ContentRecovery { .. } | WarningKind::Other(_) => None, } } } @@ -256,6 +256,16 @@ pub enum WarningKind { expected: &'static str, }, + /// Requested source content or substitutions were omitted, replaced, or recovered. + /// + /// Rendering can continue, but consumers that require complete source content + /// must handle this condition before using the recovered document. + #[error("{message}")] + ContentRecovery { + /// The recovery and the source content it affected. + message: Cow<'static, str>, + }, + /// Ad-hoc message not yet categorised into a typed variant. #[error("{0}")] Other(Cow<'static, str>), diff --git a/acdc-parser/tests/conditional_allocations.rs b/acdc-parser/tests/conditional_allocations.rs index b4c8e4d8..7f85484d 100644 --- a/acdc-parser/tests/conditional_allocations.rs +++ b/acdc-parser/tests/conditional_allocations.rs @@ -6,9 +6,9 @@ //! underlying work directly: allocation and reallocation counts plus requested //! bytes. Keep this file to one test because allocator regions are process-wide. -use std::{alloc::System, fmt::Write as _, hint::black_box}; +use std::{alloc::System, fmt::Write as _, hint::black_box, path::Path}; -use acdc_parser::{Error, Options, parse}; +use acdc_parser::{Error, Options, parse, parse_file}; use stats_alloc::{INSTRUMENTED_SYSTEM, Region, Stats, StatsAlloc}; #[global_allocator] @@ -156,6 +156,47 @@ fn measure_steady_state(input: &str, options: &Options) -> Result }) } +fn measure_file(path: &Path, options: &Options) -> Result { + let region = Region::new(GLOBAL); + let parsed = parse_file(black_box(path), black_box(options))?; + black_box(&parsed); + let stats = region.change(); + drop(parsed); + Ok(stats) +} + +fn conditional_file_paths_have_constant_allocation_overhead() +-> Result<(), Box> { + let path = std::env::temp_dir().join(format!( + "acdc-conditional-allocation-{}.adoc", + std::process::id() + )); + drop(std::fs::File::create_new(&path)?); + let options = Options::default(); + let mut first_overhead = None; + for count in [1, 1_000] { + let input = "ifdef::missing[]\nendif::missing[]\n".repeat(count); + std::fs::write(&path, &input)?; + let _ = parse_file(&path, &options)?; + let from_string = measure_steady_state(&input, &options)?; + let first = measure_file(&path, &options)?; + let second = measure_file(&path, &options)?; + let file_allocations = first.allocations.min(second.allocations); + let overhead = file_allocations.saturating_sub(from_string.allocations); + let baseline = *first_overhead.get_or_insert(overhead); + assert!( + overhead <= baseline + 4, + "file path allocation overhead grew with {count} conditionals: {overhead} > {baseline} + 4" + ); + eprintln!( + "conditional file allocations/{count}: string={} file={file_allocations}", + from_string.allocations + ); + } + std::fs::remove_file(path)?; + Ok(()) +} + fn assert_within_budget(case: &str, line_count: usize, stats: Stats, budget: Budget) { let allocated_byte_limit = budget .bytes_allocated @@ -188,7 +229,7 @@ fn assert_within_budget(case: &str, line_count: usize, stats: Stats, budget: Bud } #[test] -fn conditional_allocation_work_stays_within_budget() -> Result<(), Error> { +fn conditional_allocation_work_stays_within_budget() -> Result<(), Box> { let active_options = Options::builder() .with_attribute("bench-active", true) .build()?; @@ -231,5 +272,5 @@ fn conditional_allocation_work_stays_within_budget() -> Result<(), Error> { budget.slow_control, ); } - Ok(()) + conditional_file_paths_have_constant_allocation_overhead() } diff --git a/acdc-parser/tests/include_base_dir.rs b/acdc-parser/tests/include_base_dir.rs index 8846c86f..d39f04e9 100644 --- a/acdc-parser/tests/include_base_dir.rs +++ b/acdc-parser/tests/include_base_dir.rs @@ -164,7 +164,15 @@ fn safe_and_server_confinement_use_overridden_base() -> TestResult { return Err(format!("unexpected warnings: {:?}", result.warnings()).into()); }; assert_eq!(warning.kind.to_string(), ANCESTOR_RECOVERY_WARNING); - assert!(warning.source_location().is_none()); + assert!(matches!( + warning.kind, + acdc_parser::WarningKind::ContentRecovery { .. } + )); + let location = warning + .source_location() + .ok_or("missing recovery location")?; + assert_eq!(location.file.as_deref(), Some(main.as_path())); + assert_eq!(location.location.start.line, 1); } Ok(()) } diff --git a/acdc-parser/tests/include_max_depth.rs b/acdc-parser/tests/include_max_depth.rs index e3322a93..c4bad8e5 100644 --- a/acdc-parser/tests/include_max_depth.rs +++ b/acdc-parser/tests/include_max_depth.rs @@ -130,7 +130,11 @@ fn assert_depth_warning(result: &ParseResult, max: usize, file: &Path, line: u32 }; assert_eq!( warning.kind.to_string(), - format!("maximum include depth of {max} exceeded") + if max == 0 { + "include not read because maximum include depth is zero".to_owned() + } else { + format!("maximum include depth of {max} exceeded") + } ); let Some(location) = warning.source_location() else { return Err("expected depth warning to have a source location".into()); @@ -177,14 +181,14 @@ fn default_depth_is_defined_for_conditionals_without_being_explicit() -> TestRes } #[test] -fn zero_disables_built_in_includes_without_a_diagnostic() -> TestResult { +fn zero_disables_built_in_includes_with_a_recovery_diagnostic() -> TestResult { let tree = IncludeTree::chain("")?; let result = parse_file(&tree.main, &options("0")?)?; assert_max_depth(&result, "0")?; assert_chain(&result, "0", Expansion::BlockedAtMain)?; - assert!(result.warnings().is_empty()); + assert_depth_warning(&result, 0, &tree.main, 5)?; Ok(()) } @@ -313,7 +317,7 @@ fn boolean_true_disables_includes_without_crashing() -> TestResult { ); // A boolean has no string form, so the reference substitutes to nothing. assert_chain(&result, "", Expansion::BlockedAtMain)?; - assert!(result.warnings().is_empty()); + assert_depth_warning(&result, 0, &tree.main, 5)?; Ok(()) } diff --git a/acdc-parser/tests/include_safe_server_confinement.rs b/acdc-parser/tests/include_safe_server_confinement.rs index bcfb3889..5935c532 100644 --- a/acdc-parser/tests/include_safe_server_confinement.rs +++ b/acdc-parser/tests/include_safe_server_confinement.rs @@ -89,15 +89,25 @@ fn paragraph_texts(result: &ParseResult) -> Result, Box> { .collect() } -fn assert_single_unlocated_warning( +fn assert_single_recovery_warning( result: &ParseResult, expected: &str, + file: &Path, + line: u32, ) -> Result<(), Box> { let [warning] = result.warnings() else { return Err(format!("expected one warning, got {:?}", result.warnings()).into()); }; assert_eq!(warning.kind.to_string(), expected); - assert!(warning.source_location().is_none()); + assert!(matches!( + warning.kind, + acdc_parser::WarningKind::ContentRecovery { .. } + )); + let location = warning + .source_location() + .ok_or("missing recovery location")?; + assert_eq!(location.file.as_deref(), Some(file)); + assert_eq!(location.location.start.line, line); Ok(()) } @@ -127,7 +137,7 @@ fn ancestor_traversal_is_moved_inside_the_entry_directory() -> TestResult { for safe_mode in [SafeMode::Safe, SafeMode::Server] { let result = parse_file(&tree.main, &options(safe_mode)?)?; assert_eq!(paragraph_texts(&result)?, ["REBASED ENTRY OUTSIDE"]); - assert_single_unlocated_warning(&result, ANCESTOR_RECOVERY_WARNING)?; + assert_single_recovery_warning(&result, ANCESTOR_RECOVERY_WARNING, &tree.main, 1)?; } Ok(()) @@ -157,7 +167,7 @@ fn absolute_outside_targets_are_moved_inside_the_entry_directory() -> TestResult paragraph_texts(&result)?, ["REBASED ABSOLUTE OUTSIDE", "ABSOLUTE INSIDE"] ); - assert_single_unlocated_warning(&result, OUTSIDE_RECOVERY_WARNING)?; + assert_single_recovery_warning(&result, OUTSIDE_RECOVERY_WARNING, &tree.main, 1)?; } Ok(()) @@ -179,7 +189,12 @@ fn nested_includes_keep_the_entry_directory_boundary() -> TestResult { paragraph_texts(&result)?, ["INNER START", "ENTRY OUTSIDE", "INNER END"] ); - assert_single_unlocated_warning(&result, ANCESTOR_RECOVERY_WARNING)?; + assert_single_recovery_warning( + &result, + ANCESTOR_RECOVERY_WARNING, + &tree.entry_dir.join("sub/inner.adoc"), + 3, + )?; } Ok(()) @@ -193,7 +208,7 @@ fn optional_missing_recovered_target_keeps_only_the_recovery_warning() -> TestRe for safe_mode in [SafeMode::Safe, SafeMode::Server] { let result = parse_file(&tree.main, &options(safe_mode)?)?; assert!(result.document().blocks.is_empty()); - assert_single_unlocated_warning(&result, ANCESTOR_RECOVERY_WARNING)?; + assert_single_recovery_warning(&result, ANCESTOR_RECOVERY_WARNING, &tree.main, 1)?; } Ok(()) diff --git a/acdc-parser/tests/include_secure_mode.rs b/acdc-parser/tests/include_secure_mode.rs index f9719bb7..ca2e0840 100644 --- a/acdc-parser/tests/include_secure_mode.rs +++ b/acdc-parser/tests/include_secure_mode.rs @@ -45,7 +45,24 @@ fn secure_mode_preserves_local_and_uri_includes_without_reading_them() assert_eq!(paragraph.location.start.line, expected_line); assert!(paragraph.location.start.file.is_none()); } - assert!(result.warnings().is_empty()); + assert_eq!(result.warnings().len(), 2); + for (warning, line) in result.warnings().iter().zip([3, 5]) { + assert!(matches!( + warning.kind, + acdc_parser::WarningKind::ContentRecovery { .. } + )); + let location = warning + .source_location() + .ok_or("missing recovery location")?; + assert_eq!(location.location.start.line, line); + assert_eq!( + location + .file + .as_deref() + .and_then(std::path::Path::file_name), + Some(std::ffi::OsStr::new("secure_include_main.adoc")) + ); + } Ok(()) } diff --git a/acdc-parser/tests/include_uri_authority.rs b/acdc-parser/tests/include_uri_authority.rs index 870232dd..40593898 100644 --- a/acdc-parser/tests/include_uri_authority.rs +++ b/acdc-parser/tests/include_uri_authority.rs @@ -175,7 +175,18 @@ fn assert_include_fallback(result: &ParseResult, target: &str) -> TestResult { link.attributes.get_string("role").as_deref(), Some("include") ); - assert!(result.warnings().is_empty()); + let [warning] = result.warnings() else { + return Err("expected a content-recovery warning".into()); + }; + assert!(matches!( + warning.kind, + acdc_parser::WarningKind::ContentRecovery { .. } + )); + let location = warning + .source_location() + .ok_or("missing recovery location")?; + assert!(location.file.is_some()); + assert_eq!(location.location.start.line, paragraph.location.start.line); Ok(()) } diff --git a/acdc-parser/tests/include_uri_classification.rs b/acdc-parser/tests/include_uri_classification.rs index c6eb55e7..39dc65a6 100644 --- a/acdc-parser/tests/include_uri_classification.rs +++ b/acdc-parser/tests/include_uri_classification.rs @@ -92,7 +92,18 @@ fn denied_non_http_uri_uses_link_fallback_instead_of_local_file_handling() -> Te link.attributes.get_string("role").as_deref(), (!compat_mode).then_some("include") ); - assert!(result.warnings().is_empty()); + let [warning] = result.warnings() else { + return Err("expected a content-recovery warning".into()); + }; + assert!(matches!( + warning.kind, + acdc_parser::WarningKind::ContentRecovery { .. } + )); + let location = warning + .source_location() + .ok_or("missing recovery location")?; + assert_eq!(location.file.as_deref(), Some(document.main.as_path())); + assert_eq!(location.location.start.line, 3); } Ok(()) } diff --git a/acdc-parser/tests/include_uri_denied.rs b/acdc-parser/tests/include_uri_denied.rs index 02181897..883ba6f6 100644 --- a/acdc-parser/tests/include_uri_denied.rs +++ b/acdc-parser/tests/include_uri_denied.rs @@ -5,7 +5,9 @@ use std::{ sync::atomic::{AtomicU64, Ordering}, }; -use acdc_parser::{Block, InlineMacro, InlineNode, Options, ParseResult, SafeMode, parse_file}; +use acdc_parser::{ + Block, InlineMacro, InlineNode, Options, ParseResult, SafeMode, WarningKind, parse_file, +}; type TestResult = Result<(), Box>; @@ -70,7 +72,19 @@ fn assert_denied_uri_fallback(result: &ParseResult, target: &str) -> TestResult ); assert_eq!(fallback.location.start.line, 3); assert!(fallback.location.start.file.is_none()); - assert!(result.warnings().is_empty()); + let [warning] = result.warnings() else { + return Err("expected a content-recovery warning".into()); + }; + assert!(matches!(warning.kind, WarningKind::ContentRecovery { .. })); + assert_eq!( + warning + .source_location() + .ok_or("missing warning location")? + .location + .start + .line, + 3 + ); Ok(()) } @@ -106,6 +120,14 @@ fn compat_mode_omits_include_role_from_denied_uri_fallback() -> TestResult { }; assert_eq!(link.target.to_string(), target); assert_eq!(link.attributes.iter().count(), 0); - assert!(result.warnings().is_empty()); + let [warning] = result.warnings() else { + return Err("expected a content-recovery warning".into()); + }; + assert!(matches!(warning.kind, WarningKind::ContentRecovery { .. })); + let location = warning + .source_location() + .ok_or("missing warning location")?; + assert_eq!(location.file.as_deref(), Some(document.path.as_path())); + assert_eq!(location.location.start.line, 1); Ok(()) } diff --git a/acdc-parser/tests/retained_source.rs b/acdc-parser/tests/retained_source.rs new file mode 100644 index 00000000..022f41bd --- /dev/null +++ b/acdc-parser/tests/retained_source.rs @@ -0,0 +1,408 @@ +use std::{ + error::Error, + path::{Path, PathBuf}, +}; + +use acdc_parser::{ + Block, DelimitedBlock, DelimitedBlockType, Options, Paragraph, ParseResult, SafeMode, + WarningKind, parse, parse_file, +}; + +type TestResult = Result<(), Box>; + +fn listing(parsed: &ParseResult) -> Result<&DelimitedBlock<'_>, Box> { + parsed + .document() + .blocks + .iter() + .find_map(|block| { + let Block::DelimitedBlock(block) = block else { + return None; + }; + matches!(block.inner, DelimitedBlockType::DelimitedListing(_)).then_some(block) + }) + .ok_or_else(|| "expected listing".into()) +} + +fn fixtures() -> PathBuf { + Path::new(env!("CARGO_MANIFEST_DIR")).join("fixtures/preprocessor") +} + +#[rstest::rstest] +#[case("[source,bash]\n", "----")] +#[case("[source,bash]\n", "--")] +#[case("", "```bash")] +fn retained_listing_body_precedes_inline_substitutions( + #[case] metadata: &str, + #[case] delimiter: &str, +) -> TestResult { + let body = "cat <<'EOF'\n<1>\n\\<2>\n\n{value}\nEOF\n\n"; + let closing = if delimiter.starts_with('`') { + "```" + } else { + delimiter + }; + let input = format!(":value: replaced\n\n{metadata}{delimiter}\n{body}{closing}\n"); + let parsed = parse(&input, &Options::default())?; + assert_eq!(listing(&parsed)?.source_text(), Some(body)); + Ok(()) +} + +#[rstest::rstest] +#[case("normal")] +#[case("attributes")] +#[case("none")] +fn retained_source_paragraph_is_literal(#[case] substitutions: &str) -> TestResult { + let body = "printf '%s' '{value} *bold* \\<1>'"; + let input = format!(":value: replaced\n\n[source,bash,subs={substitutions}]\n{body}"); + let parsed = parse(&input, &Options::default())?; + let paragraph = parsed + .document() + .blocks + .iter() + .find_map(|block| { + let Block::Paragraph(paragraph) = block else { + return None; + }; + Some(paragraph) + }) + .ok_or("expected source paragraph")?; + assert_eq!(paragraph.source_text(), Some(body)); + Ok(()) +} + +#[test] +fn retained_text_does_not_change_serialization_or_equality() -> TestResult { + let parsed = parse( + "[source,bash]\n----\necho hi <1>\n----\n\n[source,bash]\necho hi", + &Options::default(), + )?; + let original = listing(&parsed)?; + let mut synthetic = DelimitedBlock::new( + original.inner.clone(), + original.delimiter, + original.location.clone(), + ) + .with_metadata(original.metadata.clone()) + .with_title(original.title.clone()); + synthetic + .open_delimiter_location + .clone_from(&original.open_delimiter_location); + synthetic + .close_delimiter_location + .clone_from(&original.close_delimiter_location); + assert!(synthetic.source_text().is_none()); + assert_eq!(original, &synthetic); + assert_eq!( + serde_json::to_value(original)?, + serde_json::to_value(synthetic)? + ); + + let Some(Block::Paragraph(original)) = parsed.document().blocks.last() else { + return Err("expected final source paragraph".into()); + }; + let synthetic = Paragraph::new(original.content.clone(), original.location.clone()) + .with_metadata(original.metadata.clone()) + .with_title(original.title.clone()); + assert!(synthetic.source_text().is_none()); + assert_eq!(original, &synthetic); + assert_eq!( + serde_json::to_value(original)?, + serde_json::to_value(synthetic)? + ); + Ok(()) +} + +#[test] +fn retained_table_cell_source_uses_its_nested_input() -> TestResult { + let parsed = parse( + "[cols=a]\n|===\n|[source,bash]\n----\ncat <<'EOF'\n<1>\nEOF\n----\n|===", + &Options::default(), + )?; + let Some(Block::DelimitedBlock(outer)) = parsed.document().blocks.first() else { + return Err("expected table".into()); + }; + let DelimitedBlockType::DelimitedTable(table) = &outer.inner else { + return Err("expected table content".into()); + }; + let inner = table + .rows + .iter() + .flat_map(|row| &row.columns) + .flat_map(|cell| &cell.content) + .find_map(|block| { + let Block::DelimitedBlock(inner) = block else { + return None; + }; + Some(inner) + }) + .ok_or("expected nested listing")?; + assert_eq!(inner.source_text(), Some("cat <<'EOF'\n<1>\nEOF\n")); + Ok(()) +} + +#[test] +fn retained_include_body_contains_transformed_content() -> TestResult { + let parsed = parse_file( + fixtures().join("include_indent_main.adoc"), + &Options::default(), + )?; + assert_eq!(listing(&parsed)?.source_text(), Some(" TARGETLINE\n")); + Ok(()) +} + +#[test] +fn retained_comment_in_list_continuation_includes_final_newline() -> TestResult { + let parsed = parse("* item\n+\n////\ncomment body\n////\n", &Options::default())?; + let Some(Block::UnorderedList(list)) = parsed.document().blocks.first() else { + return Err("expected list".into()); + }; + let comment = list + .items + .iter() + .flat_map(|item| &item.blocks) + .find_map(|block| { + let Block::DelimitedBlock(block) = block else { + return None; + }; + matches!(block.inner, DelimitedBlockType::DelimitedComment(_)).then_some(block) + }) + .ok_or("expected attached comment")?; + assert_eq!(comment.source_text(), Some("comment body\n")); + Ok(()) +} + +#[test] +fn source_location_resolves_nested_selected_content() -> TestResult { + let directory = fixtures(); + let primary = directory.join("include_tag_diagnostics_main.adoc"); + let parsed = parse_file(&primary, &Options::default())?; + let selected = parsed + .document() + .blocks + .iter() + .find_map(|block| { + let Block::Paragraph(paragraph) = block else { + return None; + }; + (paragraph.source_text() == Some("Selected.")).then_some(paragraph) + }) + .ok_or("expected selected paragraph")?; + let source = parsed.source_location(&selected.location); + assert_eq!( + source.file, + Some(directory.join("include_tag_diagnostics_target.adoc")) + ); + assert_eq!(source.location.start.line, 2); + let Some(Block::Paragraph(first)) = parsed.document().blocks.first() else { + return Err("expected primary paragraph".into()); + }; + assert_eq!(parsed.source_location(&first.location).file, Some(primary)); + Ok(()) +} + +#[test] +fn unknown_include_chain_does_not_resolve_to_the_primary_file() -> TestResult { + let parsed = parse_file( + fixtures().join("include_indent_main.adoc"), + &Options::default(), + )?; + let mut location = acdc_parser::Location::default(); + location.start.file = Some(std::sync::Arc::new(vec!["unknown.adoc".into()])); + assert!(parsed.source_location(&location).file.is_none()); + Ok(()) +} + +#[rstest::rstest] +#[case("include::__missing_recovery_test__.adoc[]", SafeMode::Unsafe)] +#[case("include::include_indent_target.rb[tag=missing]", SafeMode::Unsafe)] +#[case("include::include_indent_target.rb[lines=0]", SafeMode::Unsafe)] +#[case("include::include_indent_target.rb[]", SafeMode::Secure)] +#[case("include::https://example.invalid/script.adoc[]", SafeMode::Safe)] +fn source_loss_has_a_typed_located_warning( + #[case] input: &str, + #[case] safe_mode: SafeMode, +) -> TestResult { + let options = Options::builder() + .with_base_dir(fixtures()) + .with_safe_mode(safe_mode) + .build()?; + let parsed = parse(input, &options)?; + let warning = parsed + .warnings() + .iter() + .find(|warning| matches!(warning.kind, WarningKind::ContentRecovery { .. })) + .ok_or("expected content recovery")?; + assert_eq!( + warning + .source_location() + .ok_or("expected source location")? + .location + .start + .line, + 1 + ); + Ok(()) +} + +#[test] +fn absent_optional_include_does_not_recover_content() -> TestResult { + let options = Options::builder().with_base_dir(fixtures()).build()?; + let parsed = parse( + "include::__missing_recovery_test__.adoc[opts=optional]", + &options, + )?; + assert!(parsed.warnings().is_empty()); + Ok(()) +} + +#[test] +fn disabled_includes_record_content_recovery() -> TestResult { + let options = Options::builder() + .with_base_dir(fixtures()) + .with_attribute("max-include-depth", "0") + .build()?; + let parsed = parse("include::include_indent_target.rb[]", &options)?; + assert!( + parsed + .warnings() + .iter() + .any(|warning| matches!(warning.kind, WarningKind::ContentRecovery { .. })) + ); + Ok(()) +} + +#[test] +fn malformed_include_line_selection_is_a_parse_error() -> TestResult { + let options = Options::builder().with_base_dir(fixtures()).build()?; + let Err(error) = parse( + "include::include_indent_target.rb[lines=not-a-number]", + &options, + ) else { + return Err("malformed selection must not widen the include".into()); + }; + assert!(matches!(error, acdc_parser::Error::InvalidLineRange(_, _))); + assert_eq!( + error + .source_location() + .ok_or("missing error location")? + .location + .start + .line, + 1 + ); + Ok(()) +} + +#[rstest::rstest] +#[case("ifdef::missing[]")] +#[case("ifndef::missing[]")] +fn unclosed_conditionals_record_content_recovery(#[case] directive: &str) -> TestResult { + let parsed = parse( + &format!("intro\n\n{directive}\nremaining content"), + &Options::default(), + )?; + let warning = parsed + .warnings() + .iter() + .find(|warning| matches!(warning.kind, WarningKind::ContentRecovery { .. })) + .ok_or("expected content recovery")?; + assert_eq!( + warning + .source_location() + .ok_or("expected source location")? + .location + .start + .line, + 3 + ); + Ok(()) +} + +#[rstest::rstest] +#[case("----\nunterminated")] +#[case("[cols=\"1,1\"]\n|===\n|incomplete row\n|===")] +#[case("ifdef::missing[]\nremaining content")] +fn source_recovery_remains_available_after_warning_routing(#[case] input: &str) -> TestResult { + let mut parsed = parse(input, &Options::default())?; + let kind = parsed + .source_recovery() + .ok_or("expected source recovery")? + .kind + .clone(); + let warnings = parsed.take_warnings(); + assert!(!warnings.is_empty()); + assert!(parsed.warnings().is_empty()); + assert_eq!( + parsed.source_recovery().ok_or("lost source recovery")?.kind, + kind + ); + Ok(()) +} + +#[test] +fn presentation_warning_is_not_a_source_recovery() -> TestResult { + let mut parsed = parse("See <>.", &Options::default())?; + assert!(!parsed.take_warnings().is_empty()); + assert!(parsed.source_recovery().is_none()); + Ok(()) +} + +#[rstest::rstest] +#[case(None, false)] +#[case(Some("attributes"), true)] +#[case(Some("+attributes"), true)] +#[case(Some("attributes+"), true)] +#[case(Some("none"), false)] +#[case(Some("-attributes"), false)] +#[case(Some("+normal,-attributes"), false)] +#[case(Some("-normal,+attributes"), true)] +#[case(Some("+attributes,-normal"), false)] +#[case(Some("-attributes,attributes+"), true)] +fn block_substitutions_use_the_supplied_baseline( + #[case] subs: Option<&str>, + #[case] attributes_enabled: bool, +) -> TestResult { + let attribute = subs + .map(|value| format!(",subs=\"{value}\"")) + .unwrap_or_default(); + let input = format!("[source,sh{attribute}]\n----\n{{shell}} --version\n----\n"); + let parsed = parse(&input, &Options::default())?; + let metadata = &listing(&parsed)?.metadata; + assert_eq!( + metadata.uses_substitution( + &acdc_parser::Substitution::Attributes, + acdc_parser::VERBATIM + ), + cfg!(feature = "pre-spec-subs") && attributes_enabled, + ); + Ok(()) +} + +#[cfg(not(feature = "pre-spec-subs"))] +#[test] +fn ignored_substitutions_remain_source_recovery_after_warning_routing() -> TestResult { + let mut parsed = parse( + "[source,sh,subs=attributes]\n----\n{shell} --version\n----\n", + &Options::default(), + )?; + let warnings = parsed.take_warnings(); + assert!( + warnings + .iter() + .any(|warning| matches!(warning.kind, WarningKind::ContentRecovery { .. })) + ); + let recovery = parsed + .source_recovery() + .ok_or("lost ignored substitutions")?; + assert_eq!( + recovery + .source_location() + .ok_or("missing source position")? + .location + .start + .line, + 1 + ); + Ok(()) +} diff --git a/converters/core/CHANGELOG.md b/converters/core/CHANGELOG.md index f58f12e7..72faafcc 100644 --- a/converters/core/CHANGELOG.md +++ b/converters/core/CHANGELOG.md @@ -122,6 +122,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 Cell catalogs share preceding terms with the containing document; later terms remain excluded. +- Escaped attribute references such as `\{name}` stay literal when expanding + source-block text with `subs=attributes`, matching asciidoctor. + - References resolved within an included document keep local fallback text, including IDs with punctuation. Document-top references use `[^top]` when the document has no title or reference label. diff --git a/converters/core/src/substitutions.rs b/converters/core/src/substitutions.rs index 1895cbb7..d8e914b6 100644 --- a/converters/core/src/substitutions.rs +++ b/converters/core/src/substitutions.rs @@ -17,7 +17,7 @@ use bitflags::bitflags; /// Expand known attribute references, leaving unresolved references unchanged. /// -/// The result borrows `text` when no reference is replaced. +/// The result borrows `text` when no reference is replaced or escape removed. #[must_use] pub fn substitute_attributes<'text>( text: &'text str,