diff --git a/CHANGELOG.md b/CHANGELOG.md index 959c836..9b5c49b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -92,9 +92,10 @@ releases may contain breaking changes. auto-detected or set with `--langs`. `validate` supports `--format json` (machine-readable findings), `--strict` (warnings become errors), `--no-check-media` (skip the filesystem media-presence check), - `--require-ids` (error on entries/senses missing a stable id), and `-` to - read from stdin; `stats` also takes `--format json`. `validate`'s exit codes - and JSON schema are a supported interface. + `--allow CODE[,CODE...]` (report the given codes' findings but leave them out + of the pass/fail decision), `--require-ids` (error on entries/senses missing + a stable id), and `-` to read from stdin; `stats` also takes `--format json`. + `validate`'s exit codes and JSON schema are a supported interface. - Container image and GitHub Action wrapping `sil-lift validate`, so a non-Python CI pipeline can run the conformance check with no local Python toolchain (`Dockerfile`, `action.yml`, `docker-entrypoint.sh`). diff --git a/action.yml b/action.yml index 1c3a789..03859f7 100644 --- a/action.yml +++ b/action.yml @@ -18,6 +18,12 @@ inputs: description: "Skip the filesystem media-presence check." required: false default: "false" + allow: + description: >- + Problem codes to report but leave out of the pass/fail decision, + separated by commas, spaces, or newlines (e.g. "uri-not-rfc schema"). + required: false + default: "" runs: using: "docker" image: "Dockerfile" diff --git a/docker-entrypoint.sh b/docker-entrypoint.sh index ac6be29..660a690 100644 --- a/docker-entrypoint.sh +++ b/docker-entrypoint.sh @@ -29,4 +29,12 @@ if [ "$no_check_media" = "true" ]; then set -- "$@" --no-check-media fi +# -d '' reads past newlines (a YAML block-scalar list) and returns 1 at EOF. +IFS=$', \t\r\n' read -r -d '' -a allow_codes <<< "${INPUT_ALLOW:-}" || true +for code in "${allow_codes[@]}"; do + if [ -n "$code" ]; then + set -- "$@" --allow "$code" + fi +done + exec sil-lift "$@" diff --git a/docs/en/guides/cli.md b/docs/en/guides/cli.md index 341a7a9..1c1e64d 100644 --- a/docs/en/guides/cli.md +++ b/docs/en/guides/cli.md @@ -3,7 +3,8 @@ Installing the package (`pip install sil-lift`) also installs the `sil-lift` command — a supported tool in the spirit of LiftTools that ships with the package (and, for `validate`, a worked example of the library API). ``` -sil-lift validate PATH [--format {text,json}] [--strict] [--no-check-media] [--require-ids] +sil-lift validate PATH [--format {text,json}] [--strict] [--no-check-media] [--allow CODE[,CODE...]]... + [--require-ids] all problems, with file/entry/line; exit 1 on errors sil-lift stats PATH [--format {text,json}] entry/sense/language counts (streaming; any size) @@ -15,6 +16,8 @@ sil-lift export PATH [-o OUT] [--langs L] [--tsv] `--format json` writes a single JSON object to stdout (and nothing else) for CI/automation consumption; see the schema in the example below. `--strict` treats warnings as errors, exiting 1 if any are found — use it to gate a build on no warnings at all rather than on errors alone. `--no-check-media` skips the filesystem media-presence check (suppressing `missing-media` findings), which is useful when validating a freshly generated export whose audio/photo files live elsewhere rather than in the same folder. `--require-ids` additionally fails (a `missing-id` error) on any entry lacking a `guid` or sense lacking an `id` — stricter than LIFT, for workflows that re-import by a stable id. Passing `-` as the path reads the document from stdin (a piped document has no folder, so its companion `.lift-ranges` and media are not resolved). `stats` likewise takes `--format json`, emitting the counts as a single JSON object. +`--allow CODE[,CODE...]` (repeatable) reports the given codes' findings but leaves them out of the pass/fail decision. It applies to errors and warnings alike, so an allowed warning does not trip `--strict` either. The summary line tallies allowed findings per code; the JSON summary's `allowed` gives their total. A code that never occurs is ignored, so an allow list keeps working after a check is retired. + !!! note `validate`'s exit codes and `--format json` schema are a supported automation interface: both are covered by tests and change only under SemVer. @@ -55,7 +58,8 @@ $ sil-lift validate dictionary.lift --format json ], "summary": { "errors": 1, - "warnings": 1 + "warnings": 1, + "allowed": 0 } } @@ -69,4 +73,4 @@ $ sil-lift export dictionary.lift --langs en,fr -o dictionary.csv All output is UTF-8, on every platform and whether it goes to a console, a pipe, or a `>` redirect — never the locale encoding (cp1252 on Windows, ASCII under a C/POSIX locale), which cannot represent LIFT content. `sil-lift export dictionary.lift > dictionary.csv` therefore writes exactly the bytes `-o dictionary.csv` writes, CRLF row terminators included. -Exit codes: `0` success (warnings allowed, unless `--strict`), `1` findings (validation errors / missing media / warnings under `--strict`), `2` an I/O failure at either end — input that cannot be read, or output that cannot be written (a reader like `head` closing the pipe, a full disk). +Exit codes: `0` success (warnings do not fail the run unless `--strict`), `1` findings (validation errors / missing media / warnings under `--strict`, not counting codes given to `--allow`), `2` an I/O failure at either end — input that cannot be read, or output that cannot be written (a reader like `head` closing the pipe, a full disk). diff --git a/docs/en/guides/lift-export-interop.md b/docs/en/guides/lift-export-interop.md index f5e3c84..07251dd 100644 --- a/docs/en/guides/lift-export-interop.md +++ b/docs/en/guides/lift-export-interop.md @@ -31,6 +31,7 @@ sil-lift validate export.lift --strict --no-check-media --format json - `--strict` makes warnings (not just errors) fail the run. - `--no-check-media` skips the filesystem media-presence check, whose `missing-media` findings are noise when the audio/photo files are not in the same folder as the `.lift` in CI. - `--format json` prints a single JSON object (`{"problems": [...], "summary": {...}}`) instead of human text; its exit codes and schema are a supported, SemVer-covered interface (see [the command line guide](cli.md)). +- `--allow CODE[,CODE...]` (repeatable) keeps a tolerated code, such as FLEx's `uri-not-rfc`, from failing the run. Its findings are printed but not counted. - `--require-ids` additionally errors on entries missing a `guid` or senses missing an `id` — useful when a later re-import must update rather than duplicate. Guard against silent data loss (the failure mode that makes flat CSV export lossy) by asserting counts with `stats --format json` against your source model: @@ -51,6 +52,7 @@ A TypeScript or C# project's CI can run the same check without installing Python path: export.lift strict: "true" no-check-media: "true" + allow: "uri-not-rfc" format: json ``` diff --git a/docs/en/guides/validate.md b/docs/en/guides/validate.md index 729ae05..9d52ccb 100644 --- a/docs/en/guides/validate.md +++ b/docs/en/guides/validate.md @@ -28,7 +28,7 @@ Each `Problem` carries `level` (`"error"`/`"warning"`), a stable `code`, `messag ## Problem codes -Every finding carries one of these, whichever layer produced it — `schema` and `uri-not-rfc` come from the schema layers, the other twelve are semantic checks. The strings are a supported interface; `--strict` promotes every warning to an error. +Every finding carries one of these, whichever layer produced it — `schema` and `uri-not-rfc` come from the schema layers, the other twelve are semantic checks. The strings are a supported interface; `--strict` promotes every warning to an error. `sil-lift validate --allow CODE` leaves a code out of the pass/fail decision. | code | level | what it flags | | ------------------------ | ------- | -------------------------------------------------------------------------- | diff --git a/src/sil_lift/_cli.py b/src/sil_lift/_cli.py index 1dcd3c1..c4da71e 100644 --- a/src/sil_lift/_cli.py +++ b/src/sil_lift/_cli.py @@ -72,16 +72,26 @@ def _collect_problems(args: argparse.Namespace) -> list[Problem]: ] +def _code_list(value: str) -> list[str]: + return [code for part in value.split(",") if (code := part.strip())] + + def _cmd_validate(args: argparse.Namespace) -> int: problems = _collect_problems(args) - errors = sum(1 for problem in problems if problem.level == "error") - warnings = len(problems) - errors + allowed = Counter(problem.code for problem in problems if problem.code in args.allow) + counted = [problem for problem in problems if problem.code not in args.allow] + errors = sum(1 for problem in counted if problem.level == "error") + warnings = len(counted) - errors failed = bool(errors) or (args.strict and bool(warnings)) if args.format == "json": json.dump( { "problems": [_problem_json(problem) for problem in problems], - "summary": {"errors": errors, "warnings": warnings}, + "summary": { + "errors": errors, + "warnings": warnings, + "allowed": allowed.total(), + }, }, sys.stdout, indent=2, @@ -91,7 +101,12 @@ def _cmd_validate(args: argparse.Namespace) -> int: for problem in problems: print(problem) strict_note = " (strict: warnings treated as errors)" if args.strict and warnings else "" - print(f"{errors} error(s), {warnings} warning(s){strict_note}") + allowed_note = ( + "; allowed: " + ", ".join(f"{code} {n}" for code, n in sorted(allowed.items())) + if allowed + else "" + ) + print(f"{errors} error(s), {warnings} warning(s){allowed_note}{strict_note}") return 1 if failed else 0 @@ -342,6 +357,15 @@ def main(argv: Sequence[str] | None = None) -> int: action="store_true", help="skip the filesystem media-presence check (suppresses missing-media findings)", ) + validate.add_argument( + "--allow", + action="extend", + type=_code_list, + default=[], + metavar="CODE[,CODE...]", + help="report these codes' findings but leave them out of the pass/fail decision " + "(comma-separated; repeatable)", + ) validate.add_argument( "--require-ids", action="store_true", diff --git a/tests/test_cli.py b/tests/test_cli.py index 1712365..8ac08d4 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -60,7 +60,7 @@ def test_validate_json_clean_file(capsys: pytest.CaptureFixture[str]) -> None: path = CORPUS_DIR / "ranges" / "test20080407.lift" assert main(["validate", str(path), "--format", "json"]) == 0 payload = json.loads(capsys.readouterr().out) - assert payload == {"problems": [], "summary": {"errors": 0, "warnings": 0}} + assert payload == {"problems": [], "summary": {"errors": 0, "warnings": 0, "allowed": 0}} def test_validate_strict_treats_warnings_as_errors(capsys: pytest.CaptureFixture[str]) -> None: @@ -78,6 +78,46 @@ def test_validate_no_check_media(capsys: pytest.CaptureFixture[str]) -> None: assert "[missing-media]" not in capsys.readouterr().out +def test_validate_allow_excludes_a_warning_from_strict(capsys: pytest.CaptureFixture[str]) -> None: + path = CORPUS_DIR / "negative" / "flex-quirks.lift" + assert main(["validate", str(path), "--strict", "--allow", "uri-not-rfc"]) == 0 + out = capsys.readouterr().out + assert "[uri-not-rfc]" in out # still reported, just not counted + assert "0 error(s), 0 warning(s); allowed: uri-not-rfc 1" in out + assert "strict:" not in out + + +def test_validate_allow_applies_to_errors(capsys: pytest.CaptureFixture[str]) -> None: + path = CORPUS_DIR / "negative" / "schema-invalid.lift" + assert main(["validate", str(path), "--allow", "schema"]) == 1 # form-missing-lang remains + assert "1 error(s), 0 warning(s); allowed: schema 2" in capsys.readouterr().out + argv = ["validate", str(path), "--allow", "schema", "--allow", "form-missing-lang"] + assert main(argv) == 0 + assert "allowed: form-missing-lang 1, schema 2" in capsys.readouterr().out + + +def test_validate_allow_comma_list(capsys: pytest.CaptureFixture[str]) -> None: + path = CORPUS_DIR / "negative" / "schema-invalid.lift" + assert main(["validate", str(path), "--allow", "schema, form-missing-lang,"]) == 0 + assert "allowed: form-missing-lang 1, schema 2" in capsys.readouterr().out + + +def test_validate_allow_unknown_code_is_a_no_op(capsys: pytest.CaptureFixture[str]) -> None: + path = CORPUS_DIR / "negative" / "flex-quirks.lift" + assert main(["validate", str(path), "--strict", "--allow", "no-such-code"]) == 1 + out = capsys.readouterr().out + assert "0 error(s), 1 warning(s) (strict: warnings treated as errors)" in out + assert "allowed" not in out + + +def test_validate_allow_json(capsys: pytest.CaptureFixture[str]) -> None: + path = CORPUS_DIR / "negative" / "schema-invalid.lift" + assert main(["validate", str(path), "--format", "json", "--allow", "schema"]) == 1 + payload = json.loads(capsys.readouterr().out) + assert payload["summary"] == {"errors": 1, "warnings": 0, "allowed": 2} + assert [problem["code"] for problem in payload["problems"]].count("schema") == 2 + + def test_validate_require_ids(tmp_path: Path, capsys: pytest.CaptureFixture[str]) -> None: path = tmp_path / "noid.lift" path.write_bytes( @@ -108,7 +148,7 @@ class _Stdin: monkeypatch.setattr("sys.stdin", _Stdin()) assert main(["validate", "-", "--format", "json"]) == 0 payload = json.loads(capsys.readouterr().out) - assert payload == {"problems": [], "summary": {"errors": 0, "warnings": 0}} + assert payload == {"problems": [], "summary": {"errors": 0, "warnings": 0, "allowed": 0}} def test_stats_json(capsys: pytest.CaptureFixture[str]) -> None: