Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
173 changes: 173 additions & 0 deletions scripts/gen_conformance_corpus.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,173 @@
#!/usr/bin/env python3
"""Generate the behavioural conformance corpus from this implementation.

The Python package is the authority; `@liminate/ts-validator` is a hand-written
port of five of its stages. The existing parity gate compares the two
implementations' *reserved-word lists* against a frozen fixture — which is why
it stayed green while three minor versions of behaviour diverged. Adding `date`
to `_require_comparable` (v29, Calendar Era) changed what programs are accepted
and changed no word at all, so a word-list gate could not see it. Downstream,
`commongage` carried a code comment recording "a DATE RANGE cannot be written in
a Liminate sentence at this language version" for two months after it could.

This emits what a word list cannot: for each program in the corpus, what the
validation pipeline actually does with it.

The pipeline mirrored here is exactly the one the TypeScript package assembles
in `_run_line` — tokenize, reorder, parse, analyze, render — and stops where it
stops. No execution: the port has no interpreter, so runtime behaviour is not a
parity surface and is deliberately not recorded.

Its boundary, stated rather than left to be discovered: this is the *per-line*
surface. `validate()` wraps it with two things this corpus does not reach — the
first-line handling of `about`, and when-block buffering — so a program using
either is not a case here. They are a real parity surface and an honest gap,
not a passing one: a case that silently exercised `_run_line` instead would
report agreement about a path neither side took.

Usage:
python3 scripts/gen_conformance_corpus.py > tests/fixtures/conformance-<version>.json
"""

from __future__ import annotations

import json
import sys
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))

from liminate.analyzer import analyze # noqa: E402
from liminate.lexer import LexError, tokenize # noqa: E402
from liminate.parser import parse # noqa: E402
from liminate.renderer import render # noqa: E402
from liminate.reorderer import reorder # noqa: E402
from liminate.result import LiminateResult, ResultStatus # noqa: E402

ROOT = Path(__file__).resolve().parents[1]
CORPUS_PATH = ROOT / "tests" / "fixtures" / "conformance_corpus.txt"


def language_version() -> str:
"""The version of the source being read, not of whatever is installed.

`importlib.metadata.version("liminate")` answers for the installed
distribution, which on any machine with an editable checkout and a released
wheel is a different number from the tree this script just imported. A
corpus labelled with the wrong version is worse than an unlabelled one: the
port would compare itself against a file claiming a parity it never had.
"""
for line in (ROOT / "pyproject.toml").read_text(encoding="utf-8").splitlines():
if line.startswith("version"):
return line.split("=", 1)[1].strip().strip('"')
raise SystemExit("pyproject.toml states no version")

_ERROR_KIND = {
ResultStatus.ERROR_PARSE: "parse",
ResultStatus.ERROR_SEMANTIC: "semantic",
}


def validate_line(line: str, symtab: dict) -> dict:
"""One line through the ported stages, reported the way the port reports it.

Deliberately shaped as `ValidationResult` from the TypeScript package —
`status`, `canonical`, `errorKind`, `errorMessage` — so a case can be
compared field for field without either side reshaping the other's output.
"""
try:
tokens = tokenize(line)
except LexError as e:
return {"status": "error", "errorKind": "parse", "errorMessage": str(e)}
if not tokens:
return {"status": "success", "canonical": ""}

reordered = reorder(tokens)
if isinstance(reordered, LiminateResult):
return _from_result(reordered)

ast = parse(reordered)
if isinstance(ast, LiminateResult):
return _from_result(ast)

analysis = analyze(ast, symtab)
if isinstance(analysis, LiminateResult):
return {**_from_result(analysis), "canonical": render(ast)}

return {"status": "success", "canonical": render(ast)}


def _from_result(result: LiminateResult) -> dict:
if result.status in (ResultStatus.AMBER_PRECEDENCE, ResultStatus.AMBER_AMBIGUITY):
return {"status": "amber", "amberMessage": result.message}
return {
"status": "error",
"errorKind": _ERROR_KIND.get(result.status, result.status.value),
"errorMessage": result.message,
}


def read_corpus() -> list[dict]:
"""Read the corpus file.

A case is a `#` comment block naming it, then one or more program lines.
Multi-line cases share a symbol table, because a program's later lines
depend on what its earlier lines declared — a single-line corpus could not
reach any condition over a remembered value, which is most of the surface
that drifts.
"""
cases: list[dict] = []
name: str | None = None
lines: list[str] = []
for raw in CORPUS_PATH.read_text(encoding="utf-8").splitlines():
if raw.startswith("# "):
if name is not None and lines:
cases.append({"id": name, "source": "\n".join(lines)})
name, lines = raw[2:].strip(), []
elif raw.strip():
lines.append(raw)
if name is not None and lines:
cases.append({"id": name, "source": "\n".join(lines)})
return cases


def main() -> None:
"""Validate each line, then execute it so the next line sees the state.

The two implementations reach that state by different routes and this is
the one place the difference has to be handled. TypeScript has no
interpreter, so it simulates just enough with `update_symbol_table`; Python
has one, so it runs the line. What gets *recorded* is the validation result
either way — execution here only advances the symbol table, and its own
statuses (a fired prohibition, a runtime error) are never written to the
corpus, because the port cannot produce them and a parity file must not
contain a field one side can never match.
"""
from liminate.run import Session # noqa: PLC0415 — keeps the import local

cases = []
for case in read_corpus():
session = Session()
results = []
for line in case["source"].splitlines():
result = validate_line(line, session.symtab)
results.append(result)
if result["status"] == "success":
session.run_line(line)
cases.append({"id": case["id"], "source": case["source"], "results": results})
json.dump(
{
"language_version": language_version(),
"generator": "scripts/gen_conformance_corpus.py",
"surface": "tokenize -> reorder -> parse -> analyze -> render",
"cases": cases,
},
sys.stdout,
indent=2,
sort_keys=True,
)
sys.stdout.write("\n")


if __name__ == "__main__":
main()
16 changes: 15 additions & 1 deletion src/liminate/reorderer.py
Original file line number Diff line number Diff line change
Expand Up @@ -165,8 +165,22 @@ def _validate_where_head(tokens: list[Token]) -> ReorderOutput:
head.type is TokenType.UNKNOWN
or (head.type is TokenType.VERB and head.value == "each")
)
# `is` opens a comparison; `includes` opens a membership test,
# and `not includes` is the same test negated. All three are
# conditions the parser builds and the analyzer types — `includes`
# and `not_includes` are accepted there explicitly as
# list-membership over any operand types. Admitting only `is` here
# meant this stage reported a condition as unparseable without
# ever handing it to the stage that parses it.
second_ok = (
second.type is TokenType.OPERATOR and second.value == "is"
(second.type is TokenType.OPERATOR and second.value == "is")
or (second.type is TokenType.CONNECTIVE
and second.value == "includes")
or (second.type is TokenType.OPERATOR
and second.value == "not"
and len(rest) > 2
and rest[2].type is TokenType.CONNECTIVE
and rest[2].value == "includes")
)
if not (head_ok and second_ok):
return LiminateResult(
Expand Down
28 changes: 24 additions & 4 deletions src/liminate/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -233,6 +233,20 @@ def record_result(self, result: LiminateResult | None) -> None:
ResultStatus.ERROR_SEMANTIC,
ResultStatus.ERROR_RUNTIME,
ResultStatus.PACK_VERB_FAILURE,
# A fired prohibition or an unmet requirement is the deontic core
# answering, and it is the answer a caller most needs. Omitting
# them made `forbid` less consequential to a shell than `cite`,
# which is a pack verb: a failed cite exited 1 and a violated
# forbid exited 0. Every shell consumer had to string-match
# "Prohibition violated" to learn the verdict, and one that
# forgot reported a denial as an admission.
#
# This is the exit status only. Execution still continues past a
# fired rule, so a program reports every violation rather than
# the first — which is the more useful behaviour and is why the
# fix is here rather than a halt.
ResultStatus.PROHIBITION_VIOLATED,
ResultStatus.REQUIREMENT_NOT_MET,
):
self.had_any_error = True

Expand Down Expand Up @@ -565,10 +579,16 @@ def _emit(
lines = source.splitlines()

# Phase 2 D-4 — run contradiction detection once over the whole program
# before execution, so a warning surfaces even if a later `require`/`forbid`
# halts the run before reaching the conflicting statement. Warning-only:
# emitted as an informational SUCCESS result, never blocks execution and
# never sets had_error.
# before execution, so the warning is independent of how far execution
# gets. Warning-only: emitted as an informational SUCCESS result, never
# blocks execution and never sets had_error.
#
# This used to say "even if a later `require`/`forbid` halts the run
# before reaching the conflicting statement." Neither halts the run. A
# fired rule unwinds its own statement — `executed=False` on that result —
# and execution carries on to the next, which is what lets a program
# report every violation rather than only the first. Running the check up
# front is still right; the reason given for it was not true.
contradiction_warnings = detect_contradictions(_collect_deontic_statements(lines))
if contradiction_warnings:
warn_result = LiminateResult(
Expand Down
Loading