diff --git a/docs/system_miner.md b/docs/system_miner.md index 889ad0c..eada9c1 100644 --- a/docs/system_miner.md +++ b/docs/system_miner.md @@ -904,6 +904,23 @@ small model was right), misses (keeping an answer it got wrong), calibration, and how often unusual inputs were sent up. The live escalations view on the network page shows every request as it is decided. +### Where your quality comes from + +Validators also split each certified system's result into its parts, so you can +see which part to improve next. The split explains the result; it does not +change pay, which follows the whole system's place on the frontier. + +| Part | Measured as | +|---|---| +| Small model | Its quality alone against the arena floor, the untrained base model under the harness | +| Harness | On the verification sample, the archived small model rerun with the plain task prompt, compared with the same model through your harness; plus what your output step added on every task | +| Router and escalation | The quality escalations added; rescues (escalations that turned a wrong answer right), waste and misses; escalation spend per rescue | + +The small model, output step and escalation gains add up exactly to your end to +end quality. `mt miner status` prints the split for the current round, +`mt miner simulate` prints the router and output step parts locally, and the +published card carries the table. + ### After the round | When | What happens to your system | diff --git a/microtensor/archive/intake.py b/microtensor/archive/intake.py index 35a5d1b..3b54213 100644 --- a/microtensor/archive/intake.py +++ b/microtensor/archive/intake.py @@ -196,6 +196,46 @@ def mirror(model: str, revision: str, org: str, token: str) -> str: return f"mirrored {model}@{revision} to {repo_id} (tag {tag})" +def _part(value: Any) -> str: + return "not measured" if value is None else f"{float(value):+.4f}" + + +def _breakdown_table(parts: dict[str, Any]) -> list[str]: + if not parts: + return [] + small = dict(parts.get("small_model") or {}) + harness = dict(parts.get("harness") or {}) + router = dict(parts.get("router") or {}) + floor = small.get("floor") + per_rescue = router.get("usd_per_rescue") + return [ + "", + "## Where the quality comes from", + "", + f"End to end quality {float(parts.get('end_to_end') or 0.0):.4f}. Measured by the " + "validators from the round's traces; it explains the result and does not change pay.", + "", + "| Part | Measured |", + "|---|---|", + f"| Small model | {float(small.get('quality') or 0.0):.4f} alone" + + ( + f", {_part(small.get('lift'))} over the arena floor {float(floor):.4f}" + if floor is not None + else "" + ) + + " |", + f"| Harness | {_part(harness.get('lift'))} over a plain prompt on " + f"{int(harness.get('sample') or 0)} sampled tasks; " + f"output step {_part(harness.get('output_gain'))} |", + f"| Router and escalation | {_part(router.get('escalation_gain'))} from escalations; " + f"rescues {float(router.get('rescues') or 0.0):.1%}, " + f"waste {float(router.get('waste') or 0.0):.1%}, " + f"misses {float(router.get('misses') or 0.0):.1%}" + + (f"; ${float(per_rescue):.6f} per rescue" if per_rescue is not None else "") + + " |", + ] + + def _card(entry: dict[str, Any], manifest: Any, licence: str) -> str: from microtensor.archive.push import _front_matter, _licence_section @@ -230,6 +270,7 @@ def _card(entry: dict[str, Any], manifest: Any, licence: str) -> str: lines.append(f"| Router | features {', '.join(system.router_features)} |") if system is not None and system.escalation is not None: lines.append(f"| Escalation | `{system.escalation.key}` from the arena allowlist |") + lines += _breakdown_table(dict(entry.get("breakdown") or {})) lines += [ "", f"System digest `{entry.get('system_id')}`.", diff --git a/microtensor/cli/miner.py b/microtensor/cli/miner.py index 9b91d4b..477cdf4 100644 --- a/microtensor/cli/miner.py +++ b/microtensor/cli/miner.py @@ -5,6 +5,7 @@ import logging import os from pathlib import Path +from typing import Any from microtensor.chain.client import ChainError from microtensor.chain.rounds import ( @@ -986,6 +987,38 @@ def submit() -> None: return 0 if ok else 1 +def _signed(value: Any) -> str: + return "n/a" if value is None else f"{float(value):+.4f}" + + +def breakdown_lines(parts: dict[str, Any]) -> list[str]: + if not parts: + return ["breakdown not measured yet this round"] + small = dict(parts.get("small_model") or {}) + harness = dict(parts.get("harness") or {}) + router = dict(parts.get("router") or {}) + floor = small.get("floor") + per_rescue = router.get("usd_per_rescue") + return [ + f"breakdown end to end {float(parts.get('end_to_end') or 0.0):.4f}, " + "information only, pay follows the whole system", + f" small {float(small.get('quality') or 0.0):.4f} alone, " + + ( + f"{_signed(small.get('lift'))} over the {float(floor):.4f} floor" + if floor is not None + else "no floor set" + ), + f" harness {_signed(harness.get('lift'))} over a plain prompt on " + f"{int(harness.get('sample') or 0)} tasks, " + f"output step {_signed(harness.get('output_gain'))}", + f" router escalation {_signed(router.get('escalation_gain'))}, " + f"rescues {float(router.get('rescues') or 0.0):.1%}, " + f"waste {float(router.get('waste') or 0.0):.1%}, " + f"misses {float(router.get('misses') or 0.0):.1%}, " + + (f"${float(per_rescue):.6f} per rescue" if per_rescue is not None else "no rescues"), + ] + + def _status(args: argparse.Namespace) -> int: try: config = _config(args) @@ -1034,6 +1067,19 @@ def _status(args: argparse.Namespace) -> int: ) print(f"contribution {parts}") + if manifest.system is not None and manifest.system.full: + from microtensor.miner.standing import fetch_breakdown + + measured = fetch_breakdown( + args.server, + manifest.track, + manifest.hardware_class, + manifest.digest(), + manifest.round_index, + ) + for line in breakdown_lines(measured): + print(line) + if standing.milestone: target_quality = standing.milestone.get("target_quality") target_cost = standing.milestone.get("target_cost") diff --git a/microtensor/harness/sdk.py b/microtensor/harness/sdk.py index 75cca64..f2b6bf4 100644 --- a/microtensor/harness/sdk.py +++ b/microtensor/harness/sdk.py @@ -142,6 +142,11 @@ def tool(name: str, argument: Any) -> Any: steps.append(_step("hook", event, at, started)) return dict(found) if isinstance(found, Mapping) else payload + def finish(self, output: Any) -> Any: + return self._hook("after", {"output": output}, [], time.perf_counter()).get( + "output", output + ) + def run(self, round_index: int, task_ref: str, prompt: str, inputs: Mapping[str, Any]) -> Trace: started = time.perf_counter() steps: list[HarnessStep] = [] diff --git a/microtensor/miner/simulate.py b/microtensor/miner/simulate.py index 01cad7e..f42c31b 100644 --- a/microtensor/miner/simulate.py +++ b/microtensor/miner/simulate.py @@ -166,6 +166,11 @@ def full_report(traces: Sequence[Any], score: Any, gold: dict[str, Any], metric: f"escalation rate {s.escalation_rate:.1%}", f"waste {s.waste:.1%} escalated when the small model was right", f"misses {s.misses:.1%} kept an answer the small model got wrong", + f"rescues {s.rescues:.1%} escalations that turned a wrong answer right", + f"escalation gain {s.escalation_gain:+.4f} quality the escalations added", + f"output hook gain {s.output_gain:+.4f} quality the harness output step added", + "cost per rescue " + + (f"${s.usd_per_rescue:.6f}" if s.usd_per_rescue is not None else "no rescues"), f"calibration error {s.calibration.get('ece', 0.0):.4f}", f"cost per 1k tasks ${s.cost_usd * 1000:.4f} " f"(small ${s.small_usd * 1000:.4f}, escalation ${s.escalation_usd * 1000:.4f})", diff --git a/microtensor/miner/standing.py b/microtensor/miner/standing.py index a2859de..2be8f23 100644 --- a/microtensor/miner/standing.py +++ b/microtensor/miner/standing.py @@ -30,6 +30,7 @@ class Standing: contribution: dict[str, float] = field(default_factory=dict) release_version: str = "" milestone: dict[str, Any] = field(default_factory=dict) + breakdown: dict[str, Any] = field(default_factory=dict) reachable: bool = False reason: str = "" @@ -45,6 +46,20 @@ def _get(base: str, path: str) -> Any: return json.loads(response.read() or b"null") +def fetch_breakdown( + base: str, track: str, hardware_class: str, system_digest: str, round_index: int | None = None +) -> dict[str, Any]: + query = f"?round={int(round_index)}" if round_index is not None else "" + try: + found = _get(base, f"/v1/arenas/{track}/{hardware_class}/escalations{query}") or {} + except (urllib.error.HTTPError, urllib.error.URLError, TimeoutError, OSError, ValueError): + return {} + for system in found.get("systems") or (): + if str(system.get("system", "")) == system_digest: + return dict((system.get("summary") or {}).get("breakdown") or {}) + return {} + + def fetch(base: str, track: str, hardware_class: str, system_digest: str) -> Standing: """Read this system's position from the public frontier and release. diff --git a/microtensor/scoring/system.py b/microtensor/scoring/system.py index f264530..9ea9f45 100644 --- a/microtensor/scoring/system.py +++ b/microtensor/scoring/system.py @@ -13,6 +13,12 @@ REFERENCE_CPU_USD_PER_HOUR: Final[float] = 0.04 COST_UNITS_PER_USD: Final[float] = 1_000_000.0 CORRECT_AT: Final[float] = 0.5 +UNMEASURED_HARNESS: Final[dict[str, Any]] = { + "sample": 0, + "with_harness": None, + "plain_prompt": None, + "lift": None, +} DIGITS: Final[int] = 6 @@ -28,11 +34,20 @@ class SystemScore: escalation_usd: float calibration: dict[str, Any] = field(default_factory=dict) escalation_by_profile: dict[str, float] = field(default_factory=dict) + rescues: float = 0.0 + escalation_gain: float = 0.0 + output_gain: float = 0.0 @property def cost_usd(self) -> float: return self.small_usd + self.escalation_usd + @property + def usd_per_rescue(self) -> float | None: + if self.rescues <= 0: + return None + return round(self.escalation_usd / self.rescues, 12) + def to_dict(self) -> dict[str, Any]: return { "tasks": self.tasks, @@ -46,6 +61,9 @@ def to_dict(self) -> dict[str, Any]: "cost_usd": round(self.cost_usd, 9), "calibration": dict(self.calibration), "escalation_by_profile": dict(self.escalation_by_profile), + "rescues": self.rescues, + "escalation_gain": self.escalation_gain, + "output_gain": self.output_gain, } @@ -67,7 +85,8 @@ def score_system( if count == 0: return SystemScore(0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0) by_ref = {trace.task_ref: trace for trace in traces if trace.task_ref in golds} - final_total = small_total = escalated = waste = misses = 0.0 + final_total = small_total = escalated = waste = misses = rescues = 0.0 + escalation_gain = output_gain = 0.0 escalation_cost = 0.0 small_times: list[float] = [] judged: list[tuple[float, bool]] = [] @@ -85,15 +104,20 @@ def score_system( judged.append((trace.small.confidence, small >= CORRECT_AT)) if trace.escalation is not None: escalated += 1 + escalation_gain += final - small if small >= CORRECT_AT: waste += 1 + elif final >= CORRECT_AT: + rescues += 1 price = allowlist.get(f"{trace.escalation.model}@{trace.escalation.revision}") if price is not None: escalation_cost += price.cost_usd( trace.escalation.prompt_tokens, trace.escalation.completion_tokens ) - elif small < CORRECT_AT: - misses += 1 + else: + output_gain += final - small + if small < CORRECT_AT: + misses += 1 by_profile: dict[str, list[bool]] = {} for ref, profile in (profiles or {}).items(): trace = by_ref.get(ref) @@ -120,4 +144,61 @@ def score_system( profile: round(sum(flags) / len(flags), DIGITS) for profile, flags in sorted(by_profile.items()) }, + rescues=round(rescues / count, DIGITS), + escalation_gain=round(escalation_gain / count, DIGITS), + output_gain=round(output_gain / count, DIGITS), ) + + +def harness_part( + probe: Mapping[str, Mapping[str, Any]], golds: Mapping[str, Any], metric: str +) -> dict[str, Any] | None: + pairs = [ + ( + score_task(metric, row.get("harnessed"), golds[ref]), + score_task(metric, row.get("plain"), golds[ref]), + ) + for ref, row in probe.items() + if ref in golds and "error" not in row + ] + if not pairs: + return None + harnessed = math.fsum(h for h, _ in pairs) / len(pairs) + plain = math.fsum(p for _, p in pairs) / len(pairs) + return { + "sample": len(pairs), + "with_harness": round(harnessed, DIGITS), + "plain_prompt": round(plain, DIGITS), + "lift": round(harnessed - plain, DIGITS), + } + + +def breakdown( + score: SystemScore, + *, + floor: float | None = None, + harness: Mapping[str, Any] | None = None, +) -> dict[str, Any]: + escalated = score.escalation_rate + return { + "end_to_end": score.quality, + "small_model": { + "quality": score.small_quality, + "floor": None if floor is None else round(floor, DIGITS), + "lift": None if floor is None else round(score.small_quality - floor, DIGITS), + }, + "harness": { + **(dict(harness) if harness else UNMEASURED_HARNESS), + "output_gain": score.output_gain, + }, + "router": { + "escalation_rate": escalated, + "escalation_gain": score.escalation_gain, + "rescues": score.rescues, + "waste": score.waste, + "misses": score.misses, + "precision": round(score.rescues / escalated, DIGITS) if escalated > 0 else None, + "escalation_usd": score.escalation_usd, + "usd_per_rescue": score.usd_per_rescue, + }, + } diff --git a/microtensor/validator/coordinated.py b/microtensor/validator/coordinated.py index e51fd0b..6f60bce 100644 --- a/microtensor/validator/coordinated.py +++ b/microtensor/validator/coordinated.py @@ -124,6 +124,7 @@ class RoundBudget: max_rss_bytes: int = 0 max_p95_ms: int = 0 reference_cost_ms: int = 0 + quality_floor: float | None = None def budgets_from(config: Mapping[str, Any]) -> dict[tuple[str, str], RoundBudget]: @@ -152,6 +153,9 @@ def budgets_from(config: Mapping[str, Any]) -> dict[tuple[str, str], RoundBudget max_rss_bytes=int(ceilings.get("max_rss_bytes") or 0), max_p95_ms=int(ceilings.get("max_p95_ms") or 0), reference_cost_ms=int(block.get("reference_cost_ms") or 0), + quality_floor=( + float(block["quality_floor"]) if block.get("quality_floor") is not None else None + ), ) return out diff --git a/microtensor/validator/evaluate.py b/microtensor/validator/evaluate.py index 04672ea..286b084 100644 --- a/microtensor/validator/evaluate.py +++ b/microtensor/validator/evaluate.py @@ -310,15 +310,11 @@ def run_system( """Execute the whole system, front then router then escalation.""" system = participant.system load = participant.manifest.load.to_dict() - requests = to_requests( - tasks.all, tasks.seed, tasks.track, participant.manifest.artifact_digest - ) + requests = to_requests(tasks.all, tasks.seed, tasks.track, participant.manifest.artifact_digest) front_path = artifact / system.locate(Role.FRONT) if not system.degenerate else artifact router_path = str(artifact / system.locate(Role.ROUTER)) if system.router else "" - specialist_path = ( - str(artifact / system.locate(Role.SPECIALIST)) if system.specialist else "" - ) + specialist_path = str(artifact / system.locate(Role.SPECIALIST)) if system.specialist else "" result = run_jailed( run_cascade, @@ -335,9 +331,7 @@ def run_system( if not result.ok: if result.fault is Fault.INFRASTRUCTURE: - raise Abstain( - f"{participant.hotkey}: execution infrastructure failed — {result.error}" - ) + raise Abstain(f"{participant.hotkey}: execution infrastructure failed — {result.error}") if not result.partial: return None, f"execution failed: {result.error}" log.info( @@ -367,9 +361,7 @@ def outcomes_from( for task in tasks.all: leg = by_ref.get(task.ref) partition = partition_of(tasks, task.ref) - end_to_end.append( - _outcome(task, leg.response if leg else None, metric, partition, track) - ) + end_to_end.append(_outcome(task, leg.response if leg else None, metric, partition, track)) front_only.append( _outcome(task, leg.front_response if leg else None, metric, partition, track) ) @@ -387,9 +379,7 @@ def score( ) -> tuple[tuple[TaskOutcome, ...], dict[str, Response], str]: track = get_track(tasks.track) metric = track.metric - requests = to_requests( - tasks.all, tasks.seed, tasks.track, participant.manifest.artifact_digest - ) + requests = to_requests(tasks.all, tasks.seed, tasks.track, participant.manifest.artifact_digest) result = run_jailed( run_tasks, str(artifact), @@ -480,9 +470,11 @@ def _extraction_partition_scores( """Entity micro-F1 per partition, aggregated over the whole document set.""" from microtensor.scoring.extraction import gold_entities, micro_f1, parse_entities - buckets: dict[ - str, tuple[list[set[tuple[str, str]] | None], list[set[tuple[str, str]]]] - ] = {ROTATING: ([], []), FIXED: ([], []), NOVEL: ([], [])} + buckets: dict[str, tuple[list[set[tuple[str, str]] | None], list[set[tuple[str, str]]]]] = { + ROTATING: ([], []), + FIXED: ([], []), + NOVEL: ([], []), + } for task in tasks.all: response = by_ref.get(task.ref) preds = parse_entities(response.output) if response and response.ok else None @@ -500,7 +492,6 @@ def _extraction_partition_scores( ) - def _decision_partition_scores( tasks: RoundTasks, by_ref: dict[str, Response] ) -> tuple[float, float, float, int, int, int]: @@ -539,6 +530,7 @@ def evaluate_participant( cpu_seconds: int = 0, hardware: HardwareClass | None = None, escalations: Mapping[str, Any] | None = None, + floor: float | None = None, ) -> Evaluation: hardware = hardware or get_class(participant.competition[1]) artifact = materialise(context, participant) @@ -577,6 +569,7 @@ def evaluate_participant( measured, dict(escalations or {}), _cpu_budget(context, cpu_seconds), + floor, ) cascade: CascadeResult | None = None @@ -588,9 +581,7 @@ def evaluate_participant( ) if failure or cascade is None: log.info("%s scored zero: %s", participant.hotkey, failure) - return _evaluation( - participant, tasks, gate=_verdict(gate, failure), measured=measured - ) + return _evaluation(participant, tasks, gate=_verdict(gate, failure), measured=measured) metric = get_track(tasks.track).metric outcomes, front_outcomes = outcomes_from(cascade.legs, tasks, metric) front_only_score = combine_partitions(*partition_scores(front_outcomes)) @@ -601,9 +592,7 @@ def evaluate_participant( ) if failure: log.info("%s scored zero: %s", participant.hotkey, failure) - return _evaluation( - participant, tasks, gate=_verdict(gate, failure), measured=measured - ) + return _evaluation(participant, tasks, gate=_verdict(gate, failure), measured=measured) metric = get_track(tasks.track).metric dataset_scorer = _DATASET_METRICS.get(metric) @@ -646,11 +635,17 @@ def _evaluate_full( measured: MeasuredEnvelope, escalations: dict[str, Any], cpu_seconds: int, + floor: float | None = None, ) -> Evaluation: from microtensor.scoring.metrics import score_task - from microtensor.scoring.system import COST_UNITS_PER_USD, score_system + from microtensor.scoring.system import ( + COST_UNITS_PER_USD, + breakdown, + harness_part, + score_system, + ) from microtensor.validator.live import AxonSystemClient, chain_verifier, run_live - from microtensor.validator.verify_system import jailed_verify + from microtensor.validator.verify_system import jailed_harness_probe, jailed_verify config = context.config system = participant.system @@ -721,6 +716,29 @@ def _evaluate_full( small_ms=_expected_ms(None, measured), profiles={task.ref: task.profile for task in tasks.all}, ) + golds = {task.ref: task.gold for task in tasks.all} + harness: dict[str, Any] | None = None + try: + probe = run_jailed( + jailed_harness_probe, + str(artifact), + system.body(), + str(artifact), + participant.manifest.load.to_dict(), + [trace.to_dict() for trace in live.traces], + {task.ref: (task.prompt, dict(task.inputs)) for task in tasks.all}, + tasks.seed, + track.chat, + limits=_limits(hardware, cpu_seconds), + allow_unsandboxed=config.allow_unsandboxed, + ) + if probe.ok: + harness = harness_part(dict(probe.value or {}), golds, track.metric) + else: + log.info("%s harness probe unavailable: %s", participant.hotkey, probe.error) + except Exception as exc: + log.info("%s harness probe unavailable: %s", participant.hotkey, exc) + parts = breakdown(score, floor=floor, harness=harness) answered = live.answered triggers = dict(verdict.get("triggers") or {}) rows = [] @@ -800,7 +818,12 @@ def _evaluate_full( n_fixed=n_fixed, n_novel=n_novel, front_only=score.small_quality, - calibration={**score.to_dict(), "p95_ms": round(p95, 1), "live": rows}, + calibration={ + **score.to_dict(), + "p95_ms": round(p95, 1), + "breakdown": parts, + "live": rows, + }, expected_ms=score.cost_usd * COST_UNITS_PER_USD, ) @@ -820,6 +843,7 @@ def evaluate_competition( hardware: HardwareClass | None = None, on_evaluated: Callable[[Evaluation, Participant], None] | None = None, escalations: Mapping[str, Any] | None = None, + floor: float | None = None, ) -> CompetitionResult: evaluations: list[Evaluation] = [] @@ -835,6 +859,7 @@ def evaluate_competition( cpu_seconds=cpu_seconds, hardware=hardware, escalations=escalations, + floor=floor, ) break except ArtifactMismatch as exc: diff --git a/microtensor/validator/round.py b/microtensor/validator/round.py index f352b54..2f8ad6a 100644 --- a/microtensor/validator/round.py +++ b/microtensor/validator/round.py @@ -372,6 +372,7 @@ def publish(evaluation: Evaluation, participant: Participant, _plan: Plan = plan hardware=hardware, on_evaluated=publish, escalations=plan.escalations.get((track, hardware_class), {}), + floor=arena.quality_floor if arena else None, ) except Abstain as exc: if leasing and holding: @@ -715,9 +716,7 @@ def abstain(reason: str, roster: Roster | None = None) -> RoundOutcome: try: require_engines() - roster = discover( - context, snapshot, round_, plan.allowlists, escalations=plan.escalations - ) + roster = discover(context, snapshot, round_, plan.allowlists, escalations=plan.escalations) except Abstain as exc: return abstain(str(exc)) except ProvenanceUnavailable as exc: diff --git a/microtensor/validator/verify_system.py b/microtensor/validator/verify_system.py index 057fb3e..0f0fe83 100644 --- a/microtensor/validator/verify_system.py +++ b/microtensor/validator/verify_system.py @@ -26,6 +26,8 @@ CONFIDENCE_TOLERANCE: Final[float] = 0.02 FEATURE_TOLERANCE: Final[float] = 0.05 SAMPLE: Final[int] = 8 +PLAIN_MIN_TOKENS: Final[int] = 64 +PLAIN_MAX_TOKENS: Final[int] = 512 class Replayer(Protocol): @@ -191,6 +193,80 @@ def verify_system( return Verdict(certified=not reasons, checked=len(chosen), reasons=tuple(reasons)) +def harness_probe( + artifact: Path, + system: SystemManifest, + engine: Any, + traces: Sequence[Trace], + tasks: Mapping[str, tuple[str, Mapping[str, Any]]], + *, + seed: str, + chat: bool, + size: int = SAMPLE, +) -> dict[str, dict[str, Any]]: + if system.harness is None or system.escalation is None or system.router is None: + return {} + + def refuse(_: str) -> EscalationCall: + raise RuntimeError("the harness probe never escalates") + + runtime = Runtime( + artifact / system.harness.path, + load_router(artifact / system.locate(Role.ROUTER), system.router_features), + system.router_features, + engine_small(engine, chat=chat), + refuse, + escalation=system.escalation, + system_digest="probe", + hotkey="probe", + ) + found: dict[str, dict[str, Any]] = {} + for trace in sample(traces, seed, size): + if trace.task_ref not in tasks: + continue + prompt, _ = tasks[trace.task_ref] + budget = min(PLAIN_MAX_TOKENS, max(PLAIN_MIN_TOKENS, 2 * len(trace.small.tokens))) + try: + plain = engine_small(engine, chat=chat, max_output_tokens=budget)(prompt).output + harnessed = runtime.finish(trace.small.output) + except Exception as exc: + found[trace.task_ref] = {"error": f"{type(exc).__name__}: {exc}"[:200]} + continue + found[trace.task_ref] = {"harnessed": harnessed, "plain": plain} + return found + + +def jailed_harness_probe( + artifact: str, + system: dict[str, Any], + weights: str, + load: dict[str, Any], + traces: list[dict[str, Any]], + tasks: dict[str, tuple[str, dict[str, Any]]], + seed: str, + chat: bool, + size: int = SAMPLE, +) -> dict[str, dict[str, Any]]: + from microtensor.harness.engines.gguf import GgufEngine + from microtensor.harness.execute import rebuild_manifest + + engine = GgufEngine() + engine.load(Path(weights), rebuild_manifest(load)) + try: + return harness_probe( + Path(artifact), + SystemManifest.from_dict(system), + engine, + [Trace.from_dict(raw) for raw in traces], + tasks, + seed=seed, + chat=chat, + size=size, + ) + finally: + engine.unload() + + def jailed_verify( artifact: str, system: dict[str, Any],