Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 17 additions & 0 deletions docs/system_miner.md
Original file line number Diff line number Diff line change
Expand Up @@ -904,6 +904,23 @@ small model was right), misses (keeping an answer it got wrong), calibration,
and how often unusual inputs were sent up. The live escalations view on the
network page shows every request as it is decided.

### Where your quality comes from

Validators also split each certified system's result into its parts, so you can
see which part to improve next. The split explains the result; it does not
change pay, which follows the whole system's place on the frontier.

| Part | Measured as |
|---|---|
| Small model | Its quality alone against the arena floor, the untrained base model under the harness |
| Harness | On the verification sample, the archived small model rerun with the plain task prompt, compared with the same model through your harness; plus what your output step added on every task |
| Router and escalation | The quality escalations added; rescues (escalations that turned a wrong answer right), waste and misses; escalation spend per rescue |

The small model, output step and escalation gains add up exactly to your end to
end quality. `mt miner status` prints the split for the current round,
`mt miner simulate` prints the router and output step parts locally, and the
published card carries the table.

### After the round

| When | What happens to your system |
Expand Down
41 changes: 41 additions & 0 deletions microtensor/archive/intake.py
Original file line number Diff line number Diff line change
Expand Up @@ -196,6 +196,46 @@ def mirror(model: str, revision: str, org: str, token: str) -> str:
return f"mirrored {model}@{revision} to {repo_id} (tag {tag})"


def _part(value: Any) -> str:
return "not measured" if value is None else f"{float(value):+.4f}"


def _breakdown_table(parts: dict[str, Any]) -> list[str]:
if not parts:
return []
small = dict(parts.get("small_model") or {})
harness = dict(parts.get("harness") or {})
router = dict(parts.get("router") or {})
floor = small.get("floor")
per_rescue = router.get("usd_per_rescue")
return [
"",
"## Where the quality comes from",
"",
f"End to end quality {float(parts.get('end_to_end') or 0.0):.4f}. Measured by the "
"validators from the round's traces; it explains the result and does not change pay.",
"",
"| Part | Measured |",
"|---|---|",
f"| Small model | {float(small.get('quality') or 0.0):.4f} alone"
+ (
f", {_part(small.get('lift'))} over the arena floor {float(floor):.4f}"
if floor is not None
else ""
)
+ " |",
f"| Harness | {_part(harness.get('lift'))} over a plain prompt on "
f"{int(harness.get('sample') or 0)} sampled tasks; "
f"output step {_part(harness.get('output_gain'))} |",
f"| Router and escalation | {_part(router.get('escalation_gain'))} from escalations; "
f"rescues {float(router.get('rescues') or 0.0):.1%}, "
f"waste {float(router.get('waste') or 0.0):.1%}, "
f"misses {float(router.get('misses') or 0.0):.1%}"
+ (f"; ${float(per_rescue):.6f} per rescue" if per_rescue is not None else "")
+ " |",
]


def _card(entry: dict[str, Any], manifest: Any, licence: str) -> str:
from microtensor.archive.push import _front_matter, _licence_section

Expand Down Expand Up @@ -230,6 +270,7 @@ def _card(entry: dict[str, Any], manifest: Any, licence: str) -> str:
lines.append(f"| Router | features {', '.join(system.router_features)} |")
if system is not None and system.escalation is not None:
lines.append(f"| Escalation | `{system.escalation.key}` from the arena allowlist |")
lines += _breakdown_table(dict(entry.get("breakdown") or {}))
lines += [
"",
f"System digest `{entry.get('system_id')}`.",
Expand Down
46 changes: 46 additions & 0 deletions microtensor/cli/miner.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@
import logging
import os
from pathlib import Path
from typing import Any

from microtensor.chain.client import ChainError
from microtensor.chain.rounds import (
Expand Down Expand Up @@ -986,6 +987,38 @@ def submit() -> None:
return 0 if ok else 1


def _signed(value: Any) -> str:
return "n/a" if value is None else f"{float(value):+.4f}"


def breakdown_lines(parts: dict[str, Any]) -> list[str]:
if not parts:
return ["breakdown not measured yet this round"]
small = dict(parts.get("small_model") or {})
harness = dict(parts.get("harness") or {})
router = dict(parts.get("router") or {})
floor = small.get("floor")
per_rescue = router.get("usd_per_rescue")
return [
f"breakdown end to end {float(parts.get('end_to_end') or 0.0):.4f}, "
"information only, pay follows the whole system",
f" small {float(small.get('quality') or 0.0):.4f} alone, "
+ (
f"{_signed(small.get('lift'))} over the {float(floor):.4f} floor"
if floor is not None
else "no floor set"
),
f" harness {_signed(harness.get('lift'))} over a plain prompt on "
f"{int(harness.get('sample') or 0)} tasks, "
f"output step {_signed(harness.get('output_gain'))}",
f" router escalation {_signed(router.get('escalation_gain'))}, "
f"rescues {float(router.get('rescues') or 0.0):.1%}, "
f"waste {float(router.get('waste') or 0.0):.1%}, "
f"misses {float(router.get('misses') or 0.0):.1%}, "
+ (f"${float(per_rescue):.6f} per rescue" if per_rescue is not None else "no rescues"),
]


def _status(args: argparse.Namespace) -> int:
try:
config = _config(args)
Expand Down Expand Up @@ -1034,6 +1067,19 @@ def _status(args: argparse.Namespace) -> int:
)
print(f"contribution {parts}")

if manifest.system is not None and manifest.system.full:
from microtensor.miner.standing import fetch_breakdown

measured = fetch_breakdown(
args.server,
manifest.track,
manifest.hardware_class,
manifest.digest(),
manifest.round_index,
)
for line in breakdown_lines(measured):
print(line)

if standing.milestone:
target_quality = standing.milestone.get("target_quality")
target_cost = standing.milestone.get("target_cost")
Expand Down
5 changes: 5 additions & 0 deletions microtensor/harness/sdk.py
Original file line number Diff line number Diff line change
Expand Up @@ -142,6 +142,11 @@ def tool(name: str, argument: Any) -> Any:
steps.append(_step("hook", event, at, started))
return dict(found) if isinstance(found, Mapping) else payload

def finish(self, output: Any) -> Any:
return self._hook("after", {"output": output}, [], time.perf_counter()).get(
"output", output
)

def run(self, round_index: int, task_ref: str, prompt: str, inputs: Mapping[str, Any]) -> Trace:
started = time.perf_counter()
steps: list[HarnessStep] = []
Expand Down
5 changes: 5 additions & 0 deletions microtensor/miner/simulate.py
Original file line number Diff line number Diff line change
Expand Up @@ -166,6 +166,11 @@ def full_report(traces: Sequence[Any], score: Any, gold: dict[str, Any], metric:
f"escalation rate {s.escalation_rate:.1%}",
f"waste {s.waste:.1%} escalated when the small model was right",
f"misses {s.misses:.1%} kept an answer the small model got wrong",
f"rescues {s.rescues:.1%} escalations that turned a wrong answer right",
f"escalation gain {s.escalation_gain:+.4f} quality the escalations added",
f"output hook gain {s.output_gain:+.4f} quality the harness output step added",
"cost per rescue "
+ (f"${s.usd_per_rescue:.6f}" if s.usd_per_rescue is not None else "no rescues"),
f"calibration error {s.calibration.get('ece', 0.0):.4f}",
f"cost per 1k tasks ${s.cost_usd * 1000:.4f} "
f"(small ${s.small_usd * 1000:.4f}, escalation ${s.escalation_usd * 1000:.4f})",
Expand Down
15 changes: 15 additions & 0 deletions microtensor/miner/standing.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ class Standing:
contribution: dict[str, float] = field(default_factory=dict)
release_version: str = ""
milestone: dict[str, Any] = field(default_factory=dict)
breakdown: dict[str, Any] = field(default_factory=dict)
reachable: bool = False
reason: str = ""

Expand All @@ -45,6 +46,20 @@ def _get(base: str, path: str) -> Any:
return json.loads(response.read() or b"null")


def fetch_breakdown(
base: str, track: str, hardware_class: str, system_digest: str, round_index: int | None = None
) -> dict[str, Any]:
query = f"?round={int(round_index)}" if round_index is not None else ""
try:
found = _get(base, f"/v1/arenas/{track}/{hardware_class}/escalations{query}") or {}
except (urllib.error.HTTPError, urllib.error.URLError, TimeoutError, OSError, ValueError):
return {}
for system in found.get("systems") or ():
if str(system.get("system", "")) == system_digest:
return dict((system.get("summary") or {}).get("breakdown") or {})
return {}


def fetch(base: str, track: str, hardware_class: str, system_digest: str) -> Standing:
"""Read this system's position from the public frontier and release.

Expand Down
87 changes: 84 additions & 3 deletions microtensor/scoring/system.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,12 @@
REFERENCE_CPU_USD_PER_HOUR: Final[float] = 0.04
COST_UNITS_PER_USD: Final[float] = 1_000_000.0
CORRECT_AT: Final[float] = 0.5
UNMEASURED_HARNESS: Final[dict[str, Any]] = {
"sample": 0,
"with_harness": None,
"plain_prompt": None,
"lift": None,
}
DIGITS: Final[int] = 6


Expand All @@ -28,11 +34,20 @@ class SystemScore:
escalation_usd: float
calibration: dict[str, Any] = field(default_factory=dict)
escalation_by_profile: dict[str, float] = field(default_factory=dict)
rescues: float = 0.0
escalation_gain: float = 0.0
output_gain: float = 0.0

@property
def cost_usd(self) -> float:
return self.small_usd + self.escalation_usd

@property
def usd_per_rescue(self) -> float | None:
if self.rescues <= 0:
return None
return round(self.escalation_usd / self.rescues, 12)

def to_dict(self) -> dict[str, Any]:
return {
"tasks": self.tasks,
Expand All @@ -46,6 +61,9 @@ def to_dict(self) -> dict[str, Any]:
"cost_usd": round(self.cost_usd, 9),
"calibration": dict(self.calibration),
"escalation_by_profile": dict(self.escalation_by_profile),
"rescues": self.rescues,
"escalation_gain": self.escalation_gain,
"output_gain": self.output_gain,
}


Expand All @@ -67,7 +85,8 @@ def score_system(
if count == 0:
return SystemScore(0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0)
by_ref = {trace.task_ref: trace for trace in traces if trace.task_ref in golds}
final_total = small_total = escalated = waste = misses = 0.0
final_total = small_total = escalated = waste = misses = rescues = 0.0
escalation_gain = output_gain = 0.0
escalation_cost = 0.0
small_times: list[float] = []
judged: list[tuple[float, bool]] = []
Expand All @@ -85,15 +104,20 @@ def score_system(
judged.append((trace.small.confidence, small >= CORRECT_AT))
if trace.escalation is not None:
escalated += 1
escalation_gain += final - small
if small >= CORRECT_AT:
waste += 1
elif final >= CORRECT_AT:
rescues += 1
price = allowlist.get(f"{trace.escalation.model}@{trace.escalation.revision}")
if price is not None:
escalation_cost += price.cost_usd(
trace.escalation.prompt_tokens, trace.escalation.completion_tokens
)
elif small < CORRECT_AT:
misses += 1
else:
output_gain += final - small
if small < CORRECT_AT:
misses += 1
by_profile: dict[str, list[bool]] = {}
for ref, profile in (profiles or {}).items():
trace = by_ref.get(ref)
Expand All @@ -120,4 +144,61 @@ def score_system(
profile: round(sum(flags) / len(flags), DIGITS)
for profile, flags in sorted(by_profile.items())
},
rescues=round(rescues / count, DIGITS),
escalation_gain=round(escalation_gain / count, DIGITS),
output_gain=round(output_gain / count, DIGITS),
)


def harness_part(
probe: Mapping[str, Mapping[str, Any]], golds: Mapping[str, Any], metric: str
) -> dict[str, Any] | None:
pairs = [
(
score_task(metric, row.get("harnessed"), golds[ref]),
score_task(metric, row.get("plain"), golds[ref]),
)
for ref, row in probe.items()
if ref in golds and "error" not in row
]
if not pairs:
return None
harnessed = math.fsum(h for h, _ in pairs) / len(pairs)
plain = math.fsum(p for _, p in pairs) / len(pairs)
return {
"sample": len(pairs),
"with_harness": round(harnessed, DIGITS),
"plain_prompt": round(plain, DIGITS),
"lift": round(harnessed - plain, DIGITS),
}


def breakdown(
score: SystemScore,
*,
floor: float | None = None,
harness: Mapping[str, Any] | None = None,
) -> dict[str, Any]:
escalated = score.escalation_rate
return {
"end_to_end": score.quality,
"small_model": {
"quality": score.small_quality,
"floor": None if floor is None else round(floor, DIGITS),
"lift": None if floor is None else round(score.small_quality - floor, DIGITS),
},
"harness": {
**(dict(harness) if harness else UNMEASURED_HARNESS),
"output_gain": score.output_gain,
},
"router": {
"escalation_rate": escalated,
"escalation_gain": score.escalation_gain,
"rescues": score.rescues,
"waste": score.waste,
"misses": score.misses,
"precision": round(score.rescues / escalated, DIGITS) if escalated > 0 else None,
"escalation_usd": score.escalation_usd,
"usd_per_rescue": score.usd_per_rescue,
},
}
4 changes: 4 additions & 0 deletions microtensor/validator/coordinated.py
Original file line number Diff line number Diff line change
Expand Up @@ -124,6 +124,7 @@ class RoundBudget:
max_rss_bytes: int = 0
max_p95_ms: int = 0
reference_cost_ms: int = 0
quality_floor: float | None = None


def budgets_from(config: Mapping[str, Any]) -> dict[tuple[str, str], RoundBudget]:
Expand Down Expand Up @@ -152,6 +153,9 @@ def budgets_from(config: Mapping[str, Any]) -> dict[tuple[str, str], RoundBudget
max_rss_bytes=int(ceilings.get("max_rss_bytes") or 0),
max_p95_ms=int(ceilings.get("max_p95_ms") or 0),
reference_cost_ms=int(block.get("reference_cost_ms") or 0),
quality_floor=(
float(block["quality_floor"]) if block.get("quality_floor") is not None else None
),
)
return out

Expand Down
Loading
Loading