diff --git a/docs/system_miner.md b/docs/system_miner.md index 6cb4b0f..9b6af88 100644 --- a/docs/system_miner.md +++ b/docs/system_miner.md @@ -790,7 +790,7 @@ What your artifact must be: a GGUF with a chat template, whose tokenizer makes e ### Train ```bash -python scripts/train_decider.py \ +python scripts/train_specialist.py \ --base Qwen/Qwen3-1.7B --revision \ --corpus train.jsonl --out decider/ \ --teacher teacher.jsonl @@ -817,7 +817,7 @@ Use `--token-embedding-type` on models with tied embeddings, such as the small Q python scripts/fold_temperature.py --model decider.gguf --corpus train.jsonl --out decider-cal.gguf ``` -This fits one temperature on your train split and folds it into the GGUF by scaling `output_norm.weight`. Every logit scales by the same factor, so your top answer never changes; only your stated confidence moves towards the truth. It is confirmed exact on Qwen2, Qwen3 and Llama, and refuses everything else, including anything with an output bias or a logit soft cap. **Qwen3.5 is refused**: measured, folding a temperature of 2 moved its probabilities by up to 0.025 against the true scaled values, where Qwen3 stays within one millionth. On Qwen3.5, calibrate through training instead, with the Brier term in `train_decider.py`. +This fits one temperature on your train split and folds it into the GGUF by scaling `output_norm.weight`. Every logit scales by the same factor, so your top answer never changes; only your stated confidence moves towards the truth. It is confirmed exact on Qwen2, Qwen3 and Llama, and refuses everything else, including anything with an output bias or a logit soft cap. **Qwen3.5 is refused**: measured, folding a temperature of 2 moved its probabilities by up to 0.025 against the true scaled values, where Qwen3 stays within one millionth. On Qwen3.5, calibrate through training instead, with the Brier term in `train_specialist.py`. ### Check before you submit diff --git a/pyproject.toml b/pyproject.toml index 1a151d6..0987180 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -113,7 +113,7 @@ ignore = [ "microtensor/serving/supervise.py" = ["S310", "S603"] "microtensor/serving/audit.py" = ["S310"] "microtensor/harness/chat_template.py" = ["S701"] -"scripts/train_decider.py" = ["S311"] +"scripts/train_specialist.py" = ["S311"] "tests/*" = ["S", "PLR2004"] "scripts/optimise_harness.py" = ["S311"] "microtensor/rigs/**" = ["N", "A", "C4", "RUF", "S", "B", "SIM", "UP", "E501", "W"] diff --git a/scripts/train_decider.py b/scripts/train_specialist.py similarity index 75% rename from scripts/train_decider.py rename to scripts/train_specialist.py index dc554c4..545eb52 100644 --- a/scripts/train_decider.py +++ b/scripts/train_specialist.py @@ -11,7 +11,7 @@ from microtensor.core.tracks import DECIDE, get_track from microtensor.harness import decision_prompt -from microtensor.scoring.metrics import expected_answers, gold_label +from microtensor.scoring.metrics import expected_answers, gold_label, score_task from microtensor.tasks.corpus import load_corpus @@ -189,6 +189,62 @@ def generation_loss(model: Any, tokenizer: Any, batch: Sequence[Generation], cha return total / len(batch) +def rlcr_reward(score: float, confidence: float, weight: float) -> float: + correct = 1.0 if score >= 0.5 else 0.0 + return score - weight * (confidence - correct) ** 2 + + +def rlcr_step( + model: Any, + tokenizer: Any, + tasks: Sequence[Any], + metric: str, + chat: bool, + samples: int, + temperature: float, + weight: float, +) -> tuple[Any, float]: + import torch + + total = torch.zeros((), device=model.device) + rewards_seen: list[float] = [] + for task in tasks: + head = tokenizer(prompt_text(tokenizer, task.prompt, chat), add_special_tokens=False) + prompt_ids = torch.tensor([head["input_ids"]], device=model.device) + with torch.no_grad(): + drawn = model.generate( + prompt_ids, + do_sample=True, + temperature=temperature, + max_new_tokens=task.max_output_tokens, + num_return_sequences=samples, + pad_token_id=tokenizer.eos_token_id, + ) + completions = drawn[:, prompt_ids.shape[1] :] + log_means = [] + rewards = [] + for row in completions: + kept = row[row != tokenizer.pad_token_id] if tokenizer.pad_token_id is not None else row + if kept.numel() == 0: + continue + ids = torch.cat([prompt_ids[0], kept]).unsqueeze(0) + logits = model(input_ids=ids).logits[0, prompt_ids.shape[1] - 1 : -1].float() + chosen = torch.log_softmax(logits, dim=-1).gather(1, kept.unsqueeze(1)).squeeze(1) + mean = chosen.mean() + text = tokenizer.decode(kept, skip_special_tokens=True) + score = float(score_task(metric, text, task.gold)) + rewards.append(rlcr_reward(score, float(mean.detach().exp()), weight)) + log_means.append(mean) + if len(rewards) < 2: + continue + values = torch.tensor(rewards, device=model.device) + advantage = (values - values.mean()) / (values.std() + 1e-6) + total = total - (advantage * torch.stack(log_means)).mean() + rewards_seen.extend(rewards) + count = max(1, len(tasks)) + return total / count, (sum(rewards_seen) / len(rewards_seen) if rewards_seen else 0.0) + + def main(argv: Sequence[str] | None = None) -> int: parser = argparse.ArgumentParser( description=( @@ -217,6 +273,15 @@ def main(argv: Sequence[str] | None = None) -> int: parser.add_argument("--full", action="store_true", help="full fine tune instead of LoRA") parser.add_argument("--rank", type=int, default=16) parser.add_argument("--seed", type=int, default=0) + parser.add_argument( + "--rlcr-steps", type=int, default=0, help="RL steps rewarding right and honest answers" + ) + parser.add_argument("--rlcr-samples", type=int, default=4, help="answers drawn per task") + parser.add_argument("--rlcr-temperature", type=float, default=0.8) + parser.add_argument( + "--rlcr-weight", type=float, default=1.0, help="weight of the calibration penalty" + ) + parser.add_argument("--rlcr-batch", type=int, default=4) args = parser.parse_args(argv) import torch @@ -278,6 +343,29 @@ def main(argv: Sequence[str] | None = None) -> int: steps = max(1, math.ceil(len(data) / args.batch)) print(f"epoch {epoch + 1}/{args.epochs} examples {len(data)} loss {running / steps:.4f}") + if args.rlcr_steps and not deciding: + pool = list(load_corpus(args.corpus, args.track).tasks) + rng = random.Random(args.seed) + for step in range(1, args.rlcr_steps + 1): + batch = rng.sample(pool, min(args.rlcr_batch, len(pool))) + loss, reward = rlcr_step( + model, + tokenizer, + batch, + track.metric, + track.chat, + args.rlcr_samples, + args.rlcr_temperature, + args.rlcr_weight, + ) + if loss.requires_grad: + optimiser.zero_grad() + loss.backward() + optimiser.step() + print(f"rlcr {step}/{args.rlcr_steps} mean reward {reward:.4f}") + elif args.rlcr_steps: + print("decision arenas already train on the Brier rule, which rewards honest confidence") + if not args.full: model = model.merge_and_unload() args.out.mkdir(parents=True, exist_ok=True)