From 603b7ae78eb710a62616eb7a7cdb979eba81fffc Mon Sep 17 00:00:00 2001 From: Yusheng Su Date: Sun, 4 Oct 2026 19:34:04 +0900 Subject: [PATCH 1/2] Add sequence packing for EAGLE3, DFlash, and DFlash2 training --- docs/sections/basic_usage/training.md | 78 + .../benchmarks/dflash-sequence-packing.md | 180 + .../benchmarks/eagle3-sequence-packing.md | 151 + .../benchmarks/full-model-sequence-packing.md | 181 + .../benchmarks/online-sequence-packing.md | 147 + .../sequence-packing-results/README.md | 78 + .../dflash-family-128anchors-bf16.json | 1070 ++++++ .../dflash-family-512anchors-bf16.json | 563 +++ .../dflash-family-tiny-fp32.json | 214 ++ .../eagle3-large-bf16.json | 235 ++ .../eagle3-medium-bf16.json | 605 ++++ .../eagle3-tiny-fp32.json | 175 + .../full-model-analysis.json | 899 +++++ .../full-model-audit.json | 1107 ++++++ .../online-32step.json | 568 +++ .../target-parameter-inventory.json | 3086 +++++++++++++++++ examples/configs/README.md | 1 + scripts/benchmark_dflash_sequence_packing.py | 7 + scripts/benchmark_online_sequence_packing.py | 7 + scripts/benchmark_sequence_packing.py | 7 + .../algorithms/common/dflash_family_model.py | 147 +- .../algorithms/common/hidden_states_data.py | 43 + specforge/algorithms/common/providers.py | 10 + specforge/algorithms/contracts.py | 2 + specforge/algorithms/dflash/providers.py | 4 + specforge/algorithms/eagle3/data.py | 69 + specforge/algorithms/eagle3/model.py | 65 +- specforge/algorithms/eagle3/providers.py | 7 +- specforge/application/planning.py | 18 + .../benchmark_dflash_sequence_packing.py | 465 +++ .../benchmark_online_sequence_packing.py | 1166 +++++++ .../benchmarks/benchmark_sequence_packing.py | 460 +++ specforge/config/schema.py | 11 + specforge/launch.py | 37 +- specforge/modeling/draft/llama3_eagle.py | 32 +- specforge/modeling/packed_dflash.py | 158 + specforge/modeling/packed_sequence.py | 76 + specforge/training/assembly.py | 1 + specforge/training/disaggregated.py | 2 + specforge/training/strategies/base.py | 100 +- tests/test_config/test_sequence_packing.py | 148 + tests/test_data/test_sequence_packing.py | 239 ++ .../test_dflash_sequence_packing.py | 360 ++ .../test_eagle3_sequence_packing.py | 235 ++ .../test_online_packing_benchmark.py | 211 ++ .../test_online_sequence_packing_gate.py | 298 ++ .../test_packing_strategy_metadata.py | 94 + .../test_sequence_packing_lifecycle.py | 118 + .../test_sequence_packing_wiring.py | 236 ++ 49 files changed, 14140 insertions(+), 31 deletions(-) create mode 100644 docs/sections/benchmarks/dflash-sequence-packing.md create mode 100644 docs/sections/benchmarks/eagle3-sequence-packing.md create mode 100644 docs/sections/benchmarks/full-model-sequence-packing.md create mode 100644 docs/sections/benchmarks/online-sequence-packing.md create mode 100644 docs/sections/benchmarks/sequence-packing-results/README.md create mode 100644 docs/sections/benchmarks/sequence-packing-results/dflash-family-128anchors-bf16.json create mode 100644 docs/sections/benchmarks/sequence-packing-results/dflash-family-512anchors-bf16.json create mode 100644 docs/sections/benchmarks/sequence-packing-results/dflash-family-tiny-fp32.json create mode 100644 docs/sections/benchmarks/sequence-packing-results/eagle3-large-bf16.json create mode 100644 docs/sections/benchmarks/sequence-packing-results/eagle3-medium-bf16.json create mode 100644 docs/sections/benchmarks/sequence-packing-results/eagle3-tiny-fp32.json create mode 100644 docs/sections/benchmarks/sequence-packing-results/full-model-analysis.json create mode 100644 docs/sections/benchmarks/sequence-packing-results/full-model-audit.json create mode 100644 docs/sections/benchmarks/sequence-packing-results/online-32step.json create mode 100644 docs/sections/benchmarks/sequence-packing-results/target-parameter-inventory.json create mode 100755 scripts/benchmark_dflash_sequence_packing.py create mode 100755 scripts/benchmark_online_sequence_packing.py create mode 100755 scripts/benchmark_sequence_packing.py create mode 100644 specforge/benchmarks/benchmark_dflash_sequence_packing.py create mode 100644 specforge/benchmarks/benchmark_online_sequence_packing.py create mode 100644 specforge/benchmarks/benchmark_sequence_packing.py create mode 100644 specforge/modeling/packed_dflash.py create mode 100644 specforge/modeling/packed_sequence.py create mode 100644 tests/test_config/test_sequence_packing.py create mode 100644 tests/test_data/test_sequence_packing.py create mode 100644 tests/test_modeling/test_dflash_sequence_packing.py create mode 100644 tests/test_runtime/test_eagle3_sequence_packing.py create mode 100644 tests/test_runtime/test_online_packing_benchmark.py create mode 100644 tests/test_runtime/test_online_sequence_packing_gate.py create mode 100644 tests/test_runtime/test_packing_strategy_metadata.py create mode 100644 tests/test_runtime/test_sequence_packing_lifecycle.py create mode 100644 tests/test_runtime/test_sequence_packing_wiring.py diff --git a/docs/sections/basic_usage/training.md b/docs/sections/basic_usage/training.md index c7c055b63..81b8ead88 100644 --- a/docs/sections/basic_usage/training.md +++ b/docs/sections/basic_usage/training.md @@ -549,6 +549,84 @@ a complete checkpoint and points `-best` at it, even when `training.save_interval` is zero. `-latest` continues to identify the newest complete checkpoint. +## Sequence packing + +Text EAGLE3, DFlash, and DFlash2 can remove context padding across the samples +of each microbatch, with offline features or online server capture: + +```bash +specforge train \ + --config examples/configs/offline/colocated/qwen3-8b-eagle3-offline.yaml \ + training.batch_size=4 \ + training.attention_backend=flex_attention \ + training.sequence_packing=true +``` + +Packing defaults to `false`. Add the same two training overrides to an online +config or a DFlash/DFlash2 config. DFlash2 uses `training.strategy=dflash` with +a DFlash2 draft-model config. Offline evaluation loaders also honor packing. +The implementation requires FlexAttention and does not support USP, other +algorithms, multimodal positions, `compact_teacher`, or `trim_loss_positions`. +DFlash/DFlash2 LK objectives and D-PACE are supported; EAGLE3 LK is not. + +`batch_size` still counts original samples per rank and microbatch. Packing +concatenates those samples into one row, resets positions at each document, +isolates attention, and prevents labels from crossing document boundaries. +It does not change sample order, +gradient accumulation, reference acknowledgement, or the optimizer schedule. +`data.max_length` remains the truncation limit for each original sample; a +packed row can be longer than that limit. + +EAGLE3 prevents the initial teacher shift and every subsequent TTT shift from +crossing document boundaries. Its loss keeps the original +`batch_size * longest_sample_length` denominator so packing does not implicitly +increase the learning rate on batches that previously had substantial padding. + +DFlash/DFlash2 preserve the original per-document anchor counts and random +sampling, including invalid anchor slots. Context and proposal positions reset +for each document; full and sliding attention, block labels, and teacher +predecessors respect document boundaries. After the backbone, the original +`[batch, anchors, block]` axes are restored for the loss, D-PACE, selector, and +metrics. Packing does not reduce the number of sampled proposal tokens. +When CPU loss masks provide per-document valid-anchor counts, invalid padded +proposal slots skip the backbone; outputs are restored to the original loss +layout. This saves computation without discarding valid sampled anchors. + +Online packing happens in the consumer after each target capture is fetched. +The producer still captures individual prompts, and the consumer acknowledges +the original sample IDs. Target capture, queue order, and the durable cursor +retain their existing behavior. + +For lengths `[2048, 512, 256, 256]`, padded execution processes 8192 rows per TTT +step and packed execution processes 3072. This removes 62.5% of those rows, but +does not imply a 2.67x end-to-end speedup: attention mask construction, kernel +occupancy, feature I/O, and optimizer work still cost time. There is no padding +to remove at batch size one or when every sample has the same length, and +packing can be slower in those cases. Compare equal samples and settings using +effective (unpadded) tokens/s, not packed steps/s. Packing changes training +execution; it is not a speculative-serving speedup. + +The reproducible GPU benchmarks in `scripts/benchmark_sequence_packing.py` +(EAGLE3) and `scripts/benchmark_dflash_sequence_packing.py` (DFlash/DFlash2) +check production loss/gradient agreement before measuring training steps. +See its `--help` for model dimensions and length profiles. Generated features +measure training compute, not target capture, dataset quality, or a full epoch. +See the [H200 measurements and validation scope](../benchmarks/eagle3-sequence-packing.md) +for a controlled EAGLE3 padded-versus-packed comparison, and the +[DFlash/DFlash2 measurements and online validation](../benchmarks/dflash-sequence-packing.md) +for those models. The speedup depends on context lengths, valid proposal counts, +and mask construction; it is not implied by the padding fraction alone. + +For a concurrent target-capture and training comparison, use +`scripts/benchmark_online_sequence_packing.py`. The +[real Qwen3-4B online benchmark](../benchmarks/online-sequence-packing.md) +measured 4.1% DFlash and 4.5% DFlash2 pipeline throughput gains on 128 ShareGPT +conversations; it reports final checkpoint time separately. +The subsequent [256-step full-model comparison](../benchmarks/full-model-sequence-packing.md) +adds periodic diagnostics/checkpoints and four runs per mode: DFlash/DFlash2 +training steps improved 1.044×/1.052×, with full completion including both +checkpoints improving 1.037×/1.033× on that workload. + ## Compact offline teacher Offline text EAGLE3 can project teacher targets in exact vocabulary chunks diff --git a/docs/sections/benchmarks/dflash-sequence-packing.md b/docs/sections/benchmarks/dflash-sequence-packing.md new file mode 100644 index 000000000..5c97cb051 --- /dev/null +++ b/docs/sections/benchmarks/dflash-sequence-packing.md @@ -0,0 +1,180 @@ +# DFlash/DFlash2 sequence packing and online validation + +Based on upstream [`53398a8f01ae47175bee8459c5b5cca3848c8a7e`](https://github.com/sgl-project/SpecForge/tree/53398a8f01ae47175bee8459c5b5cca3848c8a7e) +plus the local `codex/sequence-packing` changes, tested on 2026-10-02. + +The final H200 benchmark with 512 anchors per document and 64% context padding +measured **1.39x DFlash** and **1.47x DFlash2** training-step speedups, with about +29% lower peak allocated memory. These are consumer-compute results. A separate +[real Qwen3-4B online pipeline benchmark](online-sequence-packing.md) measured +**1.041× DFlash** and **1.045× DFlash2** on 128 ShareGPT conversations, including +target capture and transport but excluding startup and final checkpoint. + +## Supported execution + +Text EAGLE3, DFlash, and DFlash2 support packing for offline features and online +server capture. DFlash2 remains `training.strategy: dflash` with a +`DFlash2DraftModel` draft config. Enable packing on an existing supported config: + +```yaml +training: + attention_backend: flex_attention + sequence_packing: true +``` + +For example, add those overrides to +`examples/configs/online/disaggregated/external/qwen3-4b-dflash-online.yaml`. +Packing is opt-in. It packs the existing microbatch and does not reorder samples +or change the number of samples per optimizer step. Other algorithms, USP, +multimodal positions, compact teacher, and loss-position trimming are unsupported. +DFlash/DFlash2 LK and D-PACE objectives are supported; EAGLE3 LK is unsupported. + +## Call chain and numerical contract + +```text +SGLang /generate capture (one request per original sample) + -> MooncakeFeatureStore + SampleRef + -> online consumer / FeatureDataLoader + -> provider.build_packed_collator: [B, max(L)] -> [1, sum(L)] + -> DFlashTrainStrategy: host document lengths and valid-anchor counts + -> OnlineDFlashModel: original sampler, isolated context/proposal positions + -> DFlashDraftModel or DFlash2DraftModel + -> restore [B, anchors, block] for original losses and metrics + -> trainer optimizer / original sample-ID acknowledgements / checkpoint +``` + +The original sampler runs once on the original `[B, max(L)]` loss-mask shape, +preserving its random draws, each document's anchor budget, anchor order, and +keep mask. Only the small sampling mask is padded. Context hidden states, token +IDs, and optional target final states are concatenated without padding. + +Attention never crosses a document boundary. Full and sliding masks respect +per-document starts; target labels cannot pass document ends, and teacher +predecessors cannot precede document starts. RoPE positions reset per document. +DFlash2 convolutions retain complete proposal blocks. Loss reductions, D-PACE +sequence weights, selector objectives, and diagnostics retain their original +batch and anchor axes. + +The packed attention mask uses conservative sparse tile ranges, with the exact +per-token predicate for partial tiles. This avoids constructing a dense +`total_queries * total_keys` mask across unrelated documents. Where integer +features are on the CPU, the strategy also supplies per-document valid-anchor +counts: invalid padded proposal slots skip the backbone and outputs are +scattered back into the original loss layout. Direct GPU-only callers without +those counts retain all proposal slots. No CUDA `nonzero()` or host readback is +needed to size the compact proposals. + +Online target requests, capture tensors, transport protocol, queue order, and +acknowledged IDs are unchanged. Packing affects the consumer's draft training; +it does not itself accelerate target capture or speculative serving. + +## Why concatenation alone was slower + +The first correct implementation removed context padding but still generated +a dense FlexAttention mask before converting it to sparse blocks. With four +samples, the flattened query/key grid also contained cross-document regions +that were entirely masked. A CUDA profiler measured DFlash mask construction +at 8.61 ms padded versus 29.43 ms packed in the 512-anchor profile, while the +backbone remained roughly 22 ms. The initial packed implementation regressed +median whole-step time by about 20%. This motivated the sparse tile builder and +invalid-proposal elimination; concatenation alone is not a reliable speedup. + +## Validation and benchmark scope + +The production comparison checks exact sampled anchors and keep masks, loss, +all loss terms, and every trainable gradient before timing. Model tests also +compare all detailed metrics, cover FP32/BF16, full/hybrid sliding attention, +D-PACE variants, LK lambda/TV, DFlash2 selectors and nonzero convolutions, short +and unsupervised documents, and cross-document perturbation isolation. + +Final regression runs passed 204 CPU tests with 540 subtests (nine GPU/live +tests skipped in that CPU run), and 57 GPU model/host-sync tests with 45 +subtests. The final live gate and strategy-metadata checks passed five tests +with seven subtests. These counts describe separate suites, not unique tests +summed across repeated runs. + +A separate two-rank `FULL_SHARD` probe also passed for EAGLE3, DFlash, and +DFlash2. The ranks used different document lengths, and the packed local loss +and all trainable gradients matched the padded reference after its gradients +were averaged across ranks. This verifies sharded forwards/backwards and host +packing metadata, not multi-node online throughput. The following command uses +the retained local probe, which is not committed with this report: + +```bash +CUDA_VISIBLE_DEVICES=0,1 PYTHONPATH=. python -m torch.distributed.run \ + --standalone --nproc-per-node=2 \ + artifacts/sequence-packing/check_packed_fsdp.py +``` + +The live online gate uses a tiny eight-layer Llama target and an FSDP training +consumer, with an isolated copy of SGLang 0.5.18 and the repository's capture +patch. The initial gate used separate H200s; the final optimized gate colocated +both processes on one H200 after unrelated work occupied the original devices. +Actual captures pass through Mooncake TCP, +`RefDistributor`, a SQLite durable ledger, and the packed feature loader. +EAGLE3, DFlash, and DFlash2 each execute two optimizer steps/four microsteps, +acknowledge all eight original sample IDs, and write checkpoints. This is +functional online evidence, not an end-to-end throughput benchmark, RDMA +validation, or pretrained-model convergence evidence. + +The training-step benchmark uses one H200, BF16, two actual draft layers, +hidden size 2048, intermediate size 8192, 16 attention heads / 4 KV heads, +vocabulary 32000, block size 16, microbatch four, and 25% prompt masking. +Frozen target weights and features are synthetic. It includes the production +strategy's CPU integer-feature processing/transfers, forward, backward, and +`BF16Optimizer` update. Hidden features are GPU resident. Target capture, +hidden-feature I/O/H2D, distributed communication, and serving are excluded. + +Five warmup steps include compilation; 20 synchronized wall-clock steps measure +steady-state performance. Useful tokens/s divides the original unpadded input +tokens by mean step time. The benchmark GPU was dedicated to these measurements; +other GPUs on the node also ran unrelated validation workloads. + +| Algorithm | Anchors/document | Lengths | Padding | P50 ms padded → packed | P50 speedup | Mean ms padded → packed | Useful tokens/s padded → packed | Peak GiB padded → packed | +| --- | ---: | --- | ---: | ---: | ---: | ---: | ---: | ---: | +| DFlash | 512 | 128, 256, 512, 2048 | 64.06% | 102.68 → 73.68 | **1.39x** | 102.74 → 73.74 | 28,655 → 39,923 | 11.82 → 8.34 | +| DFlash2 | 512 | 128, 256, 512, 2048 | 64.06% | 105.68 → 72.01 | **1.47x** | 105.65 → 72.00 | 27,865 → 40,891 | 13.28 → 9.44 | +| DFlash | 128 | 128, 256, 512, 2048 | 64.06% | 35.38 → 30.50 | 1.16x | 35.42 → 30.50 | 83,107 → 96,540 | 5.81 → 5.43 | +| DFlash2 | 128 | 128, 256, 512, 2048 | 64.06% | 35.85 → 30.81 | 1.16x | 36.10 → 30.87 | 81,542 → 95,379 | 5.08 → 4.80 | +| DFlash | 128 | 1024, 1024, 1024, 1024 | 0% | 33.35 → 31.23 | 1.07x | 33.56 → 31.22 | 122,061 → 131,179 | 5.67 → 5.59 | +| DFlash2 | 128 | 1024, 1024, 1024, 1024 | 0% | 34.63 → 31.35 | 1.10x | 34.71 → 31.55 | 118,005 → 129,824 | 4.94 → 4.98 | + +The 512-anchor mixed-length case preserves all 1178 valid sampled blocks and +skips 870 invalid slots from the padded 2048-slot grid. The backbone therefore +processes 18,848 proposal tokens instead of 32,768. Context rows fall from 8192 +to 2944. Equal-length gains mainly reflect the sparse mask builder; they are +not evidence of removed context padding. + +Numerical checks use exact anchor/keep-mask comparisons and tolerance-based +loss/gradient comparisons. BF16 execution is not bitwise equivalent: one +DFlash2 128-anchor run's loss differed by about 2% after 25 optimizer updates, +despite passing initial loss/all-gradient checks. This experiment does not +establish convergence, checkpoint quality, or serving acceptance equivalence. +Evaluate those separately on a representative pretrained target and dataset. + +```bash +PYTHONPATH=. python scripts/benchmark_dflash_sequence_packing.py \ + --preset tiny --dtype float32 --correctness-only \ + --output artifacts/sequence-packing/dflash-tiny.json +PYTHONPATH=. python scripts/benchmark_dflash_sequence_packing.py \ + --preset medium --anchors 512 --warmup 5 --steps 20 \ + --output artifacts/sequence-packing/dflash-medium.json +``` + +Committed evidence includes the [512-anchor BF16 results](sequence-packing-results/dflash-family-512anchors-bf16.json), +[128-anchor BF16 results](sequence-packing-results/dflash-family-128anchors-bf16.json), +and [tiny-model FP32 comparisons](sequence-packing-results/dflash-family-tiny-fp32.json). +These preserve all timed samples, memory, settings, numerical comparisons and +measured source hashes. The [evidence inventory](sequence-packing-results/README.md) +also links the subsequent real-target online measurements. + +Exact source snapshots, source-verification files, capture/consumer service logs, +regression logs and the ad hoc `check_packed_fsdp.py` probe remain local under +`artifacts/sequence-packing/`; they are not committed with this report. The +two-rank command above describes that retained local probe, not a file shipped +in this repository. The committed runtime tests cover the supported model, +strategy and lifecycle behavior. + +Do not extrapolate the [EAGLE3 measurements](eagle3-sequence-packing.md) to +DFlash/DFlash2. Their proposal workload differs, and a producer or network +bottleneck can limit end-to-end online gains even when consumer compute improves. diff --git a/docs/sections/benchmarks/eagle3-sequence-packing.md b/docs/sections/benchmarks/eagle3-sequence-packing.md new file mode 100644 index 000000000..0695c1722 --- /dev/null +++ b/docs/sections/benchmarks/eagle3-sequence-packing.md @@ -0,0 +1,151 @@ +# EAGLE3 sequence packing: implementation and H200 measurements + +Measured on 2026-10-02, based on upstream +[`53398a8f01ae47175bee8459c5b5cca3848c8a7e`](https://github.com/sgl-project/SpecForge/tree/53398a8f01ae47175bee8459c5b5cca3848c8a7e), +with the local `codex/sequence-packing` changes. + +## Result and scope + +Packing improves training throughput when a microbatch contains substantially +different sequence lengths. In the final H200 run below, 47% and 64% padding +produced 1.50x and 1.96x median-step speedups. The mean-based throughput gains +were 1.44x and 1.61x, including observed scheduling outliers. Equal-length +samples showed only a small timing difference. This is not a universal +"biggest lever": batch size one has no inter-sample padding to remove. + +This report measures **offline text EAGLE3 with FlexAttention**. The switch +also supports text DFlash/DFlash2 and online server-capture consumers; see the +[current support contract](../basic_usage/training.md#sequence-packing). +The EAGLE3 measurements below do not establish DFlash/DFlash2 or end-to-end +online throughput. FA/USP, multimodal positions, compact teacher, trimmed loss +positions, and EAGLE3 LK objectives remain unsupported. + +## How it works + +```text +training.sequence_packing + → offline provider.build_packed_collator + → DataCollatorWithPacking: [B, max(L)] → [1, sum(L)] + → Eagle3TrainStrategy: shift teacher/input/mask within each document + → OnlineEagle3Model: per-document TTT shifts + original loss denominator + → LlamaFlexAttention: isolated causal prefixes + diagonal TTT cache suffixes +``` + +The collator keeps sample order and logical microbatch size, and emits document +lengths, reset positions, and the original padded loss denominator. The attention +mask prevents information flow between documents. RoPE uses the maximum +individual document length, preserving dynamic NTK behavior. Every TTT shift +zeros each document's tail rather than importing the next document's tokens. +The plain masked loss is rescaled by `sum(L) / (B * max(L))`, preserving the +existing padded objective and effective learning rate. The number of samples, +optimizer steps, accumulation steps, and checkpoint cursor do not change. + +The implementation packs only the existing logical microbatch. It does not +reorder examples or greedily combine additional microbatches. `data.max_length` +continues to limit individual documents; packed rows can exceed that length. + +## Reproducible training-step benchmark + +- One NVIDIA H200, Python 3.12.3, Torch 2.13.0+cu130, CUDA 13.0, + Transformers 5.12.1. +- One real EAGLE3 draft layer, hidden size 2048, intermediate size 8192, + 16 attention heads / 4 KV heads, target and draft vocabulary 32000. +- BF16, TTT length 7, microbatch 4, seed 1729, 25% prompt mask. +- Same features and initial weights for both paths. Frozen target-head + projection, production loss/backward and `BF16Optimizer` update are included. +- Five warmup steps (including compilation), then 20 synchronized wall-clock + measurements per mode. Compilation is excluded from steady-state timing. +- Synthetic features are already on the GPU. Capture, file I/O, bulk H2D, + distributed communication, serving, and training convergence are excluded. +- Useful tokens/s counts original unpadded input tokens once and uses the + **mean** step time. It is not derived from the median or multiplied by TTT. + Memory is peak allocated CUDA memory, including model/optimizer/input state. + +| Original sequence lengths | Padding | P50 step ms, padded → packed | P50 speedup | Mean step ms, padded → packed | Useful tokens/s, padded → packed | Peak GiB, padded → packed | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| 1024, 1024, 1024, 1024 | 0.00% | 80.03 → 76.61 | 1.04x | 87.82 → 76.68 | 46,643 → 53,419 | 13.09 → 9.17 | +| 512, 768, 1024, 2048 | 46.88% | 135.94 → 90.35 | 1.50x | 147.45 → 102.27 | 29,515 → 42,554 | 23.98 → 9.60 | +| 128, 256, 512, 2048 | 64.06% | 136.79 → 69.67 | 1.96x | 142.04 → 88.27 | 20,726 → 33,354 | 23.91 → 7.21 | + +The full samples retain outliers; for example CPU scheduling produced occasional +long steps. A separate 50-step repeat before the final FSDP metadata conversion +measured median speedups of 1.03x, 1.51x and 1.94x for the same three profiles, +with mean speedups of 1.04x, 1.57x and 1.87x. Treat the difference between median +and mean as part of the measurement, not guaranteed deployment throughput. + +The equal-length case also reduces peak memory. Packing changes teacher-table +layout to batch dimension one; the TTT adapter can retain contiguous slices +instead of making per-depth copies across padded batch strides. Thus observed +memory savings are not solely proportional to removed padding. + +An additional larger draft (hidden 4096, intermediate 14336, 32 heads / 8 KV +heads) with `[128, 256, 512, 2048]` measured 249.50 → 113.15 ms P50 (2.21x), +271.96 → 115.71 ms mean, and 33.03 → 12.97 GiB peak. This measurement predates +the final FSDP tuple/int metadata conversion; the numerical path is the same, +but use the table above for the final-source timing evidence. + +Commands from the repository root: + +```bash +PYTHONPATH=. python scripts/benchmark_sequence_packing.py \ + --preset tiny --dtype float32 --correctness-only \ + --output artifacts/sequence-packing/tiny-fp32.json +PYTHONPATH=. python scripts/benchmark_sequence_packing.py \ + --preset medium --warmup 5 --steps 20 \ + --output artifacts/sequence-packing/medium-bf16-current.json +``` + +## Correctness and integration evidence + +- Latest FP32 benchmark: three profiles passed, including documents shorter + than TTT. Scalar loss matched exactly; maximum trainable-gradient absolute + difference was approximately 4.1e-10. +- BF16 medium: loss, every depth's loss, and all trainable parameter gradients + passed the recorded tolerances. Maximum gradient relative L2 difference was + approximately 0.00658; BF16 equality is numerical, not bitwise. +- Seven production GPU tests passed, including FP32/BF16/dynamic-NTK parity, + cross-document isolation, mixed CPU/GPU fields, zero supervision, short + documents, and FSDP metadata compatibility. +- CPU regression: 162 tests and 523 subtests passed; the GPU lifecycle test was + skipped in this CPU run and executed separately. +- Full offline lifecycle passed: variable-length files, packed train/eval + loaders, 2 optimizer steps / 4 microsteps, partial eval batch, finite metrics, + both checkpoints, and original sample counters 4 and 8. +- This EAGLE3 lifecycle used single-rank FSDP, which selects NO_SHARD. The + subsequent [DFlash/online validation report](dflash-sequence-packing.md) + includes a two-rank FULL_SHARD comparison and live Mooncake coverage for + EAGLE3, DFlash and DFlash2. Long-run convergence/acceptance and speculative + serving performance remain outside these measurements. + +Committed evidence includes the [medium BF16 results](sequence-packing-results/eagle3-medium-bf16.json), +[tiny FP32 comparisons](sequence-packing-results/eagle3-tiny-fp32.json), and +[larger-draft results](sequence-packing-results/eagle3-large-bf16.json). +These JSON files contain settings, individual timings, numerical comparisons +and measured source hashes. See the [evidence inventory](sequence-packing-results/README.md) +for provenance and limitations. + +Additional validation logs, `source-verification.json` and exact measured source +snapshots remain local under `artifacts/sequence-packing/`; they are not committed +with this report. The local source verification found all seven measured source +ASTs equivalent to the final implementation; changes at that point were +formatting only. + +## Enabling packing + +```bash +specforge train \ + --config examples/configs/offline/colocated/qwen3-8b-eagle3-offline.yaml \ + training.batch_size=4 \ + training.attention_backend=flex_attention \ + training.sequence_packing=true +``` + +Choose the same batch size and gradient accumulation for the baseline and +packed run. Increasing batch size at the same time changes the training +schedule and invalidates a simple A/B comparison. + +Text DFlash/DFlash2 packing is also implemented, with document-aware context +attention and positions, boundary-safe labels and the same per-document anchor +sampling budget. Its [implementation and measurements](dflash-sequence-packing.md) +use a separate attention and proposal path; EAGLE3 timing gains do not predict +DFlash/DFlash2 gains. diff --git a/docs/sections/benchmarks/full-model-sequence-packing.md b/docs/sections/benchmarks/full-model-sequence-packing.md new file mode 100644 index 000000000..d006631e3 --- /dev/null +++ b/docs/sections/benchmarks/full-model-sequence-packing.md @@ -0,0 +1,181 @@ +# Full-size Qwen3-4B online training: longer sequence-packing comparison + +This extends the [32-step online measurement](online-sequence-packing.md) to +256 optimizer steps over 1,024 real ShareGPT conversations, with normal periodic +training diagnostics and checkpoints. It uses base +`53398a8f01ae47175bee8459c5b5cca3848c8a7e` plus the local `codex/sequence-packing` +changes. The previous online measurement already used full model dimensions; +this experiment increases run length, repeat count, and model-training evidence. + +## Results + +The complete-model training step improved **1.044× for DFlash** and **1.052× +for DFlash2**. End-to-end completion including both checkpoints improved +**1.037× and 1.033×**, respectively. The values below are medians of four runs +per mode; each run processes 1,024 conversations in 256 optimizer steps. + +| Model | Padded training step | Packed training step | Training-step ratio | Padded completion | Packed completion | Completion ratio | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| DFlash | 452.79 ms | 433.64 ms | **1.0442×** | 145.526 s | 140.378 s | **1.0367×** | +| DFlash2 | 426.98 ms | 405.76 ms | **1.0523×** | 137.320 s | 132.905 s | **1.0332×** | + +Training-step values are the existing host-wall diagnostics over the first +250 steps, including the detailed metric steps. Completion covers all 256 +steps, live target capture/transport/waiting, acknowledgements, and checkpoints +at steps 128 and 256. Both exclude startup and separate full-corpus warmup. + +| Model | Padded completion range | Packed completion range | Matched-pair completion ratios | +| --- | ---: | ---: | ---: | +| DFlash | 144.905–145.755 s | 138.331–140.510 s | 1.0328–1.0533× | +| DFlash2 | 136.638–137.591 s | 132.296–133.598 s | 1.0258–1.0377× | + +All chronological A/B pairs favored packing; the ranges do not overlap. +These are observed ranges from four runs per mode, not confidence intervals. +Checkpoint totals were approximately 18.46–19.79 s per run. Consumer fetch +waits also varied: 9.52–10.09 s padded versus 8.04–9.89 s packed for DFlash, +and 7.60–8.33 s versus 8.40–9.15 s for DFlash2. The full-run ratio therefore +includes normal pipeline and storage variation, not solely GPU computation. + +The pipeline timer ending at the final durable acknowledgement, including the +intermediate checkpoint but excluding the final one, measured +135.774→130.651 s (**1.0392×**) for DFlash and 127.556→122.898 s (**1.0379×**) +for DFlash2. This boundary differs from the previous 32-step experiment, which +had no intermediate checkpoint. + +All 16 measured runs and four warmups passed the independent full-model audit. +Every measured run had zero new compiler-counter activity. All 58 DFlash and +81 DFlash2 trainable tensors received exactly 256 AdamW updates, all five +draft layers had observed sampled weight changes, and all trainable elements +were finite. Final losses repeated exactly within each mode; they were +7.263832/7.263966 padded/packed for DFlash and 7.968346/7.968497 for DFlash2. +These short training comparisons do not establish equal final model quality +or time to convergence. + +## Complete model and workload + +- Actual Qwen3-4B target checkpoint: **4,022,468,096 saved parameters**, all 36 + decoder layers, hidden size 2560, vocabulary 151936. All layer indices were + verified from the checkpoint tensor headers. The target runs full prefill; + it is frozen, as required by speculative draft training. +- Complete repository `configs/qwen3-4b-dflash.json` draft: five layers, + intermediate size 9728, 32 attention heads, eight KV heads, head dimension + 128, block size 16, 512 sampled anchors per document. DFlash2 uses the same + dimensions with its convolution and selector modules enabled. +- Runtime parameter inventories confirm **537,427,200 trainable parameters in + 58 tensors for DFlash**, and **558,918,912 in 81 tensors for DFlash2**. The + frozen target embedding/head share 388,956,160 parameters, counted once. +- Fresh, identically seeded draft initialization for every arm. All trainable + draft parameters use the production backward and BF16 optimizer with FP32 + AdamW master parameters. Frozen target embeddings and the LM head are loaded + from the real target checkpoint. Objective chunk size 128 processes every + sampled block and the entire vocabulary. +- Two H200 GPUs on one host: target on GPU1, consumer on GPU0. Torch + 2.13.0+cu130, Transformers 5.12.1, isolated SGLang 0.5.18 with the repository's + capture patch. The reserved host has eight GPUs; the experiment uses two. +- BF16, batch four, accumulation one, learning rate 0.0001, optimizer warmup + ratio zero, gradient clipping 0.5, teacher metrics enabled. Detailed metrics + run every 50 steps; checkpoints run at steps 128 and 256. +- 1,024 real ShareGPT examples, deterministic seed 1729, production Qwen parser + and actual tokenizer, original assistant supervision masks, maximum length + 2048. There are **1,309,869 input tokens** and **1,026,019 supervised tokens**. + Batch-four context padding is **32.55%**. Invalid proposal removal reduces + slots from 523,408 to 454,930 (**13.08%**). +- Each mode receives a full untimed 256-step warmup before measurement. Two + ABBA blocks provide four measured runs per mode per architecture. Every arm + starts with identical model/optimizer initialization, prompt order and masks. + +## Timing and verification + +The producer and consumer use canonical SpecForge online builders and run +concurrently, with real target captures for every arm. Features travel through +Mooncake TCP host buffers. The producer has one worker, capture batch four and +an eight-reference high watermark. Radix caching is disabled, and fresh request +namespaces force full prefill. Data preparation, process/model startup and JIT +warmup are outside the main timer. + +The full completion timer starts at the first actual capture dispatch with an +empty channel and ends after `trainer.fit()`, including the intermediate and +final checkpoints and trainer cleanup. The pipeline timer ends at the final +optimizer update and durable sample acknowledgement with CUDA synchronization; +it includes the step-128 checkpoint, but not the final step-256 checkpoint. +These boundaries must not be compared directly with a compute-only benchmark. + +Existing trainer diagnostics also report training-thread wall time spent inside +`TrainerCore.train_step` for the five 50-step logging windows. This includes +strategy preparation and host-to-device transfers, forward, +objectives/diagnostics, backward and optimizer work; it excludes data +waiting, acknowledgements and checkpoint calls. It is a host-observed training +metric, not a CUDA-event kernel profile, and covers the first 250 of 256 steps. + +Every run records exact parameter inventory, optimizer identity coverage, +per-parameter AdamW step counters, a complete finiteness scan of trainable +weights, and sampled before/after weight values for each layer. Tied target +weights are deduplicated. The warmup alone observes first-step gradient presence; +measured steps have no gradient-observation hooks. Sampling and scans occur +outside timing. Sample hashes demonstrate observed updates, not a full-tensor +numerical comparison or an assertion that every element must change. + +The benchmark also checks identical prompt/request hashes, model initialization, +publication and consumption order, optimizer sample grouping, durable +acknowledgements and final checkpoint counters. Saved Dynamo counter deltas +identify any compilation during measured runs. Finite losses and updated layers +do not establish long-run convergence or serving quality. + +This experiment uses the complete model and real training path, with one +consumer rank. It does not exercise the YAML CLI orchestration, evaluation, +resume, multi-node scaling or final speculative-serving quality. The logger +collects metrics in memory rather than publishing to an external dashboard. + +## Why the full-model gain is modest + +Packing compacts context rows and valid proposal blocks through the draft +backbone. `_forward_draft_blocks` then scatters hidden states back to the +original batch/anchor/block layout before the objective. The normal LM-head +path projects that restored layout across the full vocabulary and applies the +loss mask after cross-entropy. DFlash2's existing fused head skips the clean +anchor slot of every block, but still processes the restored invalid blocks. +Neither objective path compacts all invalid proposal rows in this change. + +These are source facts in `specforge/algorithms/common/dflash_family_model.py` +and `specforge/core/dflash_head_triton.py`. Together with the corpus's 13.08% +removable proposal slots and full 151,936-token vocabulary, they explain why +removing context padding does not remove an equivalent fraction of all training +work. This is a structural explanation, not a measured kernel-time breakdown. +Packing the objective could be a separate optimization; it is not implemented +or benchmarked by this experiment. + +## Reproduction and artifacts + +The benchmark requires a capture-enabled SGLang server, Mooncake master and the +preprocessed public corpus described above. From the repository root, run: + +```bash +python scripts/benchmark_online_sequence_packing.py \ + --server-url http://127.0.0.1:31012 \ + --target-model /cluster-storage/models/Qwen3-4B \ + --draft-config configs/qwen3-4b-dflash.json \ + --prompts-path /scratch/specforge-packing-full-model-20261002/sharegpt-prompts-1024.jsonl \ + --algorithm both --steps 256 --warmup-steps 256 --repeats 2 \ + --log-interval 50 --save-interval 128 \ + --work-dir /scratch/specforge-packing-full-model-20261002/long_v2 \ + --output /scratch/specforge-packing-full-model-20261002/long_v2.json +``` + +Committed evidence includes the [per-run timing summary](sequence-packing-results/full-model-analysis.json), +[independent full-model audit](sequence-packing-results/full-model-audit.json), +and [target tensor inventory](sequence-packing-results/target-parameter-inventory.json). +The summaries preserve every measured run's timing values, checks, parameter +counts and optimizer-update evidence. The [evidence inventory](sequence-packing-results/README.md) +describes their provenance and the raw configuration's unused CLI defaults. + +The 9.65 MB raw `long_v2.json`, its log, exact source archive and hashes, detailed +data manifest and preprocessing provenance, runtime/service launch records and +cleanup receipts remain local under `artifacts/sequence-packing/full-model/`. +These local files, including `run_benchmark.py`, are not committed with this +report. The raw result SHA256 is +`f8895e415d1baaef7ddaa74a7f8e322d83d13e008d2f7606b0378fb0b4861179`. +The 636-file measured source archive was verified against the remote snapshot; +the final results documentation was written afterward. Raw conversation text +stays on the devbox, and checkpoints were validated before deletion between +runs to bound disk use. All benchmark-owned services were stopped successfully; +the existing devbox reservation was retained. diff --git a/docs/sections/benchmarks/online-sequence-packing.md b/docs/sections/benchmarks/online-sequence-packing.md new file mode 100644 index 000000000..3d487b2a6 --- /dev/null +++ b/docs/sections/benchmarks/online-sequence-packing.md @@ -0,0 +1,147 @@ +# Online sequence packing: real Qwen3-4B pipeline benchmark + +This benchmark measures actual pretrained-target capture and concurrent draft +training, extending the [consumer-compute measurements](dflash-sequence-packing.md). +It is based on upstream `53398a8f01ae47175bee8459c5b5cca3848c8a7e` plus the local +`codex/sequence-packing` changes, tested on 2026-10-02. + +A subsequent [256-step full-model comparison](full-model-sequence-packing.md) +uses 1,024 real conversations, four runs per mode, periodic diagnostics and +checkpoints, and verifies every trainable parameter's optimizer participation. +It measures 1.044×/1.052× training-step improvements and 1.037×/1.033× full +completion improvements for DFlash/DFlash2. + +## Measured results + +On this workload, packing improved online pipeline throughput by **4.1% for +DFlash** and **4.5% for DFlash2**. Each time below is the median of two measured +runs processing the same 128 conversations in 32 optimizer steps. + +| Model | Padded pipeline | Packed pipeline | Throughput speedup | Useful input tokens/s, padded → packed | +| --- | ---: | ---: | ---: | ---: | +| DFlash | 15.3356 s | 14.7329 s | **1.0409× (+4.09%)** | 10,811 → 11,253 | +| DFlash2 | 14.3867 s | 13.7686 s | **1.0449× (+4.49%)** | 11,524 → 12,041 | + +Both packed runs were faster than both padded runs for each model. DFlash +ranges were 15.2822–15.3889 s padded and 14.6678–14.7980 s packed; DFlash2 +ranges were 14.3754–14.3980 s padded and 13.7193–13.8178 s packed. These are +short repeated measurements, not a confidence interval or a whole-epoch result. + +Including the final checkpoint changes the comparison: + +| Model | Padded completion with checkpoint | Packed completion with checkpoint | Ratio | +| --- | ---: | ---: | ---: | +| DFlash | 25.2150 s | 24.9084 s | 1.0123× | +| DFlash2 | 25.1072 s | 24.0008 s | 1.0461× | + +Each checkpoint took approximately 9.87–10.88 s. Storage timing varied between +arms: about 0.488 s of DFlash2's 1.106 s total-completion gap came from checkpoint +I/O. The pipeline measurement is therefore the cleaner estimate of packing's +effect. Checkpoint frequency will affect the realized full-run improvement. + +All eight measured runs passed sample-order, HTTP-payload, initialization, +optimizer-grouping, durable-acknowledgement, and checkpoint-counter checks. +Every measured run had **zero new Dynamo compilations** and captured 30 of its +32 batches after its first optimizer acknowledgement, confirming concurrent +online feature production. Final losses were finite and repeatable within each +arm; this short run does not establish long-run convergence or serving quality. + +The earlier 1.39×/1.47× consumer-compute results used synthetic lengths with +64% context padding and 42.5% invalid proposal slots. This real corpus has +33.28% context padding and only 13.08% removable proposal slots, and the online +timer includes target capture and transport. The synthetic speedups should not +be used as an estimate of end-to-end training gains. + +## Workload and timing boundaries + +- Two NVIDIA H200 GPUs on one host: a patched SGLang 0.5.18 Qwen3-4B target on + GPU1 and a single-rank FSDP consumer on GPU0. The draft is freshly initialized; + the target, frozen embeddings, and frozen LM head use the actual pretrained + weights from `/cluster-storage/models/Qwen3-4B`. +- The repository's `configs/qwen3-4b-dflash.json`: five draft layers, hidden size + 2560, intermediate size 9728, 32 attention heads / 8 KV heads, head dimension + 128, vocabulary 151936, block size 16, and target capture layers + `[1, 9, 17, 25, 33]`. DFlash2 uses the same base dimensions with its convolution + and selector modules enabled. +- BF16, batch four, accumulation one, 512 anchors per document, objective chunks + of 128 blocks, learning rate 0.0001, gradient clipping 0.5, teacher metrics + enabled, and log interval 50. With 32 steps per arm, no periodic detailed-metric + step runs. Both arms use identical settings apart from sequence packing. +- 128 real ShareGPT conversations, deterministically shuffled with seed 1729 + and prepared with SpecForge's Qwen parser and the target tokenizer. The + original assistant supervision masks are retained. Maximum length is 2048; + there are 165,788 input tokens and 131,810 supervised tokens. +- Lengths range from 35 to 2048, mean 1295.22 and median 1536. Batch-four context + padding is **33.28%**. With the original per-batch anchor cap, proposal slots + fall from 65,388 to 56,834, removing **13.08%** invalid slots. This is a natural + corpus slice, not the previous synthetic 64%-padding length pattern. +- Mooncake uses TCP and host tensors, with pinned receive buffers. One canonical + producer worker captures batches of four, concurrently with the canonical + online consumer, with an eight-reference high watermark. One active capture + batch may overshoot that watermark. Target capture and feature supply are + rerun for every arm; no feature cache is replayed. + +The primary timer starts at the first capture HTTP dispatch with an empty +channel, and ends after the last optimizer update, synchronous durable sample +acknowledgement, and CUDA synchronization. It includes pipeline fill/drain, +target prefill/capture, feature transport/fetch, data waiting, collation, model +forward/backward, optimizer work, and acknowledgement. It excludes data +preparation, model/server startup, and JIT warmup. + +Each mode first performs an untimed replay of the complete ordered corpus. +Measured runs then use A → B → B → A order, where A is padded and B is packed. +Every run starts from the same seeded draft initialization and resets optimizer +state. The target adapter's fresh request namespaces force full prefill, and +the benchmark server also disables radix caching. Dynamo counters are saved +for every run to verify that warmed timing did not include new graph compilation. + +Canonical final checkpoints are saved and validated in every run. Their cost, +and total completion time including checkpoint/cleanup, are reported separately +from the primary pipeline timer. Checkpoint files are deleted after validation +so repeated benchmarking does not retain many copies of the same initial run. +This is a warmed online training pipeline comparison, not complete cold CLI +startup, multi-node scaling, RDMA, convergence, or speculative-serving performance. + +## Reproduction and evidence + +The benchmark requires an already running capture-enabled SGLang server and +Mooncake master. The prompts are prepared once from the cached public +`anon8231489123/ShareGPT_Vicuna_unfiltered` dataset snapshot +`192ab2185289094fc556ec8ce5ce1e8e587154ca`; preparation is outside the timer. + +```bash +CUDA_VISIBLE_DEVICES=0 \ +MOONCAKE_MASTER_SERVER_ADDR=127.0.0.1:50212 \ +MOONCAKE_METADATA_SERVER=http://127.0.0.1:8112/metadata \ +MOONCAKE_LOCAL_HOSTNAME=127.0.0.1 \ +MOONCAKE_PROTOCOL=tcp \ +TORCH_LOGS=recompiles \ +PYTHONPATH=/scratch/sglang-sequence-packing-online-20261002:. \ +python scripts/benchmark_online_sequence_packing.py \ + --server-url http://127.0.0.1:31012 \ + --target-model /cluster-storage/models/Qwen3-4B \ + --draft-config configs/qwen3-4b-dflash.json \ + --prompts-path /scratch/specforge-packing-e2e-20261002/sharegpt-prompts.jsonl \ + --algorithm both --warmup-steps 32 --steps 32 --repeats 1 \ + --work-dir /scratch/specforge-packing-e2e-20261002/full_v1 \ + --output /scratch/specforge-packing-e2e-20261002/full_v1.json +``` + +The [committed 32-step benchmark summary](sequence-packing-results/online-32step.json) +preserves every run's timing samples across eight measured runs and four warmups, +plus settings, prompt/request hashes and lifecycle checks. Detailed per-capture +events remain in the local raw report. See the [evidence inventory](sequence-packing-results/README.md) +for the longer full-model comparison and provenance. The original raw report's +SHA256 is +`d0284b6ab87ad26bba3c6624c4d4b203cb4f18fef08609476729d650b55f9be4`. + +The detailed prompt manifest, preprocessing/service-start scripts, exact source +snapshots, service logs and cleanup receipts remain local under +`artifacts/sequence-packing/e2e/`; they are not committed with this report. +Original conversation text was not copied into the local evidence directory. + +The benchmark verifies actual HTTP payload hashes, publication order, consumed +sample order and optimizer grouping, all durable acknowledgements, finite final +loss, producer completion, and the final checkpoint's step and sample counters. +The benchmark-specific CPU tests and package architecture checks passed +21 tests with eight subtests before the GPU run. diff --git a/docs/sections/benchmarks/sequence-packing-results/README.md b/docs/sections/benchmarks/sequence-packing-results/README.md new file mode 100644 index 000000000..a5d5fe7aa --- /dev/null +++ b/docs/sections/benchmarks/sequence-packing-results/README.md @@ -0,0 +1,78 @@ +# Sequence-packing benchmark evidence + +These files support the [EAGLE3](../eagle3-sequence-packing.md), +[DFlash/DFlash2](../dflash-sequence-packing.md), +[32-step online](../online-sequence-packing.md) and +[256-step full-model](../full-model-sequence-packing.md) reports measured on +2026-10-02. The implementation was based on upstream +`53398a8f01ae47175bee8459c5b5cca3848c8a7e` plus the sequence-packing changes. +Each report defines its workload and timing boundary; the results are not +interchangeable estimates of end-to-end gains. + +## Committed files + +| File | Evidence | +| --- | --- | +| [full-model-analysis.json](full-model-analysis.json) | Every measured run's timing values and medians for the 256-step comparison. | +| [full-model-audit.json](full-model-audit.json) | Independent checks for all 16 measured runs and four warmups: model dimensions, optimizer participation, updates, finiteness, compilation and pipeline ordering. | +| [target-parameter-inventory.json](target-parameter-inventory.json) | Qwen3-4B checkpoint tensor names, shapes and counts; verifies all 36 target layers and 4,022,468,096 saved parameters. | +| [online-32step.json](online-32step.json) | Derived summary preserving every run's timing samples, settings and lifecycle checks from the shorter online experiment. | +| [eagle3-medium-bf16.json](eagle3-medium-bf16.json) | Final-source medium EAGLE3 compute timings, memory, numerical checks and source hashes. | +| [eagle3-large-bf16.json](eagle3-large-bf16.json) | Larger EAGLE3 compute case, measured before the final FSDP metadata conversion. | +| [eagle3-tiny-fp32.json](eagle3-tiny-fp32.json) | Derived FP32 correctness summary, including the number of gradient tensors and maximum differences. | +| [dflash-family-128anchors-bf16.json](dflash-family-128anchors-bf16.json) | Final optimized DFlash/DFlash2 compute measurements with 128 anchors per document. | +| [dflash-family-512anchors-bf16.json](dflash-family-512anchors-bf16.json) | Final optimized compute measurements with 512 anchors per document. | +| [dflash-family-tiny-fp32.json](dflash-family-tiny-fp32.json) | Derived FP32 correctness summary for DFlash/DFlash2, including exact sampled-anchor checks. | + +The two tiny FP32 summaries replace per-parameter gradient lists with tensor +counts, the all-pass flag and maximum absolute/relative differences. Their other +fields are unchanged, and each includes the original report SHA256. The 32-step +summary removes detailed per-capture event lists while preserving individual +run timings, including warmups, and records the original report SHA256. All +other JSON files preserve their corresponding local evidence; only the target +inventory's missing final newline was normalized by the repository hooks. + +## Configuration and interpretation + +The online benchmark's raw CLI settings retain `draft_layers=2`, a default used +only when no draft config is supplied. Both online experiments supplied +`configs/qwen3-4b-dflash.json`; the resolved config and runtime model inventory +confirm **five draft layers**. Likewise, supplying `prompts_path` bypasses the +synthetic `lengths` and `prompt_fraction` defaults. These experiments use real +ShareGPT tokens and their actual assistant supervision masks. + +The target checkpoint has 36 layers and is frozen. All trainable draft parameters +are updated: 537,427,200 parameters for DFlash and 558,918,912 for DFlash2. The +256-step evidence checks each trainable tensor's optimizer step count and scans +all trainable elements for finiteness. Layer-update hashes sample weight values; +they do not assert that every element changed or that packed and padded training +trajectories are numerically identical. + +The full-model training-step diagnostics cover the first 250 of 256 steps and +include periodic metrics. Pipeline timing includes the step-128 checkpoint; +completion timing includes both step-128 and step-256 checkpoints. Startup, +data preparation and separate corpus warmup are excluded. The tests do not +establish convergence, final model quality or speculative-serving throughput. + +## Evidence retained locally + +The original 256-step raw report, about 9.65 MB, is retained locally as +`artifacts/sequence-packing/full-model/long_v2.json`; it is **not committed**. +Its SHA256 is +`f8895e415d1baaef7ddaa74a7f8e322d83d13e008d2f7606b0378fb0b4861179`. +The original 32-step raw report is retained locally as +`artifacts/sequence-packing/e2e/full_v1.json`; its SHA256 is +`d0284b6ab87ad26bba3c6624c4d4b203cb4f18fef08609476729d650b55f9be4`. + +Detailed source snapshots, source-verification records, dataset manifests, +preparation and service-launch helpers, service logs, ad hoc distributed probes, +reservation metadata and cleanup receipts also remain local. They are not part +of this directory or promised as repository downloads. Original conversation +text was not copied into these artifacts. The public dataset revision is +`anon8231489123/ShareGPT_Vicuna_unfiltered@192ab2185289094fc556ec8ce5ce1e8e587154ca`; +the 1,024-prompt tokenized file SHA256 is +`f90478e0d3e32f0d3c45c515381b9bc42951cdd6931830ac1a7380f4459f24b1`. + +The full-model measured source manifest matches all 17 changed/new production +files in the implementation at packaging time. Benchmark documentation and +public evidence packaging were finalized after the measurements. diff --git a/docs/sections/benchmarks/sequence-packing-results/dflash-family-128anchors-bf16.json b/docs/sections/benchmarks/sequence-packing-results/dflash-family-128anchors-bf16.json new file mode 100644 index 000000000..630e15b7c --- /dev/null +++ b/docs/sections/benchmarks/sequence-packing-results/dflash-family-128anchors-bf16.json @@ -0,0 +1,1070 @@ +{ + "timestamp_utc": "2026-10-02T04:43:38.703421+00:00", + "source": { + "files_sha256": { + "scripts/benchmark_dflash_sequence_packing.py": "0372ddf408fddc99c95138ab448025e37600c51004bceb6fd51d0432afb82e46", + "specforge/benchmarks/benchmark_dflash_sequence_packing.py": "00b9ff628d145d3f986eeb8c43f62671183a197f4c4575faf3f432f16ef0641c", + "specforge/algorithms/common/dflash_family_model.py": "86be39addbc5c50036f37c5d5282c53aaa8f35ab5ae1f8cfabf33078494e1421", + "specforge/algorithms/common/hidden_states_data.py": "fa09c5e5037c31b1165c2f8d1774e2cd4a40fde8b08f1f6deddf87f4f4af7f24", + "specforge/modeling/draft/dflash.py": "97af112a6ecf66d4b397aae77f0f97c08a49759c6305a4c70577a4f78b45ca90", + "specforge/modeling/draft/dflash2.py": "819c6b3d8d6d8ffc31a843ed40d539ddf635f58b148a01c8c43d0bff1dd02b09", + "specforge/modeling/packed_dflash.py": "d763e0ad3c609602323b9e3413968860cc680a8be8c986cc7612a956556ab96e", + "specforge/training/strategies/base.py": "276932ad331e95d5b91eddc67cf3acb0b5447ee5973f9bbc757e54093a82939f" + }, + "head": "unavailable (copied snapshot is identified by file hashes)" + }, + "settings": { + "preset": "medium", + "algorithm": "both", + "hidden_size": 2048, + "intermediate_size": 8192, + "layers": 2, + "heads": 16, + "kv_heads": 4, + "vocab_size": 32000, + "block_size": 16, + "anchors": 128, + "capture_layers": 2, + "conv_group_size": 32, + "conv_kernel_size": 4, + "selector_rank": 16, + "selector_top_k": 16, + "lengths": [ + [ + 1024, + 1024, + 1024, + 1024 + ], + [ + 128, + 256, + 512, + 2048 + ] + ], + "dtype": "bfloat16", + "sliding_window": null, + "warmup": 5, + "steps": 20, + "seed": 1729, + "learning_rate": 0.0001, + "objective_chunk_blocks": 128, + "correctness_only": false, + "skip_correctness": false, + "atol": 0.002, + "rtol": 0.02, + "output": "artifacts/sequence-packing/dflash-family-medium-bf16-optimized.json" + }, + "environment": { + "torch": "2.13.0+cu130", + "cuda": "13.0", + "gpu": "NVIDIA H200", + "cuda_visible_devices": "0" + }, + "scope": "Single GPU; synthetic frozen target components and captured features; GPU-resident hidden features plus CPU integer features; production strategy forward/backward/BF16Optimizer; excludes capture, hidden-feature I/O/H2D, distributed training and serving.", + "cases": [ + { + "algorithm": "dflash", + "lengths": [ + 1024, + 1024, + 1024, + 1024 + ], + "padding_fraction": 0.0, + "useful_tokens": 4096, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 3.528594970703125e-05, + "relative_l2_diff": 3.344875867136468e-06 + }, + "loss_terms": { + "pass": true, + "max_abs_diff": 0.1171875, + "relative_l2_diff": 3.27596571374284e-06 + }, + "loss_values": [ + 10.54925537109375, + 10.549220085144043 + ], + "identical_sampled_anchors": true, + "sampled_anchor_count": 512, + "gradients": { + "draft_model.layers.0.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.004035275282337427 + }, + "draft_model.layers.0.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.00501776531413531 + }, + "draft_model.layers.0.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.004276743031545815 + }, + "draft_model.layers.0.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0030888238226688147 + }, + "draft_model.layers.0.self_attn.q_norm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.0037907497753986714 + }, + "draft_model.layers.0.self_attn.k_norm.weight": { + "pass": true, + "max_abs_diff": 1.1920928955078125e-06, + "relative_l2_diff": 0.004571789916409457 + }, + "draft_model.layers.0.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.002867089280638763 + }, + "draft_model.layers.0.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.0027648728928907685 + }, + "draft_model.layers.0.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.002450745866595975 + }, + "draft_model.layers.0.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 7.152557373046875e-07, + "relative_l2_diff": 0.004297563642974598 + }, + "draft_model.layers.0.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0031199611716563143 + }, + "draft_model.layers.1.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0029851964017069575 + }, + "draft_model.layers.1.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.004035853911087431 + }, + "draft_model.layers.1.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0033796309921081966 + }, + "draft_model.layers.1.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0024074416791069384 + }, + "draft_model.layers.1.self_attn.q_norm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.0035275305395718825 + }, + "draft_model.layers.1.self_attn.k_norm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.002868279839387132 + }, + "draft_model.layers.1.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0021278734955892174 + }, + "draft_model.layers.1.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0020574586175487694 + }, + "draft_model.layers.1.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0019429627771338416 + }, + "draft_model.layers.1.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 4.76837158203125e-07, + "relative_l2_diff": 0.0030072922824688204 + }, + "draft_model.layers.1.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0026339067731564235 + }, + "draft_model.norm.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0014766025487033196 + }, + "draft_model.fc.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.005959122214185738 + }, + "draft_model.hidden_norm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.0060941715028472844 + } + }, + "pass": true + }, + "padded": { + "mean_step_ms": 33.55688191950321, + "p50_step_ms": 33.35478808730841, + "useful_tokens_per_second": 122061.40039547031, + "p50_useful_tokens_per_second": 122800.96006841488, + "peak_allocated_gib": 5.674890518188477, + "baseline_allocated_gib": 2.144904136657715, + "warmup_seconds_including_compile": 0.19184968434274197, + "step_ms": [ + 33.65137241780758, + 33.180927857756615, + 33.23202021420002, + 33.49055536091328, + 33.38956832885742, + 33.32914970815182, + 33.664412796497345, + 33.452145755290985, + 33.31155702471733, + 36.49650141596794, + 33.26563164591789, + 33.242642879486084, + 33.677954226732254, + 33.29920209944248, + 33.28862413764, + 33.3385169506073, + 33.35254080593586, + 33.37463736534119, + 33.742642030119896, + 33.357035368680954 + ], + "final_loss": 8.091357231140137 + }, + "packed": { + "mean_step_ms": 31.22454872354865, + "p50_step_ms": 31.23283013701439, + "useful_tokens_per_second": 131178.83740337024, + "p50_useful_tokens_per_second": 131144.05521470125, + "peak_allocated_gib": 5.594754219055176, + "baseline_allocated_gib": 2.035337448120117, + "warmup_seconds_including_compile": 0.15607250295579433, + "step_ms": [ + 31.093496829271317, + 31.08426183462143, + 31.171666458249092, + 31.09334409236908, + 31.18988499045372, + 31.12914226949215, + 31.23560920357704, + 31.254412606358528, + 31.43681026995182, + 31.332312151789665, + 31.14699199795723, + 31.310075893998146, + 31.280379742383957, + 31.256863847374916, + 31.24155104160309, + 31.199980527162552, + 31.363260000944138, + 31.230051070451736, + 31.19666315615177, + 31.244216486811638 + ], + "final_loss": 8.090627670288086 + }, + "p50_speedup": 1.0679399830558185, + "mean_speedup": 1.0746954973346206 + }, + { + "algorithm": "dflash", + "lengths": [ + 128, + 256, + 512, + 2048 + ], + "padding_fraction": 0.640625, + "useful_tokens": 2944, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 7.43865966796875e-05, + "relative_l2_diff": 7.076170071937434e-06 + }, + "loss_terms": { + "pass": true, + "max_abs_diff": 0.232421875, + "relative_l2_diff": 7.069888761538474e-06 + }, + "loss_values": [ + 10.51226806640625, + 10.51219367980957 + ], + "identical_sampled_anchors": true, + "sampled_anchor_count": 478, + "gradients": { + "draft_model.layers.0.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0017025760377892554 + }, + "draft_model.layers.0.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0024859584683051485 + }, + "draft_model.layers.0.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0020038922260345268 + }, + "draft_model.layers.0.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0013343185538530963 + }, + "draft_model.layers.0.self_attn.q_norm.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0017520098353950937 + }, + "draft_model.layers.0.self_attn.k_norm.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0018270023601644342 + }, + "draft_model.layers.0.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.002008949811204485 + }, + "draft_model.layers.0.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.0020283964646443113 + }, + "draft_model.layers.0.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.001880820308327072 + }, + "draft_model.layers.0.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0018999089901229125 + }, + "draft_model.layers.0.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0018916088807969396 + }, + "draft_model.layers.1.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0016530553335780654 + }, + "draft_model.layers.1.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0024914118832995687 + }, + "draft_model.layers.1.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0019648644067848447 + }, + "draft_model.layers.1.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0013547066257487894 + }, + "draft_model.layers.1.self_attn.q_norm.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0021836037110825788 + }, + "draft_model.layers.1.self_attn.k_norm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.0017769949609473883 + }, + "draft_model.layers.1.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.001961136128200421 + }, + "draft_model.layers.1.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.0018879952579109367 + }, + "draft_model.layers.1.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0019155058101548376 + }, + "draft_model.layers.1.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.001667171626780272 + }, + "draft_model.layers.1.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0018729544534959166 + }, + "draft_model.norm.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0018962999532875571 + }, + "draft_model.fc.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.003503925527484017 + }, + "draft_model.hidden_norm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.0033592604071104744 + } + }, + "pass": true + }, + "padded": { + "mean_step_ms": 35.424255300313234, + "p50_step_ms": 35.38447059690952, + "useful_tokens_per_second": 83106.8987912914, + "p50_useful_tokens_per_second": 83200.34044135534, + "peak_allocated_gib": 5.810401916503906, + "baseline_allocated_gib": 2.178708076477051, + "warmup_seconds_including_compile": 0.17471044324338436, + "step_ms": [ + 35.73337569832802, + 35.41872464120388, + 35.288020968437195, + 35.35628691315651, + 35.33848002552986, + 35.40989197790623, + 35.713041201233864, + 35.328615456819534, + 35.29147431254387, + 35.351455211639404, + 35.28880886733532, + 35.373954102396965, + 35.680223256349564, + 35.31438298523426, + 35.409245640039444, + 35.39498709142208, + 35.41380725800991, + 35.35588085651398, + 35.61149537563324, + 35.41295416653156 + ], + "final_loss": 4.466716766357422 + }, + "packed": { + "mean_step_ms": 30.4952384904027, + "p50_step_ms": 30.499404296278954, + "useful_tokens_per_second": 96539.66145982168, + "p50_useful_tokens_per_second": 96526.4754485444, + "peak_allocated_gib": 5.433377742767334, + "baseline_allocated_gib": 2.026548385620117, + "warmup_seconds_including_compile": 0.1535922773182392, + "step_ms": [ + 30.398793518543243, + 30.40284849703312, + 30.38313053548336, + 30.465159565210342, + 30.46681173145771, + 30.445296317338943, + 30.444307252764702, + 30.531708151102066, + 30.521946027874947, + 30.687423422932625, + 30.53887188434601, + 30.485061928629875, + 30.561374500393867, + 30.53414449095726, + 30.521035194396973, + 30.50977550446987, + 30.489033088088036, + 30.482830479741096, + 30.513176694512367, + 30.522041022777557 + ], + "final_loss": 4.467428207397461 + }, + "p50_speedup": 1.1601692365259266, + "mean_speedup": 1.161632341765806 + }, + { + "algorithm": "dflash2", + "lengths": [ + 1024, + 1024, + 1024, + 1024 + ], + "padding_fraction": 0.0, + "useful_tokens": 4096, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "loss_terms": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "loss_values": [ + 10.544198989868164, + 10.544198989868164 + ], + "identical_sampled_anchors": true, + "sampled_anchor_count": 512, + "gradients": { + "draft_model.layers.0.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.0.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.0.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.0.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.0.self_attn.q_norm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.0.self_attn.k_norm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.0.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.0.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.0.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.0.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.0.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.0.attention_conv.base_kernel": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.0.attention_conv.kernel_projection.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.0.mlp_conv.base_kernel": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.0.mlp_conv.kernel_projection.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.self_attn.q_norm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.self_attn.k_norm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.attention_conv.base_kernel": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.attention_conv.kernel_projection.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.mlp_conv.base_kernel": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.layers.1.mlp_conv.kernel_projection.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.norm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.fc.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.hidden_norm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.candidate_selector.predecessor_codebook": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.candidate_selector.successor_codebook": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.candidate_selector.hidden_projection.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + } + }, + "pass": true + }, + "padded": { + "mean_step_ms": 34.710309375077486, + "p50_step_ms": 34.62953958660364, + "useful_tokens_per_second": 118005.28643345911, + "p50_useful_tokens_per_second": 118280.52145355487, + "peak_allocated_gib": 4.944029808044434, + "baseline_allocated_gib": 2.10475492477417, + "warmup_seconds_including_compile": 0.17944882810115814, + "step_ms": [ + 35.65160930156708, + 35.805532708764076, + 35.28404049575329, + 34.837789833545685, + 34.69961881637573, + 34.849222749471664, + 34.734124317765236, + 34.90187227725983, + 34.501735121011734, + 34.55946035683155, + 34.328700974583626, + 34.35760922729969, + 34.44538451731205, + 34.72575172781944, + 34.35399569571018, + 34.364184364676476, + 34.35469605028629, + 34.3801137059927, + 34.369586035609245, + 34.701159223914146 + ], + "final_loss": 8.115375518798828 + }, + "packed": { + "mean_step_ms": 31.55035898089409, + "p50_step_ms": 31.34923055768013, + "useful_tokens_per_second": 129824.19637381652, + "p50_useful_tokens_per_second": 130657.11429387974, + "peak_allocated_gib": 4.976554870605469, + "baseline_allocated_gib": 2.105029582977295, + "warmup_seconds_including_compile": 0.1659725233912468, + "step_ms": [ + 32.492758706212044, + 32.25332498550415, + 32.10993483662605, + 31.620940193533897, + 31.54403530061245, + 31.51683136820793, + 31.574249267578125, + 31.407151371240616, + 31.366491690278053, + 31.331969425082207, + 31.20245411992073, + 31.18046373128891, + 31.29424713551998, + 31.234296038746834, + 31.18448704481125, + 31.143737956881523, + 31.179973855614662, + 31.152334064245224, + 32.96910971403122, + 31.248388811945915 + ], + "final_loss": 8.115375518798828 + }, + "p50_speedup": 1.1046376249295178, + "mean_speedup": 1.1001557667250939 + }, + { + "algorithm": "dflash2", + "lengths": [ + 128, + 256, + 512, + 2048 + ], + "padding_fraction": 0.640625, + "useful_tokens": 2944, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 2.09808349609375e-05, + "relative_l2_diff": 1.9918210395875337e-06 + }, + "loss_terms": { + "pass": true, + "max_abs_diff": 0.0625, + "relative_l2_diff": 1.897349817350434e-06 + }, + "loss_values": [ + 10.533493995666504, + 10.533514976501465 + ], + "identical_sampled_anchors": true, + "sampled_anchor_count": 478, + "gradients": { + "draft_model.layers.0.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0016685134832223076 + }, + "draft_model.layers.0.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.002552931532002346 + }, + "draft_model.layers.0.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0019390143575875442 + }, + "draft_model.layers.0.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.00134887764556812 + }, + "draft_model.layers.0.self_attn.q_norm.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.002203571142238923 + }, + "draft_model.layers.0.self_attn.k_norm.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0022542591514036923 + }, + "draft_model.layers.0.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.00190230944130773 + }, + "draft_model.layers.0.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.0018909972477091912 + }, + "draft_model.layers.0.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.001779518924017613 + }, + "draft_model.layers.0.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0014711605963159716 + }, + "draft_model.layers.0.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0019131030566731015 + }, + "draft_model.layers.0.attention_conv.base_kernel": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0014278181873818977 + }, + "draft_model.layers.0.attention_conv.kernel_projection.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.0015595749262699054 + }, + "draft_model.layers.0.mlp_conv.base_kernel": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.001959983152873123 + }, + "draft_model.layers.0.mlp_conv.kernel_projection.weight": { + "pass": true, + "max_abs_diff": 0.0001220703125, + "relative_l2_diff": 0.0021964532089766165 + }, + "draft_model.layers.1.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0016678913604391784 + }, + "draft_model.layers.1.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.002632562918780912 + }, + "draft_model.layers.1.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.001934458644005553 + }, + "draft_model.layers.1.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0013250733268872269 + }, + "draft_model.layers.1.self_attn.q_norm.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0022117848109090665 + }, + "draft_model.layers.1.self_attn.k_norm.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0024255432314647576 + }, + "draft_model.layers.1.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0019073348519206132 + }, + "draft_model.layers.1.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.0018211302868553522 + }, + "draft_model.layers.1.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.001875822437668495 + }, + "draft_model.layers.1.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0014387342277127767 + }, + "draft_model.layers.1.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0019749076950236724 + }, + "draft_model.layers.1.attention_conv.base_kernel": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0014293144980451072 + }, + "draft_model.layers.1.attention_conv.kernel_projection.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0014537681874349571 + }, + "draft_model.layers.1.mlp_conv.base_kernel": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0020733078021590995 + }, + "draft_model.layers.1.mlp_conv.kernel_projection.weight": { + "pass": true, + "max_abs_diff": 6.103515625e-05, + "relative_l2_diff": 0.002429591455188707 + }, + "draft_model.norm.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0014566753276068754 + }, + "draft_model.fc.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.003518625850596219 + }, + "draft_model.hidden_norm.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0033256846885824734 + }, + "draft_model.candidate_selector.predecessor_codebook": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.candidate_selector.successor_codebook": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.candidate_selector.hidden_projection.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + } + }, + "pass": true + }, + "padded": { + "mean_step_ms": 36.104287300258875, + "p50_step_ms": 35.84901336580515, + "useful_tokens_per_second": 81541.56251628573, + "p50_useful_tokens_per_second": 82122.20431171355, + "peak_allocated_gib": 5.079184532165527, + "baseline_allocated_gib": 2.13600492477417, + "warmup_seconds_including_compile": 0.18093057349324226, + "step_ms": [ + 36.35081835091114, + 36.468904465436935, + 36.12946905195713, + 36.11389920115471, + 36.354729905724525, + 35.956135019659996, + 35.82063689827919, + 35.81521473824978, + 35.775020718574524, + 35.78690066933632, + 36.06811165809631, + 38.790395483374596, + 35.74402630329132, + 35.74172966182232, + 35.803671926259995, + 35.80854274332523, + 36.143653094768524, + 35.87738983333111, + 35.764576867222786, + 35.771919414401054 + ], + "final_loss": 3.7936742305755615 + }, + "packed": { + "mean_step_ms": 30.86625747382641, + "p50_step_ms": 30.81074357032776, + "useful_tokens_per_second": 95379.23418465673, + "p50_useful_tokens_per_second": 95551.08572047624, + "peak_allocated_gib": 4.800002574920654, + "baseline_allocated_gib": 2.0974764823913574, + "warmup_seconds_including_compile": 0.16166752576828003, + "step_ms": [ + 31.302351504564285, + 31.165508553385735, + 31.053470447659492, + 30.985631048679352, + 30.950207263231277, + 30.96415288746357, + 30.836161226034164, + 30.886171385645866, + 30.823593959212303, + 30.811427161097527, + 30.788477510213852, + 30.81005997955799, + 30.732905492186546, + 30.733147636055946, + 30.7193323969841, + 30.75392358005047, + 30.746370553970337, + 30.771011486649513, + 30.745219439268112, + 30.74602596461773 + ], + "final_loss": 3.716245174407959 + }, + "p50_speedup": 1.1635231484750497, + "mean_speedup": 1.1697008401771465 + } + ] +} diff --git a/docs/sections/benchmarks/sequence-packing-results/dflash-family-512anchors-bf16.json b/docs/sections/benchmarks/sequence-packing-results/dflash-family-512anchors-bf16.json new file mode 100644 index 000000000..151037d09 --- /dev/null +++ b/docs/sections/benchmarks/sequence-packing-results/dflash-family-512anchors-bf16.json @@ -0,0 +1,563 @@ +{ + "timestamp_utc": "2026-10-02T04:44:09.597425+00:00", + "source": { + "files_sha256": { + "scripts/benchmark_dflash_sequence_packing.py": "0372ddf408fddc99c95138ab448025e37600c51004bceb6fd51d0432afb82e46", + "specforge/benchmarks/benchmark_dflash_sequence_packing.py": "00b9ff628d145d3f986eeb8c43f62671183a197f4c4575faf3f432f16ef0641c", + "specforge/algorithms/common/dflash_family_model.py": "86be39addbc5c50036f37c5d5282c53aaa8f35ab5ae1f8cfabf33078494e1421", + "specforge/algorithms/common/hidden_states_data.py": "fa09c5e5037c31b1165c2f8d1774e2cd4a40fde8b08f1f6deddf87f4f4af7f24", + "specforge/modeling/draft/dflash.py": "97af112a6ecf66d4b397aae77f0f97c08a49759c6305a4c70577a4f78b45ca90", + "specforge/modeling/draft/dflash2.py": "819c6b3d8d6d8ffc31a843ed40d539ddf635f58b148a01c8c43d0bff1dd02b09", + "specforge/modeling/packed_dflash.py": "d763e0ad3c609602323b9e3413968860cc680a8be8c986cc7612a956556ab96e", + "specforge/training/strategies/base.py": "276932ad331e95d5b91eddc67cf3acb0b5447ee5973f9bbc757e54093a82939f" + }, + "head": "unavailable (copied snapshot is identified by file hashes)" + }, + "settings": { + "preset": "medium", + "algorithm": "both", + "hidden_size": 2048, + "intermediate_size": 8192, + "layers": 2, + "heads": 16, + "kv_heads": 4, + "vocab_size": 32000, + "block_size": 16, + "anchors": 512, + "capture_layers": 2, + "conv_group_size": 32, + "conv_kernel_size": 4, + "selector_rank": 16, + "selector_top_k": 16, + "lengths": [ + [ + 128, + 256, + 512, + 2048 + ] + ], + "dtype": "bfloat16", + "sliding_window": null, + "warmup": 5, + "steps": 20, + "seed": 1729, + "learning_rate": 0.0001, + "objective_chunk_blocks": 128, + "correctness_only": false, + "skip_correctness": false, + "atol": 0.002, + "rtol": 0.02, + "output": "artifacts/sequence-packing/dflash-family-medium-512anchors-bf16-optimized.json" + }, + "environment": { + "torch": "2.13.0+cu130", + "cuda": "13.0", + "gpu": "NVIDIA H200", + "cuda_visible_devices": "0" + }, + "scope": "Single GPU; synthetic frozen target components and captured features; GPU-resident hidden features plus CPU integer features; production strategy forward/backward/BF16Optimizer; excludes capture, hidden-feature I/O/H2D, distributed training and serving.", + "cases": [ + { + "algorithm": "dflash", + "lengths": [ + 128, + 256, + 512, + 2048 + ], + "padding_fraction": 0.640625, + "useful_tokens": 2944, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 5.91278076171875e-05, + "relative_l2_diff": 5.61251249915586e-06 + }, + "loss_terms": { + "pass": true, + "max_abs_diff": 0.453125, + "relative_l2_diff": 5.5553522850025985e-06 + }, + "loss_values": [ + 10.534997940063477, + 10.535057067871094 + ], + "identical_sampled_anchors": true, + "sampled_anchor_count": 1178, + "gradients": { + "draft_model.layers.0.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0021281991404804583 + }, + "draft_model.layers.0.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0030810157665098494 + }, + "draft_model.layers.0.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0026500859492456135 + }, + "draft_model.layers.0.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0017068411541985204 + }, + "draft_model.layers.0.self_attn.q_norm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.0019287543064844075 + }, + "draft_model.layers.0.self_attn.k_norm.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0022529095517634795 + }, + "draft_model.layers.0.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.0021051663862464735 + }, + "draft_model.layers.0.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0020911271359065784 + }, + "draft_model.layers.0.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0018525617885881578 + }, + "draft_model.layers.0.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.002017398682331893 + }, + "draft_model.layers.0.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0024711602964318856 + }, + "draft_model.layers.1.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0017866582661795515 + }, + "draft_model.layers.1.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0028495004711782653 + }, + "draft_model.layers.1.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.002261717485471004 + }, + "draft_model.layers.1.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.001464825932831344 + }, + "draft_model.layers.1.self_attn.q_norm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.0017715359662342583 + }, + "draft_model.layers.1.self_attn.k_norm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.003161722116251433 + }, + "draft_model.layers.1.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0017373591305729018 + }, + "draft_model.layers.1.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0016626714244595531 + }, + "draft_model.layers.1.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0016341453129451533 + }, + "draft_model.layers.1.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.0018407111772094009 + }, + "draft_model.layers.1.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.002002886579895657 + }, + "draft_model.norm.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0013523338859282566 + }, + "draft_model.fc.weight": { + "pass": true, + "max_abs_diff": 1.0728836059570312e-06, + "relative_l2_diff": 0.004442411498462378 + }, + "draft_model.hidden_norm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.004556106678282558 + } + }, + "pass": true + }, + "padded": { + "mean_step_ms": 102.73919776082039, + "p50_step_ms": 102.67521720379591, + "useful_tokens_per_second": 28655.080671874734, + "p50_useful_tokens_per_second": 28672.936665491274, + "peak_allocated_gib": 11.8163743019104, + "baseline_allocated_gib": 2.180513858795166, + "warmup_seconds_including_compile": 0.5427277218550444, + "step_ms": [ + 103.04645448923111, + 102.31764800846577, + 102.57496125996113, + 102.48782113194466, + 102.40579582750797, + 102.40877233445644, + 102.65973210334778, + 102.4135909974575, + 102.76232659816742, + 103.21440547704697, + 102.89289988577366, + 102.41799987852573, + 102.82001830637455, + 102.51341387629509, + 102.5018971413374, + 102.69070230424404, + 103.15846651792526, + 103.27554307878017, + 103.38030196726322, + 102.8412040323019 + ], + "final_loss": 5.586607933044434 + }, + "packed": { + "mean_step_ms": 73.74237161129713, + "p50_step_ms": 73.68302810937166, + "useful_tokens_per_second": 39922.773510975436, + "p50_useful_tokens_per_second": 39954.9268744773, + "peak_allocated_gib": 8.335425853729248, + "baseline_allocated_gib": 2.026548385620117, + "warmup_seconds_including_compile": 0.3671108912676573, + "step_ms": [ + 73.44117760658264, + 74.23617132008076, + 73.869489133358, + 73.47449846565723, + 73.62687028944492, + 73.83274286985397, + 73.57754744589329, + 73.6483745276928, + 73.83186556398869, + 73.71129095554352, + 73.57209734618664, + 73.83942790329456, + 73.66623729467392, + 73.60509596765041, + 74.06741566956043, + 73.6998189240694, + 73.65257479250431, + 73.98152723908424, + 73.88443686068058, + 73.62877205014229 + ], + "final_loss": 5.585000514984131 + }, + "p50_speedup": 1.3934717374995718, + "mean_speedup": 1.3932179765300772 + }, + { + "algorithm": "dflash2", + "lengths": [ + 128, + 256, + 512, + 2048 + ], + "padding_fraction": 0.640625, + "useful_tokens": 2944, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 6.67572021484375e-06, + "relative_l2_diff": 6.336308441356832e-07 + }, + "loss_terms": { + "pass": true, + "max_abs_diff": 0.0546875, + "relative_l2_diff": 6.704316832987997e-07 + }, + "loss_values": [ + 10.535661697387695, + 10.53565502166748 + ], + "identical_sampled_anchors": true, + "sampled_anchor_count": 1178, + "gradients": { + "draft_model.layers.0.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.002155811162419652 + }, + "draft_model.layers.0.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.003115263876083677 + }, + "draft_model.layers.0.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0024893414012672446 + }, + "draft_model.layers.0.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0016564023704723882 + }, + "draft_model.layers.0.self_attn.q_norm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.0018726661822179513 + }, + "draft_model.layers.0.self_attn.k_norm.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0029641507160095156 + }, + "draft_model.layers.0.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.0020104000601537755 + }, + "draft_model.layers.0.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.001965311158535 + }, + "draft_model.layers.0.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 3.0517578125e-05, + "relative_l2_diff": 0.0017894700855622925 + }, + "draft_model.layers.0.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.001847257034798985 + }, + "draft_model.layers.0.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.002091376779893997 + }, + "draft_model.layers.0.attention_conv.base_kernel": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0018613340828882604 + }, + "draft_model.layers.0.attention_conv.kernel_projection.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0019815838381331673 + }, + "draft_model.layers.0.mlp_conv.base_kernel": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.002203816458103695 + }, + "draft_model.layers.0.mlp_conv.kernel_projection.weight": { + "pass": true, + "max_abs_diff": 6.103515625e-05, + "relative_l2_diff": 0.0021897462902934943 + }, + "draft_model.layers.1.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.001935534452710976 + }, + "draft_model.layers.1.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0029778696938854822 + }, + "draft_model.layers.1.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.002231563848236419 + }, + "draft_model.layers.1.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.0013835182743699866 + }, + "draft_model.layers.1.self_attn.q_norm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.002283248567519543 + }, + "draft_model.layers.1.self_attn.k_norm.weight": { + "pass": true, + "max_abs_diff": 1.9073486328125e-06, + "relative_l2_diff": 0.002612348363653426 + }, + "draft_model.layers.1.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.001615087994551397 + }, + "draft_model.layers.1.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0015551054945898413 + }, + "draft_model.layers.1.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 1.52587890625e-05, + "relative_l2_diff": 0.0015438830204557216 + }, + "draft_model.layers.1.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.0019442038075880707 + }, + "draft_model.layers.1.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0020446608565408367 + }, + "draft_model.layers.1.attention_conv.base_kernel": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.0017227739116648922 + }, + "draft_model.layers.1.attention_conv.kernel_projection.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.001838657839446509 + }, + "draft_model.layers.1.mlp_conv.base_kernel": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.002047133300927381 + }, + "draft_model.layers.1.mlp_conv.kernel_projection.weight": { + "pass": true, + "max_abs_diff": 6.103515625e-05, + "relative_l2_diff": 0.0023054439329669245 + }, + "draft_model.norm.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.001168930954718689 + }, + "draft_model.fc.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.004321447013206923 + }, + "draft_model.hidden_norm.weight": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 0.004421665559420483 + }, + "draft_model.candidate_selector.predecessor_codebook": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.candidate_selector.successor_codebook": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "draft_model.candidate_selector.hidden_projection.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + } + }, + "pass": true + }, + "padded": { + "mean_step_ms": 105.65225137397647, + "p50_step_ms": 105.6781467050314, + "useful_tokens_per_second": 27865.00014636835, + "p50_useful_tokens_per_second": 27858.172117810565, + "peak_allocated_gib": 13.28050422668457, + "baseline_allocated_gib": 2.1340060234069824, + "warmup_seconds_including_compile": 0.540035929530859, + "step_ms": [ + 106.40191659331322, + 106.54045268893242, + 106.38073086738586, + 106.02187551558018, + 105.94053752720356, + 105.82047514617443, + 105.60824908316135, + 106.06533102691174, + 106.33190535008907, + 105.74804432690144, + 105.80088756978512, + 105.59522919356823, + 105.08263297379017, + 105.43076135218143, + 104.99388724565506, + 105.23213259875774, + 104.83885742723942, + 105.21486029028893, + 104.66686449944973, + 105.32939620316029 + ], + "final_loss": 4.916550636291504 + }, + "packed": { + "mean_step_ms": 71.99693070724607, + "p50_step_ms": 72.01073691248894, + "useful_tokens_per_second": 40890.63201834108, + "p50_useful_tokens_per_second": 40882.79229217855, + "peak_allocated_gib": 9.44128131866455, + "baseline_allocated_gib": 2.0951266288757324, + "warmup_seconds_including_compile": 0.36581393890082836, + "step_ms": [ + 71.98422029614449, + 72.08905182778835, + 72.05628231167793, + 72.11560942232609, + 72.03729264438152, + 72.03725352883339, + 72.13184051215649, + 72.09071330726147, + 71.98000326752663, + 72.12523184716702, + 72.10700213909149, + 72.47685827314854, + 71.98372855782509, + 71.88108563423157, + 71.81445695459843, + 71.6748759150505, + 71.81508652865887, + 71.88605889678001, + 71.80662639439106, + 71.84533588588238 + ], + "final_loss": 4.9150261878967285 + }, + "p50_speedup": 1.467533193466091, + "mean_speedup": 1.467454936427494 + } + ] +} diff --git a/docs/sections/benchmarks/sequence-packing-results/dflash-family-tiny-fp32.json b/docs/sections/benchmarks/sequence-packing-results/dflash-family-tiny-fp32.json new file mode 100644 index 000000000..32ce6ae4a --- /dev/null +++ b/docs/sections/benchmarks/sequence-packing-results/dflash-family-tiny-fp32.json @@ -0,0 +1,214 @@ +{ + "timestamp_utc": "2026-10-02T04:43:26.200384+00:00", + "source": { + "files_sha256": { + "scripts/benchmark_dflash_sequence_packing.py": "0372ddf408fddc99c95138ab448025e37600c51004bceb6fd51d0432afb82e46", + "specforge/benchmarks/benchmark_dflash_sequence_packing.py": "00b9ff628d145d3f986eeb8c43f62671183a197f4c4575faf3f432f16ef0641c", + "specforge/algorithms/common/dflash_family_model.py": "86be39addbc5c50036f37c5d5282c53aaa8f35ab5ae1f8cfabf33078494e1421", + "specforge/algorithms/common/hidden_states_data.py": "fa09c5e5037c31b1165c2f8d1774e2cd4a40fde8b08f1f6deddf87f4f4af7f24", + "specforge/modeling/draft/dflash.py": "97af112a6ecf66d4b397aae77f0f97c08a49759c6305a4c70577a4f78b45ca90", + "specforge/modeling/draft/dflash2.py": "819c6b3d8d6d8ffc31a843ed40d539ddf635f58b148a01c8c43d0bff1dd02b09", + "specforge/modeling/packed_dflash.py": "d763e0ad3c609602323b9e3413968860cc680a8be8c986cc7612a956556ab96e", + "specforge/training/strategies/base.py": "276932ad331e95d5b91eddc67cf3acb0b5447ee5973f9bbc757e54093a82939f" + }, + "head": "unavailable (copied snapshot is identified by file hashes)" + }, + "settings": { + "preset": "tiny", + "algorithm": "both", + "hidden_size": 64, + "intermediate_size": 128, + "layers": 2, + "heads": 4, + "kv_heads": 2, + "vocab_size": 128, + "block_size": 4, + "anchors": 8, + "capture_layers": 2, + "conv_group_size": 4, + "conv_kernel_size": 2, + "selector_rank": 4, + "selector_top_k": 8, + "lengths": [ + [ + 9, + 17, + 31, + 64 + ], + [ + 32, + 32, + 32, + 32 + ] + ], + "dtype": "float32", + "sliding_window": null, + "warmup": 5, + "steps": 20, + "seed": 1729, + "learning_rate": 0.0001, + "objective_chunk_blocks": 128, + "correctness_only": true, + "skip_correctness": false, + "atol": 2e-05, + "rtol": 0.0002, + "output": "artifacts/sequence-packing/dflash-family-tiny-fp32-optimized.json" + }, + "environment": { + "torch": "2.13.0+cu130", + "cuda": "13.0", + "gpu": "NVIDIA H200", + "cuda_visible_devices": "0" + }, + "scope": "Single GPU; synthetic frozen target components and captured features; GPU-resident hidden features plus CPU integer features; production strategy forward/backward/BF16Optimizer; excludes capture, hidden-feature I/O/H2D, distributed training and serving.", + "cases": [ + { + "algorithm": "dflash", + "lengths": [ + 9, + 17, + 31, + 64 + ], + "padding_fraction": 0.52734375, + "useful_tokens": 121, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "loss_terms": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "loss_values": [ + 4.920759201049805, + 4.920759201049805 + ], + "identical_sampled_anchors": true, + "sampled_anchor_count": 29, + "pass": true, + "gradient_summary": { + "parameter_tensors": 25, + "all_pass": true, + "max_abs_diff": 3.725290298461914e-09, + "max_relative_l2_diff": 2.936993432616088e-07 + } + } + }, + { + "algorithm": "dflash", + "lengths": [ + 32, + 32, + 32, + 32 + ], + "padding_fraction": 0.0, + "useful_tokens": 128, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "loss_terms": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "loss_values": [ + 4.951769828796387, + 4.951769828796387 + ], + "identical_sampled_anchors": true, + "sampled_anchor_count": 32, + "pass": true, + "gradient_summary": { + "parameter_tensors": 25, + "all_pass": true, + "max_abs_diff": 1.862645149230957e-09, + "max_relative_l2_diff": 2.8535136873839174e-07 + } + } + }, + { + "algorithm": "dflash2", + "lengths": [ + 9, + 17, + 31, + 64 + ], + "padding_fraction": 0.52734375, + "useful_tokens": 121, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "loss_terms": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "loss_values": [ + 5.123039722442627, + 5.123039722442627 + ], + "identical_sampled_anchors": true, + "sampled_anchor_count": 29, + "pass": true, + "gradient_summary": { + "parameter_tensors": 36, + "all_pass": true, + "max_abs_diff": 3.725290298461914e-09, + "max_relative_l2_diff": 2.952709571251088e-07 + } + } + }, + { + "algorithm": "dflash2", + "lengths": [ + 32, + 32, + 32, + 32 + ], + "padding_fraction": 0.0, + "useful_tokens": 128, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "loss_terms": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0 + }, + "loss_values": [ + 5.035518169403076, + 5.035518169403076 + ], + "identical_sampled_anchors": true, + "sampled_anchor_count": 32, + "pass": true, + "gradient_summary": { + "parameter_tensors": 36, + "all_pass": true, + "max_abs_diff": 9.313225746154785e-10, + "max_relative_l2_diff": 2.501790571376381e-07 + } + } + } + ], + "evidence_kind": "Derived summary: per-parameter gradient entries reduced to count, pass flag and maxima; all other fields preserved.", + "source_report_sha256": "18a6e3096731c0fbf87e7965ab5473770c6713b0a763821a299cca69b82a9e8f" +} diff --git a/docs/sections/benchmarks/sequence-packing-results/eagle3-large-bf16.json b/docs/sections/benchmarks/sequence-packing-results/eagle3-large-bf16.json new file mode 100644 index 000000000..3140021bd --- /dev/null +++ b/docs/sections/benchmarks/sequence-packing-results/eagle3-large-bf16.json @@ -0,0 +1,235 @@ +{ + "timestamp_utc": "2026-10-02T04:10:49.449132+00:00", + "source": { + "files_sha256": { + "scripts/benchmark_sequence_packing.py": "0d11416603f81d5b97726abd299244029aa65b12beccb6ff38f9fb82c89dfe47", + "specforge/benchmarks/benchmark_sequence_packing.py": "39bf66ed00bee5cd272b6c8f6a562eae54537d98ea7cb41d066ad271eb816d65", + "specforge/algorithms/eagle3/data.py": "84d03344c68ba2e3f3a452db12c8b9cef95c931445ab8645b0444ec2580a8fe2", + "specforge/algorithms/eagle3/model.py": "6988a06491728c04929ee3d1471c91adcc44bc173de8faec7543810420d401d0", + "specforge/modeling/draft/llama3_eagle.py": "1488e38e36f1ff3bb11c0355da776122ef9c779105d38dee404b8e109aa06f4a", + "specforge/modeling/packed_sequence.py": "fd5207429606166aa4c8f124e5d4a68cd0eacbb1514d879921a5c53d15a938b4", + "specforge/training/strategies/base.py": "1f3eb9916fb399d9d5dc89214a7fad687db768bd74a27a18cba940c1216e1c7e" + }, + "head": "unavailable (file hashes identify copied snapshot)" + }, + "environment": { + "python": "3.12.3", + "torch": "2.13.0+cu130", + "cuda": "13.0", + "gpu": "NVIDIA H200", + "cuda_visible_devices": "0", + "transformers": "5.12.1" + }, + "settings": { + "preset": "large", + "hidden_size": 4096, + "intermediate_size": 14336, + "num_heads": 32, + "num_kv_heads": 8, + "vocab_size": 32000, + "draft_vocab_size": 32000, + "target_hidden_size": 4096, + "lengths": [ + [ + 128, + 256, + 512, + 2048 + ] + ], + "ttt_length": 7, + "dtype": "bfloat16", + "warmup": 5, + "steps": 20, + "seed": 1729, + "learning_rate": 0.0001, + "prompt_fraction": 0.25, + "correctness_only": false, + "skip_correctness": false, + "atol": 0.002, + "rtol": 0.02, + "output": "artifacts/sequence-packing/large-bf16-final.json" + }, + "scope": "single-GPU synthetic offline features; resident inputs; production forward/backward/BF16Optimizer; excludes capture, I/O, transfer, distributed communication, and serving", + "cases": [ + { + "lengths": [ + 128, + 256, + 512, + 2048 + ], + "useful_tokens": 2944, + "padded_tokens": 8192, + "padding_fraction": 0.640625, + "raw_supervised_tokens": 2204, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 8.015076037423352e-08, + "reference_l2": 11.898506164550781 + }, + "plosses": { + "pass": true, + "max_abs_diff": 4.76837158203125e-07, + "relative_l2_diff": 1.0790098863577887e-07, + "reference_l2": 7.96684455871582 + }, + "loss_padded": 11.898506164550781, + "loss_packed": 11.898505210876465, + "gradients": { + "draft_model.midlayer.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 1.7881393432617188e-07, + "relative_l2_diff": 0.008098428375980581, + "reference_l2": 0.018883956596255302 + }, + "draft_model.midlayer.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 3.5762786865234375e-07, + "relative_l2_diff": 0.008347887328420777, + "reference_l2": 0.019043944776058197 + }, + "draft_model.midlayer.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 1.7881393432617188e-07, + "relative_l2_diff": 0.007748204675858391, + "reference_l2": 0.009764185175299644 + }, + "draft_model.midlayer.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 1.1920928955078125e-07, + "relative_l2_diff": 0.0071544088294053345, + "reference_l2": 0.009779798798263073 + }, + "draft_model.midlayer.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 1.1920928955078125e-07, + "relative_l2_diff": 0.006828965463057812, + "reference_l2": 0.0157613605260849 + }, + "draft_model.midlayer.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 1.1920928955078125e-07, + "relative_l2_diff": 0.00662542103430625, + "reference_l2": 0.015876401215791702 + }, + "draft_model.midlayer.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 1.1920928955078125e-07, + "relative_l2_diff": 0.006333277633854752, + "reference_l2": 0.016165578737854958 + }, + "draft_model.midlayer.hidden_norm.weight": { + "pass": true, + "max_abs_diff": 2.384185791015625e-07, + "relative_l2_diff": 0.009126329141738658, + "reference_l2": 0.00040416946285404265 + }, + "draft_model.midlayer.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 1.7881393432617188e-07, + "relative_l2_diff": 0.008737133519347783, + "reference_l2": 0.00040028218063525856 + }, + "draft_model.midlayer.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 2.384185791015625e-07, + "relative_l2_diff": 0.00702031283790539, + "reference_l2": 0.00044801118201576173 + }, + "draft_model.fc.weight": { + "pass": true, + "max_abs_diff": 1.862645149230957e-07, + "relative_l2_diff": 0.008248254998890576, + "reference_l2": 0.027148302644491196 + }, + "draft_model.norm.weight": { + "pass": true, + "max_abs_diff": 3.814697265625e-06, + "relative_l2_diff": 0.0007218484337602093, + "reference_l2": 0.02720426581799984 + }, + "draft_model.lm_head.weight": { + "pass": true, + "max_abs_diff": 1.7881393432617188e-07, + "relative_l2_diff": 0.004232173488441028, + "reference_l2": 0.01552735734730959 + } + }, + "pass": true + }, + "padded": { + "mean_step_ms": 271.955263055861, + "p50_step_ms": 249.50438179075718, + "stdev_step_ms": 58.620318163146024, + "useful_tokens_per_second": 10825.309894426597, + "useful_ttt_positions_per_second": 75777.16926098618, + "peak_allocated_gib": 33.02751874923706, + "baseline_allocated_gib": 6.385047912597656, + "peak_increment_gib": 26.642470836639404, + "warmup_seconds_including_compile": 1.318971425294876, + "step_ms": [ + 248.97141940891743, + 248.47774393856525, + 248.4574057161808, + 249.15488995611668, + 248.9372342824936, + 249.21941943466663, + 258.02627205848694, + 511.20829954743385, + 248.72448295354843, + 249.78934414684772, + 247.9896154254675, + 247.11078964173794, + 247.17947468161583, + 253.61876748502254, + 264.03690315783024, + 293.2693623006344, + 289.70682993531227, + 261.5733686834574, + 293.9461972564459, + 279.70744110643864 + ], + "final_loss": 11.393714904785156 + }, + "packed": { + "mean_step_ms": 115.71002416312695, + "p50_step_ms": 113.14941477030516, + "stdev_step_ms": 7.805935035036344, + "useful_tokens_per_second": 25442.912325811765, + "useful_ttt_positions_per_second": 178100.38628068237, + "peak_allocated_gib": 12.972045421600342, + "baseline_allocated_gib": 6.1793012619018555, + "peak_increment_gib": 6.792744159698486, + "warmup_seconds_including_compile": 0.5725000947713852, + "step_ms": [ + 112.05416917800903, + 112.56376467645168, + 114.90379646420479, + 113.56857605278492, + 112.49293573200703, + 112.36764304339886, + 113.49863186478615, + 112.70488612353802, + 112.50686645507812, + 146.31127193570137, + 123.74899163842201, + 121.32737971842289, + 112.57140710949898, + 112.80683055520058, + 113.80611918866634, + 112.85982467234135, + 112.63692378997803, + 113.43900486826897, + 113.50067704916, + 114.5307831466198 + ], + "final_loss": 11.393773078918457 + }, + "speedup": 2.3503172263836096, + "peak_memory_reduction_fraction": 0.6072352416149942 + } + ] +} diff --git a/docs/sections/benchmarks/sequence-packing-results/eagle3-medium-bf16.json b/docs/sections/benchmarks/sequence-packing-results/eagle3-medium-bf16.json new file mode 100644 index 000000000..e18d53fdf --- /dev/null +++ b/docs/sections/benchmarks/sequence-packing-results/eagle3-medium-bf16.json @@ -0,0 +1,605 @@ +{ + "timestamp_utc": "2026-10-02T04:13:13.890368+00:00", + "source": { + "files_sha256": { + "scripts/benchmark_sequence_packing.py": "0d11416603f81d5b97726abd299244029aa65b12beccb6ff38f9fb82c89dfe47", + "specforge/benchmarks/benchmark_sequence_packing.py": "39bf66ed00bee5cd272b6c8f6a562eae54537d98ea7cb41d066ad271eb816d65", + "specforge/algorithms/eagle3/data.py": "84d03344c68ba2e3f3a452db12c8b9cef95c931445ab8645b0444ec2580a8fe2", + "specforge/algorithms/eagle3/model.py": "aadd395819e3292ab3ccde0f0200fdd6df90f0de1de8a03fb91d6640394ca022", + "specforge/modeling/draft/llama3_eagle.py": "1488e38e36f1ff3bb11c0355da776122ef9c779105d38dee404b8e109aa06f4a", + "specforge/modeling/packed_sequence.py": "ad057f15c08fb1c5cc80994ac799106536ec0a41c28f88f33d5f1bce34c75670", + "specforge/training/strategies/base.py": "a6e3aa2961d9d55f96c816fa2f6ce764672699e02005d89688eddec357120894" + }, + "head": "unavailable (file hashes identify copied snapshot)" + }, + "environment": { + "python": "3.12.3", + "torch": "2.13.0+cu130", + "cuda": "13.0", + "gpu": "NVIDIA H200", + "cuda_visible_devices": "0", + "transformers": "5.12.1" + }, + "settings": { + "preset": "medium", + "hidden_size": 2048, + "intermediate_size": 8192, + "num_heads": 16, + "num_kv_heads": 4, + "vocab_size": 32000, + "draft_vocab_size": 32000, + "target_hidden_size": 2048, + "lengths": [ + [ + 1024, + 1024, + 1024, + 1024 + ], + [ + 512, + 768, + 1024, + 2048 + ], + [ + 128, + 256, + 512, + 2048 + ] + ], + "ttt_length": 7, + "dtype": "bfloat16", + "warmup": 5, + "steps": 20, + "seed": 1729, + "learning_rate": 0.0001, + "prompt_fraction": 0.25, + "correctness_only": false, + "skip_correctness": false, + "atol": 0.002, + "rtol": 0.02, + "output": "artifacts/sequence-packing/medium-bf16-current.json" + }, + "scope": "single-GPU synthetic offline features; resident inputs; production forward/backward/BF16Optimizer; excludes capture, I/O, transfer, distributed communication, and serving", + "cases": [ + { + "lengths": [ + 1024, + 1024, + 1024, + 1024 + ], + "useful_tokens": 4096, + "padded_tokens": 4096, + "padding_fraction": 0.0, + "raw_supervised_tokens": 3068, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 31.914281845092773 + }, + "plosses": { + "pass": true, + "max_abs_diff": 9.5367431640625e-07, + "relative_l2_diff": 6.311522082349717e-08, + "reference_l2": 21.36884117126465 + }, + "loss_padded": 31.914281845092773, + "loss_packed": 31.914281845092773, + "gradients": { + "draft_model.midlayer.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 5.960464477539063e-08, + "relative_l2_diff": 0.003765711493526104, + "reference_l2": 0.0036451490595936775 + }, + "draft_model.midlayer.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 1.1920928955078125e-07, + "relative_l2_diff": 0.004440347213036328, + "reference_l2": 0.003682289272546768 + }, + "draft_model.midlayer.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 8.940696716308594e-08, + "relative_l2_diff": 0.00396974587705717, + "reference_l2": 0.002882526256144047 + }, + "draft_model.midlayer.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 5.960464477539063e-08, + "relative_l2_diff": 0.0033154438597801433, + "reference_l2": 0.0029025187250226736 + }, + "draft_model.midlayer.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 1.1920928955078125e-07, + "relative_l2_diff": 0.00329089278355116, + "reference_l2": 0.012741141952574253 + }, + "draft_model.midlayer.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 1.1920928955078125e-07, + "relative_l2_diff": 0.0031612475656136022, + "reference_l2": 0.01308818906545639 + }, + "draft_model.midlayer.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 1.1920928955078125e-07, + "relative_l2_diff": 0.0030215318893036898, + "reference_l2": 0.013826857320964336 + }, + "draft_model.midlayer.hidden_norm.weight": { + "pass": true, + "max_abs_diff": 4.470348358154297e-08, + "relative_l2_diff": 0.004985117251834739, + "reference_l2": 8.83292086655274e-05 + }, + "draft_model.midlayer.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 5.960464477539063e-08, + "relative_l2_diff": 0.005409839549368207, + "reference_l2": 7.969953730935231e-05 + }, + "draft_model.midlayer.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 2.384185791015625e-07, + "relative_l2_diff": 0.0033067118185063616, + "reference_l2": 0.00036759089562110603 + }, + "draft_model.fc.weight": { + "pass": true, + "max_abs_diff": 1.1920928955078125e-07, + "relative_l2_diff": 0.003179768823534524, + "reference_l2": 0.019182570278644562 + }, + "draft_model.norm.weight": { + "pass": true, + "max_abs_diff": 7.62939453125e-06, + "relative_l2_diff": 0.0004027977723562092, + "reference_l2": 0.05357325077056885 + }, + "draft_model.lm_head.weight": { + "pass": true, + "max_abs_diff": 1.1920928955078125e-07, + "relative_l2_diff": 0.0021179105111969712, + "reference_l2": 0.021415317431092262 + } + }, + "pass": true + }, + "padded": { + "mean_step_ms": 87.81536612659693, + "p50_step_ms": 80.03105316311121, + "stdev_step_ms": 26.06948085014276, + "useful_tokens_per_second": 46643.31745875886, + "useful_ttt_positions_per_second": 326503.22221131204, + "peak_allocated_gib": 13.090556621551514, + "baseline_allocated_gib": 2.298463821411133, + "peak_increment_gib": 10.79209280014038, + "warmup_seconds_including_compile": 0.48302813060581684, + "step_ms": [ + 79.99945804476738, + 82.36873708665371, + 81.48440718650818, + 79.86084371805191, + 80.0857711583376, + 80.15882037580013, + 80.10819368064404, + 80.06264828145504, + 79.99635487794876, + 79.9578819423914, + 79.92365770041943, + 80.3301278501749, + 80.08425869047642, + 191.34998694062233, + 121.55988439917564, + 79.96164448559284, + 79.73956502974033, + 79.78175953030586, + 79.893684014678, + 79.59963753819466 + ], + "final_loss": 30.858375549316406 + }, + "packed": { + "mean_step_ms": 76.67620368301868, + "p50_step_ms": 76.61013770848513, + "stdev_step_ms": 0.3185837712334884, + "useful_tokens_per_second": 53419.44179882673, + "useful_ttt_positions_per_second": 373936.0925917871, + "peak_allocated_gib": 9.165019989013672, + "baseline_allocated_gib": 2.268096923828125, + "peak_increment_gib": 6.896923065185547, + "warmup_seconds_including_compile": 0.4739191196858883, + "step_ms": [ + 76.65209099650383, + 76.59146375954151, + 76.6055267304182, + 76.77727565169334, + 76.62097364664078, + 76.59219950437546, + 76.76399871706963, + 76.5523910522461, + 76.61474868655205, + 76.77584514021873, + 76.5397660434246, + 76.64117217063904, + 76.64411887526512, + 76.56820304691792, + 76.97880268096924, + 76.55680924654007, + 76.49008370935917, + 77.8732467442751, + 76.46557316184044, + 76.21978409588337 + ], + "final_loss": 30.858409881591797 + }, + "speedup": 1.1452753515240246, + "peak_memory_reduction_fraction": 0.2998754557216514 + }, + { + "lengths": [ + 512, + 768, + 1024, + 2048 + ], + "useful_tokens": 4352, + "padded_tokens": 8192, + "padding_fraction": 0.46875, + "raw_supervised_tokens": 3260, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 16.955839157104492 + }, + "plosses": { + "pass": true, + "max_abs_diff": 4.76837158203125e-07, + "relative_l2_diff": 9.391610213704716e-08, + "reference_l2": 11.35311508178711 + }, + "loss_padded": 16.955839157104492, + "loss_packed": 16.955839157104492, + "gradients": { + "draft_model.midlayer.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 0.001796242082491517 + }, + "draft_model.midlayer.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 0.001820462173782289 + }, + "draft_model.midlayer.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 0.0014368355041369796 + }, + "draft_model.midlayer.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 0.0014455055352300406 + }, + "draft_model.midlayer.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 0.006567297503352165 + }, + "draft_model.midlayer.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 0.006763069424778223 + }, + "draft_model.midlayer.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 0.007175843231379986 + }, + "draft_model.midlayer.hidden_norm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 4.3062595068477094e-05 + }, + "draft_model.midlayer.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 4.0413109672954306e-05 + }, + "draft_model.midlayer.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 0.00018638117762748152 + }, + "draft_model.fc.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 0.009890600107610226 + }, + "draft_model.norm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 0.02846393920481205 + }, + "draft_model.lm_head.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 0.011251452378928661 + } + }, + "pass": true + }, + "padded": { + "mean_step_ms": 147.45035851374269, + "p50_step_ms": 135.93942299485207, + "stdev_step_ms": 42.45054906683103, + "useful_tokens_per_second": 29515.018097391632, + "useful_ttt_positions_per_second": 206605.12668174144, + "peak_allocated_gib": 23.980101108551025, + "baseline_allocated_gib": 2.403548240661621, + "peak_increment_gib": 21.576552867889404, + "warmup_seconds_including_compile": 0.8475298807024956, + "step_ms": [ + 138.96236196160316, + 135.83327271044254, + 136.9408555328846, + 136.9077805429697, + 135.7897948473692, + 135.7873361557722, + 135.86034625768661, + 135.84019988775253, + 135.85438393056393, + 324.67854768037796, + 171.35234735906124, + 135.78341156244278, + 135.83547621965408, + 136.48096099495888, + 136.54042035341263, + 136.01849973201752, + 135.85083559155464, + 135.7721146196127, + 136.49668917059898, + 136.4215351641178 + ], + "final_loss": 16.449453353881836 + }, + "packed": { + "mean_step_ms": 102.26912191137671, + "p50_step_ms": 90.35401325672865, + "stdev_step_ms": 30.748070235129727, + "useful_tokens_per_second": 42554.38903417309, + "useful_ttt_positions_per_second": 297880.7232392116, + "peak_allocated_gib": 9.601574420928955, + "baseline_allocated_gib": 2.2720112800598145, + "peak_increment_gib": 7.329563140869141, + "warmup_seconds_including_compile": 0.4550774786621332, + "step_ms": [ + 90.31735174357891, + 90.39067476987839, + 91.11217595636845, + 90.1272390037775, + 89.55635502934456, + 90.43884836137295, + 90.76938405632973, + 90.1151355355978, + 90.11649154126644, + 89.33848328888416, + 90.56887403130531, + 90.10537527501583, + 90.23720771074295, + 90.10511636734009, + 181.663291528821, + 193.6410814523697, + 132.5615793466568, + 93.92490051686764, + 90.40801785886288, + 89.88485485315323 + ], + "final_loss": 16.449453353881836 + }, + "speedup": 1.4417876653083874, + "peak_memory_reduction_fraction": 0.5996024212964997 + }, + { + "lengths": [ + 128, + 256, + 512, + 2048 + ], + "useful_tokens": 2944, + "padded_tokens": 8192, + "padding_fraction": 0.640625, + "raw_supervised_tokens": 2204, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 11.463168144226074 + }, + "plosses": { + "pass": true, + "max_abs_diff": 2.384185791015625e-07, + "relative_l2_diff": 3.1062749860994195e-08, + "reference_l2": 7.675385475158691 + }, + "loss_padded": 11.463168144226074, + "loss_packed": 11.463168144226074, + "gradients": { + "draft_model.midlayer.self_attn.q_proj.weight": { + "pass": true, + "max_abs_diff": 2.9802322387695312e-08, + "relative_l2_diff": 0.005071662953144663, + "reference_l2": 0.0015770653262734413 + }, + "draft_model.midlayer.self_attn.k_proj.weight": { + "pass": true, + "max_abs_diff": 5.960464477539063e-08, + "relative_l2_diff": 0.0055092919147389, + "reference_l2": 0.0016009289538487792 + }, + "draft_model.midlayer.self_attn.v_proj.weight": { + "pass": true, + "max_abs_diff": 2.9802322387695312e-08, + "relative_l2_diff": 0.004675468360171435, + "reference_l2": 0.0013346055056899786 + }, + "draft_model.midlayer.self_attn.o_proj.weight": { + "pass": true, + "max_abs_diff": 2.9802322387695312e-08, + "relative_l2_diff": 0.003780963952307209, + "reference_l2": 0.0013365180930122733 + }, + "draft_model.midlayer.mlp.gate_proj.weight": { + "pass": true, + "max_abs_diff": 5.960464477539063e-08, + "relative_l2_diff": 0.0035285330381317264, + "reference_l2": 0.005381997209042311 + }, + "draft_model.midlayer.mlp.up_proj.weight": { + "pass": true, + "max_abs_diff": 5.960464477539063e-08, + "relative_l2_diff": 0.0033779653002713973, + "reference_l2": 0.00546617154031992 + }, + "draft_model.midlayer.mlp.down_proj.weight": { + "pass": true, + "max_abs_diff": 5.960464477539063e-08, + "relative_l2_diff": 0.002940032226982802, + "reference_l2": 0.005697569809854031 + }, + "draft_model.midlayer.hidden_norm.weight": { + "pass": true, + "max_abs_diff": 2.9802322387695312e-08, + "relative_l2_diff": 0.00615527538571967, + "reference_l2": 3.794665462919511e-05 + }, + "draft_model.midlayer.input_layernorm.weight": { + "pass": true, + "max_abs_diff": 1.862645149230957e-08, + "relative_l2_diff": 0.006584594031334515, + "reference_l2": 3.436251063249074e-05 + }, + "draft_model.midlayer.post_attention_layernorm.weight": { + "pass": true, + "max_abs_diff": 5.960464477539063e-08, + "relative_l2_diff": 0.0038776387553202765, + "reference_l2": 0.00015499240544158965 + }, + "draft_model.fc.weight": { + "pass": true, + "max_abs_diff": 5.960464477539063e-08, + "relative_l2_diff": 0.004189682020519722, + "reference_l2": 0.008074476383626461 + }, + "draft_model.norm.weight": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 0.019233308732509613 + }, + "draft_model.lm_head.weight": { + "pass": true, + "max_abs_diff": 2.9802322387695312e-08, + "relative_l2_diff": 3.4444987028199235e-05, + "reference_l2": 0.008254210464656353 + } + }, + "pass": true + }, + "padded": { + "mean_step_ms": 142.04363320022821, + "p50_step_ms": 136.79443392902613, + "stdev_step_ms": 17.873005002610697, + "useful_tokens_per_second": 20726.025754706407, + "useful_ttt_positions_per_second": 145082.18028294484, + "peak_allocated_gib": 23.906234741210938, + "baseline_allocated_gib": 2.329681873321533, + "peak_increment_gib": 21.576552867889404, + "warmup_seconds_including_compile": 0.6789059638977051, + "step_ms": [ + 135.30797697603703, + 135.27469523251057, + 136.8685495108366, + 155.8755338191986, + 137.39495538175106, + 141.58212766051292, + 215.13050608336926, + 135.47670654952526, + 136.57748885452747, + 137.78152875602245, + 135.2911926805973, + 135.3690456598997, + 136.72031834721565, + 135.68365201354027, + 142.4194872379303, + 137.2075453400612, + 143.5843240469694, + 137.1347662061453, + 135.08125953376293, + 135.111004114151 + ], + "final_loss": 11.069210052490234 + }, + "packed": { + "mean_step_ms": 88.26659778133035, + "p50_step_ms": 69.67359222471714, + "stdev_step_ms": 32.720125563363545, + "useful_tokens_per_second": 33353.500350080314, + "useful_ttt_positions_per_second": 233474.5024505622, + "peak_allocated_gib": 7.208748817443848, + "baseline_allocated_gib": 2.2495083808898926, + "peak_increment_gib": 4.959240436553955, + "warmup_seconds_including_compile": 0.33644894510507584, + "step_ms": [ + 84.10226367413998, + 67.03308410942554, + 74.08362068235874, + 116.3476463407278, + 154.4363684952259, + 154.60862964391708, + 152.0404890179634, + 128.29125113785267, + 73.5000278800726, + 67.39425659179688, + 67.12955050170422, + 67.01149977743626, + 82.40591175854206, + 70.29546052217484, + 68.965008482337, + 69.05172392725945, + 68.15902143716812, + 66.88438914716244, + 66.8429471552372, + 66.74880534410477 + ], + "final_loss": 11.069201469421387 + }, + "speedup": 1.6092569190456831, + "peak_memory_reduction_fraction": 0.6984573733388055 + } + ] +} diff --git a/docs/sections/benchmarks/sequence-packing-results/eagle3-tiny-fp32.json b/docs/sections/benchmarks/sequence-packing-results/eagle3-tiny-fp32.json new file mode 100644 index 000000000..1cb48a255 --- /dev/null +++ b/docs/sections/benchmarks/sequence-packing-results/eagle3-tiny-fp32.json @@ -0,0 +1,175 @@ +{ + "timestamp_utc": "2026-10-02T04:12:33.131611+00:00", + "source": { + "files_sha256": { + "scripts/benchmark_sequence_packing.py": "0d11416603f81d5b97726abd299244029aa65b12beccb6ff38f9fb82c89dfe47", + "specforge/benchmarks/benchmark_sequence_packing.py": "39bf66ed00bee5cd272b6c8f6a562eae54537d98ea7cb41d066ad271eb816d65", + "specforge/algorithms/eagle3/data.py": "84d03344c68ba2e3f3a452db12c8b9cef95c931445ab8645b0444ec2580a8fe2", + "specforge/algorithms/eagle3/model.py": "aadd395819e3292ab3ccde0f0200fdd6df90f0de1de8a03fb91d6640394ca022", + "specforge/modeling/draft/llama3_eagle.py": "1488e38e36f1ff3bb11c0355da776122ef9c779105d38dee404b8e109aa06f4a", + "specforge/modeling/packed_sequence.py": "ad057f15c08fb1c5cc80994ac799106536ec0a41c28f88f33d5f1bce34c75670", + "specforge/training/strategies/base.py": "a6e3aa2961d9d55f96c816fa2f6ce764672699e02005d89688eddec357120894" + }, + "head": "unavailable (file hashes identify copied snapshot)" + }, + "environment": { + "python": "3.12.3", + "torch": "2.13.0+cu130", + "cuda": "13.0", + "gpu": "NVIDIA H200", + "cuda_visible_devices": "0", + "transformers": "5.12.1" + }, + "settings": { + "preset": "tiny", + "hidden_size": 128, + "intermediate_size": 256, + "num_heads": 4, + "num_kv_heads": 2, + "vocab_size": 512, + "draft_vocab_size": 256, + "target_hidden_size": 128, + "lengths": [ + [ + 8, + 17, + 31, + 64 + ], + [ + 2, + 3, + 5, + 17 + ], + [ + 32, + 32, + 32, + 32 + ] + ], + "ttt_length": 7, + "dtype": "float32", + "warmup": 5, + "steps": 20, + "seed": 1729, + "learning_rate": 0.0001, + "prompt_fraction": 0.25, + "correctness_only": true, + "skip_correctness": false, + "atol": 2e-05, + "rtol": 0.0002, + "output": "artifacts/sequence-packing/tiny-fp32-latest-metadata.json" + }, + "scope": "single-GPU synthetic offline features; resident inputs; production forward/backward/BF16Optimizer; excludes capture, I/O, transfer, distributed communication, and serving", + "cases": [ + { + "lengths": [ + 8, + 17, + 31, + 64 + ], + "useful_tokens": 120, + "padded_tokens": 256, + "padding_fraction": 0.53125, + "raw_supervised_tokens": 87, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 3.5112829208374023 + }, + "plosses": { + "pass": true, + "max_abs_diff": 5.960464477539063e-08, + "relative_l2_diff": 4.462841585418779e-08, + "reference_l2": 2.3132855892181396 + }, + "loss_padded": 3.5112829208374023, + "loss_packed": 3.5112829208374023, + "pass": true, + "gradient_summary": { + "parameter_tensors": 13, + "all_pass": true, + "max_abs_diff": 2.9103830456733704e-10, + "max_relative_l2_diff": 3.381816025687098e-07 + } + } + }, + { + "lengths": [ + 2, + 3, + 5, + 17 + ], + "useful_tokens": 27, + "padded_tokens": 68, + "padding_fraction": 0.6029411764705883, + "raw_supervised_tokens": 18, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 1.8310911655426025 + }, + "plosses": { + "pass": true, + "max_abs_diff": 2.9802322387695312e-08, + "relative_l2_diff": 4.682924771319063e-08, + "reference_l2": 1.1472936868667603 + }, + "loss_padded": 1.8310911655426025, + "loss_packed": 1.8310911655426025, + "pass": true, + "gradient_summary": { + "parameter_tensors": 13, + "all_pass": true, + "max_abs_diff": 4.0745362639427185e-10, + "max_relative_l2_diff": 4.456337620781133e-07 + } + } + }, + { + "lengths": [ + 32, + 32, + 32, + 32 + ], + "useful_tokens": 128, + "padded_tokens": 128, + "padding_fraction": 0.0, + "raw_supervised_tokens": 92, + "correctness": { + "loss": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 6.87973690032959 + }, + "plosses": { + "pass": true, + "max_abs_diff": 0.0, + "relative_l2_diff": 0.0, + "reference_l2": 4.606232166290283 + }, + "loss_padded": 6.87973690032959, + "loss_packed": 6.87973690032959, + "pass": true, + "gradient_summary": { + "parameter_tensors": 13, + "all_pass": true, + "max_abs_diff": 1.1641532182693481e-10, + "max_relative_l2_diff": 1.7126517309899537e-07 + } + } + } + ], + "evidence_kind": "Derived summary: per-parameter gradient entries reduced to count, pass flag and maxima; all other fields preserved.", + "source_report_sha256": "86a3ddd7e307768881d6a0ec594a2eb1f18405dae7e387b12fce9edd88d5c702" +} diff --git a/docs/sections/benchmarks/sequence-packing-results/full-model-analysis.json b/docs/sections/benchmarks/sequence-packing-results/full-model-analysis.json new file mode 100644 index 000000000..70816ed2e --- /dev/null +++ b/docs/sections/benchmarks/sequence-packing-results/full-model-analysis.json @@ -0,0 +1,899 @@ +{ + "summary": { + "dflash": { + "padded": { + "runs": 4, + "metrics": { + "pipeline_seconds": { + "median": 135.77420610096306, + "min": 135.39080221019685, + "max": 136.07322796620429, + "values": [ + 136.07322796620429, + 136.05652987398207, + 135.49188232794404, + 135.39080221019685 + ] + }, + "trainer_fit_seconds_from_first_capture": { + "median": 145.5256609506905, + "min": 144.90472558513284, + "max": 145.755035309121, + "values": [ + 145.70392679609358, + 145.755035309121, + 145.34739510528743, + 144.90472558513284 + ] + }, + "checkpoint_seconds": { + "median": 19.004825842566788, + "min": 18.59124900586903, + "max": 19.063026294112206, + "values": [ + 18.94984213076532, + 19.059809554368258, + 19.063026294112206, + 18.59124900586903 + ] + }, + "useful_tokens_per_second": { + "median": 9647.448519004163, + "min": 9626.206562288098, + "max": 9674.726632954009, + "values": [ + 9626.206562288098, + 9627.387977726783, + 9667.509060281545, + 9674.726632954009 + ] + }, + "final_loss": { + "median": 7.263832092285156, + "min": 7.263832092285156, + "max": 7.263832092285156, + "values": [ + 7.263832092285156, + 7.263832092285156, + 7.263832092285156, + 7.263832092285156 + ] + }, + "perf/train_compute_time_s": { + "median": 0.4527946563698352, + "min": 0.4526336301639676, + "max": 0.45302082094550133, + "values": [ + 0.4526336301639676, + 0.4528050641492009, + 0.4527842485904694, + 0.45302082094550133 + ] + }, + "perf/data_wait_time_s": { + "median": 0.039065446685999636, + "min": 0.038275998421013355, + "max": 0.04061508445441723, + "values": [ + 0.04061508445441723, + 0.03976557278633118, + 0.038365320585668085, + 0.038275998421013355 + ] + }, + "perf/durable_ack_time_s": { + "median": 0.002404910206794739, + "min": 0.002309505730867386, + "max": 0.002473393775522709, + "values": [ + 0.002309505730867386, + 0.002416390419006348, + 0.002473393775522709, + 0.0023934299945831297 + ] + } + }, + "checks": [ + { + "run_id": "bd0233465065-dflash-repeat00-arm0-padded", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.320475473999977 + }, + { + "step": 256, + "seconds": 9.629366656765342 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + }, + { + "run_id": "1e720cd3e271-dflash-repeat00-arm3-padded", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.36259664222598 + }, + { + "step": 256, + "seconds": 9.697212912142277 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + }, + { + "run_id": "1b5f8203cf73-dflash-repeat01-arm0-padded", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.2087761182338 + }, + { + "step": 256, + "seconds": 9.854250175878406 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + }, + { + "run_id": "31ea763f3d18-dflash-repeat01-arm3-padded", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.078649101778865 + }, + { + "step": 256, + "seconds": 9.512599904090166 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + } + ] + }, + "packed": { + "runs": 4, + "metrics": { + "pipeline_seconds": { + "median": 130.65109005570412, + "min": 128.88156617432833, + "max": 130.91351471282542, + "values": [ + 128.88156617432833, + 130.78667958825827, + 130.51550052314997, + 130.91351471282542 + ] + }, + "trainer_fit_seconds_from_first_capture": { + "median": 140.37849941663444, + "min": 138.3309756219387, + "max": 140.5097789634019, + "values": [ + 138.3309756219387, + 140.4588652085513, + 140.5097789634019, + 140.2981336247176 + ] + }, + "checkpoint_seconds": { + "median": 18.75084599200636, + "min": 18.463733648881316, + "max": 19.00793844088912, + "values": [ + 18.463733648881316, + 19.00793844088912, + 18.919086307287216, + 18.582605676725507 + ] + }, + "useful_tokens_per_second": { + "median": 10025.713602590113, + "min": 10005.605631117272, + "max": 10163.35414661426, + "values": [ + 10163.35414661426, + 10015.308929959234, + 10036.11827522099, + 10005.605631117272 + ] + }, + "final_loss": { + "median": 7.263965606689453, + "min": 7.263965606689453, + "max": 7.263965606689453, + "values": [ + 7.263965606689453, + 7.263965606689453, + 7.263965606689453, + 7.263965606689453 + ] + }, + "perf/train_compute_time_s": { + "median": 0.43363652409240605, + "min": 0.43351573960483075, + "max": 0.4339526748508215, + "values": [ + 0.433688922919333, + 0.4339526748508215, + 0.43351573960483075, + 0.43358412526547907 + ] + }, + "perf/data_wait_time_s": { + "median": 0.03900999794527889, + "min": 0.03261461492627859, + "max": 0.0398359164968133, + "values": [ + 0.03261461492627859, + 0.038341693453490734, + 0.039678302437067034, + 0.0398359164968133 + ] + }, + "perf/durable_ack_time_s": { + "median": 0.002406857404857874, + "min": 0.002359313905239105, + "max": 0.0024157476499676706, + "values": [ + 0.002359313905239105, + 0.0024157476499676706, + 0.0024075431749224665, + 0.0024061716347932817 + ] + } + }, + "checks": [ + { + "run_id": "0db157a5b805-dflash-repeat00-arm1-packed", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.016127996146679 + }, + { + "step": 256, + "seconds": 9.447605652734637 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + }, + { + "run_id": "beb7bee09b70-dflash-repeat00-arm2-packed", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.337019385769963 + }, + { + "step": 256, + "seconds": 9.670919055119157 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + }, + { + "run_id": "d016ac38de6c-dflash-repeat01-arm1-packed", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 8.926233634352684 + }, + { + "step": 256, + "seconds": 9.992852672934532 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + }, + { + "run_id": "4dc06bfceb8e-dflash-repeat01-arm2-packed", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.199242942035198 + }, + { + "step": 256, + "seconds": 9.383362734690309 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + } + ] + }, + "speedups": { + "pipeline_seconds": 1.0392121951915951, + "trainer_fit_seconds_from_first_capture": 1.0366663096944755, + "perf/train_compute_time_s": 1.0441801629083869 + } + }, + "dflash2": { + "padded": { + "runs": 4, + "metrics": { + "pipeline_seconds": { + "median": 127.55567608494312, + "min": 127.01265624165535, + "max": 127.72959364019334, + "values": [ + 127.72959364019334, + 127.55552425421774, + 127.55582791566849, + 127.01265624165535 + ] + }, + "trainer_fit_seconds_from_first_capture": { + "median": 137.32001544442028, + "min": 136.6376235689968, + "max": 137.59124981798232, + "values": [ + 137.59124981798232, + 137.353132288903, + 137.28689859993756, + 136.6376235689968 + ] + }, + "checkpoint_seconds": { + "median": 19.090700599364936, + "min": 18.831517465412617, + "max": 19.25768494606018, + "values": [ + 19.159075815230608, + 19.25768494606018, + 19.022325383499265, + 18.831517465412617 + ] + }, + "useful_tokens_per_second": { + "median": 10268.99813638693, + "min": 10255.015792893093, + "max": 10312.901397068905, + "values": [ + 10255.015792893093, + 10269.010359672353, + 10268.98591310151, + 10312.901397068905 + ] + }, + "final_loss": { + "median": 7.968346118927002, + "min": 7.968346118927002, + "max": 7.968346118927002, + "values": [ + 7.968346118927002, + 7.968346118927002, + 7.968346118927002, + 7.968346118927002 + ] + }, + "perf/train_compute_time_s": { + "median": 0.42698242781683804, + "min": 0.4268245469406247, + "max": 0.42719867677241563, + "values": [ + 0.4269536115154624, + 0.4268245469406247, + 0.42719867677241563, + 0.4270112441182136 + ] + }, + "perf/data_wait_time_s": { + "median": 0.03236039846017957, + "min": 0.031001201815903188, + "max": 0.03348202735185623, + "values": [ + 0.03348202735185623, + 0.03243206156045198, + 0.03228873535990715, + 0.031001201815903188 + ] + }, + "perf/durable_ack_time_s": { + "median": 0.0024311989322304724, + "min": 0.002365817494690418, + "max": 0.0025716573223471643, + "values": [ + 0.0024104075357317925, + 0.0024519903287291527, + 0.0025716573223471643, + 0.002365817494690418 + ] + } + }, + "checks": [ + { + "run_id": "ca60b5621436-dflash2-repeat00-arm0-padded", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.298717353492975 + }, + { + "step": 256, + "seconds": 9.860358461737633 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + }, + { + "run_id": "e8afd561b8f6-dflash2-repeat00-arm3-padded", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.461436005309224 + }, + { + "step": 256, + "seconds": 9.796248940750957 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + }, + { + "run_id": "b026b309f6bf-dflash2-repeat01-arm0-padded", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.292883981019258 + }, + { + "step": 256, + "seconds": 9.729441402480006 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + }, + { + "run_id": "fd539583a214-dflash2-repeat01-arm3-padded", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.208016926422715 + }, + { + "step": 256, + "seconds": 9.623500538989902 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + } + ] + }, + "packed": { + "runs": 4, + "metrics": { + "pipeline_seconds": { + "median": 122.89778250828385, + "min": 122.41715203598142, + "max": 123.31176270730793, + "values": [ + 123.19319461472332, + 122.41715203598142, + 122.60237040184438, + 123.31176270730793 + ] + }, + "trainer_fit_seconds_from_first_capture": { + "median": 132.90509969182312, + "min": 132.2964224666357, + "max": 133.59795146621764, + "values": [ + 133.59795146621764, + 132.60274993814528, + 132.2964224666357, + 133.20744944550097 + ] + }, + "checkpoint_seconds": { + "median": 19.39351878874004, + "min": 19.042769499123096, + "max": 19.794485840946436, + "values": [ + 19.794485840946436, + 19.42600578442216, + 19.042769499123096, + 19.36103179305792 + ] + }, + "useful_tokens_per_second": { + "median": 10658.260397991853, + "min": 10622.41728803356, + "max": 10700.044709543621, + "values": [ + 10632.640902742303, + 10700.044709543621, + 10683.879893241403, + 10622.41728803356 + ] + }, + "final_loss": { + "median": 7.968497276306152, + "min": 7.968497276306152, + "max": 7.968497276306152, + "values": [ + 7.968497276306152, + 7.968497276306152, + 7.968497276306152, + 7.968497276306152 + ] + }, + "perf/train_compute_time_s": { + "median": 0.4057552154362202, + "min": 0.40529941400140523, + "max": 0.4060684394985437, + "values": [ + 0.40529941400140523, + 0.40570894527435303, + 0.4058014855980873, + 0.4060684394985437 + ] + }, + "perf/data_wait_time_s": { + "median": 0.03559140999987721, + "min": 0.03429850439727306, + "max": 0.03726461844146252, + "values": [ + 0.03726461844146252, + 0.03429850439727306, + 0.03459509003907442, + 0.03658772996068001 + ] + }, + "perf/durable_ack_time_s": { + "median": 0.0024793629013001917, + "min": 0.00236849345266819, + "max": 0.0025612031146883965, + "values": [ + 0.0025612031146883965, + 0.00236849345266819, + 0.002535621479153633, + 0.0024231043234467504 + ] + } + }, + "checks": [ + { + "run_id": "bf1e0bae505a-dflash2-repeat00-arm1-packed", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.391200210899115 + }, + { + "step": 256, + "seconds": 10.403285630047321 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + }, + { + "run_id": "33c10f6d8e0a-dflash2-repeat00-arm2-packed", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.241760902106762 + }, + { + "step": 256, + "seconds": 10.184244882315397 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + }, + { + "run_id": "1874cfdf56c1-dflash2-repeat01-arm1-packed", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.35001527518034 + }, + { + "step": 256, + "seconds": 9.692754223942757 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + }, + { + "run_id": "e269784cb040-dflash2-repeat01-arm2-packed", + "samples": 1024, + "steps": 256, + "compiler_counter_delta": { + "stats": {}, + "frames": {}, + "unimplemented": {}, + "graph_break": {}, + "inductor": {}, + "aot_autograd": {} + }, + "checkpoint_events": [ + { + "step": 128, + "seconds": 9.466601584106684 + }, + { + "step": 256, + "seconds": 9.894430208951235 + } + ], + "logged_steps": [ + 50, + 100, + 150, + 200, + 250 + ], + "sustained_overlap": true + } + ] + }, + "speedups": { + "pipeline_seconds": 1.037900550210052, + "trainer_fit_seconds_from_first_capture": 1.0332185579246722, + "perf/train_compute_time_s": 1.0523153161637044 + } + } + }, + "notes": [ + "Training-thread compute diagnostics cover five 50-step windows (250 steps), use host wall time, and include detailed metrics at window ends.", + "Pipeline includes intermediate checkpoint at step 128; full completion includes the final step-256 checkpoint.", + "Ratios are medians of four runs per arm when complete; no claim of convergence or serving quality." + ] +} diff --git a/docs/sections/benchmarks/sequence-packing-results/full-model-audit.json b/docs/sections/benchmarks/sequence-packing-results/full-model-audit.json new file mode 100644 index 000000000..22957abf6 --- /dev/null +++ b/docs/sections/benchmarks/sequence-packing-results/full-model-audit.json @@ -0,0 +1,1107 @@ +{ + "audit_utc": "2026-10-02T09:08:41.738677+00:00", + "source_report_remote": "/scratch/specforge-packing-full-model-20261002/long_v2.json", + "source_report_sha256": "f8895e415d1baaef7ddaa74a7f8e322d83d13e008d2f7606b0378fb0b4861179", + "dataset_sha256": "f90478e0d3e32f0d3c45c515381b9bc42951cdd6931830ac1a7380f4459f24b1", + "all_run_assertions_passed": true, + "measured_runs": 16, + "warmup_runs": 4, + "runs_per_architecture_per_mode": 4, + "target_architecture": { + "model": "Qwen3-4B", + "layers": 36, + "hidden_size": 2560, + "vocabulary": 151936, + "role": "frozen real target forward in separate SGLang producer" + }, + "validated_trainable_counts": { + "dflash": { + "elements": 537427200, + "tensors": 58, + "decoder_layers": 5, + "optimizer_updates_per_tensor_per_run": 256 + }, + "dflash2": { + "elements": 558918912, + "tensors": 81, + "decoder_layers": 5, + "optimizer_updates_per_tensor_per_run": 256 + } + }, + "frozen_tied_target_embedding_head_unique_elements_in_consumer": 388956160, + "checks": { + "exact_preprocessed_dataset_hash": true, + "same_actual_http_inputs_and_masks_per_architecture": true, + "exact_capture_publication_consumption_and_ack_order": true, + "1024_samples_and_256_optimizer_steps_per_run": true, + "all_optimizer_original_parameters_and_fp32_masters_covered": true, + "all_trainable_tensors_have_256_adam_updates": true, + "all_first_warmup_gradients_present_and_finite_global_norm": true, + "no_gradient_observation_hook_in_measured_runs": true, + "all_five_decoder_layers_have_nonempty_trainable_names_and_sampled_updates": true, + "all_trainable_elements_finite_after_training": true, + "frozen_target_parameter_samples_unchanged": true, + "logging_at_steps_50_100_150_200_250": true, + "checkpoint_events_at_128_and_256": true, + "final_checkpoint_step_and_sample_count_match_acks": true, + "all_measured_compiler_counter_deltas_empty": true, + "continued_live_capture_after_first_optimizer_ack": true, + "final_ack_equals_pipeline_endpoint": true, + "all_final_losses_finite": true + }, + "summary": { + "dflash": { + "arms": { + "padded": { + "runs": 4, + "metrics": { + "pipeline_seconds": { + "median": 135.77420610096306, + "min": 135.39080221019685, + "max": 136.07322796620429, + "values": [ + 136.07322796620429, + 136.05652987398207, + 135.49188232794404, + 135.39080221019685 + ] + }, + "fit_with_final_checkpoint_seconds": { + "median": 145.5256609506905, + "min": 144.90472558513284, + "max": 145.755035309121, + "values": [ + 145.70392679609358, + 145.755035309121, + 145.34739510528743, + 144.90472558513284 + ] + }, + "checkpoint_total_seconds": { + "median": 19.004825842566788, + "min": 18.59124900586903, + "max": 19.063026294112206, + "values": [ + 18.94984213076532, + 19.059809554368258, + 19.063026294112206, + 18.59124900586903 + ] + }, + "host_train_compute_seconds_per_step": { + "median": 0.4527946563698352, + "min": 0.4526336301639676, + "max": 0.45302082094550133, + "values": [ + 0.4526336301639676, + 0.4528050641492009, + 0.4527842485904694, + 0.45302082094550133 + ] + }, + "host_data_wait_seconds_per_step": { + "median": 0.039065446685999636, + "min": 0.038275998421013355, + "max": 0.04061508445441723, + "values": [ + 0.04061508445441723, + 0.03976557278633118, + 0.038365320585668085, + 0.038275998421013355 + ] + }, + "host_durable_ack_seconds_per_step": { + "median": 0.002404910206794739, + "min": 0.002309505730867386, + "max": 0.002473393775522709, + "values": [ + 0.002309505730867386, + 0.002416390419006348, + 0.002473393775522709, + 0.0023934299945831297 + ] + }, + "loader_wait_producer_seconds": { + "median": 0.20019448921084404, + "min": 0.20013251528143883, + "max": 0.2004451733082533, + "values": [ + 0.20013251528143883, + 0.20024952851235867, + 0.2004451733082533, + 0.20013944990932941 + ] + }, + "loader_wait_fetch_seconds": { + "median": 9.73675948008895, + "min": 9.515376891940832, + "max": 10.09336070343852, + "values": [ + 10.09336070343852, + 9.953005198389292, + 9.515376891940832, + 9.520513761788607 + ] + }, + "final_loss": { + "median": 7.263832092285156, + "min": 7.263832092285156, + "max": 7.263832092285156, + "values": [ + 7.263832092285156, + 7.263832092285156, + 7.263832092285156, + 7.263832092285156 + ] + } + } + }, + "packed": { + "runs": 4, + "metrics": { + "pipeline_seconds": { + "median": 130.65109005570412, + "min": 128.88156617432833, + "max": 130.91351471282542, + "values": [ + 128.88156617432833, + 130.78667958825827, + 130.51550052314997, + 130.91351471282542 + ] + }, + "fit_with_final_checkpoint_seconds": { + "median": 140.37849941663444, + "min": 138.3309756219387, + "max": 140.5097789634019, + "values": [ + 138.3309756219387, + 140.4588652085513, + 140.5097789634019, + 140.2981336247176 + ] + }, + "checkpoint_total_seconds": { + "median": 18.75084599200636, + "min": 18.463733648881316, + "max": 19.00793844088912, + "values": [ + 18.463733648881316, + 19.00793844088912, + 18.919086307287216, + 18.582605676725507 + ] + }, + "host_train_compute_seconds_per_step": { + "median": 0.43363652409240605, + "min": 0.43351573960483075, + "max": 0.4339526748508215, + "values": [ + 0.433688922919333, + 0.4339526748508215, + 0.43351573960483075, + 0.43358412526547907 + ] + }, + "host_data_wait_seconds_per_step": { + "median": 0.03900999794527889, + "min": 0.03261461492627859, + "max": 0.0398359164968133, + "values": [ + 0.03261461492627859, + 0.038341693453490734, + 0.039678302437067034, + 0.0398359164968133 + ] + }, + "host_durable_ack_seconds_per_step": { + "median": 0.002406857404857874, + "min": 0.002359313905239105, + "max": 0.0024157476499676706, + "values": [ + 0.002359313905239105, + 0.0024157476499676706, + 0.0024075431749224665, + 0.0024061716347932817 + ] + }, + "loader_wait_producer_seconds": { + "median": 0.2003897288814187, + "min": 0.20013267919421196, + "max": 0.30021679401397705, + "values": [ + 0.20013267919421196, + 0.30021679401397705, + 0.20041996613144875, + 0.20035949163138866 + ] + }, + "loader_wait_fetch_seconds": { + "median": 9.543545130640268, + "min": 8.035708643496037, + "max": 9.886001640930772, + "values": [ + 8.035708643496037, + 9.43034845776856, + 9.656741803511977, + 9.886001640930772 + ] + }, + "final_loss": { + "median": 7.263965606689453, + "min": 7.263965606689453, + "max": 7.263965606689453, + "values": [ + 7.263965606689453, + 7.263965606689453, + 7.263965606689453, + 7.263965606689453 + ] + } + } + } + }, + "median_speedups": { + "pipeline_seconds": 1.0392121951915951, + "fit_with_final_checkpoint_seconds": 1.0366663096944755, + "host_train_compute_seconds_per_step": 1.0441801629083869 + }, + "matched_chronological_pair_ratios": [ + { + "repeat": 0, + "padded_arm": 0, + "packed_arm": 1, + "ratios": { + "pipeline_seconds": 1.055800546232875, + "fit_with_final_checkpoint_seconds": 1.0532993506407797, + "host_train_compute_seconds_per_step": 1.0436827095262435 + } + }, + { + "repeat": 0, + "padded_arm": 3, + "packed_arm": 2, + "ratios": { + "pipeline_seconds": 1.040293478680813, + "fit_with_final_checkpoint_seconds": 1.0377062002651527, + "host_train_compute_seconds_per_step": 1.0434434222691684 + } + }, + { + "repeat": 1, + "padded_arm": 0, + "packed_arm": 1, + "ratios": { + "pipeline_seconds": 1.0381286650615986, + "fit_with_final_checkpoint_seconds": 1.0344290353139447, + "host_train_compute_seconds_per_step": 1.0444470805216963 + } + }, + { + "repeat": 1, + "padded_arm": 3, + "packed_arm": 2, + "ratios": { + "pipeline_seconds": 1.0342003459856144, + "fit_with_final_checkpoint_seconds": 1.0328343067822796, + "host_train_compute_seconds_per_step": 1.044827968893283 + } + } + ] + }, + "dflash2": { + "arms": { + "padded": { + "runs": 4, + "metrics": { + "pipeline_seconds": { + "median": 127.55567608494312, + "min": 127.01265624165535, + "max": 127.72959364019334, + "values": [ + 127.72959364019334, + 127.55552425421774, + 127.55582791566849, + 127.01265624165535 + ] + }, + "fit_with_final_checkpoint_seconds": { + "median": 137.32001544442028, + "min": 136.6376235689968, + "max": 137.59124981798232, + "values": [ + 137.59124981798232, + 137.353132288903, + 137.28689859993756, + 136.6376235689968 + ] + }, + "checkpoint_total_seconds": { + "median": 19.090700599364936, + "min": 18.831517465412617, + "max": 19.25768494606018, + "values": [ + 19.159075815230608, + 19.25768494606018, + 19.022325383499265, + 18.831517465412617 + ] + }, + "host_train_compute_seconds_per_step": { + "median": 0.42698242781683804, + "min": 0.4268245469406247, + "max": 0.42719867677241563, + "values": [ + 0.4269536115154624, + 0.4268245469406247, + 0.42719867677241563, + 0.4270112441182136 + ] + }, + "host_data_wait_seconds_per_step": { + "median": 0.03236039846017957, + "min": 0.031001201815903188, + "max": 0.03348202735185623, + "values": [ + 0.03348202735185623, + 0.03243206156045198, + 0.03228873535990715, + 0.031001201815903188 + ] + }, + "host_durable_ack_seconds_per_step": { + "median": 0.0024311989322304724, + "min": 0.002365817494690418, + "max": 0.0025716573223471643, + "values": [ + 0.0024104075357317925, + 0.0024519903287291527, + 0.0025716573223471643, + 0.002365817494690418 + ] + }, + "loader_wait_producer_seconds": { + "median": 0.20038656052201986, + "min": 0.20015659555792809, + "max": 0.3005787916481495, + "values": [ + 0.20015659555792809, + 0.2004134152084589, + 0.20035970583558083, + 0.3005787916481495 + ] + }, + "loader_wait_fetch_seconds": { + "median": 8.02211464382708, + "min": 7.595450304448605, + "max": 8.327452383935452, + "values": [ + 8.327452383935452, + 7.998322926461697, + 8.045906361192465, + 7.595450304448605 + ] + }, + "final_loss": { + "median": 7.968346118927002, + "min": 7.968346118927002, + "max": 7.968346118927002, + "values": [ + 7.968346118927002, + 7.968346118927002, + 7.968346118927002, + 7.968346118927002 + ] + } + } + }, + "packed": { + "runs": 4, + "metrics": { + "pipeline_seconds": { + "median": 122.89778250828385, + "min": 122.41715203598142, + "max": 123.31176270730793, + "values": [ + 123.19319461472332, + 122.41715203598142, + 122.60237040184438, + 123.31176270730793 + ] + }, + "fit_with_final_checkpoint_seconds": { + "median": 132.90509969182312, + "min": 132.2964224666357, + "max": 133.59795146621764, + "values": [ + 133.59795146621764, + 132.60274993814528, + 132.2964224666357, + 133.20744944550097 + ] + }, + "checkpoint_total_seconds": { + "median": 19.39351878874004, + "min": 19.042769499123096, + "max": 19.794485840946436, + "values": [ + 19.794485840946436, + 19.42600578442216, + 19.042769499123096, + 19.36103179305792 + ] + }, + "host_train_compute_seconds_per_step": { + "median": 0.4057552154362202, + "min": 0.40529941400140523, + "max": 0.4060684394985437, + "values": [ + 0.40529941400140523, + 0.40570894527435303, + 0.4058014855980873, + 0.4060684394985437 + ] + }, + "host_data_wait_seconds_per_step": { + "median": 0.03559140999987721, + "min": 0.03429850439727306, + "max": 0.03726461844146252, + "values": [ + 0.03726461844146252, + 0.03429850439727306, + 0.03459509003907442, + 0.03658772996068001 + ] + }, + "host_durable_ack_seconds_per_step": { + "median": 0.0024793629013001917, + "min": 0.00236849345266819, + "max": 0.0025612031146883965, + "values": [ + 0.0025612031146883965, + 0.00236849345266819, + 0.002535621479153633, + 0.0024231043234467504 + ] + }, + "loader_wait_producer_seconds": { + "median": 0.3002029359340668, + "min": 0.20014070346951485, + "max": 0.30046532303094864, + "values": [ + 0.20014070346951485, + 0.300190145149827, + 0.30046532303094864, + 0.30021572671830654 + ] + }, + "loader_wait_fetch_seconds": { + "median": 8.65784180443734, + "min": 8.398219551891088, + "max": 9.154761364683509, + "values": [ + 9.154761364683509, + 8.398219551891088, + 8.409457132220268, + 8.90622647665441 + ] + }, + "final_loss": { + "median": 7.968497276306152, + "min": 7.968497276306152, + "max": 7.968497276306152, + "values": [ + 7.968497276306152, + 7.968497276306152, + 7.968497276306152, + 7.968497276306152 + ] + } + } + } + }, + "median_speedups": { + "pipeline_seconds": 1.037900550210052, + "fit_with_final_checkpoint_seconds": 1.0332185579246722, + "host_train_compute_seconds_per_step": 1.0523153161637044 + }, + "matched_chronological_pair_ratios": [ + { + "repeat": 0, + "padded_arm": 0, + "packed_arm": 1, + "ratios": { + "pipeline_seconds": 1.0368234547343076, + "fit_with_final_checkpoint_seconds": 1.0298904160426026, + "host_train_compute_seconds_per_step": 1.0534276556195121 + } + }, + { + "repeat": 0, + "padded_arm": 3, + "packed_arm": 2, + "ratios": { + "pipeline_seconds": 1.0419742832828363, + "fit_with_final_checkpoint_seconds": 1.035824161663115, + "host_train_compute_seconds_per_step": 1.0520461821515734 + } + }, + { + "repeat": 1, + "padded_arm": 0, + "packed_arm": 1, + "ratios": { + "pipeline_seconds": 1.0404026243341669, + "fit_with_final_checkpoint_seconds": 1.0377219280782928, + "host_train_compute_seconds_per_step": 1.0527282228718118 + } + }, + { + "repeat": 1, + "padded_arm": 3, + "packed_arm": 2, + "ratios": { + "pipeline_seconds": 1.0300124939672775, + "fit_with_final_checkpoint_seconds": 1.0257506178353726, + "host_train_compute_seconds_per_step": 1.0515745686750053 + } + } + ] + } + }, + "per_run_checks": [ + { + "arm": "dflash/warmup/False", + "seconds": 157.2685, + "fit_seconds": 166.5949, + "host_train_compute_mean_s": 0.537832, + "checkpoint_seconds": 18.6059, + "layers": 5, + "trainable_parameters": 537427200, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": { + "stats": { + "calls_captured": 8, + "unique_graphs": 4 + }, + "frames": { + "ok": 4, + "total": 4 + }, + "aot_autograd": { + "autograd_cache_saved": 4, + "ok": 4, + "total": 4, + "autograd_cache_miss": 4 + }, + "inductor": { + "triton_bundler_save_kernel": 112, + "benchmarking.InductorBenchmarker.benchmark_gpu": 3, + "benchmarking.InductorBenchmarker.benchmark": 3, + "fxgraph_cache_miss": 8, + "async_compile_cache_hit": 3, + "async_compile_cache_miss": 21, + "triton_bundler_save_static_autotuner": 8 + } + }, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 1.0006, + "wait_fetch_s": 9.2088 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash/warmup/True", + "seconds": 136.9922, + "fit_seconds": 146.929, + "host_train_compute_mean_s": 0.46024, + "checkpoint_seconds": 19.0491, + "layers": 5, + "trainable_parameters": 537427200, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": { + "stats": { + "calls_captured": 4, + "unique_graphs": 2 + }, + "frames": { + "ok": 2, + "total": 2 + }, + "inductor": { + "async_compile_cache_miss": 12, + "triton_bundler_save_static_autotuner": 4, + "triton_bundler_save_kernel": 48, + "fxgraph_cache_miss": 4, + "async_compile_cache_hit": 2 + }, + "aot_autograd": { + "autograd_cache_saved": 2, + "ok": 2, + "total": 2, + "autograd_cache_miss": 2 + } + }, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.2003, + "wait_fetch_s": 9.4391 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash2/warmup/False", + "seconds": 129.4135, + "fit_seconds": 139.8324, + "host_train_compute_mean_s": 0.434597, + "checkpoint_seconds": 19.7909, + "layers": 5, + "trainable_parameters": 558918912, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.2006, + "wait_fetch_s": 8.0094 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash2/warmup/True", + "seconds": 123.3042, + "fit_seconds": 133.2789, + "host_train_compute_mean_s": 0.405485, + "checkpoint_seconds": 20.058, + "layers": 5, + "trainable_parameters": 558918912, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.2007, + "wait_fetch_s": 8.4689 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash/repeat00-arm0/False", + "seconds": 136.0732, + "fit_seconds": 145.7039, + "host_train_compute_mean_s": 0.452634, + "checkpoint_seconds": 18.9498, + "layers": 5, + "trainable_parameters": 537427200, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.2001, + "wait_fetch_s": 10.0934 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash/repeat00-arm1/True", + "seconds": 128.8816, + "fit_seconds": 138.331, + "host_train_compute_mean_s": 0.433689, + "checkpoint_seconds": 18.4637, + "layers": 5, + "trainable_parameters": 537427200, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.2001, + "wait_fetch_s": 8.0357 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash/repeat00-arm2/True", + "seconds": 130.7867, + "fit_seconds": 140.4589, + "host_train_compute_mean_s": 0.433953, + "checkpoint_seconds": 19.0079, + "layers": 5, + "trainable_parameters": 537427200, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.3002, + "wait_fetch_s": 9.4303 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash/repeat00-arm3/False", + "seconds": 136.0565, + "fit_seconds": 145.755, + "host_train_compute_mean_s": 0.452805, + "checkpoint_seconds": 19.0598, + "layers": 5, + "trainable_parameters": 537427200, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.2002, + "wait_fetch_s": 9.953 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash/repeat01-arm0/False", + "seconds": 135.4919, + "fit_seconds": 145.3474, + "host_train_compute_mean_s": 0.452784, + "checkpoint_seconds": 19.063, + "layers": 5, + "trainable_parameters": 537427200, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.2004, + "wait_fetch_s": 9.5154 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash/repeat01-arm1/True", + "seconds": 130.5155, + "fit_seconds": 140.5098, + "host_train_compute_mean_s": 0.433516, + "checkpoint_seconds": 18.9191, + "layers": 5, + "trainable_parameters": 537427200, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.2004, + "wait_fetch_s": 9.6567 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash/repeat01-arm2/True", + "seconds": 130.9135, + "fit_seconds": 140.2981, + "host_train_compute_mean_s": 0.433584, + "checkpoint_seconds": 18.5826, + "layers": 5, + "trainable_parameters": 537427200, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.2004, + "wait_fetch_s": 9.886 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash/repeat01-arm3/False", + "seconds": 135.3908, + "fit_seconds": 144.9047, + "host_train_compute_mean_s": 0.453021, + "checkpoint_seconds": 18.5912, + "layers": 5, + "trainable_parameters": 537427200, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.2001, + "wait_fetch_s": 9.5205 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash2/repeat00-arm0/False", + "seconds": 127.7296, + "fit_seconds": 137.5912, + "host_train_compute_mean_s": 0.426954, + "checkpoint_seconds": 19.1591, + "layers": 5, + "trainable_parameters": 558918912, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.2002, + "wait_fetch_s": 8.3275 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash2/repeat00-arm1/True", + "seconds": 123.1932, + "fit_seconds": 133.598, + "host_train_compute_mean_s": 0.405299, + "checkpoint_seconds": 19.7945, + "layers": 5, + "trainable_parameters": 558918912, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.2001, + "wait_fetch_s": 9.1548 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash2/repeat00-arm2/True", + "seconds": 122.4172, + "fit_seconds": 132.6027, + "host_train_compute_mean_s": 0.405709, + "checkpoint_seconds": 19.426, + "layers": 5, + "trainable_parameters": 558918912, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.3002, + "wait_fetch_s": 8.3982 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash2/repeat00-arm3/False", + "seconds": 127.5555, + "fit_seconds": 137.3531, + "host_train_compute_mean_s": 0.426825, + "checkpoint_seconds": 19.2577, + "layers": 5, + "trainable_parameters": 558918912, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.2004, + "wait_fetch_s": 7.9983 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash2/repeat01-arm0/False", + "seconds": 127.5558, + "fit_seconds": 137.2869, + "host_train_compute_mean_s": 0.427199, + "checkpoint_seconds": 19.0223, + "layers": 5, + "trainable_parameters": 558918912, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.2004, + "wait_fetch_s": 8.0459 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash2/repeat01-arm1/True", + "seconds": 122.6024, + "fit_seconds": 132.2964, + "host_train_compute_mean_s": 0.405801, + "checkpoint_seconds": 19.0428, + "layers": 5, + "trainable_parameters": 558918912, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.3005, + "wait_fetch_s": 8.4095 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash2/repeat01-arm2/True", + "seconds": 123.3118, + "fit_seconds": 133.2074, + "host_train_compute_mean_s": 0.406068, + "checkpoint_seconds": 19.361, + "layers": 5, + "trainable_parameters": 558918912, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.3002, + "wait_fetch_s": 8.9062 + }, + "full_model_and_pipeline_checks": "PASS" + }, + { + "arm": "dflash2/repeat01-arm3/False", + "seconds": 127.0127, + "fit_seconds": 136.6376, + "host_train_compute_mean_s": 0.427011, + "checkpoint_seconds": 18.8315, + "layers": 5, + "trainable_parameters": 558918912, + "all_tensor_adam_steps": 256, + "logs": [ + 50, + 100, + 150, + 200, + 250 + ], + "compiler_delta": {}, + "capture_after_first_ack": 254, + "loader_wait": { + "wait_producer_s": 0.3006, + "wait_fetch_s": 7.5955 + }, + "full_model_and_pipeline_checks": "PASS" + } + ], + "limits": [ + "Primary pipeline interval is first actual HTTP capture dispatch through final durable ack and CUDA completion; it includes the intermediate step-128 checkpoint.", + "Full fit includes the step-256 checkpoint and cleanup. Server/model initialization and data preparation are outside these intervals.", + "Host train_compute diagnostic means cover five 50-step windows (250 of 256 steps), including detailed metrics at window ends. They are not CUDA kernel-only timings.", + "Frozen target samples and per-layer update hashes are sampled; full finiteness scans and optimizer participation checks cover every trainable tensor.", + "Observed timing results from four runs per mode; no claim of convergence, model quality or serving throughput." + ] +} diff --git a/docs/sections/benchmarks/sequence-packing-results/online-32step.json b/docs/sections/benchmarks/sequence-packing-results/online-32step.json new file mode 100644 index 000000000..0ab362017 --- /dev/null +++ b/docs/sections/benchmarks/sequence-packing-results/online-32step.json @@ -0,0 +1,568 @@ +{ + "created_utc": "2026-10-02T06:02:31.254497+00:00", + "torch_version": "2.13.0+cu130", + "gpu": "NVIDIA H200", + "timing_contract": "first capture dispatch -> final synchronous durable ack + CUDA sync; final checkpoint separately timed", + "scope": "actual target weights and target embeddings/head; freshly initialized draft; single-rank FSDP NO_SHARD", + "prompt_source": "/scratch/specforge-packing-e2e-20261002/sharegpt-prompts.jsonl", + "prompt_file_sha256": "d99771e31b4d6bc79bdeae630b689e95abbfbc5f858db7b6f39383cbd2461e20", + "capture_cache_policy": "canonical adapter generates a fresh extra_key per request attempt, forcing full prefill", + "producer": "canonical drive_producer in parallel thread, one worker/concurrency=1, bounded ref backlog", + "async_ack": false, + "summary": { + "dflash": { + "padded": { + "runs": 2, + "median_pipeline_seconds": 15.335569551214576, + "median_useful_tokens_per_second": 10810.815020745238, + "median_fit_seconds_with_checkpoint": 25.21502728294581 + }, + "packed": { + "runs": 2, + "median_pipeline_seconds": 14.732898173853755, + "median_useful_tokens_per_second": 11253.13099697949, + "median_fit_seconds_with_checkpoint": 24.90839832369238 + }, + "pipeline_speedup": 1.0409065053086686, + "fit_speedup_with_checkpoint": 1.0123102640028754 + }, + "dflash2": { + "padded": { + "runs": 2, + "median_pipeline_seconds": 14.386700802482665, + "median_useful_tokens_per_second": 11523.70540099725, + "median_fit_seconds_with_checkpoint": 25.107247687876225 + }, + "packed": { + "runs": 2, + "median_pipeline_seconds": 13.76857764646411, + "median_useful_tokens_per_second": 12041.194447433445, + "median_fit_seconds_with_checkpoint": 24.000815202482045 + }, + "pipeline_speedup": 1.0448937553239055, + "fit_speedup_with_checkpoint": 1.0460997876972011 + } + }, + "evidence_kind": "Derived 32-step report; preserves all per-run timing samples, including warmups, plus settings and lifecycle checks. Detailed per-capture events and original raw report are retained locally.", + "source_report_sha256": "d0284b6ab87ad26bba3c6624c4d4b203cb4f18fef08609476729d650b55f9be4", + "settings": { + "server_url": "http://127.0.0.1:31012", + "target_model": "/cluster-storage/models/Qwen3-4B", + "draft_config": "configs/qwen3-4b-dflash.json", + "prompts_path": "/scratch/specforge-packing-e2e-20261002/sharegpt-prompts.jsonl", + "algorithm": "both", + "capture_layers": null, + "draft_layers": 2, + "lengths": [ + 128, + 256, + 512, + 2048 + ], + "batch_size": 4, + "accumulation_steps": 1, + "anchors": 512, + "steps": 32, + "warmup_steps": 32, + "repeats": 1, + "seed": 1729, + "prompt_fraction": 0.25, + "learning_rate": 0.0001, + "objective_chunk_blocks": 128, + "capture_batch_size": 4, + "backlog": 8, + "log_interval": 50, + "save_interval": 0, + "teacher_metrics": true, + "dataloader_workers": 4, + "receive_buffers": "pinned", + "segment_mib": 1024, + "local_buffer_mib": 256, + "request_timeout": 300, + "dist_port": 29712, + "keep_checkpoints": false, + "work_dir": "/scratch/specforge-packing-e2e-20261002/full_v1", + "output": "/scratch/specforge-packing-e2e-20261002/full_v1.json" + }, + "settings_resolution_note": "draft_config overrides unused draft_layers=2; every resolved draft has five layers. prompts_path overrides synthetic lengths/prompt_fraction defaults and uses actual assistant masks.", + "target_dimensions": { + "model_type": "qwen3", + "num_hidden_layers": 36, + "hidden_size": 2560, + "intermediate_size": 9728, + "num_attention_heads": 32, + "num_key_value_heads": 8, + "vocab_size": 151936 + }, + "warmup_runs": [ + { + "architecture": "dflash", + "packing": false, + "label": "warmup", + "optimizer_steps": 32, + "microsteps": 32, + "samples": 128, + "durable_acked_samples": 128, + "useful_tokens": 165788, + "supervised_tokens": 131810, + "pipeline_seconds": 32.0647142175585, + "useful_tokens_per_second": 5170.418762354513, + "samples_per_second": 3.9919270488900143, + "capture_finished_seconds": 31.251980060711503, + "producer_finished_seconds": 31.253749769181013, + "trainer_fit_seconds_from_first_capture": 42.36139240115881, + "fit_wall_seconds": 42.36776242032647, + "model_and_runtime_setup_seconds": 9.31502272374928, + "checkpoint_seconds": 10.295483123511076, + "final_loss": 8.513822555541992, + "capture_calls_after_first_optimizer": 30, + "sustained_live_overlap_established": true, + "compiler_counter_delta": { + "inductor": { + "async_compile_cache_miss": 23, + "triton_bundler_save_static_autotuner": 6, + "triton_bundler_save_kernel": 72, + "fxgraph_cache_hit": 2, + "triton_bundler_load_static_autotuner": 3, + "fxgraph_cache_miss": 6, + "async_compile_cache_hit": 6 + }, + "unimplemented": {}, + "frames": { + "ok": 4, + "total": 4 + }, + "graph_break": {}, + "aot_autograd": { + "total": 4, + "autograd_cache_miss": 3, + "ok": 4, + "autograd_cache_saved": 3, + "autograd_cache_hit": 1 + }, + "stats": { + "calls_captured": 8, + "unique_graphs": 4 + } + }, + "full_warmup_replay": true, + "resolved_draft_layers": 5, + "publication_consumption_order_identical": true, + "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", + "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" + }, + { + "architecture": "dflash", + "packing": true, + "label": "warmup", + "optimizer_steps": 32, + "microsteps": 32, + "samples": 128, + "durable_acked_samples": 128, + "useful_tokens": 165788, + "supervised_tokens": 131810, + "pipeline_seconds": 15.39916873909533, + "useful_tokens_per_second": 10766.03567432171, + "samples_per_second": 8.312136984059032, + "capture_finished_seconds": 14.608486795797944, + "producer_finished_seconds": 14.610368017107248, + "trainer_fit_seconds_from_first_capture": 26.10293635353446, + "fit_wall_seconds": 26.1091186106205, + "model_and_runtime_setup_seconds": 9.027630560100079, + "checkpoint_seconds": 10.695947425439954, + "final_loss": 8.513858795166016, + "capture_calls_after_first_optimizer": 30, + "sustained_live_overlap_established": true, + "compiler_counter_delta": { + "stats": { + "calls_captured": 4, + "unique_graphs": 2 + }, + "inductor": { + "triton_bundler_load_static_autotuner": 6, + "async_compile_cache_miss": 18, + "fxgraph_cache_hit": 4, + "async_compile_cache_hit": 8 + }, + "unimplemented": {}, + "frames": { + "ok": 2, + "total": 2 + }, + "graph_break": {}, + "aot_autograd": { + "autograd_cache_hit": 2, + "total": 2, + "ok": 2 + } + }, + "full_warmup_replay": true, + "resolved_draft_layers": 5, + "publication_consumption_order_identical": true, + "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", + "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" + }, + { + "architecture": "dflash2", + "packing": false, + "label": "warmup", + "optimizer_steps": 32, + "microsteps": 32, + "samples": 128, + "durable_acked_samples": 128, + "useful_tokens": 165788, + "supervised_tokens": 131810, + "pipeline_seconds": 14.374922448769212, + "useful_tokens_per_second": 11533.140480642722, + "samples_per_second": 8.9043958641293, + "capture_finished_seconds": 13.643273144960403, + "producer_finished_seconds": 13.644904932007194, + "trainer_fit_seconds_from_first_capture": 24.717275140807033, + "fit_wall_seconds": 24.723629055544734, + "model_and_runtime_setup_seconds": 8.944378308951855, + "checkpoint_seconds": 10.341318672522902, + "final_loss": 9.105286598205566, + "capture_calls_after_first_optimizer": 30, + "sustained_live_overlap_established": true, + "compiler_counter_delta": { + "stats": {}, + "inductor": {}, + "unimplemented": {}, + "frames": {}, + "graph_break": {}, + "aot_autograd": {} + }, + "full_warmup_replay": true, + "resolved_draft_layers": 5, + "publication_consumption_order_identical": true, + "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", + "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" + }, + { + "architecture": "dflash2", + "packing": true, + "label": "warmup", + "optimizer_steps": 32, + "microsteps": 32, + "samples": 128, + "durable_acked_samples": 128, + "useful_tokens": 165788, + "supervised_tokens": 131810, + "pipeline_seconds": 13.88367784023285, + "useful_tokens_per_second": 11941.21629065541, + "samples_per_second": 9.219459099596426, + "capture_finished_seconds": 13.120847288519144, + "producer_finished_seconds": 13.122505459934473, + "trainer_fit_seconds_from_first_capture": 24.113164821639657, + "fit_wall_seconds": 24.11945860646665, + "model_and_runtime_setup_seconds": 8.892530964687467, + "checkpoint_seconds": 10.228421663865447, + "final_loss": 9.104482650756836, + "capture_calls_after_first_optimizer": 30, + "sustained_live_overlap_established": true, + "compiler_counter_delta": { + "stats": {}, + "inductor": {}, + "unimplemented": {}, + "frames": {}, + "graph_break": {}, + "aot_autograd": {} + }, + "full_warmup_replay": true, + "resolved_draft_layers": 5, + "publication_consumption_order_identical": true, + "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", + "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" + } + ], + "measured_runs": [ + { + "architecture": "dflash", + "packing": false, + "label": "repeat00-arm0", + "optimizer_steps": 32, + "microsteps": 32, + "samples": 128, + "durable_acked_samples": 128, + "useful_tokens": 165788, + "supervised_tokens": 131810, + "pipeline_seconds": 15.28223004937172, + "useful_tokens_per_second": 10848.416720883995, + "samples_per_second": 8.375740947916324, + "capture_finished_seconds": 14.491078296676278, + "producer_finished_seconds": 14.492836989462376, + "trainer_fit_seconds_from_first_capture": 25.171482319012284, + "fit_wall_seconds": 25.17781513929367, + "model_and_runtime_setup_seconds": 8.702558502554893, + "checkpoint_seconds": 9.880388440564275, + "final_loss": 8.514045715332031, + "capture_calls_after_first_optimizer": 30, + "sustained_live_overlap_established": true, + "compiler_counter_delta": { + "stats": {}, + "inductor": {}, + "unimplemented": {}, + "frames": {}, + "graph_break": {}, + "aot_autograd": {} + }, + "full_warmup_replay": true, + "resolved_draft_layers": 5, + "publication_consumption_order_identical": true, + "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", + "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" + }, + { + "architecture": "dflash", + "packing": true, + "label": "repeat00-arm1", + "optimizer_steps": 32, + "microsteps": 32, + "samples": 128, + "durable_acked_samples": 128, + "useful_tokens": 165788, + "supervised_tokens": 131810, + "pipeline_seconds": 14.797958767041564, + "useful_tokens_per_second": 11203.437082771698, + "samples_per_second": 8.649841644719626, + "capture_finished_seconds": 14.030732162296772, + "producer_finished_seconds": 14.03259969688952, + "trainer_fit_seconds_from_first_capture": 24.82643250748515, + "fit_wall_seconds": 24.83258822746575, + "model_and_runtime_setup_seconds": 8.715459797531366, + "checkpoint_seconds": 10.020536547526717, + "final_loss": 8.513858795166016, + "capture_calls_after_first_optimizer": 30, + "sustained_live_overlap_established": true, + "compiler_counter_delta": { + "stats": {}, + "inductor": {}, + "unimplemented": {}, + "frames": {}, + "graph_break": {}, + "aot_autograd": {} + }, + "full_warmup_replay": true, + "resolved_draft_layers": 5, + "publication_consumption_order_identical": true, + "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", + "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" + }, + { + "architecture": "dflash", + "packing": true, + "label": "repeat00-arm2", + "optimizer_steps": 32, + "microsteps": 32, + "samples": 128, + "durable_acked_samples": 128, + "useful_tokens": 165788, + "supervised_tokens": 131810, + "pipeline_seconds": 14.667837580665946, + "useful_tokens_per_second": 11302.82491118728, + "samples_per_second": 8.726576040678285, + "capture_finished_seconds": 13.881927194073796, + "producer_finished_seconds": 13.883811619132757, + "trainer_fit_seconds_from_first_capture": 24.99036413989961, + "fit_wall_seconds": 24.997150415554643, + "model_and_runtime_setup_seconds": 8.881739912554622, + "checkpoint_seconds": 10.314807986840606, + "final_loss": 8.513858795166016, + "capture_calls_after_first_optimizer": 30, + "sustained_live_overlap_established": true, + "compiler_counter_delta": { + "stats": {}, + "inductor": {}, + "unimplemented": {}, + "frames": {}, + "graph_break": {}, + "aot_autograd": {} + }, + "full_warmup_replay": true, + "resolved_draft_layers": 5, + "publication_consumption_order_identical": true, + "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", + "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" + }, + { + "architecture": "dflash", + "packing": false, + "label": "repeat00-arm3", + "optimizer_steps": 32, + "microsteps": 32, + "samples": 128, + "durable_acked_samples": 128, + "useful_tokens": 165788, + "supervised_tokens": 131810, + "pipeline_seconds": 15.388909053057432, + "useful_tokens_per_second": 10773.213320606481, + "samples_per_second": 8.317678631973543, + "capture_finished_seconds": 14.578291799873114, + "producer_finished_seconds": 14.579897599294782, + "trainer_fit_seconds_from_first_capture": 25.25857224687934, + "fit_wall_seconds": 25.26484971679747, + "model_and_runtime_setup_seconds": 8.783952347934246, + "checkpoint_seconds": 9.86840933188796, + "final_loss": 8.514045715332031, + "capture_calls_after_first_optimizer": 30, + "sustained_live_overlap_established": true, + "compiler_counter_delta": { + "stats": {}, + "inductor": {}, + "unimplemented": {}, + "frames": {}, + "graph_break": {}, + "aot_autograd": {} + }, + "full_warmup_replay": true, + "resolved_draft_layers": 5, + "publication_consumption_order_identical": true, + "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", + "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" + }, + { + "architecture": "dflash2", + "packing": false, + "label": "repeat00-arm0", + "optimizer_steps": 32, + "microsteps": 32, + "samples": 128, + "durable_acked_samples": 128, + "useful_tokens": 165788, + "supervised_tokens": 131810, + "pipeline_seconds": 14.375430628657341, + "useful_tokens_per_second": 11532.732777375208, + "samples_per_second": 8.904081088522851, + "capture_finished_seconds": 13.629519296810031, + "producer_finished_seconds": 13.631206966936588, + "trainer_fit_seconds_from_first_capture": 25.25617554783821, + "fit_wall_seconds": 25.263109516352415, + "model_and_runtime_setup_seconds": 8.989978183060884, + "checkpoint_seconds": 10.879672184586525, + "final_loss": 9.105286598205566, + "capture_calls_after_first_optimizer": 30, + "sustained_live_overlap_established": true, + "compiler_counter_delta": { + "stats": {}, + "inductor": {}, + "unimplemented": {}, + "frames": {}, + "graph_break": {}, + "aot_autograd": {} + }, + "full_warmup_replay": true, + "resolved_draft_layers": 5, + "publication_consumption_order_identical": true, + "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", + "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" + }, + { + "architecture": "dflash2", + "packing": true, + "label": "repeat00-arm1", + "optimizer_steps": 32, + "microsteps": 32, + "samples": 128, + "durable_acked_samples": 128, + "useful_tokens": 165788, + "supervised_tokens": 131810, + "pipeline_seconds": 13.817821264266968, + "useful_tokens_per_second": 11998.128853260645, + "samples_per_second": 9.26339960200595, + "capture_finished_seconds": 13.074846714735031, + "producer_finished_seconds": 13.076601503416896, + "trainer_fit_seconds_from_first_capture": 24.060468524694443, + "fit_wall_seconds": 24.066856909543276, + "model_and_runtime_setup_seconds": 9.183672599494457, + "checkpoint_seconds": 10.24164686910808, + "final_loss": 9.104482650756836, + "capture_calls_after_first_optimizer": 30, + "sustained_live_overlap_established": true, + "compiler_counter_delta": { + "stats": {}, + "inductor": {}, + "unimplemented": {}, + "frames": {}, + "graph_break": {}, + "aot_autograd": {} + }, + "full_warmup_replay": true, + "resolved_draft_layers": 5, + "publication_consumption_order_identical": true, + "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", + "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" + }, + { + "architecture": "dflash2", + "packing": true, + "label": "repeat00-arm2", + "optimizer_steps": 32, + "microsteps": 32, + "samples": 128, + "durable_acked_samples": 128, + "useful_tokens": 165788, + "supervised_tokens": 131810, + "pipeline_seconds": 13.719334028661251, + "useful_tokens_per_second": 12084.260041606247, + "samples_per_second": 9.329898939160854, + "capture_finished_seconds": 12.968088772147894, + "producer_finished_seconds": 12.969761220738292, + "trainer_fit_seconds_from_first_capture": 23.941161880269647, + "fit_wall_seconds": 23.947430515661836, + "model_and_runtime_setup_seconds": 8.923571102321148, + "checkpoint_seconds": 10.220690120011568, + "final_loss": 9.104482650756836, + "capture_calls_after_first_optimizer": 30, + "sustained_live_overlap_established": true, + "compiler_counter_delta": { + "stats": {}, + "inductor": {}, + "unimplemented": {}, + "frames": {}, + "graph_break": {}, + "aot_autograd": {} + }, + "full_warmup_replay": true, + "resolved_draft_layers": 5, + "publication_consumption_order_identical": true, + "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", + "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" + }, + { + "architecture": "dflash2", + "packing": false, + "label": "repeat00-arm3", + "optimizer_steps": 32, + "microsteps": 32, + "samples": 128, + "durable_acked_samples": 128, + "useful_tokens": 165788, + "supervised_tokens": 131810, + "pipeline_seconds": 14.397970976307988, + "useful_tokens_per_second": 11514.678024619294, + "samples_per_second": 8.890141549154762, + "capture_finished_seconds": 13.634843789041042, + "producer_finished_seconds": 13.637094628065825, + "trainer_fit_seconds_from_first_capture": 24.958319827914238, + "fit_wall_seconds": 24.96457778289914, + "model_and_runtime_setup_seconds": 8.878480760380626, + "checkpoint_seconds": 10.559298420324922, + "final_loss": 9.105286598205566, + "capture_calls_after_first_optimizer": 30, + "sustained_live_overlap_established": true, + "compiler_counter_delta": { + "stats": {}, + "inductor": {}, + "unimplemented": {}, + "frames": {}, + "graph_break": {}, + "aot_autograd": {} + }, + "full_warmup_replay": true, + "resolved_draft_layers": 5, + "publication_consumption_order_identical": true, + "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", + "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" + } + ] +} diff --git a/docs/sections/benchmarks/sequence-packing-results/target-parameter-inventory.json b/docs/sections/benchmarks/sequence-packing-results/target-parameter-inventory.json new file mode 100644 index 000000000..0c8be9f3a --- /dev/null +++ b/docs/sections/benchmarks/sequence-packing-results/target-parameter-inventory.json @@ -0,0 +1,3086 @@ +{ + "target_model": "/cluster-storage/models/Qwen3-4B", + "unique_saved_parameters": 4022468096, + "parameter_tensors": 398, + "layer_indices": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7, + 8, + 9, + 10, + 11, + 12, + 13, + 14, + 15, + 16, + 17, + 18, + 19, + 20, + 21, + 22, + 23, + 24, + 25, + 26, + 27, + 28, + 29, + 30, + 31, + 32, + 33, + 34, + 35 + ], + "all_36_layers_present": true, + "config_sha256": "8ba006f74fecfaaeb392872a60f4a480e7ec9860153d2e1b769ec81f9a147f8a", + "parameters": [ + { + "name": "model.embed_tokens.weight", + "shape": [ + 151936, + 2560 + ], + "numel": 388956160 + }, + { + "name": "model.layers.0.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.0.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.0.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.0.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.0.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.0.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.0.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.0.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.0.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.0.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.0.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.1.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.1.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.1.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.1.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.1.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.1.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.1.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.1.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.1.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.1.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.1.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.10.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.10.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.10.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.10.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.10.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.10.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.10.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.10.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.10.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.10.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.10.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.11.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.11.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.11.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.11.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.11.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.11.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.11.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.11.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.11.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.11.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.11.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.12.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.12.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.12.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.12.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.12.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.12.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.12.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.12.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.12.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.12.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.12.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.13.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.13.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.13.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.13.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.13.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.13.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.13.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.13.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.13.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.13.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.13.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.14.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.14.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.14.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.14.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.14.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.14.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.14.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.14.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.14.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.14.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.14.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.15.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.15.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.15.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.15.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.15.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.15.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.15.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.15.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.2.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.2.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.2.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.2.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.2.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.2.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.2.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.2.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.2.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.2.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.2.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.3.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.3.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.3.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.3.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.3.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.3.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.3.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.3.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.3.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.3.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.3.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.4.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.4.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.4.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.4.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.4.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.4.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.4.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.4.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.4.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.4.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.4.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.5.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.5.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.5.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.5.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.5.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.5.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.5.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.5.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.5.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.5.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.5.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.6.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.6.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.6.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.6.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.6.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.6.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.6.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.6.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.6.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.6.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.6.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.7.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.7.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.7.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.7.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.7.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.7.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.7.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.7.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.7.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.7.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.7.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.8.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.8.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.8.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.8.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.8.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.8.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.8.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.8.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.8.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.8.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.8.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.9.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.9.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.9.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.9.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.9.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.9.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.9.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.9.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.9.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.9.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.9.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.15.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.15.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.15.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.16.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.16.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.16.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.16.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.16.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.16.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.16.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.16.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.16.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.16.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.16.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.17.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.17.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.17.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.17.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.17.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.17.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.17.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.17.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.17.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.17.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.17.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.18.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.18.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.18.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.18.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.18.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.18.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.18.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.18.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.18.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.18.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.18.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.19.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.19.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.19.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.19.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.19.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.19.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.19.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.19.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.19.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.19.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.19.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.20.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.20.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.20.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.20.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.20.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.20.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.20.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.20.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.20.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.20.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.20.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.21.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.21.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.21.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.21.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.21.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.21.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.21.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.21.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.21.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.21.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.21.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.22.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.22.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.22.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.22.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.22.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.22.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.22.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.22.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.22.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.22.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.22.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.23.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.23.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.23.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.23.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.23.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.23.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.23.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.23.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.23.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.23.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.23.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.24.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.24.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.24.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.24.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.24.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.24.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.24.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.24.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.24.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.24.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.24.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.25.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.25.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.25.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.25.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.25.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.25.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.25.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.25.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.25.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.25.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.25.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.26.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.26.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.26.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.26.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.26.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.26.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.26.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.26.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.26.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.26.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.26.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.27.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.27.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.27.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.27.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.27.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.27.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.27.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.27.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.27.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.27.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.27.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.28.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.28.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.28.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.28.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.28.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.28.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.28.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.28.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.28.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.28.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.28.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.29.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.29.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.29.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.29.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.29.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.29.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.29.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.29.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.29.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.29.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.29.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.30.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.30.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.30.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.30.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.30.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.30.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.30.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.30.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.30.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.30.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.30.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.31.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.31.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.31.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.31.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.31.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.31.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.31.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.31.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.31.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.31.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.31.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.32.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.32.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.32.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.32.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.32.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.32.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.32.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.32.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.32.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.32.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.32.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.33.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.33.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.33.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.33.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.33.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.33.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.33.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.33.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.33.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.33.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.33.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.34.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.34.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.34.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.34.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.34.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.34.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.34.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.34.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.34.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.34.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.34.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.35.mlp.gate_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.35.self_attn.k_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.35.self_attn.k_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.35.self_attn.o_proj.weight", + "shape": [ + 2560, + 4096 + ], + "numel": 10485760 + }, + { + "name": "model.layers.35.self_attn.q_norm.weight", + "shape": [ + 128 + ], + "numel": 128 + }, + { + "name": "model.layers.35.self_attn.q_proj.weight", + "shape": [ + 4096, + 2560 + ], + "numel": 10485760 + }, + { + "name": "model.layers.35.self_attn.v_proj.weight", + "shape": [ + 1024, + 2560 + ], + "numel": 2621440 + }, + { + "name": "model.layers.35.input_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.layers.35.mlp.down_proj.weight", + "shape": [ + 2560, + 9728 + ], + "numel": 24903680 + }, + { + "name": "model.layers.35.mlp.up_proj.weight", + "shape": [ + 9728, + 2560 + ], + "numel": 24903680 + }, + { + "name": "model.layers.35.post_attention_layernorm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + }, + { + "name": "model.norm.weight", + "shape": [ + 2560 + ], + "numel": 2560 + } + ] +} diff --git a/examples/configs/README.md b/examples/configs/README.md index 905c4eb10..3628ff700 100644 --- a/examples/configs/README.md +++ b/examples/configs/README.md @@ -277,6 +277,7 @@ Common fields: | `training.max_steps` | `null` | Positive hard stop in optimizer steps. If it is set while `total_steps` is omitted, it is also the fallback schedule horizon. | | `training.total_steps` | `null` | Positive optimizer/loss schedule horizon; it does not itself stop an online stream. A finite online disaggregated run may omit both fields: the producer publishes the exact horizon derived from prepared prompts, epochs, DP size, batch size, and accumulation. | | `training.batch_size` | `1` | Per-rank microbatch size. P-EAGLE and USP require 1. | +| `training.sequence_packing` | `false` | Pack each original microbatch for text EAGLE3, DFlash, or DFlash2, offline or online. Requires FlexAttention; incompatible with compact teacher and loss-position trimming. DFlash/DFlash2 LK objectives are supported; EAGLE3 LK is not. See [sequence packing](../../docs/sections/basic_usage/training.md#sequence-packing). | | `training.accumulation_steps` | `1` | Positive microbatches per optimizer update. | | `training.fsdp_sharding` | `SHARD_GRAD_OP` | Trainer FSDP mode: `SHARD_GRAD_OP`, `FULL_SHARD`, or `NO_SHARD`. | | `training.learning_rate` | `1e-4` | Positive peak learning rate. | diff --git a/scripts/benchmark_dflash_sequence_packing.py b/scripts/benchmark_dflash_sequence_packing.py new file mode 100755 index 000000000..421fe9bee --- /dev/null +++ b/scripts/benchmark_dflash_sequence_packing.py @@ -0,0 +1,7 @@ +#!/usr/bin/env python3 +"""Entrypoint for the DFlash/DFlash2 sequence-packing benchmark.""" + +from specforge.benchmarks.benchmark_dflash_sequence_packing import main + +if __name__ == "__main__": + main() diff --git a/scripts/benchmark_online_sequence_packing.py b/scripts/benchmark_online_sequence_packing.py new file mode 100755 index 000000000..94457aa2e --- /dev/null +++ b/scripts/benchmark_online_sequence_packing.py @@ -0,0 +1,7 @@ +#!/usr/bin/env python3 +"""Entrypoint for the real online sequence-packing pipeline benchmark.""" + +from specforge.benchmarks.benchmark_online_sequence_packing import main + +if __name__ == "__main__": + main() diff --git a/scripts/benchmark_sequence_packing.py b/scripts/benchmark_sequence_packing.py new file mode 100755 index 000000000..0b33ccd5f --- /dev/null +++ b/scripts/benchmark_sequence_packing.py @@ -0,0 +1,7 @@ +#!/usr/bin/env python3 +"""Entrypoint for the single-GPU EAGLE3 sequence-packing benchmark.""" + +from specforge.benchmarks.benchmark_sequence_packing import main + +if __name__ == "__main__": + main() diff --git a/specforge/algorithms/common/dflash_family_model.py b/specforge/algorithms/common/dflash_family_model.py index 719b12232..af8ee339d 100644 --- a/specforge/algorithms/common/dflash_family_model.py +++ b/specforge/algorithms/common/dflash_family_model.py @@ -13,6 +13,10 @@ from specforge.core.chunking import checkpointed_chunk_reduce from specforge.modeling.draft.dflash import DFlashDraftModel from specforge.modeling.draft.flex_attention_backend import flex_attention_backend +from specforge.modeling.packed_dflash import ( + PackedDFlashLayout, + create_packed_dflash_block_mask, +) try: from torch.nn.attention.flex_attention import BlockMask, create_block_mask @@ -252,6 +256,7 @@ def create_dflash_sdpa_mask( block_size, device, sliding_window: Optional[int] = None, + context_start_positions: Optional[torch.Tensor] = None, ): """Construct a full or sliding dense boolean DFlash mask.""" @@ -274,6 +279,11 @@ def create_dflash_sdpa_mask( ) mask_context = (kv_indices < S) & (kv_indices < anchor_expanded) + if context_start_positions is not None: + context_starts = context_start_positions.view(B, 1, N, 1).repeat_interleave( + block_size, dim=2 + ) + mask_context = mask_context & (kv_indices >= context_starts) if sliding_window is not None: # The current draft token occupies one slot in the window. context_lower_bound = anchor_expanded + q_block_offsets - (sliding_window - 1) @@ -300,6 +310,7 @@ def create_dflash_block_mask( device: torch.device, flex_block_size=None, sliding_window: Optional[int] = None, + context_start_positions: Optional[torch.Tensor] = None, ): """Construct a full or sliding Flex Attention mask for DFlash training.""" @@ -316,6 +327,10 @@ def dflash_mask_mod(b, h, q_idx, kv_idx): # Strictly less than: matches inference where target_hidden[anchor_pos] # is not available as context. mask_context = is_context & (kv_idx < anchor_pos) + if context_start_positions is not None: + mask_context = mask_context & ( + kv_idx >= context_start_positions[b, safe_q_block_id] + ) if sliding_window is not None: # The current draft token occupies one slot in the window. context_lower_bound = anchor_pos + q_block_offset - (sliding_window - 1) @@ -336,6 +351,18 @@ def dflash_mask_mod(b, h, q_idx, kv_idx): Q_LEN = N * block_size KV_LEN = S + N * block_size + if context_start_positions is not None: + return create_packed_dflash_block_mask( + anchor_positions, + block_keep_mask, + context_start_positions, + S, + block_size, + dflash_mask_mod, + block_size=flex_block_size if flex_block_size is not None else 128, + sliding_window=sliding_window, + ) + kwargs = {} if flex_block_size is not None: kwargs["BLOCK_SIZE"] = flex_block_size @@ -552,10 +579,13 @@ def _aligned_target_hidden( self, target_last_hidden_states: torch.Tensor, safe_label_indices: torch.Tensor, + minimum_indices: Optional[torch.Tensor] = None, ) -> torch.Tensor: """Gather the frozen target state that predicts each hard label.""" target_pred_indices = (safe_label_indices - 1).clamp(min=0) + if minimum_indices is not None: + target_pred_indices = torch.maximum(target_pred_indices, minimum_indices) batch_size = target_last_hidden_states.shape[0] hidden_size = target_last_hidden_states.shape[-1] gather_indices = target_pred_indices.reshape(batch_size, -1, 1).expand( @@ -700,16 +730,42 @@ def _forward_draft_blocks( hidden_states: torch.Tensor, loss_mask: torch.Tensor, max_valid_anchors: Optional[int] = None, + packed_layout: Optional[PackedDFlashLayout] = None, + valid_anchor_counts: Optional[Tuple[int, ...]] = None, ) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]: bsz, seq_len = input_ids.shape device = input_ids.device + sampling_mask = ( + loss_mask + if packed_layout is None + else packed_layout.padded_loss_mask(loss_mask) + ) anchor_positions, block_keep_mask = self._sample_anchor_positions( - seq_len, - loss_mask, + sampling_mask.shape[1], + sampling_mask, device, max_valid_anchors=max_valid_anchors, ) + local_anchor_positions = anchor_positions + if packed_layout is not None: + anchor_positions = packed_layout.pack_anchors(anchor_positions) + block_keep_mask = block_keep_mask.reshape(1, -1) + original_anchor_positions = anchor_positions + original_block_keep_mask = block_keep_mask + compact_indices = None + if packed_layout is not None and valid_anchor_counts is not None: + width = local_anchor_positions.shape[1] + compact_indices = packed_layout.compact_anchor_indices( + valid_anchor_counts, width + ) + if compact_indices.numel() == 0: + raise ValueError("DFlash packing requires at least one valid anchor") + anchor_positions = anchor_positions.index_select(1, compact_indices) + block_keep_mask = block_keep_mask.index_select(1, compact_indices) + local_anchor_positions = local_anchor_positions.reshape(1, -1).index_select( + 1, compact_indices + ) noise_embedding = self._create_noise_embed( input_ids, anchor_positions, block_keep_mask @@ -717,8 +773,12 @@ def _forward_draft_blocks( context_position_ids = ( torch.arange(seq_len, device=device).unsqueeze(0).expand(bsz, -1) + if packed_layout is None + else packed_layout.tokens.positions.unsqueeze(0) + ) + draft_position_ids = self._create_position_ids(local_anchor_positions).reshape( + bsz, -1 ) - draft_position_ids = self._create_position_ids(anchor_positions) full_position_ids = torch.cat([context_position_ids, draft_position_ids], dim=1) mask_builder = ( @@ -733,6 +793,10 @@ def _forward_draft_blocks( "block_size": self.block_size, "device": device, } + if packed_layout is not None: + mask_args["context_start_positions"] = packed_layout.anchor_starts( + anchor_positions + ) if ( self.attention_backend == "flex_attention" and flex_attention_backend() == "FLASH" @@ -781,7 +845,17 @@ def _forward_draft_blocks( attention_mask=dflash_attn_mask, **draft_kwargs, ) - return anchor_positions, block_keep_mask, output_hidden + if compact_indices is not None: + query_rows = ( + compact_indices[:, None] * self.block_size + + torch.arange(self.block_size, device=device) + ).reshape(-1) + output_hidden = output_hidden.new_zeros( + 1, + original_anchor_positions.shape[1] * self.block_size, + output_hidden.shape[-1], + ).index_copy(1, query_rows, output_hidden) + return original_anchor_positions, original_block_keep_mask, output_hidden def _selector_chunk_terms( self, @@ -1495,6 +1569,8 @@ def forward( max_valid_anchors: Optional[int] = None, selector_loss_alpha: Optional[float] = None, collect_detailed_metrics: bool = True, + sequence_lengths=None, + valid_anchor_counts: Optional[Tuple[int, ...]] = None, ) -> Tuple[torch.Tensor, torch.Tensor, Dict[str, object]]: """Parallel block-wise training forward pass; returns (loss, accuracy, metrics) — same shape as Domino's forward.""" @@ -1505,11 +1581,38 @@ def forward( bsz, seq_len = input_ids.shape device = input_ids.device + packed_layout = None + if sequence_lengths is not None: + if self.attention_backend != "flex_attention": + raise ValueError( + "DFlash sequence packing currently requires flex_attention" + ) + if ( + bsz != 1 + or hidden_states.shape[:2] != input_ids.shape + or loss_mask.shape != input_ids.shape + ): + raise ValueError( + "DFlash sequence packing requires aligned single-row inputs" + ) + packed_layout = PackedDFlashLayout.from_lengths( + sequence_lengths, seq_len, device + ) + + block_kwargs = ( + {} + if packed_layout is None + else { + "packed_layout": packed_layout, + "valid_anchor_counts": valid_anchor_counts, + } + ) anchor_positions, block_keep_mask, output_hidden = self._forward_draft_blocks( input_ids=input_ids, hidden_states=hidden_states, loss_mask=loss_mask, max_valid_anchors=max_valid_anchors, + **block_kwargs, ) # --- Labels: same-position prediction (position k predicts token anchor+k) --- @@ -1517,6 +1620,12 @@ def forward( label_indices = anchor_positions.unsqueeze(-1) + label_offsets valid_label_mask = label_indices < seq_len safe_label_indices = label_indices.clamp(max=seq_len - 1) + if packed_layout is not None: + document_ends = packed_layout.anchor_ends(anchor_positions).unsqueeze(-1) + valid_label_mask = valid_label_mask & (label_indices < document_ends) + # Even masked tail labels and selector predecessors stay inside their + # document; none can import a neighboring document's token IDs. + safe_label_indices = torch.minimum(safe_label_indices, document_ends - 1) target_ids = torch.gather( input_ids.unsqueeze(1).expand(-1, anchor_positions.size(1), -1), @@ -1554,6 +1663,15 @@ def forward( self._aligned_target_hidden( target_last_hidden_states, safe_label_indices, + **( + { + "minimum_indices": packed_layout.anchor_starts( + anchor_positions + ).unsqueeze(-1) + } + if packed_layout is not None + else {} + ), ) if ( target_last_hidden_states is not None @@ -1562,6 +1680,27 @@ def forward( ) else None ) + if packed_layout is not None: + # Preserve the original [documents, anchors, block] reduction axes. + # D-PACE normalizes anchors per document; selector and walk metrics + # also use these axes. Packing changes only the backbone execution. + original_batch = len(packed_layout.lengths) + anchors_per_document = anchor_positions.shape[1] // original_batch + shape = (original_batch, anchors_per_document, self.block_size) + hidden_4d = hidden_4d.reshape(*shape, hidden_4d.shape[-1]) + target_ids = target_ids.reshape(shape) + predecessor_ids = predecessor_ids.reshape(shape) + weight_mask = weight_mask.reshape(shape) + if aligned_target_hidden is not None: + aligned_target_hidden = aligned_target_hidden.reshape( + *shape, aligned_target_hidden.shape[-1] + ) + anchor_positions = packed_layout.tokens.positions[anchor_positions].reshape( + original_batch, anchors_per_document + ) + block_keep_mask = block_keep_mask.reshape( + original_batch, anchors_per_document + ) sequence_anchor_scale = None if self.loss_type in _DPACE_LOSS_TYPES: sequence_anchor_scale = self._sequence_anchor_scale(weight_mask) diff --git a/specforge/algorithms/common/hidden_states_data.py b/specforge/algorithms/common/hidden_states_data.py index 82d7fd63d..1549d59e3 100644 --- a/specforge/algorithms/common/hidden_states_data.py +++ b/specforge/algorithms/common/hidden_states_data.py @@ -178,6 +178,47 @@ def build_collator(): ) +def build_packed_collator(): + """Pack DFlash/DFlash2 context features after each capture is materialized.""" + return PackedHiddenStatesCollator() + + +class PackedHiddenStatesCollator: + def __call__(self, features): + import torch + + if not features: + raise ValueError("cannot pack an empty feature batch") + required = ("input_ids", "loss_mask", "hidden_states") + keys = list(required) + teacher_key = "target_last_hidden_states" + present = [teacher_key in feature for feature in features] + if any(present) and not all(present): + raise KeyError( + f"optional feature {teacher_key!r} must be present in every sample or none" + ) + if all(present): + keys.append(teacher_key) + lengths = [] + for feature in features: + missing = set(keys) - feature.keys() + if missing: + raise KeyError(f"packed sample is missing features: {sorted(missing)}") + ids = feature["input_ids"] + if ids.ndim != 2 or ids.shape[0] != 1 or ids.shape[1] == 0: + raise ValueError("packing requires nonempty [1, length] input_ids") + length = ids.shape[1] + for key in keys: + tensor = feature[key] + ndim = 2 if key in ("input_ids", "loss_mask") else 3 + if tensor.ndim != ndim or tensor.shape[:2] != (1, length): + raise ValueError(f"packing requires aligned unbatched {key}") + lengths.append(length) + batch = {key: torch.cat([f[key] for f in features], dim=1) for key in keys} + batch["sequence_lengths"] = torch.tensor(lengths, dtype=torch.long) + return batch + + def build_dspark_collator(): return _padded_collator( ("input_ids", "loss_mask", "hidden_states", "target_last_hidden_states") @@ -252,6 +293,7 @@ def build_mtp_collator(): "DSPARK_NORMALIZER_ID", "MTP_NORMALIZER_ID", "NORMALIZER_ID", + "PackedHiddenStatesCollator", "build_collator", "build_dspark_collator", "build_dspark_offline_normalizer", @@ -261,6 +303,7 @@ def build_mtp_collator(): "build_mtp_offline_reader", "build_offline_normalizer", "build_offline_reader", + "build_packed_collator", "normalize_dspark_offline_sample", "normalize_mtp_offline_sample", "normalize_offline_sample", diff --git a/specforge/algorithms/common/providers.py b/specforge/algorithms/common/providers.py index d0c22f702..9edc83119 100644 --- a/specforge/algorithms/common/providers.py +++ b/specforge/algorithms/common/providers.py @@ -491,10 +491,15 @@ class OfflineDataProvider: build_normalizer: Factory build_collator: Factory capture_layout: OfflineCaptureLayout | None = None + build_packed_collator: Factory | None = None def __post_init__(self) -> None: _non_empty(self.modality, field_name="modality") _non_empty(self.normalizer_id, field_name="normalizer_id") + if self.build_packed_collator is not None and not callable( + self.build_packed_collator + ): + raise TypeError("build_packed_collator must be callable or None") if self.capture_layout is not None and not isinstance( self.capture_layout, OfflineCaptureLayout, @@ -581,6 +586,7 @@ class ServerStreamingProvider: build_collator: Factory build_input_adapter: Factory | None = None select_layout: Factory | None = None + build_packed_collator: Factory | None = None def __post_init__(self) -> None: _non_empty(self.modality, field_name="modality") @@ -594,6 +600,10 @@ def __post_init__(self) -> None: raise TypeError("layout must be a ServerCaptureLayout") if not callable(self.build_collator): raise TypeError("build_collator must be callable") + if self.build_packed_collator is not None and not callable( + self.build_packed_collator + ): + raise TypeError("build_packed_collator must be callable or None") if self.build_input_adapter is not None and not callable( self.build_input_adapter ): diff --git a/specforge/algorithms/contracts.py b/specforge/algorithms/contracts.py index 4d323a128..e3a8df111 100644 --- a/specforge/algorithms/contracts.py +++ b/specforge/algorithms/contracts.py @@ -244,6 +244,7 @@ class AlgorithmCapabilities: #: ``training.dflash_teacher_metrics=false`` can drop the teacher-only #: final hidden state from the streaming capture. supports_teacher_metrics_opt_out: bool = False + supports_packed_lk_loss: bool = False def __post_init__(self) -> None: attention_backends = _normalized_names( @@ -262,6 +263,7 @@ def __post_init__(self) -> None: "supports_vocab_mapping", "allows_aux_layer_override", "supports_teacher_metrics_opt_out", + "supports_packed_lk_loss", ): if not isinstance(getattr(self, field_name), bool): raise TypeError(f"{field_name} must be a bool") diff --git a/specforge/algorithms/dflash/providers.py b/specforge/algorithms/dflash/providers.py index ff5a824bf..c1a9044f0 100644 --- a/specforge/algorithms/dflash/providers.py +++ b/specforge/algorithms/dflash/providers.py @@ -14,6 +14,7 @@ build_collator, build_offline_normalizer, build_offline_reader, + build_packed_collator, ) from specforge.algorithms.common.providers import ( AlgorithmProviders, @@ -209,6 +210,7 @@ def algorithm_spec() -> AlgorithmSpec: capabilities=AlgorithmCapabilities( attention_backends={"eager", "sdpa", "flex_attention"}, supports_teacher_metrics_opt_out=True, + supports_packed_lk_loss=True, ), ) @@ -260,6 +262,7 @@ def algorithm_providers() -> AlgorithmProviders: build_reader=partial(build_offline_reader, ALGORITHM_NAME), build_normalizer=build_offline_normalizer, build_collator=collator, + build_packed_collator=build_packed_collator, ), ), server_streaming=( @@ -270,6 +273,7 @@ def algorithm_providers() -> AlgorithmProviders: layout=SERVER_CAPTURE_LAYOUT, build_collator=collator, select_layout=select_server_capture_layout, + build_packed_collator=build_packed_collator, ), ), ) diff --git a/specforge/algorithms/eagle3/data.py b/specforge/algorithms/eagle3/data.py index 7521fcfeb..c0979d77b 100644 --- a/specforge/algorithms/eagle3/data.py +++ b/specforge/algorithms/eagle3/data.py @@ -87,17 +87,86 @@ def build_offline_collator(): return DataCollatorWithPadding() +class DataCollatorWithPacking: + """Pack one logical microbatch without changing its samples or loss weight. + + Each sample is a normalized, unpadded text feature with batch dimension one. + The model uses ``sequence_lengths`` for attention and TTT boundaries. Keeping + the original padded denominator makes this an execution optimization, not + a change to the EAGLE3 objective or optimizer schedule. + """ + + def __call__(self, features): + import torch + + if not features: + raise ValueError("cannot pack an empty feature batch") + keys = ("input_ids", "loss_mask", "hidden_state", "target", "attention_mask") + lengths = [] + for feature in features: + missing = set(keys) - feature.keys() + if missing: + raise KeyError(f"packed sample is missing features: {sorted(missing)}") + ids = feature["input_ids"] + if ids.ndim != 2 or ids.shape[0] != 1 or ids.shape[1] == 0: + raise ValueError("packing requires nonempty [1, length] input_ids") + length = ids.shape[1] + for key in keys: + tensor = feature[key] + ndim = 3 if key in ("hidden_state", "target") else 2 + if tensor.ndim != ndim or tensor.shape[:2] != (1, length): + raise ValueError(f"packing requires aligned unbatched {key}") + if not bool((feature["attention_mask"] == 1).all()): + raise ValueError("packing requires unpadded samples") + if "position_ids" in feature: + expected = torch.arange(length, device=ids.device).unsqueeze(0) + if not torch.equal(feature["position_ids"], expected): + raise ValueError("packing supports standard text position_ids only") + lengths.append(length) + + batch = {key: torch.cat([f[key] for f in features], dim=1) for key in keys} + batch["position_ids"] = torch.cat( + [torch.arange(n, device=batch["input_ids"].device) for n in lengths] + ).unsqueeze(0) + # These small descriptors stay on the host until the strategy builds + # the device layout; no GPU scalar synchronization is needed. + batch["sequence_lengths"] = torch.tensor(lengths, dtype=torch.long) + batch["loss_denominator"] = torch.tensor( + len(lengths) * max(lengths), dtype=torch.long + ) + return batch + + +def build_packed_collator(): + return DataCollatorWithPacking() + + def build_server_collator(): from specforge.algorithms.common.collation import concatenate_features return concatenate_features +def build_padded_server_collator(): + """Accept ragged, unshifted EAGLE3 features from separate capture requests.""" + from specforge.algorithms.common.collation import pad_and_concatenate_features + + keys = ("input_ids", "attention_mask", "loss_mask", "hidden_state", "target") + return partial( + pad_and_concatenate_features, + sequence_axes={key: 1 for key in keys}, + required_keys=keys, + ) + + __all__ = [ + "DataCollatorWithPacking", "NORMALIZER_ID", "build_offline_collator", "build_offline_normalizer", "build_offline_reader", + "build_packed_collator", + "build_padded_server_collator", "build_server_collator", "normalize_offline_sample", ] diff --git a/specforge/algorithms/eagle3/model.py b/specforge/algorithms/eagle3/model.py index d89cf5d0c..bfcd21172 100644 --- a/specforge/algorithms/eagle3/model.py +++ b/specforge/algorithms/eagle3/model.py @@ -22,7 +22,7 @@ """EAGLE3 training model implementation.""" -from typing import Callable, List, Optional, Tuple +from typing import Callable, List, Optional, Tuple, Union import torch import torch.nn as nn @@ -37,6 +37,7 @@ from specforge.core.lk_loss import compute_acceptance_rate, compute_lk_loss from specforge.core.loss import LogSoftmaxLoss from specforge.modeling.draft import Eagle3DraftModel +from specforge.modeling.packed_sequence import PackedSequenceLayout from specforge.utils import padding @@ -262,6 +263,8 @@ def forward( target_head_weight: Optional[torch.Tensor] = None, compact_teacher_chunk_size: int = DEFAULT_VOCAB_CHUNK_SIZE, trim_loss_positions: bool = False, + sequence_lengths: Optional[Union[torch.Tensor, Tuple[int, ...]]] = None, + loss_denominator: Optional[Union[torch.Tensor, int]] = None, ) -> Tuple[ List[torch.Tensor], List[torch.Tensor], @@ -285,7 +288,42 @@ def forward( states in draft-vocab space and ``target`` is ignored. trim_loss_positions: compute the teacher, draft logits and loss only at supervised positions when the batch/objective supports it. + sequence_lengths: CPU int64 document lengths (or a Python tuple) for a packed single row. + The strategy has already shifted input and target fields per document. + loss_denominator: optional CPU scalar equal to document count times + the longest document, preserving the padded batch's mean loss. """ + packed_layout = None + if sequence_lengths is not None: + if self.attention_backend != "flex_attention": + raise ValueError("sequence packing currently requires flex_attention") + if ( + trim_loss_positions + or target_hidden_for_compact is not None + or self.lk_loss_type is not None + ): + raise ValueError( + "sequence packing does not support trim, compact teacher, or LK loss" + ) + if input_ids.shape[0] != 1 or hidden_states.shape[:2] != input_ids.shape: + raise ValueError("sequence packing requires one concatenated batch row") + packed_layout = PackedSequenceLayout.from_lengths( + sequence_lengths, input_ids.shape[1], hidden_states.device + ) + if loss_denominator is not None: + if isinstance(loss_denominator, torch.Tensor) and ( + loss_denominator.device.type != "cpu" + or loss_denominator.numel() != 1 + ): + raise ValueError("loss_denominator must be a scalar CPU tensor") + if int(loss_denominator) != packed_layout.padded_denominator: + raise ValueError( + "loss_denominator must equal document_count * longest_document" + ) + if position_ids is None: + position_ids = packed_layout.positions.unsqueeze(0) + elif loss_denominator is not None: + raise ValueError("loss_denominator requires sequence_lengths") adapter = self._make_adapter() # Step 1: handle vocab size if target_hidden_for_compact is not None: @@ -385,7 +423,9 @@ def forward( dtype=torch.bool, device=hidden_states.device, ) - if self.attention_backend == "sdpa": + if packed_layout is not None: + attention_mask = packed_layout + elif self.attention_backend == "sdpa": attention_mask = self.draft_model.prepare_decoder_attention_mask( attention_mask=attention_mask, hidden_states=hidden_states, @@ -527,6 +567,16 @@ def forward( position_mask=state.position_mask, loss_mask=state.loss_mask, adapter=adapter, + loss_scale=( + seq_length / packed_layout.padded_denominator + if packed_layout is not None + else 1.0 + ), + full_positions=( + packed_layout.padded_denominator + if packed_layout is not None + else None + ), ) acces.append(acc) acceptance_rates.append(acceptance_rate) @@ -538,9 +588,14 @@ def forward( if not is_last: # Step 5.7: we need to update the loss mask - global_input_ids = padding(global_input_ids, left=False) - position_mask = padding(position_mask, left=False) - loss_mask = padding(loss_mask, left=False) + if packed_layout is None: + global_input_ids = padding(global_input_ids, left=False) + position_mask = padding(position_mask, left=False) + loss_mask = padding(loss_mask, left=False) + else: + global_input_ids = packed_layout.shift_left(global_input_ids) + position_mask = packed_layout.shift_left(position_mask) + loss_mask = packed_layout.shift_left(loss_mask) # Flex attention mask shirnking is handled inside attention module return ( plosses, diff --git a/specforge/algorithms/eagle3/providers.py b/specforge/algorithms/eagle3/providers.py index 835be387c..c2ecea4e8 100644 --- a/specforge/algorithms/eagle3/providers.py +++ b/specforge/algorithms/eagle3/providers.py @@ -32,7 +32,8 @@ build_offline_collator, build_offline_normalizer, build_offline_reader, - build_server_collator, + build_packed_collator, + build_padded_server_collator, ) ALGORITHM_NAME = "eagle3" @@ -211,6 +212,7 @@ def algorithm_providers() -> AlgorithmProviders: build_reader=build_offline_reader, build_normalizer=build_offline_normalizer, build_collator=build_offline_collator, + build_packed_collator=build_packed_collator, ), ), server_streaming=( @@ -227,7 +229,8 @@ def algorithm_providers() -> AlgorithmProviders: ), attention_mask_feature="attention_mask", ), - build_collator=build_server_collator, + build_collator=build_padded_server_collator, + build_packed_collator=build_packed_collator, ), ), vocab_mapping_modes=frozenset({FeatureMode.OFFLINE, FeatureMode.STREAMING}), diff --git a/specforge/application/planning.py b/specforge/application/planning.py index 8298b8420..ac5d25040 100644 --- a/specforge/application/planning.py +++ b/specforge/application/planning.py @@ -78,6 +78,24 @@ def _validate_algorithm_capabilities( ) -> None: capabilities = algorithm.spec.capabilities training = cfg.training + if training.sequence_packing: + provider = ( + algorithm.providers.offline_for(cfg.model.input_modality) + if mode is FeatureMode.OFFLINE + else algorithm.providers.server_streaming_for(cfg.model.input_modality) + ) + if provider.build_packed_collator is None: + raise ValueError( + f"algorithm {algorithm.name!r} does not support training.sequence_packing " + f"for modality {cfg.model.input_modality!r}" + ) + if ( + training.lk_loss_type is not None + and not capabilities.supports_packed_lk_loss + ): + raise ValueError( + f"algorithm {algorithm.name!r} does not support sequence_packing with lk_loss_type" + ) if training.attention_backend not in capabilities.attention_backends: raise ValueError( f"algorithm {algorithm.name!r} does not support attention_backend=" diff --git a/specforge/benchmarks/benchmark_dflash_sequence_packing.py b/specforge/benchmarks/benchmark_dflash_sequence_packing.py new file mode 100644 index 000000000..b1fa26ce5 --- /dev/null +++ b/specforge/benchmarks/benchmark_dflash_sequence_packing.py @@ -0,0 +1,465 @@ +"""Measure DFlash and DFlash2 packing separately, with identical sampled anchors. + +Example (CUDA, no model downloads):: + + PYTHONPATH=. python scripts/benchmark_dflash_sequence_packing.py \ + --preset tiny --dtype float32 --correctness-only --output tiny.json + PYTHONPATH=. python scripts/benchmark_dflash_sequence_packing.py \ + --preset medium --steps 20 --output perf.json + +Production model, collators, strategy, and BF16Optimizer are used. Frozen target +embeddings/head and captured features are synthetic. The timed region includes +the strategy's CPU integer-feature processing/transfers, forward, backward, and +optimizer update. Hidden features are GPU resident. Capture, hidden-feature I/O +and H2D, distributed communication, and serving are outside this benchmark. +""" + +from __future__ import annotations + +import argparse +import gc +import hashlib +import json +import os +import statistics +import subprocess +import time +from datetime import datetime, timezone +from pathlib import Path +from unittest import mock + +import torch +from torch import nn +from transformers import Qwen3Config + +from specforge.algorithms.common.dflash_family_model import OnlineDFlashModel +from specforge.algorithms.common.hidden_states_data import ( + build_collator, + build_packed_collator, +) +from specforge.modeling.draft.dflash import DFlashDraftModel +from specforge.modeling.draft.dflash2 import DFlash2DraftModel +from specforge.optimizer import BF16Optimizer +from specforge.runtime.contracts import TrainBatch +from specforge.training.strategies.base import DFlashTrainStrategy, StepContext + +PRESETS = { + "tiny": dict( + hidden_size=64, + intermediate_size=128, + layers=2, + heads=4, + kv_heads=2, + vocab_size=128, + block_size=4, + anchors=8, + capture_layers=2, + conv_group_size=4, + conv_kernel_size=2, + selector_rank=4, + selector_top_k=8, + lengths=[[9, 17, 31, 64], [32, 32, 32, 32]], + ), + "medium": dict( + hidden_size=2048, + intermediate_size=8192, + layers=2, + heads=16, + kv_heads=4, + vocab_size=32000, + block_size=16, + anchors=128, + capture_layers=2, + conv_group_size=32, + conv_kernel_size=4, + selector_rank=16, + selector_top_k=16, + lengths=[[1024, 1024, 1024, 1024], [128, 256, 512, 2048]], + ), +} +CONTEXT = StepContext(global_step=10, total_steps=100, collect_detailed_metrics=False) + + +def parse_args(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--preset", choices=PRESETS, default="medium") + parser.add_argument( + "--algorithm", choices=["dflash", "dflash2", "both"], default="both" + ) + for key in PRESETS["tiny"]: + if key != "lengths": + parser.add_argument("--" + key.replace("_", "-"), type=int) + parser.add_argument("--lengths", action="append") + parser.add_argument("--dtype", choices=["float32", "bfloat16"], default="bfloat16") + parser.add_argument("--sliding-window", type=int) + parser.add_argument("--warmup", type=int, default=5) + parser.add_argument("--steps", type=int, default=20) + parser.add_argument("--seed", type=int, default=1729) + parser.add_argument("--learning-rate", type=float, default=1e-4) + parser.add_argument("--objective-chunk-blocks", type=int, default=128) + parser.add_argument("--correctness-only", action="store_true") + parser.add_argument("--skip-correctness", action="store_true") + parser.add_argument("--atol", type=float) + parser.add_argument("--rtol", type=float) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + for key, value in PRESETS[args.preset].items(): + if getattr(args, key) is None: + setattr(args, key, value) + if isinstance(args.lengths[0], str): + args.lengths = [ + [int(value) for value in row.split(",")] for row in args.lengths + ] + args.atol = ( + args.atol + if args.atol is not None + else (2e-5 if args.dtype == "float32" else 2e-3) + ) + args.rtol = ( + args.rtol + if args.rtol is not None + else (2e-4 if args.dtype == "float32" else 2e-2) + ) + if any(not row or min(row) < 4 for row in args.lengths): + parser.error("each batch must contain sequence lengths >= 4") + if args.steps < 1 or args.warmup < 1 or args.anchors < 1: + parser.error("steps, warmup, and anchors must be positive") + if args.hidden_size % args.heads or args.heads % args.kv_heads: + parser.error("hidden_size/heads and heads/kv_heads must be integral") + if args.correctness_only and args.skip_correctness: + parser.error("correctness-only and skip-correctness cannot be combined") + return args + + +def build_strategy(args, algorithm): + torch.manual_seed(args.seed) + config = Qwen3Config( + architectures=[ + "DFlash2DraftModel" if algorithm == "dflash2" else "DFlashDraftModel" + ], + hidden_size=args.hidden_size, + intermediate_size=args.intermediate_size, + num_hidden_layers=args.layers, + num_target_layers=args.capture_layers + 4, + num_attention_heads=args.heads, + num_key_value_heads=args.kv_heads, + head_dim=args.hidden_size // args.heads, + vocab_size=args.vocab_size, + attention_dropout=0.0, + max_position_embeddings=max(map(max, args.lengths)) + args.block_size, + layer_types=["sliding_attention" if args.sliding_window else "full_attention"] + * args.layers, + use_sliding_window=bool(args.sliding_window), + sliding_window=args.sliding_window, + dflash_config={ + "block_size": args.block_size, + "mask_token_id": args.vocab_size - 1, + "target_layer_ids": list(range(1, args.capture_layers + 1)), + "conv_group_size": args.conv_group_size, + "conv_kernel_size": args.conv_kernel_size, + "selector_rank": args.selector_rank, + "selector_top_k": args.selector_top_k, + }, + ) + config._attn_implementation = "flex_attention" + draft_type = DFlash2DraftModel if algorithm == "dflash2" else DFlashDraftModel + draft = draft_type(config) + model = OnlineDFlashModel( + draft_model=draft, + target_lm_head=nn.Linear( + args.hidden_size, args.vocab_size, bias=False + ).requires_grad_(False), + target_embed_tokens=nn.Embedding( + args.vocab_size, args.hidden_size + ).requires_grad_(False), + mask_token_id=args.vocab_size - 1, + block_size=args.block_size, + attention_backend="flex_attention", + num_anchors=args.anchors, + objective_chunk_blocks=args.objective_chunk_blocks, + loss_decay_gamma=7.0, + selector_loss_alpha=1.0 if algorithm == "dflash2" else 0.0, + teacher_metrics=False, + ).to("cuda", getattr(torch, args.dtype)) + return DFlashTrainStrategy(model.train()) + + +def make_features(args, lengths): + generator = torch.Generator().manual_seed(args.seed + 1) + features = [] + for length in lengths: + loss_mask = torch.ones(1, length, dtype=torch.long) + loss_mask[:, : length // 4] = 0 + loss_mask[:, -1] = 0 + features.append( + { + "input_ids": torch.randint( + 0, args.vocab_size - 1, (1, length), generator=generator + ), + "loss_mask": loss_mask, + "hidden_states": torch.randn( + 1, + length, + args.capture_layers * args.hidden_size, + generator=generator, + ).to(getattr(torch, args.dtype)), + } + ) + return features + + +def make_batch(features, mode, algorithm): + collator = build_collator() if mode == "padded" else build_packed_collator() + tensors = collator(features) + tensors["hidden_states"] = tensors["hidden_states"].cuda() + return TrainBatch( + sample_ids=[str(i) for i in range(len(features))], + strategy=algorithm, + tensors=tensors, + metadata={}, + ) + + +def compare(reference, actual, args): + reference, actual = reference.float(), actual.float() + delta = actual - reference + return { + "pass": bool(torch.allclose(reference, actual, atol=args.atol, rtol=args.rtol)), + "max_abs_diff": float(delta.abs().max()), + "relative_l2_diff": float(delta.norm()) / max(float(reference.norm()), 1e-30), + } + + +def correctness(strategy, features, args, algorithm): + model = strategy.trainable_module() + results = {} + for mode in ["padded", "packed"]: + model.zero_grad(set_to_none=True) + batch = make_batch(features, mode, algorithm) + recorded_anchors = [] + sampler = model._sample_anchor_positions + + def record(*pos, **kwargs): + anchors, keep = sampler(*pos, **kwargs) + recorded_anchors.append((anchors.cpu(), keep.cpu())) + return anchors, keep + + torch.cuda.manual_seed(args.seed + 100) + with mock.patch.object(model, "_sample_anchor_positions", side_effect=record): + output = strategy.forward_loss(batch, CONTEXT) + output.loss.backward() + results[mode] = { + "loss": output.loss.detach().float().cpu(), + "loss_terms": torch.stack(output.loss_terms).detach().float().cpu(), + "anchors": recorded_anchors, + "grads": { + name: None if p.grad is None else p.grad.detach().float().cpu() + for name, p in model.named_parameters() + if p.requires_grad + }, + } + del output, batch + baseline, packed = results["padded"], results["packed"] + checks = { + key: compare(baseline[key], packed[key], args) for key in ["loss", "loss_terms"] + } + checks["loss_values"] = [float(baseline["loss"]), float(packed["loss"])] + checks["identical_sampled_anchors"] = len(baseline["anchors"]) == len( + packed["anchors"] + ) and all( + torch.equal(a, b) and torch.equal(ka, kb) + for (a, ka), (b, kb) in zip(baseline["anchors"], packed["anchors"]) + ) + checks["sampled_anchor_count"] = sum( + int(keep.sum()) for _, keep in baseline["anchors"] + ) + checks["gradients"] = {} + for name, ref in baseline["grads"].items(): + value = packed["grads"][name] + checks["gradients"][name] = ( + {"pass": ref is None and value is None, "missing_gradient": True} + if ref is None or value is None + else compare(ref, value, args) + ) + checks["pass"] = ( + checks["identical_sampled_anchors"] + and all(checks[key]["pass"] for key in ["loss", "loss_terms"]) + and all(row["pass"] for row in checks["gradients"].values()) + ) + model.zero_grad(set_to_none=True) + return checks + + +def measure(strategy, initial_state, features, lengths, args, algorithm, mode): + model = strategy.trainable_module() + model.load_state_dict(initial_state) + model.zero_grad(set_to_none=True) + batch = make_batch(features, mode, algorithm) + optimizer = BF16Optimizer( + model, + lr=args.learning_rate, + warmup_ratio=0.0, + total_steps=args.warmup + args.steps + 1, + lr_scheduler="constant", + ) + torch.cuda.manual_seed(args.seed + 100) + + def step(): + output = strategy.forward_loss(batch, CONTEXT) + output.loss.backward() + optimizer.step() + return output.loss.detach() + + torch.cuda.synchronize() + start = time.perf_counter() + for _ in range(args.warmup): + step() + torch.cuda.synchronize() + warmup_seconds = time.perf_counter() - start + torch.cuda.reset_peak_memory_stats() + baseline_bytes = torch.cuda.memory_allocated() + times = [] + for _ in range(args.steps): + torch.cuda.synchronize() + start = time.perf_counter() + loss = step() + torch.cuda.synchronize() + times.append(time.perf_counter() - start) + mean, median = statistics.mean(times), statistics.median(times) + result = { + "mean_step_ms": mean * 1000, + "p50_step_ms": median * 1000, + "useful_tokens_per_second": sum(lengths) / mean, + "p50_useful_tokens_per_second": sum(lengths) / median, + "peak_allocated_gib": torch.cuda.max_memory_allocated() / 2**30, + "baseline_allocated_gib": baseline_bytes / 2**30, + "warmup_seconds_including_compile": warmup_seconds, + "step_ms": [duration * 1000 for duration in times], + "final_loss": float(loss), + } + optimizer = batch = None + model.zero_grad(set_to_none=True) + gc.collect() + torch.cuda.empty_cache() + return result + + +def provenance(): + root = Path(__file__).resolve().parents[2] + files = [ + "scripts/benchmark_dflash_sequence_packing.py", + "specforge/benchmarks/benchmark_dflash_sequence_packing.py", + "specforge/algorithms/common/dflash_family_model.py", + "specforge/algorithms/common/hidden_states_data.py", + "specforge/modeling/draft/dflash.py", + "specforge/modeling/draft/dflash2.py", + "specforge/modeling/packed_dflash.py", + "specforge/training/strategies/base.py", + ] + result = { + "files_sha256": { + name: hashlib.sha256((root / name).read_bytes()).hexdigest() + for name in files + } + } + try: + result["head"] = subprocess.check_output( + ["git", "rev-parse", "HEAD"], cwd=root, text=True, stderr=subprocess.DEVNULL + ).strip() + except (OSError, subprocess.CalledProcessError): + result["head"] = "unavailable (copied snapshot is identified by file hashes)" + return result + + +def save(report, path): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(report, indent=2) + "\n") + + +def main(): + args = parse_args() + if not torch.cuda.is_available(): + raise RuntimeError("CUDA GPU required") + torch.set_num_threads(4) + torch.backends.cuda.matmul.allow_tf32 = False + torch.backends.cudnn.allow_tf32 = False + settings = {**vars(args), "output": str(args.output)} + report = { + "timestamp_utc": datetime.now(timezone.utc).isoformat(), + "source": provenance(), + "settings": settings, + "environment": { + "torch": torch.__version__, + "cuda": torch.version.cuda, + "gpu": torch.cuda.get_device_name(), + "cuda_visible_devices": os.environ.get("CUDA_VISIBLE_DEVICES"), + }, + "scope": "Single GPU; synthetic frozen target components and captured features; GPU-resident hidden features plus CPU integer features; production strategy forward/backward/BF16Optimizer; excludes capture, hidden-feature I/O/H2D, distributed training and serving.", + "cases": [], + } + algorithms = ["dflash", "dflash2"] if args.algorithm == "both" else [args.algorithm] + for algorithm in algorithms: + strategy = build_strategy(args, algorithm) + initial_state = { + name: value.detach().cpu().clone() + for name, value in strategy.trainable_module().state_dict().items() + } + for lengths in args.lengths: + strategy.trainable_module().load_state_dict(initial_state) + features = make_features(args, lengths) + case = { + "algorithm": algorithm, + "lengths": lengths, + "padding_fraction": 1 - sum(lengths) / (len(lengths) * max(lengths)), + "useful_tokens": sum(lengths), + } + report["cases"].append(case) + print(json.dumps({"case_start": algorithm, "lengths": lengths}), flush=True) + if not args.skip_correctness: + case["correctness"] = correctness(strategy, features, args, algorithm) + save(report, args.output) + print( + json.dumps( + { + "correctness": case["correctness"]["pass"], + "anchors": case["correctness"]["sampled_anchor_count"], + } + ), + flush=True, + ) + if not case["correctness"]["pass"]: + raise AssertionError(f"Packed parity failed; inspect {args.output}") + if not args.correctness_only: + for mode in ["padded", "packed"]: + case[mode] = measure( + strategy, + initial_state, + features, + lengths, + args, + algorithm, + mode, + ) + save(report, args.output) + print( + json.dumps( + {"algorithm": algorithm, "mode": mode, **case[mode]} + ), + flush=True, + ) + case["p50_speedup"] = ( + case["padded"]["p50_step_ms"] / case["packed"]["p50_step_ms"] + ) + case["mean_speedup"] = ( + case["padded"]["mean_step_ms"] / case["packed"]["mean_step_ms"] + ) + save(report, args.output) + del features + del strategy, initial_state + gc.collect() + torch.cuda.empty_cache() + print(f"Report written to {args.output}", flush=True) + + +if __name__ == "__main__": + main() diff --git a/specforge/benchmarks/benchmark_online_sequence_packing.py b/specforge/benchmarks/benchmark_online_sequence_packing.py new file mode 100644 index 000000000..36d681cca --- /dev/null +++ b/specforge/benchmarks/benchmark_online_sequence_packing.py @@ -0,0 +1,1166 @@ +"""Measure an overlapping SGLang -> Mooncake -> online trainer pipeline. + +The server and Mooncake master must already be running. A fresh bounded producer +and consumer run concurrently for every arm; features are never precaptured. +The primary interval starts at the first HTTP capture dispatch and ends after +the final optimizer update, synchronous durable acknowledgement and CUDA sync. +Configured periodic checkpoints occur inside the primary interval; the final +checkpoint/cleanup is outside it and included in fit wall time. All checkpoint +time is also reported separately. Warmup runs precede measured ABBA arms. +""" + +from __future__ import annotations + +import argparse +import gc +import hashlib +import json +import math +import os +import random +import shutil +import sqlite3 +import statistics +import threading +import time +import uuid +from datetime import datetime, timezone +from pathlib import Path + +import torch +from transformers import AutoConfig, Qwen3Config + +from specforge.algorithms.builtin import builtin_algorithm_registry +from specforge.algorithms.common.dflash_family_model import OnlineDFlashModel +from specforge.inference.adapters.server_capture import ( + ServerCaptureSchema, + SGLangServerCaptureAdapter, +) +from specforge.launch import build_disagg_online_consumer, build_disagg_online_producer +from specforge.modeling.draft.dflash import DFlashDraftModel +from specforge.modeling.draft.dflash2 import DFlash2DraftModel +from specforge.modeling.target.target_utils import TargetEmbeddingsAndHead +from specforge.optimizer import BF16Optimizer +from specforge.runtime.data_plane.mooncake_store import MooncakeFeatureStore +from specforge.runtime.data_plane.streaming_ref_channel import StreamingRefChannel +from specforge.training.checkpoint import STATE_FILE + + +def parse_args(argv=None): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--server-url", required=True) + parser.add_argument("--target-model", required=True) + parser.add_argument("--draft-config", type=Path) + parser.add_argument( + "--prompts-path", + type=Path, + help="Pretokenized JSONL input_ids/loss_mask in the exact desired order", + ) + parser.add_argument( + "--algorithm", choices=("dflash", "dflash2", "both"), default="both" + ) + parser.add_argument( + "--capture-layers", + help="Comma-separated target layer IDs; defaults to draft config", + ) + parser.add_argument( + "--draft-layers", type=int, default=2, help="Used only without --draft-config" + ) + parser.add_argument("--lengths", default="128,256,512,2048") + parser.add_argument("--batch-size", type=int, default=4) + parser.add_argument("--accumulation-steps", type=int, default=1) + parser.add_argument("--anchors", type=int, default=512) + parser.add_argument("--steps", type=int, default=32) + parser.add_argument( + "--warmup-steps", + type=int, + help="Defaults to a full untimed replay of all measured steps", + ) + parser.add_argument( + "--repeats", type=int, default=3, help="Number of ABBA blocks (4 runs each)" + ) + parser.add_argument("--seed", type=int, default=1729) + parser.add_argument("--prompt-fraction", type=float, default=0.25) + parser.add_argument("--learning-rate", type=float, default=1e-4) + parser.add_argument("--objective-chunk-blocks", type=int, default=128) + parser.add_argument("--capture-batch-size", type=int) + parser.add_argument( + "--backlog", + type=int, + default=8, + help="High watermark refs; one capture batch may overshoot", + ) + parser.add_argument("--log-interval", type=int, default=50) + parser.add_argument("--save-interval", type=int, default=0) + parser.add_argument( + "--teacher-metrics", action=argparse.BooleanOptionalAction, default=True + ) + parser.add_argument("--dataloader-workers", type=int, default=4) + parser.add_argument( + "--receive-buffers", choices=("pageable", "pinned"), default="pinned" + ) + parser.add_argument("--segment-mib", type=int, default=1024) + parser.add_argument("--local-buffer-mib", type=int, default=256) + parser.add_argument("--request-timeout", type=float, default=300) + parser.add_argument("--dist-port", type=int, default=29712) + parser.add_argument("--keep-checkpoints", action="store_true") + parser.add_argument("--work-dir", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args(argv) + args.warmup_steps = args.steps if args.warmup_steps is None else args.warmup_steps + args.lengths = [int(value) for value in args.lengths.split(",")] + if args.capture_layers: + args.capture_layers = [int(value) for value in args.capture_layers.split(",")] + args.capture_batch_size = args.capture_batch_size or args.batch_size + positive = ( + args.batch_size, + args.accumulation_steps, + args.anchors, + args.steps, + args.warmup_steps, + args.repeats, + args.capture_batch_size, + ) + if min(positive) < 1 or not args.lengths or min(args.lengths) < 4: + parser.error("batch/step/anchor counts must be positive and lengths >= 4") + if not 0 <= args.prompt_fraction < 1: + parser.error("prompt-fraction must be in [0,1)") + quantum = args.batch_size * args.accumulation_steps + if args.backlog < 2 * quantum: + parser.error("backlog must be at least twice the optimizer sample quantum") + if args.log_interval < 1 or args.save_interval < 0: + parser.error("log-interval must be positive and save-interval nonnegative") + return args + + +def _draft_config(args, target_config, architecture): + if args.draft_config: + payload = json.loads(args.draft_config.read_text()) + else: + payload = target_config.to_dict() + payload.update( + num_hidden_layers=args.draft_layers, + num_target_layers=target_config.num_hidden_layers, + layer_types=["full_attention"] * args.draft_layers, + block_size=16, + ) + method = dict(payload.get("dflash_config") or {}) + layers = args.capture_layers or method.get("target_layer_ids") + if not layers: + raise ValueError( + "supply --capture-layers or a draft config with target_layer_ids" + ) + if min(layers) < 0 or max(layers) >= target_config.num_hidden_layers: + raise ValueError("capture layers are outside the target model") + if int(payload["hidden_size"]) != int(target_config.hidden_size) or int( + payload["vocab_size"] + ) != int(target_config.vocab_size): + raise ValueError("draft hidden size and vocabulary must match the target") + method["target_layer_ids"] = list(layers) + method.setdefault("mask_token_id", int(target_config.vocab_size) - 1) + if architecture == "dflash2": + method.setdefault("conv_group_size", 32) + method.setdefault("conv_kernel_size", 4) + method.setdefault("selector_rank", 16) + method.setdefault("selector_top_k", 16) + payload["architectures"] = [ + "DFlash2DraftModel" if architecture == "dflash2" else "DFlashDraftModel" + ] + payload["dflash_config"] = method + payload["num_target_layers"] = int(target_config.num_hidden_layers) + payload["attention_dropout"] = 0.0 + config = Qwen3Config(**payload) + config._attn_implementation = "flex_attention" + return config + + +def _fingerprint(model): + """Check every parameter's shape and deterministic boundary values cheaply.""" + digest = hashlib.sha256() + for name, value in model.state_dict().items(): + digest.update(f"{name}:{tuple(value.shape)}:{value.dtype}".encode()) + flat = value.detach().reshape(-1) + digest.update( + torch.cat((flat[:64], flat[-64:])).float().cpu().numpy().tobytes() + ) + return digest.hexdigest() + + +def _parameter_inventory(model): + """Inventory unique Parameters before wrapping; component counts may overlap.""" + named = dict(model.named_parameters()) + + def counts(parameters): + unique = {id(parameter): parameter for parameter in parameters}.values() + unique = list(unique) + total = sum(parameter.numel() for parameter in unique) + trainable = sum( + parameter.numel() for parameter in unique if parameter.requires_grad + ) + return { + "tensors": len(unique), + "total": total, + "trainable": trainable, + "frozen": total - trainable, + } + + aliases = {} + for name, parameter in model.named_parameters(remove_duplicate=False): + aliases.setdefault(id(parameter), []).append(name) + components = { + "draft": model.draft_model, + "target_embedding": model.embed_tokens, + "target_lm_head": model.lm_head, + } + return named, { + "whole_model_unique": counts(named.values()), + "components": { + name: counts(module.parameters()) for name, module in components.items() + }, + "component_count_note": "Each component deduplicates Parameters; tied embedding/head weights appear in both component counts, but only once in whole_model_unique.", + "draft_layer_count": len(model.draft_model.layers), + "parameters": [ + { + "name": name, + "aliases": aliases[id(parameter)], + "shape": list(parameter.shape), + "numel": parameter.numel(), + "dtype": str(parameter.dtype), + "trainable": parameter.requires_grad, + } + for name, parameter in named.items() + ], + } + + +def _optimizer_coverage(model, named, optimizer): + """Check original-parameter and FP32-master identities, not just counts.""" + draft_ids = { + id(parameter) + for parameter in model.draft_model.parameters() + if parameter.requires_grad + } + target_ids = { + id(parameter) + for component in (model.embed_tokens, model.lm_head) + for parameter in component.parameters() + } + trainable_ids = { + id(parameter) for parameter in named.values() if parameter.requires_grad + } + optimizer_ids = [id(parameter) for parameter in optimizer.model_params] + master_ids = [id(parameter) for parameter in optimizer.fp32_params] + adam_ids = [ + id(parameter) + for group in optimizer.optimizer.param_groups + for parameter in group["params"] + ] + if ( + set(optimizer_ids) != draft_ids + or draft_ids != trainable_ids + or len(optimizer_ids) != len(draft_ids) + or target_ids & set(optimizer_ids) + ): + raise AssertionError( + "optimizer must include every trainable draft Parameter exactly once and no target Parameter" + ) + if ( + len(master_ids) != len(optimizer_ids) + or len(adam_ids) != len(master_ids) + or len(set(master_ids)) != len(master_ids) + or set(adam_ids) != set(master_ids) + or any( + parameter.shape != master.shape or master.dtype != torch.float32 + for parameter, master in zip(optimizer.model_params, optimizer.fp32_params) + ) + ): + raise AssertionError( + "AdamW must optimize one matching FP32 master per draft Parameter" + ) + return { + "all_trainable_draft_parameters_included": True, + "target_parameters_excluded": True, + "one_fp32_adamw_master_per_trainable_parameter": True, + "optimized_tensors": len(optimizer_ids), + "optimized_elements": sum( + parameter.numel() for parameter in optimizer.model_params + ), + "parameter_names": [ + name for name, parameter in named.items() if id(parameter) in draft_ids + ], + } + + +def _parameter_sample_indices(numel, *, device="cpu"): + """Integer interpolation keeps large tensor endpoints exact and in bounds.""" + count = min(128, numel) + return ( + torch.arange(count, dtype=torch.int64, device=device) + * max(0, numel - 1) + // max(1, count - 1) + ) + + +def _parameter_samples(named): + """Sample up to 128 evenly spaced values per tensor outside pipeline timing.""" + snapshots = {} + for name, parameter in named.items(): + flat = parameter.detach().reshape(-1) + indices = _parameter_sample_indices(flat.numel(), device=flat.device) + values = flat[indices].float().cpu() + snapshots[name] = { + "sha256": hashlib.sha256(values.numpy().tobytes()).hexdigest(), + "sampled_elements": values.numel(), + "sampled_values_finite": bool(torch.isfinite(values).all()), + } + return snapshots + + +def _parameter_update_evidence(named, before, layer_count): + """Full finiteness scan plus sampled change evidence, after the timed fit.""" + after = _parameter_samples(named) + trainable = [ + (name, parameter) + for name, parameter in named.items() + if parameter.requires_grad + ] + finite = ( + torch.stack( + [torch.isfinite(parameter.detach()).all() for _, parameter in trainable] + ) + .cpu() + .tolist() + ) + if not all(finite): + raise AssertionError("training left nonfinite values in a trainable Parameter") + changed = {name: before[name]["sha256"] != after[name]["sha256"] for name in named} + frozen_changed = [ + name + for name, parameter in named.items() + if not parameter.requires_grad and changed[name] + ] + if frozen_changed: + raise AssertionError(f"frozen target/draft samples changed: {frozen_changed}") + layers = [] + for index in range(layer_count): + prefix = f"draft_model.layers.{index}." + layer_names = [name for name, _ in trainable if name.startswith(prefix)] + changed_names = [name for name in layer_names if changed[name]] + layers.append( + { + "layer": index, + "trainable_tensors": len(layer_names), + "changed_sampled_tensors": changed_names, + "sampled_update_observed": bool(changed_names), + } + ) + return { + "timing": "Snapshots before first capture and after fit; full finite scan after fit. No measured-step hooks or synchronizations.", + "semantics": "All trainable elements are checked finite. Change hashes cover at most 128 evenly spaced elements per tensor; an unchanged hash does not prove an entire tensor was unchanged. Zero initial gradients or sparse selector updates are legitimate.", + "all_trainable_elements_finite": True, + "all_frozen_parameter_samples_unchanged": True, + "all_decoder_layers_have_sampled_updates": all( + layer["sampled_update_observed"] for layer in layers + ), + "layers": layers, + "parameters": { + name: { + "before": before[name], + "after": after[name], + "sampled_update_observed": changed[name], + } + for name in named + }, + } + + +def _observe_first_warmup_step(optimizer, named): + """Observe presence without reading gradient tensors; never used in timed arms.""" + evidence = { + "performed": False, + "scope": "first optimizer step of untimed warmup only", + } + original_step = optimizer.step + names = {id(parameter): name for name, parameter in named.items()} + + def step(**kwargs): + evidence["gradient_present"] = { + names[id(parameter)]: parameter.grad is not None + for parameter in optimizer.model_params + } + result = original_step(**kwargs) + evidence["performed"] = True + evidence["_gradient_norm"] = optimizer.last_grad_norm.detach().clone() + optimizer.step = original_step + return result + + optimizer.step = step + return evidence + + +def _optimizer_step_evidence(optimizer, named, expected_steps): + names = {id(parameter): name for name, parameter in named.items()} + steps = {} + for parameter, master in zip(optimizer.model_params, optimizer.fp32_params): + value = optimizer.optimizer.state.get(master, {}).get("step", 0) + steps[names[id(parameter)]] = int( + value.item() if isinstance(value, torch.Tensor) else value + ) + return { + "expected_steps": expected_steps, + "adamw_steps_by_parameter": steps, + "all_trainable_parameters_received_every_optimizer_step": all( + value == expected_steps for value in steps.values() + ), + "semantics": "AdamW step counters prove optimizer participation, including legitimate zero gradients; they do not imply every element changed.", + } + + +def _model(args, target_config, architecture): + torch.manual_seed(args.seed) + config = _draft_config(args, target_config, architecture) + draft_type = DFlash2DraftModel if architecture == "dflash2" else DFlashDraftModel + draft = draft_type(config).to(dtype=torch.bfloat16) + fingerprint = _fingerprint(draft) + components = TargetEmbeddingsAndHead.from_pretrained( + args.target_model, + device="cuda", + dtype=torch.bfloat16, + ).requires_grad_(False) + model = OnlineDFlashModel( + draft_model=draft.cuda(), + target_lm_head=components.lm_head, + target_embed_tokens=components.embed_tokens, + mask_token_id=int(config.dflash_config["mask_token_id"]), + block_size=int(config.block_size), + attention_backend="flex_attention", + num_anchors=args.anchors, + objective_chunk_blocks=args.objective_chunk_blocks, + teacher_metrics=args.teacher_metrics, + ).cuda() + return model, config, fingerprint + + +def _prompts(args, steps, vocab_size): + count = steps * args.batch_size * args.accumulation_steps + generator = torch.Generator().manual_seed(args.seed + 1) + desired = [] + digest = hashlib.sha256() + useful_tokens = supervised_tokens = 0 + supplied = None + if args.prompts_path: + with args.prompts_path.open() as stream: + supplied = [json.loads(line) for line in stream if line.strip()] + if len(supplied) < count: + raise ValueError( + f"{args.prompts_path} has {len(supplied)} prompts; this run requires {count}" + ) + for index in range(count): + if supplied is None: + length = args.lengths[index % len(args.lengths)] + ids = torch.randint(0, vocab_size, (length,), generator=generator).tolist() + prompt_length = min(int(length * args.prompt_fraction), length - 2) + mask = [0] * prompt_length + [1] * (length - prompt_length) + else: + row = supplied[index].get("payload", supplied[index]) + ids, mask = list(row["input_ids"]), list(row["loss_mask"]) + length = len(ids) + if ( + len(mask) != length + or not ids + or not all( + isinstance(token, int) and 0 <= token < vocab_size for token in ids + ) + ): + raise ValueError(f"invalid token IDs or mask shape in prompt {index}") + if not any(a > 0.5 and b > 0.5 for a, b in zip(mask, mask[1:])): + raise ValueError( + f"prompt {index} has no two consecutive supervised tokens" + ) + payload = {"input_ids": ids, "loss_mask": mask} + digest.update(json.dumps(payload, separators=(",", ":")).encode()) + desired.append( + { + "task_id": f"prompt-{index:08d}", + "source_id": ( + "pretokenized-jsonl" + if supplied is not None + else "synthetic-length-pattern" + ), + "payload": payload, + "max_length": length, + } + ) + useful_tokens += length + supervised_tokens += sum(mask) + # Compensate for the canonical producer's deterministic shuffle so every + # warm/measured microbatch sees the same repeating length pattern. Capture + # order is checked below; an upstream ordering change fails this benchmark. + order = list(range(count)) + random.Random(args.seed).shuffle(order) + prompts = [None] * count + for position, shuffled_index in enumerate(order): + prompts[shuffled_index] = desired[position] + return prompts, digest.hexdigest(), useful_tokens, supervised_tokens + + +def _compiler_counters(): + from torch._dynamo.utils import counters + + return { + str(namespace): {str(key): int(value) for key, value in counts.items()} + for namespace, counts in counters.items() + } + + +def _counter_delta(before, after): + return { + namespace: { + key: after.get(namespace, {}).get(key, 0) + - before.get(namespace, {}).get(key, 0) + for key in set(before.get(namespace, {})) | set(after.get(namespace, {})) + if after.get(namespace, {}).get(key, 0) + != before.get(namespace, {}).get(key, 0) + } + for namespace in set(before) | set(after) + } + + +class _TimedSource: + def __init__(self, adapter): + self.adapter = adapter + self.calls = [] + self.task_ids = [] + self.first_dispatch = None + self.request_digest = hashlib.sha256() + original_post = adapter.post_fn + + def timed_post(url, *, json_body, timeout): + # Hash the actual HTTP inputs and masks after canonical request + # construction, excluding fresh transport/cache namespaces. + normalized = { + key: value + for key, value in json_body.items() + if key not in ("extra_key", "spec_capture") + } + normalized["spec_capture"] = [ + { + **{ + key: value + for key, value in capture.items() + if key not in ("store_id", "sample_id") + }, + "sample_id": capture["sample_id"].split(":", 1)[1], + } + for capture in json_body["spec_capture"] + ] + self.request_digest.update(json.dumps(normalized, sort_keys=True).encode()) + begin = time.perf_counter() + if self.first_dispatch is None: + self.first_dispatch = begin + result = original_post(url, json_body=json_body, timeout=timeout) + self.calls.append( + { + "start": begin, + "end": time.perf_counter(), + "samples": len(json_body["spec_capture"]), + } + ) + return result + + adapter.post_fn = timed_post + + def produce_refs(self, tasks, *, capture): + result = self.adapter.produce_refs(tasks, capture=capture) + self.task_ids.extend(task.task_id for task in tasks) + return result + + def __getattr__(self, name): + return getattr(self.adapter, name) + + +class _TimedChannel(StreamingRefChannel): + def __init__(self, path): + super().__init__(path) + self.publications = [] + self.max_backlog = 0 + + def begin_publish(self, refs): + transaction = super().begin_publish(refs) + channel = self + + class Transaction: + def commit(self): + result = transaction.commit() + backlog = channel.in_flight_remote() + channel.max_backlog = max(channel.max_backlog, backlog) + channel.publications.append( + { + "time": time.perf_counter(), + "samples": len(refs), + "backlog": backlog, + "task_ids": [ref.source_task_id for ref in refs], + } + ) + return result + + def __getattr__(self, name): + return getattr(transaction, name) + + return Transaction() + + +def _store(args, run_id): + required = ("MOONCAKE_MASTER_SERVER_ADDR", "MOONCAKE_METADATA_SERVER") + for name in required: + if not os.environ.get(name): + raise ValueError(f"set {name} for the existing server's Mooncake master") + return MooncakeFeatureStore( + store_id=run_id, + retain_on_release=True, + receive_buffers=args.receive_buffers, + setup_kwargs={ + "local_hostname": os.environ.get("MOONCAKE_LOCAL_HOSTNAME", "127.0.0.1"), + "metadata_server": os.environ["MOONCAKE_METADATA_SERVER"], + "master_server_addr": os.environ["MOONCAKE_MASTER_SERVER_ADDR"], + "global_segment_size": args.segment_mib << 20, + "local_buffer_size": args.local_buffer_mib << 20, + "protocol": os.environ.get("MOONCAKE_PROTOCOL", "tcp"), + "rdma_devices": os.environ.get("MOONCAKE_RDMA_DEVICES", ""), + }, + ) + + +def run_pipeline(args, target_config, architecture, packing, steps, label): + run_id = f"{uuid.uuid4().hex[:12]}-{architecture}-{label}-{'packed' if packing else 'padded'}" + work = args.work_dir / run_id + work.mkdir(parents=True, exist_ok=False) + assembled_at = time.perf_counter() + model, config, fingerprint = _model(args, target_config, architecture) + named_parameters, parameter_inventory = _parameter_inventory(model) + before_parameter_samples = _parameter_samples(named_parameters) + prompts, prompt_hash, useful_tokens, supervised_tokens = _prompts( + args, steps, config.vocab_size + ) + algorithm = builtin_algorithm_registry().resolve("dflash") + provider = algorithm.providers.server_streaming_for("text") + layout = provider.layout + schema = ServerCaptureSchema( + aux_feature=layout.aux_feature, + last_hidden_feature=( + layout.last_hidden_feature if args.teacher_metrics else None + ), + passthrough=layout.passthrough, + attention_mask_feature=layout.attention_mask_feature, + ) + store = _store(args, run_id) + channel = _TimedChannel(str(work / "refs.jsonl")) + source = _TimedSource( + SGLangServerCaptureAdapter( + args.server_url, + store, + run_id=run_id, + algorithm="dflash", + schema=schema, + timeout_s=args.request_timeout, + target_model_version=args.target_model, + ) + ) + producer_thread = None + trainer = None + stop = threading.Event() + producer_state = {} + acknowledgements = [] + checkpoints = [] + logged = [] + loader_counter_windows = [] + fit_started = False + try: + trainer = build_disagg_online_consumer( + algorithm=algorithm, + feature_store=store, + channel=channel, + draft_model=model, + optimizer_factory=lambda module: BF16Optimizer( + module, + lr=args.learning_rate, + max_grad_norm=0.5, + warmup_ratio=0.0, + total_steps=steps, + ), + run_id=run_id, + output_dir=str(work / "output"), + batch_size=args.batch_size, + accumulation_steps=args.accumulation_steps, + max_steps=steps, + sequence_packing=packing, + save_interval=args.save_interval, + log_interval=args.log_interval, + metadata_db_path=str(work / "consumer.sqlite"), + async_ack=False, + idle_timeout_s=args.request_timeout * 2, + dataloader_num_workers=args.dataloader_workers, + logger=lambda metrics, step: logged.append( + {"step": step, "metrics": dict(metrics)} + ), + ) + optimizer = trainer.backend.optimizer + optimizer_coverage = _optimizer_coverage(model, named_parameters, optimizer) + first_step_evidence = ( + _observe_first_warmup_step(optimizer, named_parameters) + if label == "warmup" + else { + "performed": False, + "scope": "measured arms have no gradient-observation hook; see matching warmup", + } + ) + backend_evidence = { + "wrapper_kind": trainer.backend._wrapper_kind, + "configured_sharding_strategy": trainer.backend.parallel_config.sharding_strategy, + "effective_sharding_strategy": str( + getattr(trainer.backend.module, "sharding_strategy", "not applicable") + ), + } + original_snapshot = trainer._loader.perf_counters_snapshot + + def tracked_snapshot(reset=False): + snapshot = original_snapshot(reset=reset) + if reset: + loader_counter_windows.append(snapshot) + return snapshot + + trainer._loader.perf_counters_snapshot = tracked_snapshot + original_ack = trainer._controller.ack_fn + + def timed_ack(ids, step): + original_ack(ids, step) + # Add a CUDA barrier only at the final timing endpoint. Intermediate + # event timestamps record canonical durable-ack completion. + if step == steps: + torch.cuda.synchronize() + acknowledgements.append( + {"step": step, "time": time.perf_counter(), "sample_ids": list(ids)} + ) + + trainer._controller.ack_fn = timed_ack + original_checkpoint = trainer._controller.save_checkpoint + + def timed_checkpoint(step): + start = time.perf_counter() + result = original_checkpoint(step) + torch.cuda.synchronize() + checkpoints.append({"step": step, "seconds": time.perf_counter() - start}) + return result + + trainer._controller.save_checkpoint = timed_checkpoint + _, drive = build_disagg_online_producer( + algorithm=algorithm, + prompts=prompts, + feature_store=store, + channel=channel, + run_id=run_id, + target_hidden_size=config.hidden_size, + target_vocab_size=config.vocab_size, + target_repr=provider.target_representation, + aux_hidden_state_layer_ids=config.dflash_config["target_layer_ids"], + feature_source=source, + lease=args.capture_batch_size, + producer_concurrency=1, + num_rollout_workers=1, + in_flight_high_watermark=args.backlog, + in_flight_low_watermark=max( + args.batch_size * args.accumulation_steps, args.backlog // 2 + ), + prompt_seed=args.seed, + prompt_ingest_batch_size=max(64, args.backlog), + backpressure_poll_s=0.01, + peer_wait_timeout_s=args.request_timeout * 2, + max_prompt_attempts=1, + max_worker_failures=1, + ) + + def produce(): + try: + producer_state["samples"] = drive( + should_stop=lambda: stop.is_set() or channel.consumer_stopped() + ) + except BaseException as exc: + producer_state["error"] = repr(exc) + channel.fail(repr(exc)) + finally: + producer_state["end"] = time.perf_counter() + + torch.manual_seed(args.seed + 2) + torch.cuda.synchronize() + compiler_before = _compiler_counters() + setup_seconds = time.perf_counter() - assembled_at + fit_begin = time.perf_counter() + producer_thread = threading.Thread( + target=produce, name=f"capture-{run_id}", daemon=True + ) + producer_thread.start() + fit_started = True + final_step = trainer.fit() + torch.cuda.synchronize() + fit_end = time.perf_counter() + producer_thread.join(timeout=args.request_timeout + 10) + if producer_thread.is_alive(): + raise RuntimeError("producer did not finish after the final optimizer step") + if "error" in producer_state: + raise RuntimeError(producer_state["error"]) + expected_count = steps * args.batch_size * args.accumulation_steps + if producer_state.get("samples") != expected_count: + raise AssertionError("producer did not publish the complete workload") + if not channel.consumer_stopped() or channel.consumer_failure() is not None: + raise AssertionError("consumer did not publish a clean completion") + expected_order = [f"prompt-{i:08d}" for i in range(expected_count)] + published_order = [ + task for event in channel.publications for task in event["task_ids"] + ] + if source.task_ids != expected_order or published_order != expected_order: + raise AssertionError( + "capture/publication ordering changed or capture retried; comparison invalid" + ) + with sqlite3.connect(work / "consumer.sqlite") as connection: + acked = connection.execute( + "SELECT sample_id FROM acked ORDER BY sample_id" + ).fetchall() + ack_ids = [ + sample for event in acknowledgements for sample in event["sample_ids"] + ] + if ( + final_step != steps + or len(ack_ids) != expected_count + or len(acked) != expected_count + ): + raise AssertionError( + "optimizer steps or durable sample acknowledgements do not match the workload" + ) + if ( + set(ack_ids) != {row[0] for row in acked} + or len(set(ack_ids)) != expected_count + ): + raise AssertionError( + "sample identities changed or duplicate acknowledgements occurred" + ) + ack_order = [sample.split(":", 1)[1] for sample in ack_ids] + if ack_order != expected_order: + raise AssertionError("consumed sample order or logical batching changed") + final_checkpoint = work / "output" / f"{run_id}-step{steps}" / STATE_FILE + if not final_checkpoint.is_file(): + raise AssertionError("canonical final checkpoint is missing") + state = torch.load( + final_checkpoint, map_location="cpu", weights_only=False, mmap=True + ) + checkpoint_proof = { + key: state[key] + for key in ("global_step", "epoch", "epoch_batch", "epoch_samples") + } + del state + if ( + checkpoint_proof["global_step"] != steps + or checkpoint_proof["epoch_samples"] != expected_count + ): + raise AssertionError( + "checkpoint step/sample progress disagrees with durable acks" + ) + loader_counter_windows.append(original_snapshot(reset=False)) + loader_counters = { + key: sum(window.get(key, 0) for window in loader_counter_windows) + for key in loader_counter_windows[-1] + } + if trainer.micro_step != steps * args.accumulation_steps: + raise AssertionError("logical microbatch count changed") + final_loss = trainer._controller._last_result.loss + final_loss = ( + float(final_loss.detach().cpu()) + if isinstance(final_loss, torch.Tensor) + else float(final_loss) + ) + if not math.isfinite(final_loss): + raise AssertionError("nonfinite final training loss") + start = source.first_dispatch + finish = acknowledgements[-1]["time"] + capture_end = max(call["end"] for call in source.calls) + if start is None or finish <= start: + raise AssertionError("invalid pipeline timing boundaries") + elapsed = finish - start + # These checks run after the final ack AND the checkpoint/fit endpoint; + # their tensor reads and finite scans affect neither reported interval. + parameter_updates = _parameter_update_evidence( + named_parameters, + before_parameter_samples, + parameter_inventory["draft_layer_count"], + ) + optimizer_steps = _optimizer_step_evidence(optimizer, named_parameters, steps) + if first_step_evidence["performed"]: + first_step_evidence["global_gradient_norm"] = float( + first_step_evidence.pop("_gradient_norm").cpu() + ) + first_step_evidence["all_trainable_gradients_present"] = all( + first_step_evidence["gradient_present"].values() + ) + first_step_evidence["global_gradient_norm_finite"] = math.isfinite( + first_step_evidence["global_gradient_norm"] + ) + first_step_evidence["semantics"] = ( + "Presence is observed before BF16Optimizer clears gradients. Its finite global norm check covers all present gradients. Zero gradients are valid at initialization, including DFlash2 bilinear selector factors." + ) + result = { + "run_id": run_id, + "architecture": architecture, + "packing": packing, + "label": label, + "optimizer_steps": steps, + "microsteps": trainer.micro_step, + "samples": expected_count, + "durable_acked_samples": len(acked), + "useful_tokens": useful_tokens, + "supervised_tokens": supervised_tokens, + "pipeline_seconds": elapsed, + "useful_tokens_per_second": useful_tokens / elapsed, + "samples_per_second": expected_count / elapsed, + "capture_finished_seconds": capture_end - start, + "producer_finished_seconds": producer_state["end"] - start, + "trainer_fit_seconds_from_first_capture": fit_end - start, + "fit_wall_seconds": fit_end - fit_begin, + "model_and_runtime_setup_seconds": setup_seconds, + "checkpoint_seconds": sum(item["seconds"] for item in checkpoints), + "checkpoint_events": checkpoints, + "checkpoint_proof": checkpoint_proof, + "backend": backend_evidence, + "parameter_inventory": parameter_inventory, + "optimizer_coverage": optimizer_coverage, + "first_warmup_step_gradient_evidence": first_step_evidence, + "parameter_update_evidence": parameter_updates, + "optimizer_step_evidence": optimizer_steps, + "loader_perf_counters": loader_counters, + "loader_perf_note": "wait_producer_s and wait_fetch_s are consumer blocking; fetch_s overlaps training and is not additive with wall time", + "final_loss": final_loss, + "max_observed_backlog_refs": channel.max_backlog, + "prompt_sha256": prompt_hash, + "initial_draft_boundary_fingerprint": fingerprint, + "actual_capture_task_order": source.task_ids, + "actual_request_sha256": source.request_digest.hexdigest(), + "actual_consumed_task_order": ack_order, + "capture_calls": [ + {**event, "start": event["start"] - start, "end": event["end"] - start} + for event in source.calls + ], + "publication_events": [ + {**event, "time": event["time"] - start} + for event in channel.publications + ], + "optimizer_ack_events": [ + { + "step": event["step"], + "time": event["time"] - start, + "sample_count": len(event["sample_ids"]), + "task_ids": [ + sample.split(":", 1)[1] for sample in event["sample_ids"] + ], + } + for event in acknowledgements + ], + "capture_calls_after_first_optimizer": sum( + call["start"] > acknowledgements[0]["time"] for call in source.calls + ), + "sustained_live_overlap_established": any( + call["start"] > acknowledgements[0]["time"] for call in source.calls + ), + "checkpoint_policy": { + "save_interval": args.save_interval, + "canonical_final_save": True, + }, + "resolved_draft_config": config.to_dict(), + "logged": logged, + "compiler_counters_before": compiler_before, + "compiler_counters_after": _compiler_counters(), + "compiler_counter_delta": _counter_delta( + compiler_before, _compiler_counters() + ), + "full_warmup_replay": args.warmup_steps >= args.steps, + } + (work / "result.json").write_text(json.dumps(result, indent=2, default=str)) + return result + finally: + stop.set() + if producer_thread is not None and producer_thread.is_alive(): + channel.mark_consumer_failed("benchmark consumer stopped") + producer_thread.join(timeout=args.request_timeout + 10) + if trainer is not None and not fit_started: + trainer._loader.close() + if trainer._on_fit_finally is not None: + trainer._on_fit_finally() + store.discard_external_attempts(reason="benchmark-finished") + store.close() + if not args.keep_checkpoints and (work / "output").exists(): + shutil.rmtree(work / "output") + del trainer, model + gc.collect() + torch.cuda.empty_cache() + + +def _summary(results): + summary = {} + for architecture in sorted({row["architecture"] for row in results}): + arms = {} + for packing in (False, True): + rows = [ + row + for row in results + if row["architecture"] == architecture and row["packing"] == packing + ] + if rows: + arms["packed" if packing else "padded"] = { + "runs": len(rows), + "median_pipeline_seconds": statistics.median( + row["pipeline_seconds"] for row in rows + ), + "median_useful_tokens_per_second": statistics.median( + row["useful_tokens_per_second"] for row in rows + ), + "median_fit_seconds_with_checkpoint": statistics.median( + row["trainer_fit_seconds_from_first_capture"] for row in rows + ), + } + if len(arms) == 2: + arms["pipeline_speedup"] = ( + arms["padded"]["median_pipeline_seconds"] + / arms["packed"]["median_pipeline_seconds"] + ) + arms["fit_speedup_with_checkpoint"] = ( + arms["padded"]["median_fit_seconds_with_checkpoint"] + / arms["packed"]["median_fit_seconds_with_checkpoint"] + ) + summary[architecture] = arms + return summary + + +def main(argv=None): + args = parse_args(argv) + if not torch.cuda.is_available(): + raise RuntimeError("online pipeline benchmark requires a CUDA consumer") + import requests + import torch.distributed as dist + + requests.get(args.server_url.rstrip("/") + "/health", timeout=10).raise_for_status() + args.work_dir.mkdir(parents=True, exist_ok=True) + args.output.parent.mkdir(parents=True, exist_ok=True) + own_group = not dist.is_initialized() + if own_group: + os.environ.update( + RANK="0", + WORLD_SIZE="1", + LOCAL_RANK="0", + MASTER_ADDR="127.0.0.1", + MASTER_PORT=str(args.dist_port), + ) + torch.cuda.set_device(0) + from specforge.distributed import init_distributed + + init_distributed(timeout=10, tp_size=1) + target_config = AutoConfig.from_pretrained(args.target_model) + architectures = ( + ("dflash", "dflash2") if args.algorithm == "both" else (args.algorithm,) + ) + report = { + "created_utc": datetime.now(timezone.utc).isoformat(), + "settings": { + key: str(value) if isinstance(value, Path) else value + for key, value in vars(args).items() + }, + "torch_version": torch.__version__, + "gpu": torch.cuda.get_device_name(), + "target_config": target_config.to_dict(), + "warmup_runs": [], + "measured_runs": [], + "timing_contract": "first capture dispatch -> final synchronous durable ack + CUDA sync, including configured periodic checkpoints before the final step; final checkpoint and cleanup excluded from pipeline_seconds but included in trainer_fit_seconds_from_first_capture; all checkpoint events separately timed", + "scope": "actual target weights and target embeddings/head; freshly initialized full draft; single-rank canonical trainer with actual wrapper/sharding recorded per run", + "prompt_source": ( + str(args.prompts_path) + if args.prompts_path + else "synthetic deterministic token prompts" + ), + "prompt_file_sha256": ( + hashlib.sha256(args.prompts_path.read_bytes()).hexdigest() + if args.prompts_path + else None + ), + "server_startup": "existing server: startup excluded, not measured by this process", + "capture_cache_policy": "canonical adapter generates a fresh extra_key per request attempt, forcing full prefill", + "producer": "canonical drive_producer in parallel thread, one worker/concurrency=1, bounded ref backlog", + "async_ack": False, + } + + def save(): + report["summary"] = _summary(report["measured_runs"]) + temporary = args.output.with_suffix(args.output.suffix + ".tmp") + temporary.write_text(json.dumps(report, indent=2, default=str)) + temporary.replace(args.output) + + try: + for architecture in architectures: + for packing in (False, True): + row = run_pipeline( + args, + target_config, + architecture, + packing, + args.warmup_steps, + "warmup", + ) + row["cold_first_use_for_mode"] = True + report["warmup_runs"].append(row) + save() + gc.collect() + torch.cuda.empty_cache() + expected = None + for repeat in range(args.repeats): + for index, packing in enumerate((False, True, True, False)): + label = f"repeat{repeat:02d}-arm{index}" + row = run_pipeline( + args, target_config, architecture, packing, args.steps, label + ) + identity = ( + row["prompt_sha256"], + row["initial_draft_boundary_fingerprint"], + row["actual_capture_task_order"], + row["actual_consumed_task_order"], + row["actual_request_sha256"], + ) + if expected is None: + expected = identity + elif identity != expected: + raise AssertionError( + "A/B prompt order or draft initialization differs" + ) + report["measured_runs"].append(row) + save() + gc.collect() + torch.cuda.empty_cache() + print( + json.dumps( + { + key: row[key] + for key in ( + "architecture", + "packing", + "label", + "pipeline_seconds", + "useful_tokens_per_second", + "checkpoint_seconds", + "capture_calls_after_first_optimizer", + ) + } + ), + flush=True, + ) + finally: + save() + if own_group and dist.is_initialized(): + dist.destroy_process_group() + + +if __name__ == "__main__": + main() diff --git a/specforge/benchmarks/benchmark_sequence_packing.py b/specforge/benchmarks/benchmark_sequence_packing.py new file mode 100644 index 000000000..8cac474be --- /dev/null +++ b/specforge/benchmarks/benchmark_sequence_packing.py @@ -0,0 +1,460 @@ +"""Compare padded and packed EAGLE3 training on identical synthetic features. + +This exercises the production collators, Eagle3TrainStrategy, OnlineEagle3Model, +LlamaForCausalLMEagle3, TargetHead preprocessing/projection, and BF16Optimizer. +No model downloads are needed. It measures one GPU with resident input features; +target feature generation, disk loading, H2D transfers, and distributed training +are outside the measurement. Synthetic features do not establish model quality +or speculative serving speedups. + +Run a strict small FP32 gate before the representative BF16 benchmark:: + + PYTHONPATH=. python scripts/benchmark_sequence_packing.py --preset tiny \ + --dtype float32 --correctness-only --output /tmp/packing-correctness.json + PYTHONPATH=. python scripts/benchmark_sequence_packing.py --preset medium \ + --warmup 5 --steps 20 --output /tmp/packing-perf.json + +Each --lengths argument defines one batch (repeat it for several padding ratios). +For example: --lengths 1024,1024,1024,1024 --lengths 128,256,512,2048. +""" + +from __future__ import annotations + +import argparse +import gc +import hashlib +import importlib.metadata +import json +import os +import platform +import statistics +import subprocess +import time +from datetime import datetime, timezone +from pathlib import Path +from types import SimpleNamespace + +import torch +from transformers import LlamaConfig + +from specforge.algorithms.eagle3.data import DataCollatorWithPacking +from specforge.algorithms.eagle3.model import OnlineEagle3Model +from specforge.data.utils import DataCollatorWithPadding +from specforge.modeling.draft.llama3_eagle import LlamaForCausalLMEagle3 +from specforge.modeling.target.target_head import TargetHead +from specforge.optimizer import BF16Optimizer +from specforge.runtime.contracts import TrainBatch +from specforge.training.strategies.base import Eagle3TrainStrategy + + +class SyntheticTargetHead(TargetHead): + """Random frozen head with the production forward and preprocess methods.""" + + def __init__(self, hidden_size: int, vocab_size: int): + torch.nn.Module.__init__(self) + self.hidden_size = hidden_size + self.vocab_size = vocab_size + self.config = SimpleNamespace(hidden_size=hidden_size, vocab_size=vocab_size) + self.fc = torch.nn.Linear(hidden_size, vocab_size, bias=False) + self.freeze_weights() + + +PRESETS = { + "tiny": { + "hidden_size": 128, + "intermediate_size": 256, + "num_heads": 4, + "num_kv_heads": 2, + "vocab_size": 512, + "draft_vocab_size": 256, + "lengths": [[8, 17, 31, 64], [2, 3, 5, 17], [32, 32, 32, 32]], + }, + "medium": { + "hidden_size": 2048, + "intermediate_size": 8192, + "num_heads": 16, + "num_kv_heads": 4, + "vocab_size": 32000, + "draft_vocab_size": 32000, + "lengths": [ + [1024, 1024, 1024, 1024], + [512, 768, 1024, 2048], + [128, 256, 512, 2048], + ], + }, + "large": { + "hidden_size": 4096, + "intermediate_size": 14336, + "num_heads": 32, + "num_kv_heads": 8, + "vocab_size": 32000, + "draft_vocab_size": 32000, + "lengths": [[128, 256, 512, 2048]], + }, +} + + +def parse_args(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--preset", choices=PRESETS, default="medium") + for key in PRESETS["tiny"]: + if key != "lengths": + parser.add_argument("--" + key.replace("_", "-"), type=int) + parser.add_argument("--target-hidden-size", type=int) + parser.add_argument("--lengths", action="append", help="comma-separated lengths") + parser.add_argument("--ttt-length", type=int, default=7) + parser.add_argument("--dtype", choices=["float32", "bfloat16"], default="bfloat16") + parser.add_argument("--warmup", type=int, default=5) + parser.add_argument("--steps", type=int, default=20) + parser.add_argument("--seed", type=int, default=1729) + parser.add_argument("--learning-rate", type=float, default=1e-4) + parser.add_argument("--prompt-fraction", type=float, default=0.25) + parser.add_argument("--correctness-only", action="store_true") + parser.add_argument("--skip-correctness", action="store_true") + parser.add_argument("--atol", type=float) + parser.add_argument("--rtol", type=float) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + preset = PRESETS[args.preset] + for key, value in preset.items(): + if getattr(args, key) is None: + setattr(args, key, value) + if isinstance(args.lengths[0], str): + args.lengths = [[int(item) for item in row.split(",")] for row in args.lengths] + args.target_hidden_size = args.target_hidden_size or args.hidden_size + args.atol = ( + args.atol + if args.atol is not None + else (2e-5 if args.dtype == "float32" else 2e-3) + ) + args.rtol = ( + args.rtol + if args.rtol is not None + else (2e-4 if args.dtype == "float32" else 2e-2) + ) + if any(not row or min(row) < 1 for row in args.lengths): + parser.error("each batch must contain positive sequence lengths") + if args.steps < 1 or args.warmup < 1 or args.ttt_length < 1: + parser.error("steps, warmup, and ttt-length must be positive") + if not 0 <= args.prompt_fraction < 1: + parser.error("prompt-fraction must be in [0, 1)") + if args.draft_vocab_size > args.vocab_size: + parser.error("draft-vocab-size cannot exceed vocab-size") + if args.hidden_size % args.num_heads or args.num_heads % args.num_kv_heads: + parser.error( + "hidden-size / num-heads and num-heads / num-kv-heads must be integral" + ) + if args.correctness_only and args.skip_correctness: + parser.error("correctness-only and skip-correctness are incompatible") + return args + + +def build_strategy(args): + torch.manual_seed(args.seed) + config = LlamaConfig( + hidden_size=args.hidden_size, + target_hidden_size=args.target_hidden_size, + intermediate_size=args.intermediate_size, + num_attention_heads=args.num_heads, + num_key_value_heads=args.num_kv_heads, + num_hidden_layers=1, + vocab_size=args.vocab_size, + draft_vocab_size=args.draft_vocab_size, + max_position_embeddings=max(map(max, args.lengths)) + args.ttt_length, + pad_token_id=0, + attention_dropout=0.0, + rms_norm_eps=1e-5, + tie_word_embeddings=False, + ) + draft = LlamaForCausalLMEagle3(config, attention_backend="flex_attention") + # Exercise the real target-to-draft mapping, including a reduced vocabulary. + selected = torch.randperm(args.vocab_size)[: args.draft_vocab_size].sort().values + draft.t2d.zero_() + draft.t2d[selected] = True + draft.d2t.copy_(selected - torch.arange(args.draft_vocab_size)) + draft.freeze_embedding() + dtype = getattr(torch, args.dtype) + model = OnlineEagle3Model( + draft, length=args.ttt_length, attention_backend="flex_attention" + ).to(device="cuda", dtype=dtype) + head = ( + SyntheticTargetHead(args.target_hidden_size, args.vocab_size) + .to(device="cuda", dtype=dtype) + .eval() + ) + model.train() + return Eagle3TrainStrategy(model, target_head=head) + + +def make_features(args, lengths): + generator = torch.Generator().manual_seed(args.seed + 1) + dtype = getattr(torch, args.dtype) + features = [] + for length in lengths: + loss_mask = torch.ones(1, length, dtype=torch.long) + loss_mask[:, : int(length * args.prompt_fraction)] = 0 + loss_mask[:, -1] = 0 # The production offline normalizer does this. + features.append( + { + "input_ids": torch.randint( + 1, args.vocab_size, (1, length), generator=generator + ), + "attention_mask": torch.ones(1, length, dtype=torch.long), + "loss_mask": loss_mask, + "hidden_state": torch.randn( + 1, length, 3 * args.target_hidden_size, generator=generator + ).to(dtype), + "target": torch.randn( + 1, length, args.target_hidden_size, generator=generator + ).to(dtype), + } + ) + return features + + +def make_batch(features, mode): + collator = ( + DataCollatorWithPadding() if mode == "padded" else DataCollatorWithPacking() + ) + tensors = collator(features) + # Keep packing control metadata on CPU, matching the production loader. + tensors = { + name: ( + value if name in {"sequence_lengths", "loss_denominator"} else value.cuda() + ) + for name, value in tensors.items() + } + return TrainBatch( + sample_ids=[str(i) for i in range(len(features))], + strategy="eagle3", + tensors=tensors, + metadata={"target_repr": "hidden_state"}, + ) + + +def tensor_comparison(reference, actual, args): + reference, actual = reference.float(), actual.float() + delta = actual - reference + reference_norm = float(reference.norm()) + return { + "pass": bool(torch.allclose(reference, actual, atol=args.atol, rtol=args.rtol)), + "max_abs_diff": float(delta.abs().max()), + "relative_l2_diff": float(delta.norm()) / max(reference_norm, 1e-30), + "reference_l2": reference_norm, + } + + +def check_correctness(strategy, features, args): + model = strategy.trainable_module() + outputs = {} + for mode in ("padded", "packed"): + model.zero_grad(set_to_none=True) + batch = make_batch(features, mode) + result = strategy.forward_loss(batch) + result.loss.backward() + outputs[mode] = { + "loss": result.loss.detach().float().cpu(), + "plosses": torch.stack(result.metrics["plosses"]).float().cpu(), + "grads": { + name: None if param.grad is None else param.grad.detach().float().cpu() + for name, param in model.named_parameters() + if param.requires_grad + }, + } + del batch, result + checks = { + key: tensor_comparison(outputs["padded"][key], outputs["packed"][key], args) + for key in ("loss", "plosses") + } + checks["loss_padded"] = float(outputs["padded"]["loss"]) + checks["loss_packed"] = float(outputs["packed"]["loss"]) + checks["gradients"] = {} + for name, reference in outputs["padded"]["grads"].items(): + actual = outputs["packed"]["grads"][name] + checks["gradients"][name] = ( + {"pass": reference is None and actual is None, "missing_gradient": True} + if reference is None or actual is None + else tensor_comparison(reference, actual, args) + ) + checks["pass"] = all(checks[key]["pass"] for key in ("loss", "plosses")) and all( + row["pass"] for row in checks["gradients"].values() + ) + model.zero_grad(set_to_none=True) + return checks + + +def time_mode(strategy, features, lengths, mode, args, initial_state): + model = strategy.trainable_module() + model.load_state_dict(initial_state) + model.zero_grad(set_to_none=True) + batch = make_batch(features, mode) + optimizer = BF16Optimizer( + model, + lr=args.learning_rate, + warmup_ratio=0.0, + total_steps=args.warmup + args.steps + 1, + lr_scheduler="constant", + ) + + def step(): + output = strategy.forward_loss(batch) + output.loss.backward() + optimizer.step() + return output.loss.detach() + + torch.cuda.synchronize() + warm_start = time.perf_counter() + for _ in range(args.warmup): + step() + torch.cuda.synchronize() + warmup_seconds = time.perf_counter() - warm_start + torch.cuda.reset_peak_memory_stats() + baseline_bytes = torch.cuda.memory_allocated() + durations = [] + for _ in range(args.steps): + torch.cuda.synchronize() + start = time.perf_counter() + final_loss = step() + torch.cuda.synchronize() + durations.append(time.perf_counter() - start) + peak_bytes = torch.cuda.max_memory_allocated() + mean_seconds = statistics.mean(durations) + result = { + "mean_step_ms": mean_seconds * 1000, + "p50_step_ms": statistics.median(durations) * 1000, + "stdev_step_ms": ( + statistics.stdev(durations) * 1000 if len(durations) > 1 else 0 + ), + "useful_tokens_per_second": sum(lengths) / mean_seconds, + "useful_ttt_positions_per_second": sum(lengths) + * args.ttt_length + / mean_seconds, + "peak_allocated_gib": peak_bytes / 2**30, + "baseline_allocated_gib": baseline_bytes / 2**30, + "peak_increment_gib": (peak_bytes - baseline_bytes) / 2**30, + "warmup_seconds_including_compile": warmup_seconds, + "step_ms": [duration * 1000 for duration in durations], + "final_loss": float(final_loss), + } + optimizer = batch = None + model.zero_grad(set_to_none=True) + gc.collect() + torch.cuda.empty_cache() + return result + + +def source_state(): + root = Path(__file__).resolve().parents[2] + source_paths = [ + "scripts/benchmark_sequence_packing.py", + "specforge/benchmarks/benchmark_sequence_packing.py", + "specforge/algorithms/eagle3/data.py", + "specforge/algorithms/eagle3/model.py", + "specforge/modeling/draft/llama3_eagle.py", + "specforge/modeling/packed_sequence.py", + "specforge/training/strategies/base.py", + ] + result = { + "files_sha256": { + name: hashlib.sha256((root / name).read_bytes()).hexdigest() + for name in source_paths + } + } + + def git(*command): + return subprocess.check_output( + ["git", *command], cwd=root, text=True, stderr=subprocess.DEVNULL + ).strip() + + try: + result.update(head=git("rev-parse", "HEAD"), dirty=git("status", "--short")) + except (OSError, subprocess.CalledProcessError): + result["head"] = "unavailable (file hashes identify copied snapshot)" + return result + + +def write_report(report, path): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(report, indent=2) + "\n") + + +def main(): + args = parse_args() + if not torch.cuda.is_available(): + raise RuntimeError("This benchmark requires a CUDA GPU") + torch.backends.cuda.matmul.allow_tf32 = False + torch.backends.cudnn.allow_tf32 = False + torch.set_num_threads(4) + settings = vars(args).copy() + settings["output"] = str(args.output) + report = { + "timestamp_utc": datetime.now(timezone.utc).isoformat(), + "source": source_state(), + "environment": { + "python": platform.python_version(), + "torch": torch.__version__, + "cuda": torch.version.cuda, + "gpu": torch.cuda.get_device_name(), + "cuda_visible_devices": os.environ.get("CUDA_VISIBLE_DEVICES"), + "transformers": importlib.metadata.version("transformers"), + }, + "settings": settings, + "scope": "single-GPU synthetic offline features; resident inputs; production forward/backward/BF16Optimizer; excludes capture, I/O, transfer, distributed communication, and serving", + "cases": [], + } + strategy = build_strategy(args) + initial_state = { + key: value.detach().cpu().clone() + for key, value in strategy.trainable_module().state_dict().items() + } + for lengths in args.lengths: + strategy.trainable_module().load_state_dict(initial_state) + features = make_features(args, lengths) + case = { + "lengths": lengths, + "useful_tokens": sum(lengths), + "padded_tokens": len(lengths) * max(lengths), + "padding_fraction": 1 - sum(lengths) / (len(lengths) * max(lengths)), + "raw_supervised_tokens": sum( + int(item["loss_mask"].sum()) for item in features + ), + } + report["cases"].append(case) + print( + json.dumps( + {"case_start": lengths, "padding_fraction": case["padding_fraction"]} + ), + flush=True, + ) + if not args.skip_correctness: + case["correctness"] = check_correctness(strategy, features, args) + write_report(report, args.output) + print( + json.dumps({"correctness_pass": case["correctness"]["pass"]}), + flush=True, + ) + if not case["correctness"]["pass"]: + raise AssertionError( + f"Packed loss/gradient parity failed; inspect {args.output}" + ) + if not args.correctness_only: + for mode in ("padded", "packed"): + case[mode] = time_mode( + strategy, features, lengths, mode, args, initial_state + ) + write_report(report, args.output) + print(json.dumps({"mode": mode, **case[mode]}), flush=True) + case["speedup"] = ( + case["padded"]["mean_step_ms"] / case["packed"]["mean_step_ms"] + ) + case["peak_memory_reduction_fraction"] = 1 - ( + case["packed"]["peak_allocated_gib"] + / case["padded"]["peak_allocated_gib"] + ) + write_report(report, args.output) + del features + print(f"Report written to {args.output}", flush=True) + + +if __name__ == "__main__": + main() diff --git a/specforge/config/schema.py b/specforge/config/schema.py index 32eed42cf..9685bbe5a 100644 --- a/specforge/config/schema.py +++ b/specforge/config/schema.py @@ -898,6 +898,9 @@ class TrainingConfig(StrictConfigModel): max_steps: Optional[int] = Field(default=None, gt=0) total_steps: Optional[int] = Field(default=None, gt=0) batch_size: int = Field(default=1, gt=0) + #: Concatenate the samples of each microbatch, retaining document boundaries + #: and the original loss normalization. Text EAGLE3/DFlash/DFlash2 + FlexAttention. + sequence_packing: bool = False accumulation_steps: int = Field(default=1, gt=0) fsdp_sharding: Literal["SHARD_GRAD_OP", "FULL_SHARD", "NO_SHARD"] = "SHARD_GRAD_OP" learning_rate: float = Field(default=1e-4, gt=0.0) @@ -991,6 +994,14 @@ class TrainingConfig(StrictConfigModel): @model_validator(mode="after") def _validate_training_shape(self): + if self.sequence_packing: + if self.attention_backend != "flex_attention": + raise ValueError("training.sequence_packing requires flex_attention") + if self.compact_teacher or self.trim_loss_positions: + raise ValueError( + "training.sequence_packing currently requires compact_teacher=false, " + "trim_loss_positions=false" + ) if not 0.0 <= self.dpace_alpha <= 1.0: raise ValueError("training.dpace_alpha must be in [0, 1]") if not 0.0 < self.down_sample_ratio <= 1.0: diff --git a/specforge/launch.py b/specforge/launch.py index 030e5388f..02a798ff4 100644 --- a/specforge/launch.py +++ b/specforge/launch.py @@ -163,10 +163,19 @@ def _offline_io( *, ttt_length: int, use_usp_preprocess: bool, + sequence_packing: bool = False, ): """Resolve the algorithm-owned normalizer and collator for one modality.""" provider = algorithm.providers.offline_for(modality) - return provider.build_collator(), provider.build_normalizer( + if sequence_packing: + if use_usp_preprocess or provider.build_packed_collator is None: + raise ValueError( + "sequence_packing requires a supported non-USP offline provider" + ) + collator = provider.build_packed_collator() + else: + collator = provider.build_collator() + return collator, provider.build_normalizer( max_len, ttt_length=ttt_length, use_usp_preprocess=use_usp_preprocess, @@ -252,6 +261,7 @@ def _make_offline_eval_data_factory( ttt_length: int, use_usp_preprocess: bool, dataloader_num_workers: int, + sequence_packing: bool = False, ): """Build a fresh re-iterable eval loader over the offline feature path.""" provider = algorithm.providers.offline_for(modality) @@ -261,6 +271,7 @@ def _make_offline_eval_data_factory( max_len, ttt_length=ttt_length, use_usp_preprocess=use_usp_preprocess, + sequence_packing=sequence_packing, ) eval_run_id = f"{run_id}-eval" refs = provider.build_reader( @@ -293,11 +304,22 @@ def _streaming_collate( algorithm: AlgorithmRegistration, modality: str, collate_fn, + *, + sequence_packing: bool = False, ): """Resolve an algorithm-owned server-streaming collator.""" if collate_fn is not None: + if sequence_packing: + raise ValueError( + "sequence_packing cannot be combined with a custom collate_fn" + ) return collate_fn - return algorithm.providers.server_streaming_for(modality).build_collator() + provider = algorithm.providers.server_streaming_for(modality) + if sequence_packing: + if provider.build_packed_collator is None: + raise ValueError("sequence_packing requires a supported streaming provider") + return provider.build_packed_collator() + return provider.build_collator() def _resolve_metadata_store( @@ -563,6 +585,7 @@ def build_offline_runtime( sp_ulysses_size: int = 1, sp_ring_size: int = 1, use_usp_preprocess: bool = False, + sequence_packing: bool = False, seed: int = 0, logger=None, log_interval: int = 50, @@ -586,6 +609,7 @@ def build_offline_runtime( max_len, ttt_length=ttt_length, use_usp_preprocess=use_usp_preprocess, + sequence_packing=sequence_packing, ) controller = DataFlowController( run_id, @@ -621,6 +645,7 @@ def refs_for_epoch(epoch): ttt_length=ttt_length, use_usp_preprocess=use_usp_preprocess, dataloader_num_workers=dataloader_num_workers, + sequence_packing=sequence_packing, ) return _assemble_trainer( algorithm=algorithm, @@ -689,6 +714,7 @@ def build_disagg_offline_runtime( sp_ulysses_size: int = 1, sp_ring_size: int = 1, use_usp_preprocess: bool = False, + sequence_packing: bool = False, seed: int = 0, logger=None, log_interval: int = 50, @@ -711,6 +737,7 @@ def build_disagg_offline_runtime( max_len, ttt_length=ttt_length, use_usp_preprocess=use_usp_preprocess, + sequence_packing=sequence_packing, ) source_refs = list(refs) @@ -743,6 +770,7 @@ def refs_for_epoch(epoch): ttt_length=ttt_length, use_usp_preprocess=use_usp_preprocess, dataloader_num_workers=dataloader_num_workers, + sequence_packing=sequence_packing, ) return _assemble_trainer( algorithm=algorithm, @@ -1536,6 +1564,7 @@ def build_disagg_online_consumer( eval_interval: int = 0, eval_data_factory=None, collate_fn=None, + sequence_packing: bool = False, idle_timeout_s: Optional[float] = None, metadata_store: Optional[MetadataStore] = None, metadata_db_path: Optional[str] = None, @@ -1888,7 +1917,9 @@ def stop_distributor_and_drain() -> None: eval_data_factory=eval_data_factory, logger=logger, log_interval=log_interval, - collate_fn=_streaming_collate(algorithm, modality, collate_fn), + collate_fn=_streaming_collate( + algorithm, modality, collate_fn, sequence_packing=sequence_packing + ), strategy_kwargs=strategy_kwargs, per_sample_transform=None, max_checkpoints=max_checkpoints, diff --git a/specforge/modeling/draft/llama3_eagle.py b/specforge/modeling/draft/llama3_eagle.py index 67276ec5c..6fe355c31 100644 --- a/specforge/modeling/draft/llama3_eagle.py +++ b/specforge/modeling/draft/llama3_eagle.py @@ -16,6 +16,10 @@ compile_friendly_flex_attention, generate_eagle3_mask, ) +from specforge.modeling.packed_sequence import ( + PackedSequenceLayout, + generate_packed_eagle3_mask, +) from specforge.utils import print_with_rank from ...distributed import get_sp_ring_group, get_sp_ulysses_group @@ -761,7 +765,12 @@ def forward( self.rope_scaling["mrope_section"], ) else: - cos, sin = self.rotary_emb(query_states, seq_len=q_len + lck) + rope_length = ( + attention_mask.maximum_length + if isinstance(attention_mask, PackedSequenceLayout) + else q_len + ) + cos, sin = self.rotary_emb(query_states, seq_len=rope_length + lck) cos, sin = cos.to(query_states.device), sin.to(query_states.device) # Keep positions ids aligned when padding so the KV cache is unaffected. query_states, key_states = apply_rotary_pos_emb( @@ -780,10 +789,16 @@ def forward( cache_kwargs=cache_kwargs, ) - seq_lengths = attention_mask.sum(dim=-1) - # Shrink the attention mask to align with the padding to the right. - # This is equivalent to the shrinking logic in eagle3.py - seq_lengths -= lck + if isinstance(attention_mask, PackedSequenceLayout): + mask_mod = generate_packed_eagle3_mask(attention_mask, q_len, lck) + else: + seq_lengths = attention_mask.sum(dim=-1) - lck + mask_mod = generate_eagle3_mask( + seq_lengths=seq_lengths, + Q_LEN=q_len, + KV_LEN=key_cache.shape[-2], + lck=lck, + ) # TODO: Remove the usage of uncompiled create_block_mask after # https://github.com/pytorch/pytorch/issues/160018 if q_len <= 128: @@ -794,12 +809,7 @@ def forward( flex_attention_func = compile_friendly_flex_attention block_mask = create_block_mask_func( - mask_mod=generate_eagle3_mask( - seq_lengths=seq_lengths, - Q_LEN=q_len, - KV_LEN=key_cache.shape[-2], - lck=lck, - ), + mask_mod=mask_mod, B=bsz, H=1, # Rely on broadcast Q_LEN=q_len, diff --git a/specforge/modeling/packed_dflash.py b/specforge/modeling/packed_dflash.py new file mode 100644 index 000000000..bda1c8ba2 --- /dev/null +++ b/specforge/modeling/packed_dflash.py @@ -0,0 +1,158 @@ +"""Index metadata for packing DFlash context while preserving sampled blocks.""" + +from dataclasses import dataclass + +import torch + +from specforge.modeling.packed_sequence import PackedSequenceLayout + + +@dataclass(frozen=True) +class PackedDFlashLayout: + tokens: PackedSequenceLayout + lengths: tuple[int, ...] + document_starts: torch.Tensor + padded_indices: torch.Tensor + padded_valid: torch.Tensor + + @classmethod + def from_lengths(cls, lengths, sequence_length, device): + tokens = PackedSequenceLayout.from_lengths(lengths, sequence_length, device) + lengths = torch.as_tensor(lengths, dtype=torch.long) + starts = lengths.cumsum(0) - lengths + columns = torch.arange(tokens.maximum_length) + indices = starts[:, None] + columns[None, :] + return cls( + tokens=tokens, + lengths=tuple(lengths.tolist()), + document_starts=starts.to(device), + padded_indices=indices.clamp(max=sequence_length - 1).to(device), + padded_valid=(columns[None, :] < lengths[:, None]).to(device), + ) + + def padded_loss_mask(self, loss_mask): + """Rebuild only the small sampling mask, preserving baseline RNG shape.""" + return loss_mask[0, self.padded_indices] * self.padded_valid + + def pack_anchors(self, local_anchors): + return (local_anchors + self.document_starts[:, None]).reshape(1, -1) + + def anchor_starts(self, packed_anchors): + return packed_anchors - self.tokens.positions[packed_anchors] + + def anchor_ends(self, packed_anchors): + return ( + self.anchor_starts(packed_anchors) + + self.tokens.document_lengths[packed_anchors] + ) + + def compact_anchor_indices(self, valid_anchor_counts, width): + """Indices of valid prefix slots after the unchanged sorted sampler.""" + if len(valid_anchor_counts) != len(self.lengths) or any( + not isinstance(count, int) or count < 0 for count in valid_anchor_counts + ): + raise ValueError( + "valid_anchor_counts must contain one nonnegative integer per document" + ) + return torch.tensor( + [ + document * width + index + for document, count in enumerate(valid_anchor_counts) + for index in range(min(count, width)) + ], + dtype=torch.long, + device=self.document_starts.device, + ) + + +def create_packed_dflash_block_mask( + anchor_positions, + block_keep_mask, + context_start_positions, + context_length, + proposal_size, + mask_mod, + *, + block_size=128, + sliding_window=None, +): + """Construct sparse tiles without materializing a quadratic token mask. + + A tile's context union is bounded by its earliest start and latest anchor; + its intersection identifies fully allowed context tiles. A conservative + union may include extra partial tiles, whose exact token predicate remains + ``mask_mod``. Draft tiles intersect only the proposals present in a Q tile. + """ + from torch.nn.attention.flex_attention import BlockMask + + q_block, kv_block = ( + (block_size, block_size) if isinstance(block_size, int) else block_size + ) + batch, anchors = anchor_positions.shape + q_length = anchors * proposal_size + kv_length = context_length + q_length + device = anchor_positions.device + q = torch.arange( + ((q_length + q_block - 1) // q_block) * q_block, device=device + ).reshape(-1, q_block) + anchor_index = (q // proposal_size).clamp(max=anchors - 1) + valid = (q < q_length).unsqueeze(0) & block_keep_mask[:, anchor_index] + anchor = anchor_positions[:, anchor_index] + lower = context_start_positions[:, anchor_index] + if sliding_window is not None: + lower = torch.maximum( + lower, anchor + q.remainder(proposal_size) - (sliding_window - 1) + ) + + context_min = torch.where(valid, lower, context_length).amin(-1) + context_max = torch.where(valid, anchor, 0).amax(-1) + intersection_min = torch.where(valid, lower, 0).amax(-1) + intersection_max = torch.where(valid, anchor, context_length).amin(-1) + any_valid = valid.any(-1) + all_valid = valid.all(-1) + + kv_start = ( + torch.arange((kv_length + kv_block - 1) // kv_block, device=device) * kv_block + ) + kv_end = kv_start + kv_block + context_tiles = ( + (kv_start < context_max[..., None]) + & (kv_end > context_min[..., None]) + & (kv_start < context_length) + & (context_min < context_max)[..., None] + ) + full_tiles = ( + all_valid[..., None] + & (kv_start >= intersection_min[..., None]) + & (kv_end <= intersection_max[..., None]) + & (kv_end <= context_length) + ) + draft_min = context_length + (q[:, 0] // proposal_size) * proposal_size + draft_max = context_length + torch.minimum( + (q[:, -1] // proposal_size + 1) * proposal_size, + q.new_tensor(q_length), + ) + draft_tiles = (kv_start < draft_max[:, None]) & (kv_end > draft_min[:, None]) + partial_tiles = any_valid[..., None] & (context_tiles | draft_tiles) & ~full_tiles + + def ordered(mask): + mask = mask.unsqueeze(1) + counts = mask.sum(-1, dtype=torch.int32) + indices = ( + mask.to(torch.int32) + .argsort(dim=-1, descending=True, stable=True) + .to(torch.int32) + ) + return counts, indices + + partial_counts, partial_indices = ordered(partial_tiles) + full_counts, full_indices = ordered(full_tiles) + return BlockMask.from_kv_blocks( + partial_counts, + partial_indices, + full_counts, + full_indices, + BLOCK_SIZE=(q_block, kv_block), + mask_mod=mask_mod, + seq_lengths=(q_length, kv_length), + ) diff --git a/specforge/modeling/packed_sequence.py b/specforge/modeling/packed_sequence.py new file mode 100644 index 000000000..2706f24bb --- /dev/null +++ b/specforge/modeling/packed_sequence.py @@ -0,0 +1,76 @@ +"""Document boundaries for packed, single-row EAGLE3 training batches.""" + +from dataclasses import dataclass + +import torch + + +@dataclass(frozen=True) +class PackedSequenceLayout: + """Token metadata shared by segment shifts and the Flex Attention mask. + + Lengths are small CPU metadata; token metadata lives beside the model's + tensors. Padding remains local to each document during every TTT shift. + """ + + document_ids: torch.Tensor + positions: torch.Tensor + document_lengths: torch.Tensor + padded_denominator: int + maximum_length: int + + @classmethod + def from_lengths(cls, lengths, sequence_length: int, device): + if not isinstance(lengths, torch.Tensor): + lengths = torch.tensor(lengths, dtype=torch.long) + if lengths.device.type != "cpu" or lengths.dtype != torch.long: + raise ValueError("sequence_lengths must be a CPU int64 tensor") + if lengths.ndim != 1 or not lengths.numel() or bool((lengths <= 0).any()): + raise ValueError("sequence_lengths must contain positive document lengths") + if int(lengths.sum()) != sequence_length: + raise ValueError("sequence_lengths must sum to the packed sequence length") + document_ids = torch.repeat_interleave(torch.arange(lengths.numel()), lengths) + starts = lengths.cumsum(0) - lengths + positions = torch.arange(sequence_length) - starts[document_ids] + return cls( + document_ids=document_ids.to(device), + positions=positions.to(device), + document_lengths=lengths[document_ids].to(device), + padded_denominator=int(lengths.numel() * lengths.max()), + maximum_length=int(lengths.max()), + ) + + def shift_left(self, tensor: torch.Tensor) -> torch.Tensor: + """Drop each document's first entry and append one zero to that document.""" + if tensor.shape[:2] != (1, self.positions.numel()): + raise ValueError( + "packed tensors must have shape [1, sum(sequence_lengths), ...]" + ) + shifted = torch.cat((tensor[:, 1:], torch.zeros_like(tensor[:, -1:])), dim=1) + valid = (self.positions + 1 < self.document_lengths).to(tensor.device) + return shifted * valid.reshape(1, -1, *([1] * (tensor.ndim - 2))) + + +def generate_packed_eagle3_mask( + layout: PackedSequenceLayout, query_length: int, depth: int +): + """Match independent EAGLE3 causal prefixes and diagonal TTT cache suffixes.""" + document_ids = layout.document_ids + positions = layout.positions + lengths = layout.document_lengths + + def mask_mod(_b, _h, q_idx, kv_idx): + # Flex evaluates complete tiles, including indices outside the actual Q/KV + # sizes. Clamp metadata reads; the explicit bounds mask removes those cells. + safe_q = q_idx.clamp(max=query_length - 1) + kv_row = kv_idx % query_length + same_document = document_ids[safe_q] == document_ids[kv_row] + valid_query = (q_idx < query_length) & ( + positions[safe_q] < lengths[safe_q] - depth + ) + valid_key = positions[kv_row] < lengths[kv_row] - depth + causal = (kv_idx < query_length) & (q_idx >= kv_idx) + suffix = (kv_idx >= query_length) & (kv_row == q_idx) + return same_document & valid_query & valid_key & (causal | suffix) + + return mask_mod diff --git a/specforge/training/assembly.py b/specforge/training/assembly.py index 07e63405c..26cca714f 100644 --- a/specforge/training/assembly.py +++ b/specforge/training/assembly.py @@ -645,6 +645,7 @@ def build_training_run( max_len=cfg.data.max_length, num_epochs=t.num_epochs, use_usp_preprocess=(t.attention_backend == "usp"), + sequence_packing=t.sequence_packing, seed=t.seed, resume_from=t.resume_from, **_common_launch_kwargs( diff --git a/specforge/training/disaggregated.py b/specforge/training/disaggregated.py index b30eb274c..b91c9d787 100644 --- a/specforge/training/disaggregated.py +++ b/specforge/training/disaggregated.py @@ -559,6 +559,7 @@ def produce() -> int: sp_ulysses_size=cfg.training.sp_ulysses_size, sp_ring_size=cfg.training.sp_ring_size, use_usp_preprocess=(cfg.training.attention_backend == "usp"), + sequence_packing=cfg.training.sequence_packing, seed=cfg.training.seed, dataloader_num_workers=_dataloader_num_workers(cfg, algorithm), profiling_options=_profiling_options(cfg), @@ -820,6 +821,7 @@ def produce() -> int: run_id=cfg.run_id, output_dir=cfg.output_dir, batch_size=cfg.training.batch_size, + sequence_packing=cfg.training.sequence_packing, accumulation_steps=cfg.training.accumulation_steps, max_steps=cfg.training.max_steps, total_steps=total_steps, diff --git a/specforge/training/strategies/base.py b/specforge/training/strategies/base.py index f592ccdc0..0f84981e1 100644 --- a/specforge/training/strategies/base.py +++ b/specforge/training/strategies/base.py @@ -70,7 +70,21 @@ def linear_lambda_base( return max(0.0, min(1.0, lambda_start * (1.0 - progress))) -def _cpu_max_valid_anchors(loss_mask: torch.Tensor) -> Optional[int]: +def _cpu_valid_anchor_counts( + loss_mask: torch.Tensor, sequence_lengths: Tuple[int, ...] +) -> Optional[Tuple[int, ...]]: + """Count packed document anchors on the host before feature H2D copies.""" + if loss_mask.device.type != "cpu": + return None + return tuple( + int(((mask[:-1] > 0.5) & (mask[1:] > 0.5)).sum()) + for mask in loss_mask[0].split(sequence_lengths) + ) + + +def _cpu_max_valid_anchors( + loss_mask: torch.Tensor, sequence_lengths: Optional[Tuple[int, ...]] = None +) -> Optional[int]: """Count the widest valid anchor row without synchronizing the GPU. Online/offline loaders hand strategies CPU integer features; Mooncake @@ -82,6 +96,10 @@ def _cpu_max_valid_anchors(loss_mask: torch.Tensor) -> Optional[int]: """ if loss_mask.device.type != "cpu": return None + if sequence_lengths is not None: + # Preserve the per-document anchor budget, not the sum across a packed row. + counts = _cpu_valid_anchor_counts(loss_mask, sequence_lengths) + return max(counts, default=0) num_candidates = max(loss_mask.shape[1] - 1, 0) valid = (loss_mask[:, :num_candidates] > 0.5) & ( loss_mask[:, 1 : num_candidates + 1] > 0.5 @@ -128,9 +146,9 @@ def _prepare_eagle_target( ) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]: """Normalize EAGLE-family teacher features for a training forward. - Online capture already shifts logits and input IDs. Offline capture stores - the target model's final hidden state, so the frozen target head owns the - equivalent shift and projection to full-vocabulary logits. + Raw hidden-state features from offline readers or online server capture + require the frozen target head's shift and full-vocabulary projection. + Preprocessed logits and their aligned input IDs are used as delivered. """ if target_repr == "hidden_state": if target_head is None: @@ -278,7 +296,46 @@ def forward_loss( target_repr = batch.metadata.get("target_repr") compact_kwargs: Dict[str, Any] = {} - if self.compact_teacher: + packed_kwargs: Dict[str, Any] = {} + sequence_lengths = t.get("sequence_lengths") + if sequence_lengths is not None: + if self.compact_teacher or self.trim_loss_positions: + raise ValueError( + "sequence packing does not support compact_teacher or trim_loss_positions" + ) + if target_repr != "hidden_state" or self.target_head is None: + raise ValueError( + "sequence packing requires unshifted hidden_state targets and a target head" + ) + from specforge.modeling.packed_sequence import PackedSequenceLayout + + layout = PackedSequenceLayout.from_lengths( + sequence_lengths, t["input_ids"].shape[1], t["input_ids"].device + ) + # TargetHead's ordinary global shift would import the next document's + # first token into this document's last row. Shift all three fields + # within document boundaries before projecting the frozen teacher. + input_ids = layout.shift_left(t["input_ids"]).to(device, non_blocking=True) + target_hidden = layout.shift_left(t["target"]) + target = self.target_head(target_hidden.to(device, non_blocking=True)) + loss_mask = layout.shift_left(t["loss_mask"])[..., None].to( + device, non_blocking=True + ) + loss_denominator = t.get("loss_denominator") + if loss_denominator is not None and ( + loss_denominator.device.type != "cpu" or loss_denominator.numel() != 1 + ): + raise ValueError("loss_denominator must be a scalar CPU tensor") + # FSDP moves tensor kwargs onto its compute device. Keep the small + # control metadata as Python values so it stays host-side without a + # GPU synchronization when the wrapped model builds its layout. + packed_kwargs = { + "sequence_lengths": tuple(sequence_lengths.tolist()), + "loss_denominator": ( + int(loss_denominator) if loss_denominator is not None else None + ), + } + elif self.compact_teacher: if target_repr != "hidden_state": raise ValueError( "compact teacher is offline-only and requires " @@ -326,6 +383,7 @@ def forward_loss( else None ), trim_loss_positions=self.trim_loss_positions, + **packed_kwargs, **compact_kwargs, ) weights = [self.ploss_decay**i for i in range(len(plosses))] @@ -516,7 +574,33 @@ def forward_loss( t = batch.tensors device = self._device() selector_loss_alpha = self._selector_loss_alpha(ctx) - max_valid_anchors = _cpu_max_valid_anchors(t["loss_mask"]) + sequence_lengths = t.get("sequence_lengths") + if sequence_lengths is not None: + if ( + sequence_lengths.device.type != "cpu" + or sequence_lengths.dtype != torch.long + or sequence_lengths.ndim != 1 + or not sequence_lengths.numel() + or bool((sequence_lengths <= 0).any()) + or t["input_ids"].shape[0] != 1 + or int(sequence_lengths.sum()) != t["input_ids"].shape[1] + ): + raise ValueError( + "sequence_lengths must be positive CPU int64 document lengths for one packed row" + ) + # FSDP moves tensor kwargs to the GPU; small control metadata must + # remain host-side so packing does not introduce a device sync. + sequence_lengths = tuple(sequence_lengths.tolist()) + valid_anchor_counts = ( + _cpu_valid_anchor_counts(t["loss_mask"], sequence_lengths) + if sequence_lengths is not None + else None + ) + max_valid_anchors = ( + max(valid_anchor_counts, default=0) + if valid_anchor_counts is not None + else _cpu_max_valid_anchors(t["loss_mask"]) + ) collect_detailed_metrics = ( ctx.collect_detailed_metrics if ctx is not None else True ) @@ -527,6 +611,10 @@ def forward_loss( "max_valid_anchors": max_valid_anchors, "selector_loss_alpha": selector_loss_alpha, } + if sequence_lengths is not None: + model_inputs["sequence_lengths"] = sequence_lengths + if valid_anchor_counts is not None: + model_inputs["valid_anchor_counts"] = valid_anchor_counts if ctx is not None: model_inputs["collect_detailed_metrics"] = collect_detailed_metrics target_last_hidden_states = t.get("target_last_hidden_states") diff --git a/tests/test_config/test_sequence_packing.py b/tests/test_config/test_sequence_packing.py new file mode 100644 index 000000000..99c6a0add --- /dev/null +++ b/tests/test_config/test_sequence_packing.py @@ -0,0 +1,148 @@ +"""Reject unsupported packing combinations before building any model.""" + +import unittest +from dataclasses import replace + +from pydantic import ValidationError + +from specforge.algorithms.builtin import builtin_algorithm_registry +from specforge.algorithms.eagle3.data import DataCollatorWithPacking +from specforge.application import resolve_run +from specforge.config import Config +from specforge.config.schema import TrainingConfig + + +def _offline_config(**training): + return Config.model_validate( + { + "model": { + "target_model_path": "target", + "draft_model_config": "draft.json", + "vocab_mapping_path": "mapping.pt", + }, + "data": {"hidden_states_path": "features"}, + "training": training, + } + ) + + +class SequencePackingConfigTest(unittest.TestCase): + def test_default_retains_padded_execution(self): + resolved = resolve_run(_offline_config()) + self.assertFalse(resolved.config.training.sequence_packing) + + def test_offline_eagle3_packing_is_available(self): + resolved = resolve_run(_offline_config(sequence_packing=True, batch_size=4)) + self.assertTrue(resolved.config.training.sequence_packing) + provider = resolved.algorithm.providers.offline_for("text") + self.assertIsInstance(provider.build_packed_collator(), DataCollatorWithPacking) + + def test_other_algorithms_reject_packing(self): + for strategy in ("domino", "dspark"): + with ( + self.subTest(strategy=strategy), + self.assertRaisesRegex( + ValueError, "does not support training.sequence_packing" + ), + ): + resolve_run(_offline_config(strategy=strategy, sequence_packing=True)) + + def test_rejects_non_flex_attention(self): + for attention_backend in ("eager", "sdpa", "fa", "usp"): + with ( + self.subTest(backend=attention_backend), + self.assertRaisesRegex( + ValidationError, "sequence_packing requires flex_attention" + ), + ): + TrainingConfig( + sequence_packing=True, attention_backend=attention_backend + ) + + def test_rejects_unimplemented_objective_combinations(self): + for extra in ( + {"compact_teacher": True}, + {"trim_loss_positions": True}, + ): + with ( + self.subTest(extra=extra), + self.assertRaisesRegex( + ValidationError, "sequence_packing currently requires" + ), + ): + TrainingConfig(sequence_packing=True, **extra) + + def test_lk_packing_is_algorithm_specific(self): + for lk_loss_type in ("lambda", "alpha", "tv"): + with self.subTest(lk_loss_type=lk_loss_type): + with self.assertRaisesRegex(ValueError, "LK|lk_loss"): + resolve_run( + _offline_config( + sequence_packing=True, lk_loss_type=lk_loss_type + ) + ) + resolve_run( + _offline_config( + strategy="dflash", + sequence_packing=True, + lk_loss_type=lk_loss_type, + ) + ) + + def test_supports_dflash_offline_and_dflash2_architecture(self): + resolved = resolve_run( + _offline_config(strategy="dflash", sequence_packing=True) + ) + self.assertTrue( + callable( + resolved.algorithm.providers.offline_for("text").build_packed_collator + ) + ) + self.assertIn( + "DFlash2DraftModel", resolved.algorithm.spec.draft.compatible_architectures + ) + + def test_supports_online_text_features(self): + payload = _offline_config(sequence_packing=True).model_dump() + payload["data"] = {"train_data_path": "train.jsonl"} + payload["training"]["max_steps"] = 1 + payload["training"]["role"] = "auto" + payload["deployment"] = { + "mode": "disaggregated", + "disaggregated": { + "control_dir": "outputs/packing-test/control", + "backend": "mooncake", + "server_urls": ["http://127.0.0.1:30000"], + "mooncake_metadata_server": "http://127.0.0.1:35880/metadata", + "mooncake_master_server_addr": "127.0.0.1:35551", + }, + } + for strategy in ("eagle3", "dflash"): + payload["training"]["strategy"] = strategy + with self.subTest(strategy=strategy): + resolved = resolve_run(Config.model_validate(payload)) + provider = resolved.algorithm.providers.server_streaming_for("text") + self.assertTrue(callable(provider.build_packed_collator)) + + def test_provider_rejects_noncallable_packing_factory(self): + provider = ( + builtin_algorithm_registry().resolve("eagle3").providers.offline_for("text") + ) + with self.assertRaisesRegex( + TypeError, "build_packed_collator must be callable" + ): + replace(provider, build_packed_collator=True) + + streaming = ( + builtin_algorithm_registry() + .resolve("eagle3") + .providers.server_streaming_for("text") + ) + with self.assertRaisesRegex( + TypeError, "build_packed_collator must be callable" + ): + replace(streaming, build_packed_collator=True) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_data/test_sequence_packing.py b/tests/test_data/test_sequence_packing.py new file mode 100644 index 000000000..b47c976a4 --- /dev/null +++ b/tests/test_data/test_sequence_packing.py @@ -0,0 +1,239 @@ +"""Packing preserves logical samples while removing only batch padding (CPU).""" + +import tempfile +import unittest +from pathlib import Path + +import torch + +from specforge.algorithms.eagle3.data import ( + DataCollatorWithPacking, + build_offline_normalizer, + build_packed_collator, +) +from specforge.runtime.data_plane.feature_dataloader import FeatureDataLoader +from specforge.runtime.data_plane.feature_store import LocalFeatureStore +from specforge.runtime.data_plane.offline_reader import OfflineManifestReader +from specforge.runtime.data_plane.sample_ref_queue import SampleRefQueue + + +def _feature(length, offset=0): + return { + "input_ids": (torch.arange(length) + offset).unsqueeze(0), + "attention_mask": torch.ones(1, length, dtype=torch.long), + "loss_mask": (torch.arange(length) % 2).unsqueeze(0), + "hidden_state": torch.arange(length * 6).reshape(1, length, 6) + offset, + "target": torch.arange(length * 2).reshape(1, length, 2) + offset, + } + + +class SequencePackingCollatorTest(unittest.TestCase): + def test_concatenates_features_with_document_positions_and_padded_denominator(self): + features = [_feature(2, 10), _feature(5, 20), _feature(1, 30)] + originals = [{key: value.clone() for key, value in f.items()} for f in features] + + batch = build_packed_collator()(features) + + for key in features[0]: + with self.subTest(key=key): + torch.testing.assert_close( + batch[key], torch.cat([f[key] for f in originals], dim=1) + ) + self.assertEqual(batch["input_ids"].shape, (1, 8)) + self.assertEqual(batch["position_ids"].tolist(), [[0, 1, 0, 1, 2, 3, 4, 0]]) + self.assertEqual(batch["sequence_lengths"].tolist(), [2, 5, 1]) + self.assertEqual(batch["loss_denominator"].item(), 3 * 5) + self.assertEqual(batch["sequence_lengths"].dtype, torch.long) + self.assertEqual(batch["sequence_lengths"].device.type, "cpu") + self.assertEqual(batch["loss_denominator"].device.type, "cpu") + for feature, original in zip(features, originals): + self.assertEqual(feature.keys(), original.keys()) + for key in original: + torch.testing.assert_close(feature[key], original[key]) + + def test_accepts_standard_positions_and_single_sample(self): + feature = _feature(3) + feature["position_ids"] = torch.arange(3).unsqueeze(0) + + batch = DataCollatorWithPacking()([feature]) + + self.assertEqual(batch["position_ids"].tolist(), [[0, 1, 2]]) + self.assertEqual(batch["sequence_lengths"].tolist(), [3]) + self.assertEqual(batch["loss_denominator"].item(), 3) + + def test_rejects_empty_batch_and_empty_or_batched_sample(self): + with self.assertRaisesRegex(ValueError, "empty feature batch"): + DataCollatorWithPacking()([]) + for ids in (torch.empty(1, 0), torch.zeros(2, 3), torch.zeros(3)): + feature = _feature(3) + feature["input_ids"] = ids + with ( + self.subTest(shape=ids.shape), + self.assertRaisesRegex(ValueError, "nonempty"), + ): + DataCollatorWithPacking()([feature]) + + def test_rejects_missing_or_misaligned_features(self): + for key in _feature(3): + feature = _feature(3) + del feature[key] + with self.subTest(missing=key), self.assertRaisesRegex(KeyError, key): + DataCollatorWithPacking()([feature]) + for key in ("loss_mask", "attention_mask", "hidden_state", "target"): + for wrong in (torch.zeros(1, 2), torch.zeros(3), torch.zeros(2, 3, 6)): + feature = _feature(3) + feature[key] = wrong + with ( + self.subTest(key=key, shape=wrong.shape), + self.assertRaisesRegex(ValueError, key), + ): + DataCollatorWithPacking()([feature]) + + def test_rejects_padding_and_nonstandard_positions(self): + feature = _feature(3) + feature["attention_mask"][0, -1] = 0 + with self.assertRaisesRegex(ValueError, "unpadded"): + DataCollatorWithPacking()([feature]) + for positions in ( + torch.tensor([[1, 2, 3]]), + torch.tensor([[0, 0, 1]]), + torch.arange(3), + torch.arange(3).reshape(1, 1, 3), + ): + feature = _feature(3) + feature["position_ids"] = positions + with ( + self.subTest(shape=positions.shape), + self.assertRaisesRegex(ValueError, "standard text position_ids"), + ): + DataCollatorWithPacking()([feature]) + + def test_offline_loader_preserves_sample_ids_order_and_partial_batch(self): + with tempfile.TemporaryDirectory() as directory: + for index, length in enumerate((2, 5, 3)): + torch.save( + { + "input_ids": torch.arange(length) + index * 10, + "loss_mask": torch.ones(length, dtype=torch.long), + "hidden_state": torch.full((1, length, 2), float(index)), + "aux_hidden_state": torch.full((1, length, 6), float(index)), + }, + Path(directory) / f"{index:03d}.ckpt", + ) + refs = OfflineManifestReader(directory, run_id="packing-test").read() + loader = FeatureDataLoader( + LocalFeatureStore("packing-test"), + refs=refs, + batch_size=2, + collate_fn=build_packed_collator(), + per_sample_transform=build_offline_normalizer(4), + drop_last=False, + ) + batches = list(loader) + repeated = list(loader) + + expected_ids = [[r.sample_id for r in refs[:2]], [refs[2].sample_id]] + self.assertEqual([b.sample_ids for b in batches], expected_ids) + self.assertEqual([b.sample_ids for b in repeated], expected_ids) + self.assertEqual(batches[0].tensors["sequence_lengths"].tolist(), [2, 4]) + self.assertEqual( + batches[0].tensors["input_ids"].tolist(), [[0, 1, 10, 11, 12, 13]] + ) + self.assertEqual(batches[0].tensors["loss_mask"].tolist(), [[1, 0, 1, 1, 1, 0]]) + self.assertEqual(batches[0].tensors["loss_denominator"].item(), 8) + self.assertEqual(batches[1].tensors["sequence_lengths"].tolist(), [3]) + self.assertEqual(batches[1].tensors["loss_denominator"].item(), 3) + + +class StreamingPackingDataTest(unittest.TestCase): + @staticmethod + def _dflash(length, offset=0, teacher=True): + sample = _feature(length, offset) + result = { + "input_ids": sample["input_ids"], + "loss_mask": sample["loss_mask"], + "hidden_states": sample["hidden_state"], + } + if teacher: + result["target_last_hidden_states"] = sample["target"] + return result + + def test_dflash_preserves_optional_teacher_features_without_loss_denominator(self): + from specforge.algorithms.common.hidden_states_data import build_packed_collator + + for teacher in (False, True): + features = [self._dflash(3, 10, teacher), self._dflash(5, 20, teacher)] + originals = [{k: v.clone() for k, v in f.items()} for f in features] + batch = build_packed_collator()(features) + self.assertEqual(batch["sequence_lengths"].tolist(), [3, 5]) + self.assertNotIn("loss_denominator", batch) + self.assertEqual("target_last_hidden_states" in batch, teacher) + for key in originals[0]: + torch.testing.assert_close( + batch[key], torch.cat([f[key] for f in originals], dim=1) + ) + for actual, original in zip(features, originals): + torch.testing.assert_close(actual[key], original[key]) + + def test_dflash_rejects_inconsistent_teacher_features_and_bad_shapes(self): + from specforge.algorithms.common.hidden_states_data import build_packed_collator + + collate = build_packed_collator() + with self.assertRaises(KeyError): + collate([self._dflash(3, teacher=True), self._dflash(3, teacher=False)]) + for key in self._dflash(3): + sample = self._dflash(3) + sample[key] = sample[key][:, :2] + with self.subTest(key=key), self.assertRaises(ValueError): + collate([sample]) + + def test_online_queue_keeps_logical_batch_size_identity_and_ack_count(self): + from specforge.algorithms.builtin import builtin_algorithm_registry + + for name in ("eagle3", "dflash"): + algorithm = builtin_algorithm_registry().resolve(name) + store = LocalFeatureStore(f"packing-queue-{name}") + refs = [] + for index, length in enumerate((3, 7, 4, 6)): + sample = ( + _feature(length, index * 10) + if name == "eagle3" + else self._dflash(length, index * 10) + ) + refs.append( + store.put( + sample, + sample_id=f"sample-{index}", + metadata={ + "run_id": "queue-test", + "strategy": name, + "target_repr": "hidden_state", + }, + ) + ) + queue = SampleRefQueue() + queue.put(refs) + loader = FeatureDataLoader( + store, + queue, + batch_size=2, + strategy=name, + collate_fn=algorithm.providers.server_streaming_for( + "text" + ).build_packed_collator(), + ) + batches = list(loader) + self.assertEqual( + [batch.sample_ids for batch in batches], + [["sample-0", "sample-1"], ["sample-2", "sample-3"]], + ) + self.assertEqual( + [batch.tensors["sequence_lengths"].tolist() for batch in batches], + [[3, 7], [4, 6]], + ) + self.assertEqual(queue.in_flight(), 0) + self.assertEqual(queue.depth(), 0) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_modeling/test_dflash_sequence_packing.py b/tests/test_modeling/test_dflash_sequence_packing.py new file mode 100644 index 000000000..86e245da6 --- /dev/null +++ b/tests/test_modeling/test_dflash_sequence_packing.py @@ -0,0 +1,360 @@ +"""DFlash context packing preserves per-document sampling and objectives.""" + +import copy +import unittest + +import torch +from torch import nn + +from specforge.algorithms.common.dflash_family_model import ( + OnlineDFlashModel, + create_dflash_block_mask, + create_dflash_sdpa_mask, +) +from specforge.modeling.packed_dflash import PackedDFlashLayout + + +class PackedDFlashMaskTest(unittest.TestCase): + def test_sparse_tile_metadata_covers_exact_mask_and_full_tiles_are_exact(self): + lengths = (131, 67, 3, 312) + starts = torch.tensor([0, 131, 198, 201]) + local = torch.arange(9)[None, :].expand(4, -1) * 13 + local = torch.minimum(local, torch.tensor(lengths)[:, None] - 1) + anchors = (local + starts[:, None]).reshape(1, -1) + context_starts = starts[:, None].expand(-1, 9).reshape(1, -1) + keep = torch.ones_like(anchors, dtype=torch.bool) + keep[:, 10:13] = False + for proposal in (3, 16, 33): + for window in (None, 7): + for tile_shape in ((128, 128), (256, 128)): + with self.subTest( + proposal=proposal, window=window, tile_shape=tile_shape + ): + arguments = dict( + anchor_positions=anchors, + block_keep_mask=keep, + S=sum(lengths), + block_size=proposal, + device="cpu", + sliding_window=window, + context_start_positions=context_starts, + ) + dense = create_dflash_sdpa_mask(**arguments)[0, 0] + sparse = create_dflash_block_mask( + **arguments, flex_block_size=tile_shape + ) + rows, columns = sparse.kv_indices.shape[-2:] + tiles = torch.zeros(rows, columns, dtype=torch.bool) + full = torch.zeros_like(tiles) + for row in range(rows): + tiles[ + row, + sparse.kv_indices[ + 0, 0, row, : sparse.kv_num_blocks[0, 0, row] + ], + ] = True + full[ + row, + sparse.full_kv_indices[ + 0, 0, row, : sparse.full_kv_num_blocks[0, 0, row] + ], + ] = True + padded = torch.zeros( + rows * tile_shape[0], + columns * tile_shape[1], + dtype=torch.bool, + ) + padded[: dense.shape[0], : dense.shape[1]] = dense + token_tiles = padded.reshape( + rows, tile_shape[0], columns, tile_shape[1] + ).permute(0, 2, 1, 3) + exact_any = token_tiles.any(-1).any(-1) + exact_full = token_tiles.all(-1).all(-1) + self.assertFalse(bool((exact_any & ~(tiles | full)).any())) + self.assertFalse(bool((full & ~exact_full).any())) + + def test_sampling_mask_restores_documents_and_excludes_padding(self): + layout = PackedDFlashLayout.from_lengths((3, 1, 2), 6, "cpu") + packed_mask = torch.tensor([[1, 0, 1, 1, 1, 1]]) + torch.testing.assert_close( + layout.padded_loss_mask(packed_mask), + torch.tensor([[1, 0, 1], [1, 0, 0], [1, 1, 0]]), + ) + + def test_full_and_sliding_masks_never_read_previous_documents(self): + anchors = torch.tensor([[1, 4, 6]]) + starts = torch.tensor([[0, 3, 3]]) + keep = torch.tensor([[True, False, True]]) + for window in (None, 3): + args = dict( + anchor_positions=anchors, + block_keep_mask=keep, + S=8, + block_size=4, + device="cpu", + sliding_window=window, + context_start_positions=starts, + ) + dense = create_dflash_sdpa_mask(**args)[0, 0] + sparse = create_dflash_block_mask(**args) + q = torch.arange(12)[:, None] + k = torch.arange(20)[None, :] + actual = sparse.mask_mod(torch.tensor(0), torch.tensor(0), q, k) + torch.testing.assert_close(actual, dense) + self.assertFalse(bool(dense[8:, :3].any())) + self.assertFalse(bool(dense[4:8].any())) + self.assertFalse(bool(dense[8:, 8:16].any())) + + +@unittest.skipUnless( + torch.cuda.is_available(), "production Flex Attention requires CUDA" +) +class PackedDFlashProductionTest(unittest.TestCase): + def _fixtures(self, *, dflash2, sliding, dtype, loss_type, lk_loss_type=None): + from transformers import Qwen3Config + + from specforge.modeling.draft.dflash import DFlashDraftModel + from specforge.modeling.draft.dflash2 import DFlash2DraftModel + + torch.manual_seed(912) + method = { + "block_size": 4, + "mask_token_id": 63, + "target_layer_ids": [1, 2], + } + if dflash2: + method.update( + conv_group_size=4, conv_kernel_size=3, selector_rank=8, selector_top_k=8 + ) + config = Qwen3Config( + hidden_size=64, + intermediate_size=128, + num_attention_heads=4, + num_key_value_heads=2, + head_dim=16, + num_hidden_layers=2, + num_target_layers=4, + max_position_embeddings=256, + vocab_size=64, + layer_types=( + ["sliding_attention", "full_attention"] + if sliding + else ["full_attention"] * 2 + ), + sliding_window=16 if sliding else None, + use_sliding_window=sliding, + dflash_config=method, + ) + config._attn_implementation = "flex_attention" + draft = ( + (DFlash2DraftModel if dflash2 else DFlashDraftModel)(config) + .cuda() + .to(dtype) + ) + if dflash2: + with torch.no_grad(): + for layer in draft.layers: + layer.attention_conv.kernel_projection.weight.normal_(std=0.01) + layer.mlp_conv.kernel_projection.weight.normal_(std=0.01) + head = nn.Linear(64, 64, bias=False).cuda().to(dtype).requires_grad_(False) + embedding = nn.Embedding(64, 64).cuda().to(dtype).requires_grad_(False) + model = OnlineDFlashModel( + draft, + head, + embedding, + mask_token_id=63, + block_size=4, + num_anchors=8, + attention_backend="flex_attention", + loss_type=loss_type, + lk_loss_type=lk_loss_type, + objective_chunk_blocks=3, + loss_decay_gamma=3.0, + selector_loss_alpha=0.7, + metric_top_k=8, + ) + generator = torch.Generator().manual_seed(555) + features = [] + for length in (131, 67, 3, 1): + loss_mask = torch.ones(1, length) + if length > 10: + loss_mask[:, :4] = 0 + loss_mask[:, 9:12] = 0 + features.append( + { + "input_ids": torch.randint(0, 63, (1, length), generator=generator), + "loss_mask": loss_mask, + "hidden_states": torch.randn( + 1, length, 128, generator=generator + ).to(dtype), + "target_last_hidden_states": torch.randn( + 1, length, 64, generator=generator + ).to(dtype), + } + ) + return model, features + + def _run(self, model, features, *, packed, compact=True): + lengths = [f["input_ids"].shape[1] for f in features] + if packed: + batch = { + key: torch.cat([f[key] for f in features], dim=1).cuda() + for key in features[0] + } + batch["sequence_lengths"] = tuple(lengths) + if compact: + batch["valid_anchor_counts"] = tuple( + int( + ( + (f["loss_mask"][:, :-1] > 0.5) + & (f["loss_mask"][:, 1:] > 0.5) + ).sum() + ) + for f in features + ) + else: + batch = {} + for key in features[0]: + rows = [] + for f, length in zip(features, lengths): + value = f[key] + rows.append( + torch.cat( + ( + value, + value.new_zeros( + (1, max(lengths) - length, *value.shape[2:]) + ), + ), + dim=1, + ) + ) + batch[key] = torch.cat(rows).cuda() + sampled = [] + original_sampler = model._sample_anchor_positions + + def record(*args, **kwargs): + result = original_sampler(*args, **kwargs) + sampled.append(tuple(t.detach().clone() for t in result)) + return result + + model._sample_anchor_positions = record + torch.manual_seed(999) + try: + loss, accuracy, metrics = model(**batch) + loss.backward() + finally: + model._sample_anchor_positions = original_sampler + gradients = { + name: p.grad.detach().clone() + for name, p in model.named_parameters() + if p.grad is not None + } + return (loss.detach(), accuracy.detach(), metrics), gradients, sampled[0] + + def _compare(self, left, right, tol, path=""): + if isinstance(left, dict): + self.assertEqual(left.keys(), right.keys(), path) + for name in left: + self._compare(left[name], right[name], tol, path + "/" + name) + elif isinstance(left, (tuple, list)): + for index, (a, b) in enumerate(zip(left, right)): + self._compare(a, b, tol, path + f"/{index}") + elif isinstance(left, torch.Tensor): + torch.testing.assert_close(left, right, msg=path, **tol) + else: + self.assertEqual(left, right, path) + + def test_production_anchors_losses_metrics_and_every_gradient_match(self): + cases = ( + (False, False, torch.float32, "dflash", None), + (False, True, torch.float32, "dpace", None), + (True, False, torch.float32, "dflash", None), + (True, True, torch.float32, "dpace", None), + (True, False, torch.bfloat16, "dflash", None), + (True, True, torch.bfloat16, "dpace", None), + (True, True, torch.float32, "dpace-cumulative-confidence-only", "lambda"), + (True, False, torch.float32, "dpace-continuation-value-only", "tv"), + ) + for dflash2, sliding, dtype, loss_type, lk_loss_type in cases: + with self.subTest( + dflash2=dflash2, + sliding=sliding, + dtype=dtype, + loss_type=loss_type, + lk_loss_type=lk_loss_type, + ): + model, features = self._fixtures( + dflash2=dflash2, + sliding=sliding, + dtype=dtype, + loss_type=loss_type, + lk_loss_type=lk_loss_type, + ) + packed_model = copy.deepcopy(model) + baseline, baseline_gradients, baseline_anchors = self._run( + model, features, packed=False + ) + packed, packed_gradients, packed_anchors = self._run( + packed_model, features, packed=True + ) + for expected, actual in zip(baseline_anchors, packed_anchors): + torch.testing.assert_close(actual, expected, atol=0, rtol=0) + tol = ( + dict(atol=3e-6, rtol=5e-4) + if dtype == torch.float32 + else dict(atol=7e-4, rtol=5e-2) + ) + self._compare(packed, baseline, tol) + self._compare(packed_gradients, baseline_gradients, tol, "gradients") + self.assertTrue(any("q_proj" in name for name in packed_gradients)) + if dflash2: + self.assertTrue( + any("candidate_selector" in name for name in packed_gradients) + ) + self.assertTrue( + any("attention_conv" in name for name in packed_gradients) + ) + + def test_dflash2_blocks_cannot_read_another_document(self): + model, features = self._fixtures( + dflash2=True, sliding=True, dtype=torch.float32, loss_type="dpace" + ) + changed = copy.deepcopy(features) + changed[0]["input_ids"].fill_(33) + changed[0]["hidden_states"].mul_(100) + captured = [] + handle = model.draft_model.register_forward_hook( + lambda _module, _args, output: captured.append(output.detach().clone()) + ) + try: + self._run(model, features, packed=True) + model.zero_grad(set_to_none=True) + self._run(model, changed, packed=True) + finally: + handle.remove() + self.assertEqual(len(captured), 2) + first_document_blocks = model.num_anchors * model.block_size + torch.testing.assert_close( + captured[0][:, first_document_blocks:], + captured[1][:, first_document_blocks:], + atol=0, + rtol=0, + ) + + def test_missing_host_counts_retains_equivalent_uncompacted_fallback(self): + model, features = self._fixtures( + dflash2=True, sliding=True, dtype=torch.float32, loss_type="dpace" + ) + uncompacted_model = copy.deepcopy(model) + compact, compact_grads, _ = self._run(model, features, packed=True) + uncompacted, uncompacted_grads, _ = self._run( + uncompacted_model, features, packed=True, compact=False + ) + tolerance = dict(atol=3e-6, rtol=5e-4) + self._compare(compact, uncompacted, tolerance) + self._compare(compact_grads, uncompacted_grads, tolerance, "gradients") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_runtime/test_eagle3_sequence_packing.py b/tests/test_runtime/test_eagle3_sequence_packing.py new file mode 100644 index 000000000..6440b5494 --- /dev/null +++ b/tests/test_runtime/test_eagle3_sequence_packing.py @@ -0,0 +1,235 @@ +"""Packed EAGLE3 must retain padded-batch supervision, gradients, and isolation.""" + +import copy +import tempfile +import unittest + +import torch + +from specforge.modeling.packed_sequence import ( + PackedSequenceLayout, + generate_packed_eagle3_mask, +) + + +class PackedSequenceLayoutTest(unittest.TestCase): + def test_repeated_shifts_never_import_the_next_document(self): + layout = PackedSequenceLayout.from_lengths(torch.tensor([3, 2]), 5, "cpu") + values = torch.tensor([[1, 2, 3, 4, 5]]) + for expected in ([2, 3, 0, 5, 0], [3, 0, 0, 0, 0], [0, 0, 0, 0, 0]): + values = layout.shift_left(values) + self.assertEqual(values.tolist(), [expected]) + + def test_mask_matches_independent_documents_at_every_depth(self): + lengths = [5, 2, 4] + layout = PackedSequenceLayout.from_lengths( + torch.tensor(lengths), sum(lengths), "cpu" + ) + size = sum(lengths) + for depth in range(4): + mask = generate_packed_eagle3_mask(layout, size, depth) + for q in range(size + 2): + for k in range(size * (depth + 1)): + row = k % size + valid = q < size + if valid: + valid = ( + layout.document_ids[q] == layout.document_ids[row] + and layout.positions[q] < layout.document_lengths[q] - depth + and layout.positions[row] + < layout.document_lengths[row] - depth + ) + expected = bool( + valid and ((k < size and q >= k) or (k >= size and row == q)) + ) + actual = bool(mask(0, 0, torch.tensor(q), torch.tensor(k))) + self.assertEqual(actual, expected, (depth, q, k)) + + +@unittest.skipUnless( + torch.cuda.is_available(), "production Flex Attention and loss require CUDA" +) +class Eagle3PackedProductionTest(unittest.TestCase): + def test_cpu_layout_can_shift_device_resident_hidden_features(self): + layout = PackedSequenceLayout.from_lengths(torch.tensor([3, 2]), 5, "cpu") + hidden = torch.arange(10, device="cuda").view(1, 5, 2) + torch.testing.assert_close( + layout.shift_left(hidden), + torch.tensor([[[2, 3], [4, 5], [0, 0], [8, 9], [0, 0]]], device="cuda"), + ) + + def _fixtures(self, dtype=torch.float32, rope_scaling=None): + from transformers import LlamaConfig + + from specforge.algorithms.eagle3.model import OnlineEagle3Model + from specforge.modeling.draft.llama3_eagle import LlamaForCausalLMEagle3 + from specforge.modeling.target.target_head import TargetHead + + torch.manual_seed(123) + config = LlamaConfig( + vocab_size=64, + draft_vocab_size=64, + hidden_size=64, + intermediate_size=128, + num_hidden_layers=1, + num_attention_heads=4, + num_key_value_heads=2, + max_position_embeddings=160, + rope_scaling=rope_scaling, + pad_token_id=0, + ) + draft = ( + LlamaForCausalLMEagle3(config, attention_backend="flex_attention") + .cuda() + .to(dtype) + ) + model = OnlineEagle3Model(draft, length=4, attention_backend="flex_attention") + + # Use the real frozen head methods, without downloading any checkpoint. + head = TargetHead.__new__(TargetHead) + torch.nn.Module.__init__(head) + head.fc = torch.nn.Linear(64, 64, bias=False).cuda().to(dtype) + head.freeze_weights() + features = [] + for length in (131, 67, 3): + mask = torch.ones(1, length, dtype=torch.long) + mask[:, -1] = 0 + # A prompt mask and an internal supervision gap exercise mask shifts. + if length > 10: + mask[:, :4] = 0 + mask[:, 9:12] = 0 + features.append( + { + "input_ids": torch.randint(1, 64, (1, length)), + "attention_mask": torch.ones(1, length, dtype=torch.long), + "loss_mask": mask, + "hidden_state": torch.randn(1, length, 192).to(dtype), + "target": torch.randn(1, length, 64).to(dtype), + } + ) + return model, head, features + + def _run(self, model, head, features, packed): + from specforge.algorithms.eagle3.data import DataCollatorWithPacking + from specforge.data.utils import DataCollatorWithPadding + from specforge.runtime.contracts import TrainBatch + from specforge.training.strategies.base import Eagle3TrainStrategy + + collator = DataCollatorWithPacking() if packed else DataCollatorWithPadding() + batch = TrainBatch( + sample_ids=[str(i) for i in range(len(features))], + strategy="eagle3", + tensors=collator(features), + metadata={"target_repr": "hidden_state"}, + ) + output = Eagle3TrainStrategy(model, target_head=head).forward_loss(batch) + output.loss.backward() + gradients = { + name: parameter.grad.clone() + for name, parameter in model.named_parameters() + if parameter.grad is not None + } + return output, gradients + + def test_production_loss_metrics_and_all_gradients_match_padding(self): + cases = ( + (torch.float32, None), + (torch.bfloat16, None), + (torch.float32, {"rope_type": "dynamic", "factor": 2.0}), + ) + for dtype, rope_scaling in cases: + with self.subTest(dtype=dtype, rope_scaling=rope_scaling): + model, head, features = self._fixtures(dtype, rope_scaling) + packed_model = copy.deepcopy(model) + padded, padded_grads = self._run(model, head, features, packed=False) + packed, packed_grads = self._run( + packed_model, head, features, packed=True + ) + if rope_scaling is not None: + # Initialization caches max_position_embeddings + 20 (180). + # The packed total is 201; packing must not extend that cache + # and change dynamic NTK scaling relative to the padded batch. + self.assertEqual( + packed_model.draft_model.midlayer.self_attn.rotary_emb.max_seq_len_cached, + 180, + ) + tol = ( + {"atol": 2e-6, "rtol": 2e-4} + if dtype == torch.float32 + else {"atol": 3e-4, "rtol": 3e-2} + ) + torch.testing.assert_close(packed.loss, padded.loss, **tol) + for key in padded.metrics: + for expected, actual in zip( + padded.metrics[key], packed.metrics[key] + ): + torch.testing.assert_close(actual, expected, **tol) + self.assertEqual(padded_grads.keys(), packed_grads.keys()) + for name in padded_grads: + torch.testing.assert_close( + packed_grads[name], padded_grads[name], msg=name, **tol + ) + + def test_changing_previous_document_cannot_change_later_document_logits(self): + model, head, features = self._fixtures() + changed = copy.deepcopy(features) + changed[0]["input_ids"].fill_(33) + changed[0]["hidden_state"].mul_(100) + changed[0]["target"].neg_() + captured = [] + hook = model.draft_model.lm_head.register_forward_hook( + lambda _module, _inputs, output: captured.append(output.detach().clone()) + ) + try: + self._run(model, head, features, packed=True) + first = captured[:] + captured.clear() + model.zero_grad(set_to_none=True) + self._run(model, head, changed, packed=True) + finally: + hook.remove() + self.assertEqual(len(first), 4) + self.assertEqual(len(captured), 4) + start = features[0]["input_ids"].shape[1] + for before, after in zip(first, captured): + torch.testing.assert_close( + before[:, start:], after[:, start:], atol=0, rtol=0 + ) + + def test_fully_unsupervised_batch_has_zero_loss_and_finite_zero_gradients(self): + model, head, features = self._fixtures() + for feature in features: + feature["loss_mask"].zero_() + result, gradients = self._run(model, head, features, packed=True) + self.assertEqual(float(result.loss.detach()), 0.0) + for name, gradient in gradients.items(): + self.assertTrue(bool(torch.isfinite(gradient).all()), name) + self.assertEqual(float(gradient.abs().max()), 0.0, name) + + def test_fsdp_forward_preserves_host_packing_metadata(self): + import torch.distributed as dist + from torch.distributed.fsdp import FullyShardedDataParallel as FSDP + + if dist.is_initialized(): + self.skipTest("requires its own single-rank process group") + model, head, features = self._fixtures() + with tempfile.TemporaryDirectory() as directory: + dist.init_process_group( + "nccl", + init_method=f"file://{directory}/process_group", + rank=0, + world_size=1, + ) + try: + wrapped = FSDP(model, use_orig_params=True) + result, gradients = self._run(wrapped, head, features, packed=True) + self.assertTrue(bool(torch.isfinite(result.loss))) + self.assertTrue(gradients) + for name, gradient in gradients.items(): + self.assertTrue(bool(torch.isfinite(gradient).all()), name) + finally: + dist.destroy_process_group() + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_runtime/test_online_packing_benchmark.py b/tests/test_runtime/test_online_packing_benchmark.py new file mode 100644 index 000000000..56c7eb64d --- /dev/null +++ b/tests/test_runtime/test_online_packing_benchmark.py @@ -0,0 +1,211 @@ +"""Fairness checks for the overlapping online packing benchmark.""" + +import json +import tempfile +import unittest +from pathlib import Path +from types import SimpleNamespace + +import torch + +from specforge.benchmarks.benchmark_online_sequence_packing import ( + _observe_first_warmup_step, + _optimizer_coverage, + _optimizer_step_evidence, + _parameter_inventory, + _parameter_sample_indices, + _parameter_samples, + _parameter_update_evidence, + _prompts, + _TimedSource, + parse_args, +) +from specforge.launch import _iter_epoch_online_prompt_batches +from specforge.optimizer import BF16Optimizer + + +class OnlinePackingBenchmarkTests(unittest.TestCase): + def _model(self): + model = torch.nn.Module() + model.draft_model = torch.nn.Module() + model.draft_model.layers = torch.nn.ModuleList( + [torch.nn.Linear(2, 2, bias=False) for _ in range(2)] + ) + # This factor has a legitimate zero gradient and no first-step change. + model.draft_model.zero_factor = torch.nn.Parameter(torch.zeros(2)) + model.embed_tokens = torch.nn.Embedding(4, 2).requires_grad_(False) + model.lm_head = torch.nn.Linear(2, 4, bias=False) + model.lm_head.weight = model.embed_tokens.weight + return model + + def _args(self, *extra): + return parse_args( + [ + "--server-url", + "http://localhost:31012", + "--target-model", + "unused", + "--work-dir", + "unused", + "--output", + "unused.json", + *extra, + ] + ) + + def test_default_warmup_replays_full_corpus(self): + args = self._args("--steps", "7") + self.assertEqual(args.warmup_steps, 7) + self.assertTrue(args.teacher_metrics) + self.assertEqual(args.log_interval, 50) + self.assertEqual(args.objective_chunk_blocks, 128) + + def test_sample_indices_keep_large_embedding_endpoint_in_bounds(self): + # Only allocate <=128 indices, never the 389-million-element embedding. + for numel in (0, 1, 2, 127, 128, 129, 388956160, 2**40): + with self.subTest(numel=numel): + indices = _parameter_sample_indices(numel) + self.assertEqual(indices.dtype, torch.int64) + self.assertEqual(indices.numel(), min(128, numel)) + if numel: + self.assertEqual(indices[0].item(), 0) + self.assertEqual(indices[-1].item(), numel - 1) + self.assertTrue(bool(((indices >= 0) & (indices < numel)).all())) + self.assertTrue(bool((indices[1:] > indices[:-1]).all())) + + def test_inventory_counts_tied_target_once_and_checks_optimizer_identity(self): + model = self._model() + named, inventory = _parameter_inventory(model) + self.assertEqual(inventory["draft_layer_count"], 2) + self.assertEqual(inventory["whole_model_unique"]["total"], 18) + self.assertEqual(inventory["whole_model_unique"]["trainable"], 10) + self.assertEqual(inventory["whole_model_unique"]["frozen"], 8) + self.assertEqual(inventory["components"]["target_embedding"]["total"], 8) + self.assertEqual(inventory["components"]["target_lm_head"]["total"], 8) + tied = next(row for row in inventory["parameters"] if not row["trainable"]) + self.assertEqual( + set(tied["aliases"]), {"embed_tokens.weight", "lm_head.weight"} + ) + optimizer = BF16Optimizer(model.draft_model, lr=0.01, total_steps=2) + coverage = _optimizer_coverage(model, named, optimizer) + self.assertEqual(coverage["optimized_elements"], 10) + self.assertTrue(coverage["target_parameters_excluded"]) + optimizer.model_params[-1] = model.embed_tokens.weight + with self.assertRaisesRegex(AssertionError, "no target Parameter"): + _optimizer_coverage(model, named, optimizer) + + def test_full_update_evidence_accepts_legitimate_zero_gradient_parameters(self): + model = self._model() + named, _ = _parameter_inventory(model) + before = _parameter_samples(named) + optimizer = BF16Optimizer(model.draft_model, lr=0.01, total_steps=2) + gradient_evidence = _observe_first_warmup_step(optimizer, named) + sum( + parameter.square().sum() for parameter in model.draft_model.parameters() + ).backward() + optimizer.step() + self.assertTrue(gradient_evidence["performed"]) + self.assertTrue(all(gradient_evidence["gradient_present"].values())) + self.assertTrue(torch.isfinite(gradient_evidence["_gradient_norm"])) + self.assertTrue( + all(parameter.grad is None for parameter in optimizer.model_params) + ) + updates = _parameter_update_evidence(named, before, 2) + self.assertTrue(updates["all_decoder_layers_have_sampled_updates"]) + self.assertTrue(updates["all_trainable_elements_finite"]) + self.assertTrue(updates["all_frozen_parameter_samples_unchanged"]) + self.assertFalse( + updates["parameters"]["draft_model.zero_factor"]["sampled_update_observed"] + ) + steps = _optimizer_step_evidence(optimizer, named, 1) + self.assertTrue(steps["all_trainable_parameters_received_every_optimizer_step"]) + self.assertEqual( + steps["adamw_steps_by_parameter"]["draft_model.zero_factor"], 1 + ) + + def test_evidence_rejects_nonfinite_parameters_and_changed_frozen_samples(self): + model = self._model() + named, _ = _parameter_inventory(model) + before = _parameter_samples(named) + with torch.no_grad(): + model.draft_model.zero_factor[0] = float("inf") + with self.assertRaisesRegex(AssertionError, "nonfinite"): + _parameter_update_evidence(named, before, 2) + with torch.no_grad(): + model.draft_model.zero_factor[0] = 0 + model.embed_tokens.weight[0, 0] += 1 + with self.assertRaisesRegex(AssertionError, "frozen"): + _parameter_update_evidence(named, before, 2) + + def test_real_masks_and_batch_order_survive_canonical_shuffle(self): + rows = [ + { + "input_ids": [index] * (index + 4), + "loss_mask": [0] * (index + 2) + [1, 1], + } + for index in range(8) + ] + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "prompts.jsonl" + path.write_text("\n".join(json.dumps(row) for row in rows)) + args = self._args("--prompts-path", str(path), "--steps", "2") + prompts, digest, tokens, supervised = _prompts(args, 2, 16) + ordered = [ + prompt + for batch in _iter_epoch_online_prompt_batches( + prompts, 0, 1, seed=args.seed, batch_size=3 + ) + for prompt in batch + ] + self.assertEqual([prompt["payload"] for prompt in ordered], rows) + self.assertEqual( + [prompt["task_id"] for prompt in ordered], + [f"prompt-{index:08d}" for index in range(8)], + ) + self.assertEqual(tokens, sum(len(row["input_ids"]) for row in rows)) + self.assertEqual(supervised, 16) + self.assertEqual(_prompts(args, 2, 16)[1], digest) + # Warmup prefixes use the same input order despite a different + # shuffle length, making short debugging runs deterministic too. + warm = _prompts(args, 1, 16)[0] + warm_ordered = [ + prompt + for batch in _iter_epoch_online_prompt_batches( + warm, 0, 1, seed=args.seed + ) + for prompt in batch + ] + self.assertEqual([prompt["payload"] for prompt in warm_ordered], rows[:4]) + + def test_request_hash_checks_masks_but_ignores_transport_namespaces(self): + def capture(run, mask): + sent = [] + adapter = SimpleNamespace(post_fn=lambda url, **kw: sent.append(kw) or []) + source = _TimedSource(adapter) + adapter.post_fn( + "http://localhost/generate", + timeout=1, + json_body={ + "input_ids": [[1, 2, 3]], + "extra_key": [run], + "sampling_params": {"max_new_tokens": 0}, + "spec_capture": [ + { + "store_id": run, + "sample_id": f"{run}:prompt-00000000", + "passthrough": [{"data": mask}], + } + ], + }, + ) + self.assertEqual(len(sent), 1) + self.assertEqual(source.calls[0]["samples"], 1) + self.assertLessEqual(source.first_dispatch, source.calls[0]["end"]) + return source.request_digest.hexdigest() + + self.assertEqual(capture("run-a", [0, 1, 1]), capture("run-b", [0, 1, 1])) + self.assertNotEqual(capture("run-a", [0, 1, 1]), capture("run-b", [1, 1, 1])) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_runtime/test_online_sequence_packing_gate.py b/tests/test_runtime/test_online_sequence_packing_gate.py new file mode 100644 index 000000000..d758d66d0 --- /dev/null +++ b/tests/test_runtime/test_online_sequence_packing_gate.py @@ -0,0 +1,298 @@ +"""Opt-in live SGLang -> Mooncake -> packed online consumer lifecycle gate. + +Use SPECFORGE_RUN_SERVER_CAPTURE_TESTS=1 and an isolated patched SGLang on +PYTHONPATH. CUDA_VISIBLE_DEVICES chooses the consumer; PACKING_TARGET_GPU +chooses the server. This gate owns and cleans up only its fixture processes. +""" + +import json +import math +import numbers +import os +import shutil +import sqlite3 +import subprocess +import time +from pathlib import Path + +import torch + +from tests.test_runtime import test_server_capture_gate as gate + + +class TestOnlineSequencePackingGate(gate.TestServerCaptureGate): + # Reuse the live fixture; its separate extraction tests remain in their + # original module instead of being duplicated by this derived test case. + test_eagle3_zero_copy_end_to_end = None + test_dflash_capture_same_server = None + test_dflash_capture_without_teacher_metrics_skips_last_hidden = None + + @classmethod + def setUpClass(cls): + gate.PORT = int(os.environ.get("PACKING_SERVER_PORT", "30992")) + target_gpu = os.environ.get( + "PACKING_TARGET_GPU", os.environ.get("CUDA_VISIBLE_DEVICES", "0") + ) + original_gpu = os.environ.get("CUDA_VISIBLE_DEVICES") + os.environ["CUDA_VISIBLE_DEVICES"] = target_gpu + try: + super().setUpClass() + finally: + if original_gpu is None: + os.environ.pop("CUDA_VISIBLE_DEVICES", None) + else: + os.environ["CUDA_VISIBLE_DEVICES"] = original_gpu + + @classmethod + def _ensure_mooncake_master(cls): + rpc_port = os.environ.get("PACKING_MOONCAKE_RPC_PORT", "50192") + metadata_port = os.environ.get("PACKING_MOONCAKE_METADATA_PORT", "8092") + binary = shutil.which("mooncake_master") + if binary is None: + raise RuntimeError("live packing gate requires mooncake_master") + cls.master = subprocess.Popen( + [ + binary, + "--enable-http-metadata-server=true", + f"--rpc_port={rpc_port}", + f"--http_metadata_server_port={metadata_port}", + "--metrics_port=9092", + ], + stdout=open(os.path.join(cls.workdir, "mooncake_master.log"), "w"), + stderr=subprocess.STDOUT, + start_new_session=True, + ) + time.sleep(3) + if cls.master.poll() is not None: + raise RuntimeError("packing mooncake_master exited during startup") + os.environ["MOONCAKE_MASTER_SERVER_ADDR"] = f"127.0.0.1:{rpc_port}" + os.environ["MOONCAKE_METADATA_SERVER"] = ( + f"http://127.0.0.1:{metadata_port}/metadata" + ) + os.environ["MOONCAKE_LOCAL_HOSTNAME"] = "127.0.0.1" + os.environ["MOONCAKE_PROTOCOL"] = "tcp" + + @classmethod + def _cleanup_processes(cls): + super()._cleanup_processes() + destination = os.environ.get("PACKING_ARTIFACT_DIR") + if destination and cls.workdir: + Path(destination).mkdir(parents=True, exist_ok=True) + for source in Path(cls.workdir).glob("*.log"): + shutil.copy2(source, Path(destination) / source.name) + for source in Path(cls.workdir).glob("*.json"): + if source.name.endswith("-result.json"): + shutil.copy2(source, Path(destination) / source.name) + + def _packing_store(self, run_id): + from specforge.runtime.data_plane.mooncake_store import MooncakeFeatureStore + + return MooncakeFeatureStore( + store_id=run_id, + retain_on_release=True, + setup_kwargs={ + "local_hostname": "127.0.0.1", + "metadata_server": os.environ["MOONCAKE_METADATA_SERVER"], + "global_segment_size": 1 << 28, + "local_buffer_size": 1 << 28, + "protocol": "tcp", + "rdma_devices": "", + "master_server_addr": os.environ["MOONCAKE_MASTER_SERVER_ADDR"], + }, + ) + + def _model(self, name, work): + from safetensors.torch import load_file + from torch import nn + from transformers import Qwen3Config + + from specforge.algorithms.common.dflash_family_model import OnlineDFlashModel + from specforge.modeling.draft.dflash import DFlashDraftModel + from specforge.modeling.draft.dflash2 import DFlash2DraftModel + from specforge.modeling.target.target_head import TargetHead + from tests.test_runtime import _fixtures as fx + + if name == "eagle3": + model, _ = fx.build_eagle3(str(work), ttt=3) + return model, TargetHead.from_pretrained(self.target_dir) + config = Qwen3Config( + architectures=[ + "DFlash2DraftModel" if name == "dflash2" else "DFlashDraftModel" + ], + hidden_size=gate.H, + intermediate_size=128, + num_attention_heads=4, + num_key_value_heads=2, + head_dim=16, + num_hidden_layers=1, + num_target_layers=8, + vocab_size=256, + max_position_embeddings=512, + layer_types=["full_attention"], + dflash_config={ + "block_size": 4, + "mask_token_id": 0, + "target_layer_ids": gate.AUX_LAYER_IDS, + "conv_group_size": 4, + "conv_kernel_size": 2, + "selector_rank": 4, + "selector_top_k": 4, + }, + ) + config._attn_implementation = "flex_attention" + draft = (DFlash2DraftModel if name == "dflash2" else DFlashDraftModel)(config) + weights = load_file(os.path.join(self.target_dir, "model.safetensors")) + head = nn.Linear(gate.H, 256, bias=False) + head.weight.data.copy_(weights["lm_head.weight"]) + head.requires_grad_(False) + embeddings = nn.Embedding.from_pretrained( + weights["model.embed_tokens.weight"], freeze=True + ) + model = OnlineDFlashModel( + draft, + head, + embeddings, + mask_token_id=0, + block_size=4, + num_anchors=4, + attention_backend="flex_attention", + ).to(device="cuda", dtype=torch.bfloat16) + return model, None + + def test_live_packed_online_training_and_durable_acks(self): + from specforge.algorithms.builtin import builtin_algorithm_registry + from specforge.algorithms.common.hidden_states_data import ( + PackedHiddenStatesCollator, + ) + from specforge.algorithms.eagle3.data import DataCollatorWithPacking + from specforge.inference.adapters.server_capture import ( + SGLangServerCaptureAdapter, + ) + from specforge.inference.capture import CaptureConfig + from specforge.launch import build_disagg_online_consumer + from specforge.optimizer import BF16Optimizer + from specforge.runtime.contracts import SampleRef + from specforge.runtime.data_plane.streaming_ref_channel import ( + StreamingRefChannel, + ) + from specforge.training.checkpoint import STATE_FILE + from tests.test_runtime import _fixtures as fx + + fx.build_single_rank_distributed(port="29692") + for name in ("eagle3", "dflash", "dflash2"): + with self.subTest(architecture=name): + algorithm_name = "dflash" if name == "dflash2" else name + run_id = f"packing-online-{name}" + work = Path(self.workdir) / name + work.mkdir() + store = self._packing_store(run_id) + try: + adapter = SGLangServerCaptureAdapter( + f"http://127.0.0.1:{gate.PORT}", + store, + run_id=run_id, + algorithm=algorithm_name, + schema=gate._capture_schema(algorithm_name), + ) + required = {"input_ids", "loss_mask"} | ( + {"attention_mask", "hidden_state", "target"} + if name == "eagle3" + else {"hidden_states"} + ) + contract = CaptureConfig.from_strategy( + required_features=required, + aux_hidden_state_layer_ids=tuple(gate.AUX_LAYER_IDS), + target_repr="hidden_state", + target_hidden_size=gate.H, + ) + rows = [ + list(range(3, 3 + length)) for length in (8, 24, 12, 16) * 2 + ] + tasks = self._tasks(rows) + refs = list(adapter.produce_refs(tasks, capture=contract)) + self.assertEqual(len(refs), 8) + self.assertTrue( + all(isinstance(ref, SampleRef) for ref in refs), repr(refs) + ) + channel = StreamingRefChannel(str(work / "refs.jsonl")) + channel.publish_many(refs) + channel.close() + model, target_head = self._model(name, work) + logged = [] + database = str(work / "consumer.sqlite") + trainer = build_disagg_online_consumer( + algorithm=builtin_algorithm_registry().resolve(algorithm_name), + feature_store=store, + channel=channel, + draft_model=model, + target_head=target_head, + optimizer_factory=lambda module: BF16Optimizer( + module, + lr=1e-3, + max_grad_norm=0.5, + warmup_ratio=0.0, + total_steps=2, + ), + run_id=run_id, + output_dir=str(work / "output"), + batch_size=2, + accumulation_steps=2, + max_steps=2, + sequence_packing=True, + save_interval=1, + log_interval=1, + metadata_db_path=database, + async_ack=False, + idle_timeout_s=60, + logger=lambda metrics, step: logged.append( + (dict(metrics), step) + ), + ) + self.assertIsInstance( + trainer._loader.collate_fn, + ( + DataCollatorWithPacking + if name == "eagle3" + else PackedHiddenStatesCollator + ), + ) + self.assertEqual(trainer.fit(), 2) + self.assertEqual(trainer.micro_step, 4) + self.assertEqual(channel.consumer_quantum(), 4) + self.assertTrue(channel.consumer_stopped()) + self.assertIsNone(channel.consumer_failure()) + self.assertEqual([step for _, step in logged], [1, 2]) + for metrics, _ in logged: + for key, value in metrics.items(): + if isinstance(value, numbers.Real): + self.assertTrue(math.isfinite(value), f"{key}={value}") + with sqlite3.connect(database) as connection: + acked = [ + row[0] + for row in connection.execute( + "SELECT sample_id FROM acked ORDER BY sample_id" + ) + ] + self.assertEqual(acked, sorted(ref.sample_id for ref in refs)) + for step in (1, 2): + self.assertTrue( + ( + work / "output" / f"{run_id}-step{step}" / STATE_FILE + ).is_file() + ) + result = { + "architecture": name, + "optimizer_steps": trainer.global_step, + "microsteps": trainer.micro_step, + "acked_samples": len(acked), + "sample_lengths": [len(row) for row in rows], + "consumer_quantum": channel.consumer_quantum(), + "logged_steps": [step for _, step in logged], + } + (Path(self.workdir) / f"{name}-result.json").write_text( + json.dumps(result, indent=2) + ) + del trainer, model, target_head + torch.cuda.empty_cache() + finally: + store.close() diff --git a/tests/test_runtime/test_packing_strategy_metadata.py b/tests/test_runtime/test_packing_strategy_metadata.py new file mode 100644 index 000000000..1095deebb --- /dev/null +++ b/tests/test_runtime/test_packing_strategy_metadata.py @@ -0,0 +1,94 @@ +"""DFlash packing metadata remains host-side across the wrapped model boundary.""" + +import unittest + +import torch +from torch import nn + +from specforge.runtime.contracts import TrainBatch +from specforge.training.strategies.base import DFlashTrainStrategy, StepContext + + +class _RecordingModel(nn.Module): + def __init__(self, device="cpu"): + super().__init__() + self.weight = nn.Parameter(torch.ones((), device=device)) + self.kwargs = None + + def forward(self, **kwargs): + self.kwargs = kwargs + return ( + self.weight * kwargs["hidden_states"].sum(), + self.weight.new_zeros(()), + {}, + ) + + +def _batch(loss_mask, lengths): + return TrainBatch( + sample_ids=[str(i) for i in range(len(lengths))], + strategy="dflash", + tensors={ + "input_ids": torch.zeros_like(loss_mask, dtype=torch.long), + "loss_mask": loss_mask, + "hidden_states": torch.ones(1, sum(lengths), 4, device=loss_mask.device), + "sequence_lengths": torch.tensor(lengths, dtype=torch.long), + }, + metadata={}, + ) + + +class PackingStrategyMetadataTest(unittest.TestCase): + def test_host_counts_respect_document_boundaries_and_stay_python_values(self): + model = _RecordingModel() + mask = torch.tensor([[1, 1, 1, 1, 0, 1, 1, 1, 0]]) + DFlashTrainStrategy(model).forward_loss(_batch(mask, (3, 4, 2)), StepContext()) + self.assertEqual(model.kwargs["sequence_lengths"], (3, 4, 2)) + self.assertEqual(model.kwargs["valid_anchor_counts"], (2, 1, 0)) + self.assertEqual(model.kwargs["max_valid_anchors"], 2) + self.assertIsInstance(model.kwargs["max_valid_anchors"], int) + self.assertTrue( + all(isinstance(n, int) for n in model.kwargs["valid_anchor_counts"]) + ) + + def test_zero_masks_and_cross_document_pair_have_no_anchors(self): + for mask in (torch.zeros(1, 4), torch.tensor([[0, 1, 1, 0]])): + with self.subTest(mask=mask.tolist()): + model = _RecordingModel() + DFlashTrainStrategy(model).forward_loss(_batch(mask, (2, 2))) + self.assertEqual(model.kwargs["valid_anchor_counts"], (0, 0)) + self.assertEqual(model.kwargs["max_valid_anchors"], 0) + + def test_model_rejects_no_anchor_batch_before_attention(self): + from specforge.algorithms.common.dflash_family_model import OnlineDFlashModel + + model = OnlineDFlashModel( + nn.Linear(4, 4), + nn.Linear(4, 8, bias=False), + nn.Embedding(8, 4), + mask_token_id=0, + block_size=2, + num_anchors=2, + attention_backend="flex_attention", + ) + for mask in (torch.zeros(1, 4), torch.tensor([[0, 1, 1, 0]])): + with ( + self.subTest(mask=mask.tolist()), + self.assertRaisesRegex(ValueError, "two consecutive supervised"), + ): + DFlashTrainStrategy(model).forward_loss(_batch(mask, (2, 2))) + + @unittest.skipUnless( + torch.cuda.is_available(), "GPU loss-mask fallback requires CUDA" + ) + def test_gpu_loss_mask_uses_model_fallback_without_host_counts(self): + model = _RecordingModel("cuda") + batch = _batch(torch.tensor([[1, 1, 0, 1, 1]], device="cuda"), (3, 2)) + DFlashTrainStrategy(model).forward_loss(batch) + self.assertEqual(model.kwargs["sequence_lengths"], (3, 2)) + self.assertIsNone(model.kwargs["max_valid_anchors"]) + self.assertNotIn("valid_anchor_counts", model.kwargs) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_runtime/test_sequence_packing_lifecycle.py b/tests/test_runtime/test_sequence_packing_lifecycle.py new file mode 100644 index 000000000..9bf978da1 --- /dev/null +++ b/tests/test_runtime/test_sequence_packing_lifecycle.py @@ -0,0 +1,118 @@ +"""Packed offline loader -> FSDP -> optimizer/eval/checkpoint GPU smoke test.""" + +import math +import numbers +import tempfile +import unittest +from pathlib import Path + +import torch + +from specforge.algorithms.builtin import builtin_algorithm_registry +from specforge.algorithms.eagle3.data import DataCollatorWithPacking +from tests.test_runtime import _fixtures as fx + + +def _write_variable_length_features(directory, lengths): + directory.mkdir() + generator = torch.Generator().manual_seed(17) + for index, length in enumerate(lengths): + torch.save( + { + "input_ids": torch.randint(0, fx.V, (length,), generator=generator), + "loss_mask": torch.ones(length, dtype=torch.long), + "hidden_state": torch.randn( + 1, length, fx.H, generator=generator + ).bfloat16(), + "aux_hidden_state": torch.randn( + 1, length, 3 * fx.H, generator=generator + ).bfloat16(), + }, + directory / f"{index:04d}.ckpt", + ) + return str(directory) + + +@unittest.skipUnless( + torch.cuda.is_available(), "packed trainer lifecycle requires CUDA" +) +class SequencePackingLifecycleTest(unittest.TestCase): + def test_packed_training_preserves_steps_evaluation_and_checkpoints(self): + from torch.distributed.fsdp import FullyShardedDataParallel as FSDP + + from specforge.launch import build_offline_runtime + from specforge.optimizer import BF16Optimizer + from specforge.training.checkpoint import STATE_FILE + + torch.manual_seed(17) + fx.build_single_rank_distributed(port="29687") + logged = [] + with tempfile.TemporaryDirectory(prefix="packed_lifecycle_") as workdir: + work = Path(workdir) + train_path = _write_variable_length_features( + work / "train", (8, 24, 12, 16) * 2 + ) + eval_path = _write_variable_length_features(work / "eval", (8, 24, 12)) + model, target_head = fx.build_eagle3(workdir, ttt=3) + output = work / "output" + + def optimizer_factory(draft_module): + return BF16Optimizer( + draft_module, + lr=1e-3, + max_grad_norm=0.5, + warmup_ratio=0.0, + total_steps=2, + ) + + trainer = build_offline_runtime( + algorithm=builtin_algorithm_registry().resolve("eagle3"), + hidden_states_path=train_path, + eval_hidden_states_path=eval_path, + draft_model=model, + target_head=target_head, + optimizer_factory=optimizer_factory, + run_id="packing-lifecycle", + output_dir=str(output), + ttt_length=3, + max_len=32, + batch_size=2, + sequence_packing=True, + accumulation_steps=2, + num_epochs=1, + max_steps=2, + eval_interval=1, + save_interval=1, + log_interval=1, + logger=lambda metrics, step: logged.append((dict(metrics), step)), + ) + self.assertIsInstance(trainer.core.strategy.trainable_module(), FSDP) + self.assertIsInstance(trainer._loader.collate_fn, DataCollatorWithPacking) + eval_loader = trainer._controller.eval_data_factory() + self.assertIsInstance(eval_loader.collate_fn, DataCollatorWithPacking) + self.assertEqual([len(b.sample_ids) for b in eval_loader], [2, 1]) + + self.assertEqual(trainer.fit(), 2) + self.assertEqual(trainer.global_step, 2) + self.assertEqual(trainer.micro_step, 4) + self.assertEqual(trainer.last_checkpoint_step, 2) + self.assertEqual( + [step for metrics, step in logged if "eval/avg_loss" in metrics], [1, 2] + ) + self.assertTrue(any("loss" in metrics for metrics, _ in logged)) + for metrics, _ in logged: + for name, value in metrics.items(): + if isinstance(value, numbers.Real): + self.assertTrue(math.isfinite(value), f"{name}={value}") + for step in (1, 2): + checkpoint = output / f"packing-lifecycle-step{step}" / STATE_FILE + self.assertTrue(checkpoint.is_file()) + state = torch.load(checkpoint, map_location="cpu", weights_only=False) + self.assertEqual(state["global_step"], step) + self.assertEqual(state["epoch_samples"], 4 * step) + self.assertTrue(state["draft_state_dict"]) + self.assertTrue((output / "packing-lifecycle-latest").is_dir()) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_runtime/test_sequence_packing_wiring.py b/tests/test_runtime/test_sequence_packing_wiring.py new file mode 100644 index 000000000..f7f532228 --- /dev/null +++ b/tests/test_runtime/test_sequence_packing_wiring.py @@ -0,0 +1,236 @@ +"""Packing reaches both offline training topologies and offline evaluation.""" + +import tempfile +import unittest +from unittest import mock + +from specforge.algorithms.builtin import builtin_algorithm_registry +from specforge.algorithms.eagle3.data import DataCollatorWithPacking +from specforge.config import Config +from specforge.launch import ( + _make_offline_eval_data_factory, + _offline_io, + _streaming_collate, + build_disagg_offline_runtime, + build_offline_runtime, +) +from specforge.training.assembly import ModelBundle, build_training_run +from specforge.training.disaggregated import _build_offline, _build_online + +ALGORITHM = builtin_algorithm_registry().resolve("eagle3") + + +def _config(): + return Config.model_validate( + { + "model": { + "target_model_path": "target", + "draft_model_config": "draft.json", + "vocab_mapping_path": "mapping.pt", + }, + "data": { + "hidden_states_path": "/train-features", + "eval_hidden_states_path": "/eval-features", + }, + "training": {"sequence_packing": True, "batch_size": 3, "eval_interval": 1}, + } + ) + + +def _bundle(): + return ModelBundle( + model=object(), + draft_model=object(), + draft_config=object(), + target_head=object(), + strategy_kwargs={}, + ) + + +class SequencePackingWiringTest(unittest.TestCase): + def test_io_selects_packing_and_preserves_the_normalizer(self): + packed, normalizer = _offline_io( + ALGORITHM, + "text", + 123, + ttt_length=3, + use_usp_preprocess=False, + sequence_packing=True, + ) + padded, baseline = _offline_io( + ALGORITHM, + "text", + 123, + ttt_length=3, + use_usp_preprocess=False, + ) + self.assertIsInstance(packed, DataCollatorWithPacking) + self.assertNotIsInstance(padded, DataCollatorWithPacking) + self.assertIs(normalizer.func, baseline.func) + self.assertEqual(normalizer.keywords, baseline.keywords) + + def test_direct_io_rejects_usp_and_unsupported_provider(self): + for algorithm, usp in ( + (ALGORITHM, True), + (builtin_algorithm_registry().resolve("domino"), False), + ): + with ( + self.subTest(algorithm=algorithm.name, usp=usp), + self.assertRaisesRegex( + ValueError, "supported non-USP offline provider" + ), + ): + _offline_io( + algorithm, + "text", + 123, + ttt_length=3, + use_usp_preprocess=usp, + sequence_packing=True, + ) + + def test_streaming_selects_the_registered_packing_factory(self): + for name in ("eagle3", "dflash"): + algorithm = builtin_algorithm_registry().resolve(name) + collator = _streaming_collate( + algorithm, "text", None, sequence_packing=True + ) + self.assertTrue(callable(collator)) + with self.assertRaisesRegex(ValueError, "packing"): + _streaming_collate( + builtin_algorithm_registry().resolve("domino"), + "text", + None, + sequence_packing=True, + ) + with self.assertRaisesRegex(ValueError, "collate"): + _streaming_collate( + ALGORITHM, "text", lambda samples: samples, sequence_packing=True + ) + + def test_eval_factory_selects_packing_and_keeps_partial_batches(self): + with tempfile.TemporaryDirectory() as directory: + factory = _make_offline_eval_data_factory( + algorithm=ALGORITHM, + modality="text", + hidden_states_path=directory, + run_id="packing-eval", + batch_size=3, + max_len=123, + ttt_length=3, + use_usp_preprocess=False, + dataloader_num_workers=0, + sequence_packing=True, + ) + first, second = factory(), factory() + self.assertIsNot(first, second) + self.assertIsInstance(first.collate_fn, DataCollatorWithPacking) + self.assertEqual(first.batch_size, 3) + self.assertFalse(first.drop_last) + + def test_both_offline_builders_pack_train_and_eval(self): + with tempfile.TemporaryDirectory() as directory: + for builder in (build_offline_runtime, build_disagg_offline_runtime): + with ( + self.subTest(builder=builder.__name__), + mock.patch("specforge.launch._assemble_trainer") as assemble, + ): + data_args = ( + {"hidden_states_path": directory} + if builder is build_offline_runtime + else {"feature_store": object(), "refs": []} + ) + builder( + algorithm=ALGORITHM, + draft_model=object(), + target_head=object(), + optimizer_factory=object(), + run_id="packing-test", + output_dir=directory, + batch_size=3, + accumulation_steps=2, + eval_hidden_states_path=directory, + sequence_packing=True, + **data_args, + ) + kwargs = assemble.call_args.kwargs + self.assertIsInstance(kwargs["collate_fn"], DataCollatorWithPacking) + self.assertEqual(kwargs["batch_size"], 3) + self.assertEqual(kwargs["accumulation_steps"], 2) + self.assertIsInstance( + kwargs["eval_data_factory"]().collate_fn, + DataCollatorWithPacking, + ) + + def test_unified_offline_assembly_passes_packing(self): + with ( + mock.patch( + "specforge.training.assembly.build_model_bundle", return_value=_bundle() + ), + mock.patch("specforge.launch.build_offline_runtime") as build, + ): + build_training_run(_config(), algorithm=ALGORITHM) + self.assertTrue(build.call_args.kwargs["sequence_packing"]) + self.assertEqual(build.call_args.kwargs["batch_size"], 3) + + def test_disaggregated_consumer_assembly_passes_packing(self): + cfg = _config() + with ( + mock.patch( + "specforge.training.disaggregated._env", return_value="/manifest.json" + ), + mock.patch("specforge.training.disaggregated._wait_for"), + mock.patch("specforge.training.disaggregated._offline_store"), + mock.patch( + "specforge.runtime.data_plane.disagg_ingest.read_ref_manifest", + return_value=[], + ), + mock.patch("specforge.launch.build_disagg_offline_runtime") as build, + ): + _build_offline( + cfg, + algorithm=ALGORITHM, + build_model_bundle=lambda _: _bundle(), + optimizer_factory=lambda _: object(), + logger=None, + ) + self.assertTrue(build.call_args.kwargs["sequence_packing"]) + self.assertEqual(build.call_args.kwargs["batch_size"], 3) + + def test_online_consumer_assembly_passes_packing_and_logical_batch_size(self): + payload = _config().model_dump() + payload["data"] = {"train_data_path": "train.jsonl"} + payload["training"].update(role="consumer", max_steps=2, eval_interval=0) + payload["deployment"] = { + "mode": "disaggregated", + "disaggregated": { + "control_dir": "outputs/packing-test/control", + "backend": "mooncake", + "server_urls": ["http://127.0.0.1:30000"], + "mooncake_metadata_server": "http://127.0.0.1:35880/metadata", + "mooncake_master_server_addr": "127.0.0.1:35551", + }, + } + cfg = Config.model_validate(payload) + with ( + mock.patch( + "specforge.training.disaggregated._env", + return_value="/ref-channel.jsonl", + ), + mock.patch("specforge.training.disaggregated._mooncake_store"), + mock.patch("specforge.launch.build_disagg_online_consumer") as build, + ): + _build_online( + cfg, + algorithm=ALGORITHM, + build_model_bundle=lambda _: _bundle(), + prepare_prompts=lambda *_args, **_kwargs: [], + optimizer_factory=lambda _: object(), + logger=None, + ) + self.assertTrue(build.call_args.kwargs["sequence_packing"]) + self.assertEqual(build.call_args.kwargs["batch_size"], 3) + + +if __name__ == "__main__": + unittest.main() From 9ad8ebc592b61ce11114c914464f50cf3a0c4c67 Mon Sep 17 00:00:00 2001 From: Yusheng Su Date: Sun, 4 Oct 2026 19:46:09 +0900 Subject: [PATCH 2/2] Keep sequence packing focused on implementation and tests --- docs/sections/basic_usage/training.md | 99 +- .../benchmarks/dflash-sequence-packing.md | 180 - .../benchmarks/eagle3-sequence-packing.md | 151 - .../benchmarks/full-model-sequence-packing.md | 181 - .../benchmarks/online-sequence-packing.md | 147 - .../sequence-packing-results/README.md | 78 - .../dflash-family-128anchors-bf16.json | 1070 ------ .../dflash-family-512anchors-bf16.json | 563 --- .../dflash-family-tiny-fp32.json | 214 -- .../eagle3-large-bf16.json | 235 -- .../eagle3-medium-bf16.json | 605 ---- .../eagle3-tiny-fp32.json | 175 - .../full-model-analysis.json | 899 ----- .../full-model-audit.json | 1107 ------ .../online-32step.json | 568 --- .../target-parameter-inventory.json | 3086 ----------------- scripts/benchmark_dflash_sequence_packing.py | 7 - scripts/benchmark_online_sequence_packing.py | 7 - scripts/benchmark_sequence_packing.py | 7 - .../benchmark_dflash_sequence_packing.py | 465 --- .../benchmark_online_sequence_packing.py | 1166 ------- .../benchmarks/benchmark_sequence_packing.py | 460 --- .../test_online_packing_benchmark.py | 211 -- 23 files changed, 27 insertions(+), 11654 deletions(-) delete mode 100644 docs/sections/benchmarks/dflash-sequence-packing.md delete mode 100644 docs/sections/benchmarks/eagle3-sequence-packing.md delete mode 100644 docs/sections/benchmarks/full-model-sequence-packing.md delete mode 100644 docs/sections/benchmarks/online-sequence-packing.md delete mode 100644 docs/sections/benchmarks/sequence-packing-results/README.md delete mode 100644 docs/sections/benchmarks/sequence-packing-results/dflash-family-128anchors-bf16.json delete mode 100644 docs/sections/benchmarks/sequence-packing-results/dflash-family-512anchors-bf16.json delete mode 100644 docs/sections/benchmarks/sequence-packing-results/dflash-family-tiny-fp32.json delete mode 100644 docs/sections/benchmarks/sequence-packing-results/eagle3-large-bf16.json delete mode 100644 docs/sections/benchmarks/sequence-packing-results/eagle3-medium-bf16.json delete mode 100644 docs/sections/benchmarks/sequence-packing-results/eagle3-tiny-fp32.json delete mode 100644 docs/sections/benchmarks/sequence-packing-results/full-model-analysis.json delete mode 100644 docs/sections/benchmarks/sequence-packing-results/full-model-audit.json delete mode 100644 docs/sections/benchmarks/sequence-packing-results/online-32step.json delete mode 100644 docs/sections/benchmarks/sequence-packing-results/target-parameter-inventory.json delete mode 100755 scripts/benchmark_dflash_sequence_packing.py delete mode 100755 scripts/benchmark_online_sequence_packing.py delete mode 100755 scripts/benchmark_sequence_packing.py delete mode 100644 specforge/benchmarks/benchmark_dflash_sequence_packing.py delete mode 100644 specforge/benchmarks/benchmark_online_sequence_packing.py delete mode 100644 specforge/benchmarks/benchmark_sequence_packing.py delete mode 100644 tests/test_runtime/test_online_packing_benchmark.py diff --git a/docs/sections/basic_usage/training.md b/docs/sections/basic_usage/training.md index 81b8ead88..6d55ecc06 100644 --- a/docs/sections/basic_usage/training.md +++ b/docs/sections/basic_usage/training.md @@ -551,81 +551,36 @@ newest complete checkpoint. ## Sequence packing -Text EAGLE3, DFlash, and DFlash2 can remove context padding across the samples -of each microbatch, with offline features or online server capture: +Text EAGLE3, DFlash, and DFlash2 can pack variable-length samples within each +microbatch for offline training/evaluation and online disaggregated consumers: -```bash -specforge train \ - --config examples/configs/offline/colocated/qwen3-8b-eagle3-offline.yaml \ - training.batch_size=4 \ - training.attention_backend=flex_attention \ - training.sequence_packing=true +```yaml +training: + attention_backend: flex_attention + sequence_packing: true ``` -Packing defaults to `false`. Add the same two training overrides to an online -config or a DFlash/DFlash2 config. DFlash2 uses `training.strategy=dflash` with -a DFlash2 draft-model config. Offline evaluation loaders also honor packing. -The implementation requires FlexAttention and does not support USP, other -algorithms, multimodal positions, `compact_teacher`, or `trim_loss_positions`. -DFlash/DFlash2 LK objectives and D-PACE are supported; EAGLE3 LK is not. - -`batch_size` still counts original samples per rank and microbatch. Packing -concatenates those samples into one row, resets positions at each document, -isolates attention, and prevents labels from crossing document boundaries. -It does not change sample order, -gradient accumulation, reference acknowledgement, or the optimizer schedule. -`data.max_length` remains the truncation limit for each original sample; a -packed row can be longer than that limit. - -EAGLE3 prevents the initial teacher shift and every subsequent TTT shift from -crossing document boundaries. Its loss keeps the original -`batch_size * longest_sample_length` denominator so packing does not implicitly -increase the learning rate on batches that previously had substantial padding. - -DFlash/DFlash2 preserve the original per-document anchor counts and random -sampling, including invalid anchor slots. Context and proposal positions reset -for each document; full and sliding attention, block labels, and teacher -predecessors respect document boundaries. After the backbone, the original -`[batch, anchors, block]` axes are restored for the loss, D-PACE, selector, and -metrics. Packing does not reduce the number of sampled proposal tokens. -When CPU loss masks provide per-document valid-anchor counts, invalid padded -proposal slots skip the backbone; outputs are restored to the original loss -layout. This saves computation without discarding valid sampled anchors. - -Online packing happens in the consumer after each target capture is fetched. -The producer still captures individual prompts, and the consumer acknowledges -the original sample IDs. Target capture, queue order, and the durable cursor -retain their existing behavior. - -For lengths `[2048, 512, 256, 256]`, padded execution processes 8192 rows per TTT -step and packed execution processes 3072. This removes 62.5% of those rows, but -does not imply a 2.67x end-to-end speedup: attention mask construction, kernel -occupancy, feature I/O, and optimizer work still cost time. There is no padding -to remove at batch size one or when every sample has the same length, and -packing can be slower in those cases. Compare equal samples and settings using -effective (unpadded) tokens/s, not packed steps/s. Packing changes training -execution; it is not a speculative-serving speedup. - -The reproducible GPU benchmarks in `scripts/benchmark_sequence_packing.py` -(EAGLE3) and `scripts/benchmark_dflash_sequence_packing.py` (DFlash/DFlash2) -check production loss/gradient agreement before measuring training steps. -See its `--help` for model dimensions and length profiles. Generated features -measure training compute, not target capture, dataset quality, or a full epoch. -See the [H200 measurements and validation scope](../benchmarks/eagle3-sequence-packing.md) -for a controlled EAGLE3 padded-versus-packed comparison, and the -[DFlash/DFlash2 measurements and online validation](../benchmarks/dflash-sequence-packing.md) -for those models. The speedup depends on context lengths, valid proposal counts, -and mask construction; it is not implied by the padding fraction alone. - -For a concurrent target-capture and training comparison, use -`scripts/benchmark_online_sequence_packing.py`. The -[real Qwen3-4B online benchmark](../benchmarks/online-sequence-packing.md) -measured 4.1% DFlash and 4.5% DFlash2 pipeline throughput gains on 128 ShareGPT -conversations; it reports final checkpoint time separately. -The subsequent [256-step full-model comparison](../benchmarks/full-model-sequence-packing.md) -adds periodic diagnostics/checkpoints and four runs per mode: DFlash/DFlash2 -training steps improved 1.044×/1.052×, with full completion including both -checkpoints improving 1.037×/1.033× on that workload. +Packing defaults to `false`. DFlash2 uses `training.strategy: dflash` with a +DFlash2 draft-model config. FlexAttention is required; USP, multimodal inputs, +other algorithms, `compact_teacher`, and `trim_loss_positions` are unsupported. +DFlash/DFlash2 LK and D-PACE are supported; EAGLE3 LK is not. + +Packing concatenates the existing microbatch, resets positions per document, +and isolates attention, labels, and teacher shifts. It preserves sample order, +logical batch size, loss normalization, gradient accumulation, and optimizer +schedule. `data.max_length` still limits each original sample; the packed row +can exceed that length. + +DFlash/DFlash2 preserve per-document anchor sampling. Invalid padded proposal +slots skip the backbone when host metadata is available; outputs return to the +original batch/anchor/block layout for objectives and metrics. Online packing +happens after the consumer fetches target features, preserving individual +capture requests and sample-ID acknowledgements. + +Gains depend on sequence lengths, valid proposal counts, and pipeline overhead. +Batch size one or equal-length samples have no inter-sample context padding to +remove. Compare the same samples, batch size, and accumulation using unpadded +tokens/s; training gains do not imply faster speculative serving. ## Compact offline teacher diff --git a/docs/sections/benchmarks/dflash-sequence-packing.md b/docs/sections/benchmarks/dflash-sequence-packing.md deleted file mode 100644 index 5c97cb051..000000000 --- a/docs/sections/benchmarks/dflash-sequence-packing.md +++ /dev/null @@ -1,180 +0,0 @@ -# DFlash/DFlash2 sequence packing and online validation - -Based on upstream [`53398a8f01ae47175bee8459c5b5cca3848c8a7e`](https://github.com/sgl-project/SpecForge/tree/53398a8f01ae47175bee8459c5b5cca3848c8a7e) -plus the local `codex/sequence-packing` changes, tested on 2026-10-02. - -The final H200 benchmark with 512 anchors per document and 64% context padding -measured **1.39x DFlash** and **1.47x DFlash2** training-step speedups, with about -29% lower peak allocated memory. These are consumer-compute results. A separate -[real Qwen3-4B online pipeline benchmark](online-sequence-packing.md) measured -**1.041× DFlash** and **1.045× DFlash2** on 128 ShareGPT conversations, including -target capture and transport but excluding startup and final checkpoint. - -## Supported execution - -Text EAGLE3, DFlash, and DFlash2 support packing for offline features and online -server capture. DFlash2 remains `training.strategy: dflash` with a -`DFlash2DraftModel` draft config. Enable packing on an existing supported config: - -```yaml -training: - attention_backend: flex_attention - sequence_packing: true -``` - -For example, add those overrides to -`examples/configs/online/disaggregated/external/qwen3-4b-dflash-online.yaml`. -Packing is opt-in. It packs the existing microbatch and does not reorder samples -or change the number of samples per optimizer step. Other algorithms, USP, -multimodal positions, compact teacher, and loss-position trimming are unsupported. -DFlash/DFlash2 LK and D-PACE objectives are supported; EAGLE3 LK is unsupported. - -## Call chain and numerical contract - -```text -SGLang /generate capture (one request per original sample) - -> MooncakeFeatureStore + SampleRef - -> online consumer / FeatureDataLoader - -> provider.build_packed_collator: [B, max(L)] -> [1, sum(L)] - -> DFlashTrainStrategy: host document lengths and valid-anchor counts - -> OnlineDFlashModel: original sampler, isolated context/proposal positions - -> DFlashDraftModel or DFlash2DraftModel - -> restore [B, anchors, block] for original losses and metrics - -> trainer optimizer / original sample-ID acknowledgements / checkpoint -``` - -The original sampler runs once on the original `[B, max(L)]` loss-mask shape, -preserving its random draws, each document's anchor budget, anchor order, and -keep mask. Only the small sampling mask is padded. Context hidden states, token -IDs, and optional target final states are concatenated without padding. - -Attention never crosses a document boundary. Full and sliding masks respect -per-document starts; target labels cannot pass document ends, and teacher -predecessors cannot precede document starts. RoPE positions reset per document. -DFlash2 convolutions retain complete proposal blocks. Loss reductions, D-PACE -sequence weights, selector objectives, and diagnostics retain their original -batch and anchor axes. - -The packed attention mask uses conservative sparse tile ranges, with the exact -per-token predicate for partial tiles. This avoids constructing a dense -`total_queries * total_keys` mask across unrelated documents. Where integer -features are on the CPU, the strategy also supplies per-document valid-anchor -counts: invalid padded proposal slots skip the backbone and outputs are -scattered back into the original loss layout. Direct GPU-only callers without -those counts retain all proposal slots. No CUDA `nonzero()` or host readback is -needed to size the compact proposals. - -Online target requests, capture tensors, transport protocol, queue order, and -acknowledged IDs are unchanged. Packing affects the consumer's draft training; -it does not itself accelerate target capture or speculative serving. - -## Why concatenation alone was slower - -The first correct implementation removed context padding but still generated -a dense FlexAttention mask before converting it to sparse blocks. With four -samples, the flattened query/key grid also contained cross-document regions -that were entirely masked. A CUDA profiler measured DFlash mask construction -at 8.61 ms padded versus 29.43 ms packed in the 512-anchor profile, while the -backbone remained roughly 22 ms. The initial packed implementation regressed -median whole-step time by about 20%. This motivated the sparse tile builder and -invalid-proposal elimination; concatenation alone is not a reliable speedup. - -## Validation and benchmark scope - -The production comparison checks exact sampled anchors and keep masks, loss, -all loss terms, and every trainable gradient before timing. Model tests also -compare all detailed metrics, cover FP32/BF16, full/hybrid sliding attention, -D-PACE variants, LK lambda/TV, DFlash2 selectors and nonzero convolutions, short -and unsupervised documents, and cross-document perturbation isolation. - -Final regression runs passed 204 CPU tests with 540 subtests (nine GPU/live -tests skipped in that CPU run), and 57 GPU model/host-sync tests with 45 -subtests. The final live gate and strategy-metadata checks passed five tests -with seven subtests. These counts describe separate suites, not unique tests -summed across repeated runs. - -A separate two-rank `FULL_SHARD` probe also passed for EAGLE3, DFlash, and -DFlash2. The ranks used different document lengths, and the packed local loss -and all trainable gradients matched the padded reference after its gradients -were averaged across ranks. This verifies sharded forwards/backwards and host -packing metadata, not multi-node online throughput. The following command uses -the retained local probe, which is not committed with this report: - -```bash -CUDA_VISIBLE_DEVICES=0,1 PYTHONPATH=. python -m torch.distributed.run \ - --standalone --nproc-per-node=2 \ - artifacts/sequence-packing/check_packed_fsdp.py -``` - -The live online gate uses a tiny eight-layer Llama target and an FSDP training -consumer, with an isolated copy of SGLang 0.5.18 and the repository's capture -patch. The initial gate used separate H200s; the final optimized gate colocated -both processes on one H200 after unrelated work occupied the original devices. -Actual captures pass through Mooncake TCP, -`RefDistributor`, a SQLite durable ledger, and the packed feature loader. -EAGLE3, DFlash, and DFlash2 each execute two optimizer steps/four microsteps, -acknowledge all eight original sample IDs, and write checkpoints. This is -functional online evidence, not an end-to-end throughput benchmark, RDMA -validation, or pretrained-model convergence evidence. - -The training-step benchmark uses one H200, BF16, two actual draft layers, -hidden size 2048, intermediate size 8192, 16 attention heads / 4 KV heads, -vocabulary 32000, block size 16, microbatch four, and 25% prompt masking. -Frozen target weights and features are synthetic. It includes the production -strategy's CPU integer-feature processing/transfers, forward, backward, and -`BF16Optimizer` update. Hidden features are GPU resident. Target capture, -hidden-feature I/O/H2D, distributed communication, and serving are excluded. - -Five warmup steps include compilation; 20 synchronized wall-clock steps measure -steady-state performance. Useful tokens/s divides the original unpadded input -tokens by mean step time. The benchmark GPU was dedicated to these measurements; -other GPUs on the node also ran unrelated validation workloads. - -| Algorithm | Anchors/document | Lengths | Padding | P50 ms padded → packed | P50 speedup | Mean ms padded → packed | Useful tokens/s padded → packed | Peak GiB padded → packed | -| --- | ---: | --- | ---: | ---: | ---: | ---: | ---: | ---: | -| DFlash | 512 | 128, 256, 512, 2048 | 64.06% | 102.68 → 73.68 | **1.39x** | 102.74 → 73.74 | 28,655 → 39,923 | 11.82 → 8.34 | -| DFlash2 | 512 | 128, 256, 512, 2048 | 64.06% | 105.68 → 72.01 | **1.47x** | 105.65 → 72.00 | 27,865 → 40,891 | 13.28 → 9.44 | -| DFlash | 128 | 128, 256, 512, 2048 | 64.06% | 35.38 → 30.50 | 1.16x | 35.42 → 30.50 | 83,107 → 96,540 | 5.81 → 5.43 | -| DFlash2 | 128 | 128, 256, 512, 2048 | 64.06% | 35.85 → 30.81 | 1.16x | 36.10 → 30.87 | 81,542 → 95,379 | 5.08 → 4.80 | -| DFlash | 128 | 1024, 1024, 1024, 1024 | 0% | 33.35 → 31.23 | 1.07x | 33.56 → 31.22 | 122,061 → 131,179 | 5.67 → 5.59 | -| DFlash2 | 128 | 1024, 1024, 1024, 1024 | 0% | 34.63 → 31.35 | 1.10x | 34.71 → 31.55 | 118,005 → 129,824 | 4.94 → 4.98 | - -The 512-anchor mixed-length case preserves all 1178 valid sampled blocks and -skips 870 invalid slots from the padded 2048-slot grid. The backbone therefore -processes 18,848 proposal tokens instead of 32,768. Context rows fall from 8192 -to 2944. Equal-length gains mainly reflect the sparse mask builder; they are -not evidence of removed context padding. - -Numerical checks use exact anchor/keep-mask comparisons and tolerance-based -loss/gradient comparisons. BF16 execution is not bitwise equivalent: one -DFlash2 128-anchor run's loss differed by about 2% after 25 optimizer updates, -despite passing initial loss/all-gradient checks. This experiment does not -establish convergence, checkpoint quality, or serving acceptance equivalence. -Evaluate those separately on a representative pretrained target and dataset. - -```bash -PYTHONPATH=. python scripts/benchmark_dflash_sequence_packing.py \ - --preset tiny --dtype float32 --correctness-only \ - --output artifacts/sequence-packing/dflash-tiny.json -PYTHONPATH=. python scripts/benchmark_dflash_sequence_packing.py \ - --preset medium --anchors 512 --warmup 5 --steps 20 \ - --output artifacts/sequence-packing/dflash-medium.json -``` - -Committed evidence includes the [512-anchor BF16 results](sequence-packing-results/dflash-family-512anchors-bf16.json), -[128-anchor BF16 results](sequence-packing-results/dflash-family-128anchors-bf16.json), -and [tiny-model FP32 comparisons](sequence-packing-results/dflash-family-tiny-fp32.json). -These preserve all timed samples, memory, settings, numerical comparisons and -measured source hashes. The [evidence inventory](sequence-packing-results/README.md) -also links the subsequent real-target online measurements. - -Exact source snapshots, source-verification files, capture/consumer service logs, -regression logs and the ad hoc `check_packed_fsdp.py` probe remain local under -`artifacts/sequence-packing/`; they are not committed with this report. The -two-rank command above describes that retained local probe, not a file shipped -in this repository. The committed runtime tests cover the supported model, -strategy and lifecycle behavior. - -Do not extrapolate the [EAGLE3 measurements](eagle3-sequence-packing.md) to -DFlash/DFlash2. Their proposal workload differs, and a producer or network -bottleneck can limit end-to-end online gains even when consumer compute improves. diff --git a/docs/sections/benchmarks/eagle3-sequence-packing.md b/docs/sections/benchmarks/eagle3-sequence-packing.md deleted file mode 100644 index 0695c1722..000000000 --- a/docs/sections/benchmarks/eagle3-sequence-packing.md +++ /dev/null @@ -1,151 +0,0 @@ -# EAGLE3 sequence packing: implementation and H200 measurements - -Measured on 2026-10-02, based on upstream -[`53398a8f01ae47175bee8459c5b5cca3848c8a7e`](https://github.com/sgl-project/SpecForge/tree/53398a8f01ae47175bee8459c5b5cca3848c8a7e), -with the local `codex/sequence-packing` changes. - -## Result and scope - -Packing improves training throughput when a microbatch contains substantially -different sequence lengths. In the final H200 run below, 47% and 64% padding -produced 1.50x and 1.96x median-step speedups. The mean-based throughput gains -were 1.44x and 1.61x, including observed scheduling outliers. Equal-length -samples showed only a small timing difference. This is not a universal -"biggest lever": batch size one has no inter-sample padding to remove. - -This report measures **offline text EAGLE3 with FlexAttention**. The switch -also supports text DFlash/DFlash2 and online server-capture consumers; see the -[current support contract](../basic_usage/training.md#sequence-packing). -The EAGLE3 measurements below do not establish DFlash/DFlash2 or end-to-end -online throughput. FA/USP, multimodal positions, compact teacher, trimmed loss -positions, and EAGLE3 LK objectives remain unsupported. - -## How it works - -```text -training.sequence_packing - → offline provider.build_packed_collator - → DataCollatorWithPacking: [B, max(L)] → [1, sum(L)] - → Eagle3TrainStrategy: shift teacher/input/mask within each document - → OnlineEagle3Model: per-document TTT shifts + original loss denominator - → LlamaFlexAttention: isolated causal prefixes + diagonal TTT cache suffixes -``` - -The collator keeps sample order and logical microbatch size, and emits document -lengths, reset positions, and the original padded loss denominator. The attention -mask prevents information flow between documents. RoPE uses the maximum -individual document length, preserving dynamic NTK behavior. Every TTT shift -zeros each document's tail rather than importing the next document's tokens. -The plain masked loss is rescaled by `sum(L) / (B * max(L))`, preserving the -existing padded objective and effective learning rate. The number of samples, -optimizer steps, accumulation steps, and checkpoint cursor do not change. - -The implementation packs only the existing logical microbatch. It does not -reorder examples or greedily combine additional microbatches. `data.max_length` -continues to limit individual documents; packed rows can exceed that length. - -## Reproducible training-step benchmark - -- One NVIDIA H200, Python 3.12.3, Torch 2.13.0+cu130, CUDA 13.0, - Transformers 5.12.1. -- One real EAGLE3 draft layer, hidden size 2048, intermediate size 8192, - 16 attention heads / 4 KV heads, target and draft vocabulary 32000. -- BF16, TTT length 7, microbatch 4, seed 1729, 25% prompt mask. -- Same features and initial weights for both paths. Frozen target-head - projection, production loss/backward and `BF16Optimizer` update are included. -- Five warmup steps (including compilation), then 20 synchronized wall-clock - measurements per mode. Compilation is excluded from steady-state timing. -- Synthetic features are already on the GPU. Capture, file I/O, bulk H2D, - distributed communication, serving, and training convergence are excluded. -- Useful tokens/s counts original unpadded input tokens once and uses the - **mean** step time. It is not derived from the median or multiplied by TTT. - Memory is peak allocated CUDA memory, including model/optimizer/input state. - -| Original sequence lengths | Padding | P50 step ms, padded → packed | P50 speedup | Mean step ms, padded → packed | Useful tokens/s, padded → packed | Peak GiB, padded → packed | -| --- | ---: | ---: | ---: | ---: | ---: | ---: | -| 1024, 1024, 1024, 1024 | 0.00% | 80.03 → 76.61 | 1.04x | 87.82 → 76.68 | 46,643 → 53,419 | 13.09 → 9.17 | -| 512, 768, 1024, 2048 | 46.88% | 135.94 → 90.35 | 1.50x | 147.45 → 102.27 | 29,515 → 42,554 | 23.98 → 9.60 | -| 128, 256, 512, 2048 | 64.06% | 136.79 → 69.67 | 1.96x | 142.04 → 88.27 | 20,726 → 33,354 | 23.91 → 7.21 | - -The full samples retain outliers; for example CPU scheduling produced occasional -long steps. A separate 50-step repeat before the final FSDP metadata conversion -measured median speedups of 1.03x, 1.51x and 1.94x for the same three profiles, -with mean speedups of 1.04x, 1.57x and 1.87x. Treat the difference between median -and mean as part of the measurement, not guaranteed deployment throughput. - -The equal-length case also reduces peak memory. Packing changes teacher-table -layout to batch dimension one; the TTT adapter can retain contiguous slices -instead of making per-depth copies across padded batch strides. Thus observed -memory savings are not solely proportional to removed padding. - -An additional larger draft (hidden 4096, intermediate 14336, 32 heads / 8 KV -heads) with `[128, 256, 512, 2048]` measured 249.50 → 113.15 ms P50 (2.21x), -271.96 → 115.71 ms mean, and 33.03 → 12.97 GiB peak. This measurement predates -the final FSDP tuple/int metadata conversion; the numerical path is the same, -but use the table above for the final-source timing evidence. - -Commands from the repository root: - -```bash -PYTHONPATH=. python scripts/benchmark_sequence_packing.py \ - --preset tiny --dtype float32 --correctness-only \ - --output artifacts/sequence-packing/tiny-fp32.json -PYTHONPATH=. python scripts/benchmark_sequence_packing.py \ - --preset medium --warmup 5 --steps 20 \ - --output artifacts/sequence-packing/medium-bf16-current.json -``` - -## Correctness and integration evidence - -- Latest FP32 benchmark: three profiles passed, including documents shorter - than TTT. Scalar loss matched exactly; maximum trainable-gradient absolute - difference was approximately 4.1e-10. -- BF16 medium: loss, every depth's loss, and all trainable parameter gradients - passed the recorded tolerances. Maximum gradient relative L2 difference was - approximately 0.00658; BF16 equality is numerical, not bitwise. -- Seven production GPU tests passed, including FP32/BF16/dynamic-NTK parity, - cross-document isolation, mixed CPU/GPU fields, zero supervision, short - documents, and FSDP metadata compatibility. -- CPU regression: 162 tests and 523 subtests passed; the GPU lifecycle test was - skipped in this CPU run and executed separately. -- Full offline lifecycle passed: variable-length files, packed train/eval - loaders, 2 optimizer steps / 4 microsteps, partial eval batch, finite metrics, - both checkpoints, and original sample counters 4 and 8. -- This EAGLE3 lifecycle used single-rank FSDP, which selects NO_SHARD. The - subsequent [DFlash/online validation report](dflash-sequence-packing.md) - includes a two-rank FULL_SHARD comparison and live Mooncake coverage for - EAGLE3, DFlash and DFlash2. Long-run convergence/acceptance and speculative - serving performance remain outside these measurements. - -Committed evidence includes the [medium BF16 results](sequence-packing-results/eagle3-medium-bf16.json), -[tiny FP32 comparisons](sequence-packing-results/eagle3-tiny-fp32.json), and -[larger-draft results](sequence-packing-results/eagle3-large-bf16.json). -These JSON files contain settings, individual timings, numerical comparisons -and measured source hashes. See the [evidence inventory](sequence-packing-results/README.md) -for provenance and limitations. - -Additional validation logs, `source-verification.json` and exact measured source -snapshots remain local under `artifacts/sequence-packing/`; they are not committed -with this report. The local source verification found all seven measured source -ASTs equivalent to the final implementation; changes at that point were -formatting only. - -## Enabling packing - -```bash -specforge train \ - --config examples/configs/offline/colocated/qwen3-8b-eagle3-offline.yaml \ - training.batch_size=4 \ - training.attention_backend=flex_attention \ - training.sequence_packing=true -``` - -Choose the same batch size and gradient accumulation for the baseline and -packed run. Increasing batch size at the same time changes the training -schedule and invalidates a simple A/B comparison. - -Text DFlash/DFlash2 packing is also implemented, with document-aware context -attention and positions, boundary-safe labels and the same per-document anchor -sampling budget. Its [implementation and measurements](dflash-sequence-packing.md) -use a separate attention and proposal path; EAGLE3 timing gains do not predict -DFlash/DFlash2 gains. diff --git a/docs/sections/benchmarks/full-model-sequence-packing.md b/docs/sections/benchmarks/full-model-sequence-packing.md deleted file mode 100644 index d006631e3..000000000 --- a/docs/sections/benchmarks/full-model-sequence-packing.md +++ /dev/null @@ -1,181 +0,0 @@ -# Full-size Qwen3-4B online training: longer sequence-packing comparison - -This extends the [32-step online measurement](online-sequence-packing.md) to -256 optimizer steps over 1,024 real ShareGPT conversations, with normal periodic -training diagnostics and checkpoints. It uses base -`53398a8f01ae47175bee8459c5b5cca3848c8a7e` plus the local `codex/sequence-packing` -changes. The previous online measurement already used full model dimensions; -this experiment increases run length, repeat count, and model-training evidence. - -## Results - -The complete-model training step improved **1.044× for DFlash** and **1.052× -for DFlash2**. End-to-end completion including both checkpoints improved -**1.037× and 1.033×**, respectively. The values below are medians of four runs -per mode; each run processes 1,024 conversations in 256 optimizer steps. - -| Model | Padded training step | Packed training step | Training-step ratio | Padded completion | Packed completion | Completion ratio | -| --- | ---: | ---: | ---: | ---: | ---: | ---: | -| DFlash | 452.79 ms | 433.64 ms | **1.0442×** | 145.526 s | 140.378 s | **1.0367×** | -| DFlash2 | 426.98 ms | 405.76 ms | **1.0523×** | 137.320 s | 132.905 s | **1.0332×** | - -Training-step values are the existing host-wall diagnostics over the first -250 steps, including the detailed metric steps. Completion covers all 256 -steps, live target capture/transport/waiting, acknowledgements, and checkpoints -at steps 128 and 256. Both exclude startup and separate full-corpus warmup. - -| Model | Padded completion range | Packed completion range | Matched-pair completion ratios | -| --- | ---: | ---: | ---: | -| DFlash | 144.905–145.755 s | 138.331–140.510 s | 1.0328–1.0533× | -| DFlash2 | 136.638–137.591 s | 132.296–133.598 s | 1.0258–1.0377× | - -All chronological A/B pairs favored packing; the ranges do not overlap. -These are observed ranges from four runs per mode, not confidence intervals. -Checkpoint totals were approximately 18.46–19.79 s per run. Consumer fetch -waits also varied: 9.52–10.09 s padded versus 8.04–9.89 s packed for DFlash, -and 7.60–8.33 s versus 8.40–9.15 s for DFlash2. The full-run ratio therefore -includes normal pipeline and storage variation, not solely GPU computation. - -The pipeline timer ending at the final durable acknowledgement, including the -intermediate checkpoint but excluding the final one, measured -135.774→130.651 s (**1.0392×**) for DFlash and 127.556→122.898 s (**1.0379×**) -for DFlash2. This boundary differs from the previous 32-step experiment, which -had no intermediate checkpoint. - -All 16 measured runs and four warmups passed the independent full-model audit. -Every measured run had zero new compiler-counter activity. All 58 DFlash and -81 DFlash2 trainable tensors received exactly 256 AdamW updates, all five -draft layers had observed sampled weight changes, and all trainable elements -were finite. Final losses repeated exactly within each mode; they were -7.263832/7.263966 padded/packed for DFlash and 7.968346/7.968497 for DFlash2. -These short training comparisons do not establish equal final model quality -or time to convergence. - -## Complete model and workload - -- Actual Qwen3-4B target checkpoint: **4,022,468,096 saved parameters**, all 36 - decoder layers, hidden size 2560, vocabulary 151936. All layer indices were - verified from the checkpoint tensor headers. The target runs full prefill; - it is frozen, as required by speculative draft training. -- Complete repository `configs/qwen3-4b-dflash.json` draft: five layers, - intermediate size 9728, 32 attention heads, eight KV heads, head dimension - 128, block size 16, 512 sampled anchors per document. DFlash2 uses the same - dimensions with its convolution and selector modules enabled. -- Runtime parameter inventories confirm **537,427,200 trainable parameters in - 58 tensors for DFlash**, and **558,918,912 in 81 tensors for DFlash2**. The - frozen target embedding/head share 388,956,160 parameters, counted once. -- Fresh, identically seeded draft initialization for every arm. All trainable - draft parameters use the production backward and BF16 optimizer with FP32 - AdamW master parameters. Frozen target embeddings and the LM head are loaded - from the real target checkpoint. Objective chunk size 128 processes every - sampled block and the entire vocabulary. -- Two H200 GPUs on one host: target on GPU1, consumer on GPU0. Torch - 2.13.0+cu130, Transformers 5.12.1, isolated SGLang 0.5.18 with the repository's - capture patch. The reserved host has eight GPUs; the experiment uses two. -- BF16, batch four, accumulation one, learning rate 0.0001, optimizer warmup - ratio zero, gradient clipping 0.5, teacher metrics enabled. Detailed metrics - run every 50 steps; checkpoints run at steps 128 and 256. -- 1,024 real ShareGPT examples, deterministic seed 1729, production Qwen parser - and actual tokenizer, original assistant supervision masks, maximum length - 2048. There are **1,309,869 input tokens** and **1,026,019 supervised tokens**. - Batch-four context padding is **32.55%**. Invalid proposal removal reduces - slots from 523,408 to 454,930 (**13.08%**). -- Each mode receives a full untimed 256-step warmup before measurement. Two - ABBA blocks provide four measured runs per mode per architecture. Every arm - starts with identical model/optimizer initialization, prompt order and masks. - -## Timing and verification - -The producer and consumer use canonical SpecForge online builders and run -concurrently, with real target captures for every arm. Features travel through -Mooncake TCP host buffers. The producer has one worker, capture batch four and -an eight-reference high watermark. Radix caching is disabled, and fresh request -namespaces force full prefill. Data preparation, process/model startup and JIT -warmup are outside the main timer. - -The full completion timer starts at the first actual capture dispatch with an -empty channel and ends after `trainer.fit()`, including the intermediate and -final checkpoints and trainer cleanup. The pipeline timer ends at the final -optimizer update and durable sample acknowledgement with CUDA synchronization; -it includes the step-128 checkpoint, but not the final step-256 checkpoint. -These boundaries must not be compared directly with a compute-only benchmark. - -Existing trainer diagnostics also report training-thread wall time spent inside -`TrainerCore.train_step` for the five 50-step logging windows. This includes -strategy preparation and host-to-device transfers, forward, -objectives/diagnostics, backward and optimizer work; it excludes data -waiting, acknowledgements and checkpoint calls. It is a host-observed training -metric, not a CUDA-event kernel profile, and covers the first 250 of 256 steps. - -Every run records exact parameter inventory, optimizer identity coverage, -per-parameter AdamW step counters, a complete finiteness scan of trainable -weights, and sampled before/after weight values for each layer. Tied target -weights are deduplicated. The warmup alone observes first-step gradient presence; -measured steps have no gradient-observation hooks. Sampling and scans occur -outside timing. Sample hashes demonstrate observed updates, not a full-tensor -numerical comparison or an assertion that every element must change. - -The benchmark also checks identical prompt/request hashes, model initialization, -publication and consumption order, optimizer sample grouping, durable -acknowledgements and final checkpoint counters. Saved Dynamo counter deltas -identify any compilation during measured runs. Finite losses and updated layers -do not establish long-run convergence or serving quality. - -This experiment uses the complete model and real training path, with one -consumer rank. It does not exercise the YAML CLI orchestration, evaluation, -resume, multi-node scaling or final speculative-serving quality. The logger -collects metrics in memory rather than publishing to an external dashboard. - -## Why the full-model gain is modest - -Packing compacts context rows and valid proposal blocks through the draft -backbone. `_forward_draft_blocks` then scatters hidden states back to the -original batch/anchor/block layout before the objective. The normal LM-head -path projects that restored layout across the full vocabulary and applies the -loss mask after cross-entropy. DFlash2's existing fused head skips the clean -anchor slot of every block, but still processes the restored invalid blocks. -Neither objective path compacts all invalid proposal rows in this change. - -These are source facts in `specforge/algorithms/common/dflash_family_model.py` -and `specforge/core/dflash_head_triton.py`. Together with the corpus's 13.08% -removable proposal slots and full 151,936-token vocabulary, they explain why -removing context padding does not remove an equivalent fraction of all training -work. This is a structural explanation, not a measured kernel-time breakdown. -Packing the objective could be a separate optimization; it is not implemented -or benchmarked by this experiment. - -## Reproduction and artifacts - -The benchmark requires a capture-enabled SGLang server, Mooncake master and the -preprocessed public corpus described above. From the repository root, run: - -```bash -python scripts/benchmark_online_sequence_packing.py \ - --server-url http://127.0.0.1:31012 \ - --target-model /cluster-storage/models/Qwen3-4B \ - --draft-config configs/qwen3-4b-dflash.json \ - --prompts-path /scratch/specforge-packing-full-model-20261002/sharegpt-prompts-1024.jsonl \ - --algorithm both --steps 256 --warmup-steps 256 --repeats 2 \ - --log-interval 50 --save-interval 128 \ - --work-dir /scratch/specforge-packing-full-model-20261002/long_v2 \ - --output /scratch/specforge-packing-full-model-20261002/long_v2.json -``` - -Committed evidence includes the [per-run timing summary](sequence-packing-results/full-model-analysis.json), -[independent full-model audit](sequence-packing-results/full-model-audit.json), -and [target tensor inventory](sequence-packing-results/target-parameter-inventory.json). -The summaries preserve every measured run's timing values, checks, parameter -counts and optimizer-update evidence. The [evidence inventory](sequence-packing-results/README.md) -describes their provenance and the raw configuration's unused CLI defaults. - -The 9.65 MB raw `long_v2.json`, its log, exact source archive and hashes, detailed -data manifest and preprocessing provenance, runtime/service launch records and -cleanup receipts remain local under `artifacts/sequence-packing/full-model/`. -These local files, including `run_benchmark.py`, are not committed with this -report. The raw result SHA256 is -`f8895e415d1baaef7ddaa74a7f8e322d83d13e008d2f7606b0378fb0b4861179`. -The 636-file measured source archive was verified against the remote snapshot; -the final results documentation was written afterward. Raw conversation text -stays on the devbox, and checkpoints were validated before deletion between -runs to bound disk use. All benchmark-owned services were stopped successfully; -the existing devbox reservation was retained. diff --git a/docs/sections/benchmarks/online-sequence-packing.md b/docs/sections/benchmarks/online-sequence-packing.md deleted file mode 100644 index 3d487b2a6..000000000 --- a/docs/sections/benchmarks/online-sequence-packing.md +++ /dev/null @@ -1,147 +0,0 @@ -# Online sequence packing: real Qwen3-4B pipeline benchmark - -This benchmark measures actual pretrained-target capture and concurrent draft -training, extending the [consumer-compute measurements](dflash-sequence-packing.md). -It is based on upstream `53398a8f01ae47175bee8459c5b5cca3848c8a7e` plus the local -`codex/sequence-packing` changes, tested on 2026-10-02. - -A subsequent [256-step full-model comparison](full-model-sequence-packing.md) -uses 1,024 real conversations, four runs per mode, periodic diagnostics and -checkpoints, and verifies every trainable parameter's optimizer participation. -It measures 1.044×/1.052× training-step improvements and 1.037×/1.033× full -completion improvements for DFlash/DFlash2. - -## Measured results - -On this workload, packing improved online pipeline throughput by **4.1% for -DFlash** and **4.5% for DFlash2**. Each time below is the median of two measured -runs processing the same 128 conversations in 32 optimizer steps. - -| Model | Padded pipeline | Packed pipeline | Throughput speedup | Useful input tokens/s, padded → packed | -| --- | ---: | ---: | ---: | ---: | -| DFlash | 15.3356 s | 14.7329 s | **1.0409× (+4.09%)** | 10,811 → 11,253 | -| DFlash2 | 14.3867 s | 13.7686 s | **1.0449× (+4.49%)** | 11,524 → 12,041 | - -Both packed runs were faster than both padded runs for each model. DFlash -ranges were 15.2822–15.3889 s padded and 14.6678–14.7980 s packed; DFlash2 -ranges were 14.3754–14.3980 s padded and 13.7193–13.8178 s packed. These are -short repeated measurements, not a confidence interval or a whole-epoch result. - -Including the final checkpoint changes the comparison: - -| Model | Padded completion with checkpoint | Packed completion with checkpoint | Ratio | -| --- | ---: | ---: | ---: | -| DFlash | 25.2150 s | 24.9084 s | 1.0123× | -| DFlash2 | 25.1072 s | 24.0008 s | 1.0461× | - -Each checkpoint took approximately 9.87–10.88 s. Storage timing varied between -arms: about 0.488 s of DFlash2's 1.106 s total-completion gap came from checkpoint -I/O. The pipeline measurement is therefore the cleaner estimate of packing's -effect. Checkpoint frequency will affect the realized full-run improvement. - -All eight measured runs passed sample-order, HTTP-payload, initialization, -optimizer-grouping, durable-acknowledgement, and checkpoint-counter checks. -Every measured run had **zero new Dynamo compilations** and captured 30 of its -32 batches after its first optimizer acknowledgement, confirming concurrent -online feature production. Final losses were finite and repeatable within each -arm; this short run does not establish long-run convergence or serving quality. - -The earlier 1.39×/1.47× consumer-compute results used synthetic lengths with -64% context padding and 42.5% invalid proposal slots. This real corpus has -33.28% context padding and only 13.08% removable proposal slots, and the online -timer includes target capture and transport. The synthetic speedups should not -be used as an estimate of end-to-end training gains. - -## Workload and timing boundaries - -- Two NVIDIA H200 GPUs on one host: a patched SGLang 0.5.18 Qwen3-4B target on - GPU1 and a single-rank FSDP consumer on GPU0. The draft is freshly initialized; - the target, frozen embeddings, and frozen LM head use the actual pretrained - weights from `/cluster-storage/models/Qwen3-4B`. -- The repository's `configs/qwen3-4b-dflash.json`: five draft layers, hidden size - 2560, intermediate size 9728, 32 attention heads / 8 KV heads, head dimension - 128, vocabulary 151936, block size 16, and target capture layers - `[1, 9, 17, 25, 33]`. DFlash2 uses the same base dimensions with its convolution - and selector modules enabled. -- BF16, batch four, accumulation one, 512 anchors per document, objective chunks - of 128 blocks, learning rate 0.0001, gradient clipping 0.5, teacher metrics - enabled, and log interval 50. With 32 steps per arm, no periodic detailed-metric - step runs. Both arms use identical settings apart from sequence packing. -- 128 real ShareGPT conversations, deterministically shuffled with seed 1729 - and prepared with SpecForge's Qwen parser and the target tokenizer. The - original assistant supervision masks are retained. Maximum length is 2048; - there are 165,788 input tokens and 131,810 supervised tokens. -- Lengths range from 35 to 2048, mean 1295.22 and median 1536. Batch-four context - padding is **33.28%**. With the original per-batch anchor cap, proposal slots - fall from 65,388 to 56,834, removing **13.08%** invalid slots. This is a natural - corpus slice, not the previous synthetic 64%-padding length pattern. -- Mooncake uses TCP and host tensors, with pinned receive buffers. One canonical - producer worker captures batches of four, concurrently with the canonical - online consumer, with an eight-reference high watermark. One active capture - batch may overshoot that watermark. Target capture and feature supply are - rerun for every arm; no feature cache is replayed. - -The primary timer starts at the first capture HTTP dispatch with an empty -channel, and ends after the last optimizer update, synchronous durable sample -acknowledgement, and CUDA synchronization. It includes pipeline fill/drain, -target prefill/capture, feature transport/fetch, data waiting, collation, model -forward/backward, optimizer work, and acknowledgement. It excludes data -preparation, model/server startup, and JIT warmup. - -Each mode first performs an untimed replay of the complete ordered corpus. -Measured runs then use A → B → B → A order, where A is padded and B is packed. -Every run starts from the same seeded draft initialization and resets optimizer -state. The target adapter's fresh request namespaces force full prefill, and -the benchmark server also disables radix caching. Dynamo counters are saved -for every run to verify that warmed timing did not include new graph compilation. - -Canonical final checkpoints are saved and validated in every run. Their cost, -and total completion time including checkpoint/cleanup, are reported separately -from the primary pipeline timer. Checkpoint files are deleted after validation -so repeated benchmarking does not retain many copies of the same initial run. -This is a warmed online training pipeline comparison, not complete cold CLI -startup, multi-node scaling, RDMA, convergence, or speculative-serving performance. - -## Reproduction and evidence - -The benchmark requires an already running capture-enabled SGLang server and -Mooncake master. The prompts are prepared once from the cached public -`anon8231489123/ShareGPT_Vicuna_unfiltered` dataset snapshot -`192ab2185289094fc556ec8ce5ce1e8e587154ca`; preparation is outside the timer. - -```bash -CUDA_VISIBLE_DEVICES=0 \ -MOONCAKE_MASTER_SERVER_ADDR=127.0.0.1:50212 \ -MOONCAKE_METADATA_SERVER=http://127.0.0.1:8112/metadata \ -MOONCAKE_LOCAL_HOSTNAME=127.0.0.1 \ -MOONCAKE_PROTOCOL=tcp \ -TORCH_LOGS=recompiles \ -PYTHONPATH=/scratch/sglang-sequence-packing-online-20261002:. \ -python scripts/benchmark_online_sequence_packing.py \ - --server-url http://127.0.0.1:31012 \ - --target-model /cluster-storage/models/Qwen3-4B \ - --draft-config configs/qwen3-4b-dflash.json \ - --prompts-path /scratch/specforge-packing-e2e-20261002/sharegpt-prompts.jsonl \ - --algorithm both --warmup-steps 32 --steps 32 --repeats 1 \ - --work-dir /scratch/specforge-packing-e2e-20261002/full_v1 \ - --output /scratch/specforge-packing-e2e-20261002/full_v1.json -``` - -The [committed 32-step benchmark summary](sequence-packing-results/online-32step.json) -preserves every run's timing samples across eight measured runs and four warmups, -plus settings, prompt/request hashes and lifecycle checks. Detailed per-capture -events remain in the local raw report. See the [evidence inventory](sequence-packing-results/README.md) -for the longer full-model comparison and provenance. The original raw report's -SHA256 is -`d0284b6ab87ad26bba3c6624c4d4b203cb4f18fef08609476729d650b55f9be4`. - -The detailed prompt manifest, preprocessing/service-start scripts, exact source -snapshots, service logs and cleanup receipts remain local under -`artifacts/sequence-packing/e2e/`; they are not committed with this report. -Original conversation text was not copied into the local evidence directory. - -The benchmark verifies actual HTTP payload hashes, publication order, consumed -sample order and optimizer grouping, all durable acknowledgements, finite final -loss, producer completion, and the final checkpoint's step and sample counters. -The benchmark-specific CPU tests and package architecture checks passed -21 tests with eight subtests before the GPU run. diff --git a/docs/sections/benchmarks/sequence-packing-results/README.md b/docs/sections/benchmarks/sequence-packing-results/README.md deleted file mode 100644 index a5d5fe7aa..000000000 --- a/docs/sections/benchmarks/sequence-packing-results/README.md +++ /dev/null @@ -1,78 +0,0 @@ -# Sequence-packing benchmark evidence - -These files support the [EAGLE3](../eagle3-sequence-packing.md), -[DFlash/DFlash2](../dflash-sequence-packing.md), -[32-step online](../online-sequence-packing.md) and -[256-step full-model](../full-model-sequence-packing.md) reports measured on -2026-10-02. The implementation was based on upstream -`53398a8f01ae47175bee8459c5b5cca3848c8a7e` plus the sequence-packing changes. -Each report defines its workload and timing boundary; the results are not -interchangeable estimates of end-to-end gains. - -## Committed files - -| File | Evidence | -| --- | --- | -| [full-model-analysis.json](full-model-analysis.json) | Every measured run's timing values and medians for the 256-step comparison. | -| [full-model-audit.json](full-model-audit.json) | Independent checks for all 16 measured runs and four warmups: model dimensions, optimizer participation, updates, finiteness, compilation and pipeline ordering. | -| [target-parameter-inventory.json](target-parameter-inventory.json) | Qwen3-4B checkpoint tensor names, shapes and counts; verifies all 36 target layers and 4,022,468,096 saved parameters. | -| [online-32step.json](online-32step.json) | Derived summary preserving every run's timing samples, settings and lifecycle checks from the shorter online experiment. | -| [eagle3-medium-bf16.json](eagle3-medium-bf16.json) | Final-source medium EAGLE3 compute timings, memory, numerical checks and source hashes. | -| [eagle3-large-bf16.json](eagle3-large-bf16.json) | Larger EAGLE3 compute case, measured before the final FSDP metadata conversion. | -| [eagle3-tiny-fp32.json](eagle3-tiny-fp32.json) | Derived FP32 correctness summary, including the number of gradient tensors and maximum differences. | -| [dflash-family-128anchors-bf16.json](dflash-family-128anchors-bf16.json) | Final optimized DFlash/DFlash2 compute measurements with 128 anchors per document. | -| [dflash-family-512anchors-bf16.json](dflash-family-512anchors-bf16.json) | Final optimized compute measurements with 512 anchors per document. | -| [dflash-family-tiny-fp32.json](dflash-family-tiny-fp32.json) | Derived FP32 correctness summary for DFlash/DFlash2, including exact sampled-anchor checks. | - -The two tiny FP32 summaries replace per-parameter gradient lists with tensor -counts, the all-pass flag and maximum absolute/relative differences. Their other -fields are unchanged, and each includes the original report SHA256. The 32-step -summary removes detailed per-capture event lists while preserving individual -run timings, including warmups, and records the original report SHA256. All -other JSON files preserve their corresponding local evidence; only the target -inventory's missing final newline was normalized by the repository hooks. - -## Configuration and interpretation - -The online benchmark's raw CLI settings retain `draft_layers=2`, a default used -only when no draft config is supplied. Both online experiments supplied -`configs/qwen3-4b-dflash.json`; the resolved config and runtime model inventory -confirm **five draft layers**. Likewise, supplying `prompts_path` bypasses the -synthetic `lengths` and `prompt_fraction` defaults. These experiments use real -ShareGPT tokens and their actual assistant supervision masks. - -The target checkpoint has 36 layers and is frozen. All trainable draft parameters -are updated: 537,427,200 parameters for DFlash and 558,918,912 for DFlash2. The -256-step evidence checks each trainable tensor's optimizer step count and scans -all trainable elements for finiteness. Layer-update hashes sample weight values; -they do not assert that every element changed or that packed and padded training -trajectories are numerically identical. - -The full-model training-step diagnostics cover the first 250 of 256 steps and -include periodic metrics. Pipeline timing includes the step-128 checkpoint; -completion timing includes both step-128 and step-256 checkpoints. Startup, -data preparation and separate corpus warmup are excluded. The tests do not -establish convergence, final model quality or speculative-serving throughput. - -## Evidence retained locally - -The original 256-step raw report, about 9.65 MB, is retained locally as -`artifacts/sequence-packing/full-model/long_v2.json`; it is **not committed**. -Its SHA256 is -`f8895e415d1baaef7ddaa74a7f8e322d83d13e008d2f7606b0378fb0b4861179`. -The original 32-step raw report is retained locally as -`artifacts/sequence-packing/e2e/full_v1.json`; its SHA256 is -`d0284b6ab87ad26bba3c6624c4d4b203cb4f18fef08609476729d650b55f9be4`. - -Detailed source snapshots, source-verification records, dataset manifests, -preparation and service-launch helpers, service logs, ad hoc distributed probes, -reservation metadata and cleanup receipts also remain local. They are not part -of this directory or promised as repository downloads. Original conversation -text was not copied into these artifacts. The public dataset revision is -`anon8231489123/ShareGPT_Vicuna_unfiltered@192ab2185289094fc556ec8ce5ce1e8e587154ca`; -the 1,024-prompt tokenized file SHA256 is -`f90478e0d3e32f0d3c45c515381b9bc42951cdd6931830ac1a7380f4459f24b1`. - -The full-model measured source manifest matches all 17 changed/new production -files in the implementation at packaging time. Benchmark documentation and -public evidence packaging were finalized after the measurements. diff --git a/docs/sections/benchmarks/sequence-packing-results/dflash-family-128anchors-bf16.json b/docs/sections/benchmarks/sequence-packing-results/dflash-family-128anchors-bf16.json deleted file mode 100644 index 630e15b7c..000000000 --- a/docs/sections/benchmarks/sequence-packing-results/dflash-family-128anchors-bf16.json +++ /dev/null @@ -1,1070 +0,0 @@ -{ - "timestamp_utc": "2026-10-02T04:43:38.703421+00:00", - "source": { - "files_sha256": { - "scripts/benchmark_dflash_sequence_packing.py": "0372ddf408fddc99c95138ab448025e37600c51004bceb6fd51d0432afb82e46", - "specforge/benchmarks/benchmark_dflash_sequence_packing.py": "00b9ff628d145d3f986eeb8c43f62671183a197f4c4575faf3f432f16ef0641c", - "specforge/algorithms/common/dflash_family_model.py": "86be39addbc5c50036f37c5d5282c53aaa8f35ab5ae1f8cfabf33078494e1421", - "specforge/algorithms/common/hidden_states_data.py": "fa09c5e5037c31b1165c2f8d1774e2cd4a40fde8b08f1f6deddf87f4f4af7f24", - "specforge/modeling/draft/dflash.py": "97af112a6ecf66d4b397aae77f0f97c08a49759c6305a4c70577a4f78b45ca90", - "specforge/modeling/draft/dflash2.py": "819c6b3d8d6d8ffc31a843ed40d539ddf635f58b148a01c8c43d0bff1dd02b09", - "specforge/modeling/packed_dflash.py": "d763e0ad3c609602323b9e3413968860cc680a8be8c986cc7612a956556ab96e", - "specforge/training/strategies/base.py": "276932ad331e95d5b91eddc67cf3acb0b5447ee5973f9bbc757e54093a82939f" - }, - "head": "unavailable (copied snapshot is identified by file hashes)" - }, - "settings": { - "preset": "medium", - "algorithm": "both", - "hidden_size": 2048, - "intermediate_size": 8192, - "layers": 2, - "heads": 16, - "kv_heads": 4, - "vocab_size": 32000, - "block_size": 16, - "anchors": 128, - "capture_layers": 2, - "conv_group_size": 32, - "conv_kernel_size": 4, - "selector_rank": 16, - "selector_top_k": 16, - "lengths": [ - [ - 1024, - 1024, - 1024, - 1024 - ], - [ - 128, - 256, - 512, - 2048 - ] - ], - "dtype": "bfloat16", - "sliding_window": null, - "warmup": 5, - "steps": 20, - "seed": 1729, - "learning_rate": 0.0001, - "objective_chunk_blocks": 128, - "correctness_only": false, - "skip_correctness": false, - "atol": 0.002, - "rtol": 0.02, - "output": "artifacts/sequence-packing/dflash-family-medium-bf16-optimized.json" - }, - "environment": { - "torch": "2.13.0+cu130", - "cuda": "13.0", - "gpu": "NVIDIA H200", - "cuda_visible_devices": "0" - }, - "scope": "Single GPU; synthetic frozen target components and captured features; GPU-resident hidden features plus CPU integer features; production strategy forward/backward/BF16Optimizer; excludes capture, hidden-feature I/O/H2D, distributed training and serving.", - "cases": [ - { - "algorithm": "dflash", - "lengths": [ - 1024, - 1024, - 1024, - 1024 - ], - "padding_fraction": 0.0, - "useful_tokens": 4096, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 3.528594970703125e-05, - "relative_l2_diff": 3.344875867136468e-06 - }, - "loss_terms": { - "pass": true, - "max_abs_diff": 0.1171875, - "relative_l2_diff": 3.27596571374284e-06 - }, - "loss_values": [ - 10.54925537109375, - 10.549220085144043 - ], - "identical_sampled_anchors": true, - "sampled_anchor_count": 512, - "gradients": { - "draft_model.layers.0.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.004035275282337427 - }, - "draft_model.layers.0.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.00501776531413531 - }, - "draft_model.layers.0.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.004276743031545815 - }, - "draft_model.layers.0.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0030888238226688147 - }, - "draft_model.layers.0.self_attn.q_norm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.0037907497753986714 - }, - "draft_model.layers.0.self_attn.k_norm.weight": { - "pass": true, - "max_abs_diff": 1.1920928955078125e-06, - "relative_l2_diff": 0.004571789916409457 - }, - "draft_model.layers.0.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.002867089280638763 - }, - "draft_model.layers.0.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.0027648728928907685 - }, - "draft_model.layers.0.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.002450745866595975 - }, - "draft_model.layers.0.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 7.152557373046875e-07, - "relative_l2_diff": 0.004297563642974598 - }, - "draft_model.layers.0.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0031199611716563143 - }, - "draft_model.layers.1.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0029851964017069575 - }, - "draft_model.layers.1.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.004035853911087431 - }, - "draft_model.layers.1.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0033796309921081966 - }, - "draft_model.layers.1.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0024074416791069384 - }, - "draft_model.layers.1.self_attn.q_norm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.0035275305395718825 - }, - "draft_model.layers.1.self_attn.k_norm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.002868279839387132 - }, - "draft_model.layers.1.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0021278734955892174 - }, - "draft_model.layers.1.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0020574586175487694 - }, - "draft_model.layers.1.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0019429627771338416 - }, - "draft_model.layers.1.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 4.76837158203125e-07, - "relative_l2_diff": 0.0030072922824688204 - }, - "draft_model.layers.1.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0026339067731564235 - }, - "draft_model.norm.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0014766025487033196 - }, - "draft_model.fc.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.005959122214185738 - }, - "draft_model.hidden_norm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.0060941715028472844 - } - }, - "pass": true - }, - "padded": { - "mean_step_ms": 33.55688191950321, - "p50_step_ms": 33.35478808730841, - "useful_tokens_per_second": 122061.40039547031, - "p50_useful_tokens_per_second": 122800.96006841488, - "peak_allocated_gib": 5.674890518188477, - "baseline_allocated_gib": 2.144904136657715, - "warmup_seconds_including_compile": 0.19184968434274197, - "step_ms": [ - 33.65137241780758, - 33.180927857756615, - 33.23202021420002, - 33.49055536091328, - 33.38956832885742, - 33.32914970815182, - 33.664412796497345, - 33.452145755290985, - 33.31155702471733, - 36.49650141596794, - 33.26563164591789, - 33.242642879486084, - 33.677954226732254, - 33.29920209944248, - 33.28862413764, - 33.3385169506073, - 33.35254080593586, - 33.37463736534119, - 33.742642030119896, - 33.357035368680954 - ], - "final_loss": 8.091357231140137 - }, - "packed": { - "mean_step_ms": 31.22454872354865, - "p50_step_ms": 31.23283013701439, - "useful_tokens_per_second": 131178.83740337024, - "p50_useful_tokens_per_second": 131144.05521470125, - "peak_allocated_gib": 5.594754219055176, - "baseline_allocated_gib": 2.035337448120117, - "warmup_seconds_including_compile": 0.15607250295579433, - "step_ms": [ - 31.093496829271317, - 31.08426183462143, - 31.171666458249092, - 31.09334409236908, - 31.18988499045372, - 31.12914226949215, - 31.23560920357704, - 31.254412606358528, - 31.43681026995182, - 31.332312151789665, - 31.14699199795723, - 31.310075893998146, - 31.280379742383957, - 31.256863847374916, - 31.24155104160309, - 31.199980527162552, - 31.363260000944138, - 31.230051070451736, - 31.19666315615177, - 31.244216486811638 - ], - "final_loss": 8.090627670288086 - }, - "p50_speedup": 1.0679399830558185, - "mean_speedup": 1.0746954973346206 - }, - { - "algorithm": "dflash", - "lengths": [ - 128, - 256, - 512, - 2048 - ], - "padding_fraction": 0.640625, - "useful_tokens": 2944, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 7.43865966796875e-05, - "relative_l2_diff": 7.076170071937434e-06 - }, - "loss_terms": { - "pass": true, - "max_abs_diff": 0.232421875, - "relative_l2_diff": 7.069888761538474e-06 - }, - "loss_values": [ - 10.51226806640625, - 10.51219367980957 - ], - "identical_sampled_anchors": true, - "sampled_anchor_count": 478, - "gradients": { - "draft_model.layers.0.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0017025760377892554 - }, - "draft_model.layers.0.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0024859584683051485 - }, - "draft_model.layers.0.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0020038922260345268 - }, - "draft_model.layers.0.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0013343185538530963 - }, - "draft_model.layers.0.self_attn.q_norm.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0017520098353950937 - }, - "draft_model.layers.0.self_attn.k_norm.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0018270023601644342 - }, - "draft_model.layers.0.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.002008949811204485 - }, - "draft_model.layers.0.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.0020283964646443113 - }, - "draft_model.layers.0.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.001880820308327072 - }, - "draft_model.layers.0.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0018999089901229125 - }, - "draft_model.layers.0.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0018916088807969396 - }, - "draft_model.layers.1.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0016530553335780654 - }, - "draft_model.layers.1.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0024914118832995687 - }, - "draft_model.layers.1.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0019648644067848447 - }, - "draft_model.layers.1.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0013547066257487894 - }, - "draft_model.layers.1.self_attn.q_norm.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0021836037110825788 - }, - "draft_model.layers.1.self_attn.k_norm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.0017769949609473883 - }, - "draft_model.layers.1.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.001961136128200421 - }, - "draft_model.layers.1.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.0018879952579109367 - }, - "draft_model.layers.1.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0019155058101548376 - }, - "draft_model.layers.1.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.001667171626780272 - }, - "draft_model.layers.1.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0018729544534959166 - }, - "draft_model.norm.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0018962999532875571 - }, - "draft_model.fc.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.003503925527484017 - }, - "draft_model.hidden_norm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.0033592604071104744 - } - }, - "pass": true - }, - "padded": { - "mean_step_ms": 35.424255300313234, - "p50_step_ms": 35.38447059690952, - "useful_tokens_per_second": 83106.8987912914, - "p50_useful_tokens_per_second": 83200.34044135534, - "peak_allocated_gib": 5.810401916503906, - "baseline_allocated_gib": 2.178708076477051, - "warmup_seconds_including_compile": 0.17471044324338436, - "step_ms": [ - 35.73337569832802, - 35.41872464120388, - 35.288020968437195, - 35.35628691315651, - 35.33848002552986, - 35.40989197790623, - 35.713041201233864, - 35.328615456819534, - 35.29147431254387, - 35.351455211639404, - 35.28880886733532, - 35.373954102396965, - 35.680223256349564, - 35.31438298523426, - 35.409245640039444, - 35.39498709142208, - 35.41380725800991, - 35.35588085651398, - 35.61149537563324, - 35.41295416653156 - ], - "final_loss": 4.466716766357422 - }, - "packed": { - "mean_step_ms": 30.4952384904027, - "p50_step_ms": 30.499404296278954, - "useful_tokens_per_second": 96539.66145982168, - "p50_useful_tokens_per_second": 96526.4754485444, - "peak_allocated_gib": 5.433377742767334, - "baseline_allocated_gib": 2.026548385620117, - "warmup_seconds_including_compile": 0.1535922773182392, - "step_ms": [ - 30.398793518543243, - 30.40284849703312, - 30.38313053548336, - 30.465159565210342, - 30.46681173145771, - 30.445296317338943, - 30.444307252764702, - 30.531708151102066, - 30.521946027874947, - 30.687423422932625, - 30.53887188434601, - 30.485061928629875, - 30.561374500393867, - 30.53414449095726, - 30.521035194396973, - 30.50977550446987, - 30.489033088088036, - 30.482830479741096, - 30.513176694512367, - 30.522041022777557 - ], - "final_loss": 4.467428207397461 - }, - "p50_speedup": 1.1601692365259266, - "mean_speedup": 1.161632341765806 - }, - { - "algorithm": "dflash2", - "lengths": [ - 1024, - 1024, - 1024, - 1024 - ], - "padding_fraction": 0.0, - "useful_tokens": 4096, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "loss_terms": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "loss_values": [ - 10.544198989868164, - 10.544198989868164 - ], - "identical_sampled_anchors": true, - "sampled_anchor_count": 512, - "gradients": { - "draft_model.layers.0.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.0.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.0.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.0.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.0.self_attn.q_norm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.0.self_attn.k_norm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.0.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.0.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.0.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.0.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.0.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.0.attention_conv.base_kernel": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.0.attention_conv.kernel_projection.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.0.mlp_conv.base_kernel": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.0.mlp_conv.kernel_projection.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.self_attn.q_norm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.self_attn.k_norm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.attention_conv.base_kernel": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.attention_conv.kernel_projection.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.mlp_conv.base_kernel": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.layers.1.mlp_conv.kernel_projection.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.norm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.fc.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.hidden_norm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.candidate_selector.predecessor_codebook": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.candidate_selector.successor_codebook": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.candidate_selector.hidden_projection.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - } - }, - "pass": true - }, - "padded": { - "mean_step_ms": 34.710309375077486, - "p50_step_ms": 34.62953958660364, - "useful_tokens_per_second": 118005.28643345911, - "p50_useful_tokens_per_second": 118280.52145355487, - "peak_allocated_gib": 4.944029808044434, - "baseline_allocated_gib": 2.10475492477417, - "warmup_seconds_including_compile": 0.17944882810115814, - "step_ms": [ - 35.65160930156708, - 35.805532708764076, - 35.28404049575329, - 34.837789833545685, - 34.69961881637573, - 34.849222749471664, - 34.734124317765236, - 34.90187227725983, - 34.501735121011734, - 34.55946035683155, - 34.328700974583626, - 34.35760922729969, - 34.44538451731205, - 34.72575172781944, - 34.35399569571018, - 34.364184364676476, - 34.35469605028629, - 34.3801137059927, - 34.369586035609245, - 34.701159223914146 - ], - "final_loss": 8.115375518798828 - }, - "packed": { - "mean_step_ms": 31.55035898089409, - "p50_step_ms": 31.34923055768013, - "useful_tokens_per_second": 129824.19637381652, - "p50_useful_tokens_per_second": 130657.11429387974, - "peak_allocated_gib": 4.976554870605469, - "baseline_allocated_gib": 2.105029582977295, - "warmup_seconds_including_compile": 0.1659725233912468, - "step_ms": [ - 32.492758706212044, - 32.25332498550415, - 32.10993483662605, - 31.620940193533897, - 31.54403530061245, - 31.51683136820793, - 31.574249267578125, - 31.407151371240616, - 31.366491690278053, - 31.331969425082207, - 31.20245411992073, - 31.18046373128891, - 31.29424713551998, - 31.234296038746834, - 31.18448704481125, - 31.143737956881523, - 31.179973855614662, - 31.152334064245224, - 32.96910971403122, - 31.248388811945915 - ], - "final_loss": 8.115375518798828 - }, - "p50_speedup": 1.1046376249295178, - "mean_speedup": 1.1001557667250939 - }, - { - "algorithm": "dflash2", - "lengths": [ - 128, - 256, - 512, - 2048 - ], - "padding_fraction": 0.640625, - "useful_tokens": 2944, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 2.09808349609375e-05, - "relative_l2_diff": 1.9918210395875337e-06 - }, - "loss_terms": { - "pass": true, - "max_abs_diff": 0.0625, - "relative_l2_diff": 1.897349817350434e-06 - }, - "loss_values": [ - 10.533493995666504, - 10.533514976501465 - ], - "identical_sampled_anchors": true, - "sampled_anchor_count": 478, - "gradients": { - "draft_model.layers.0.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0016685134832223076 - }, - "draft_model.layers.0.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.002552931532002346 - }, - "draft_model.layers.0.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0019390143575875442 - }, - "draft_model.layers.0.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.00134887764556812 - }, - "draft_model.layers.0.self_attn.q_norm.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.002203571142238923 - }, - "draft_model.layers.0.self_attn.k_norm.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0022542591514036923 - }, - "draft_model.layers.0.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.00190230944130773 - }, - "draft_model.layers.0.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.0018909972477091912 - }, - "draft_model.layers.0.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.001779518924017613 - }, - "draft_model.layers.0.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0014711605963159716 - }, - "draft_model.layers.0.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0019131030566731015 - }, - "draft_model.layers.0.attention_conv.base_kernel": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0014278181873818977 - }, - "draft_model.layers.0.attention_conv.kernel_projection.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.0015595749262699054 - }, - "draft_model.layers.0.mlp_conv.base_kernel": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.001959983152873123 - }, - "draft_model.layers.0.mlp_conv.kernel_projection.weight": { - "pass": true, - "max_abs_diff": 0.0001220703125, - "relative_l2_diff": 0.0021964532089766165 - }, - "draft_model.layers.1.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0016678913604391784 - }, - "draft_model.layers.1.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.002632562918780912 - }, - "draft_model.layers.1.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.001934458644005553 - }, - "draft_model.layers.1.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0013250733268872269 - }, - "draft_model.layers.1.self_attn.q_norm.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0022117848109090665 - }, - "draft_model.layers.1.self_attn.k_norm.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0024255432314647576 - }, - "draft_model.layers.1.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0019073348519206132 - }, - "draft_model.layers.1.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.0018211302868553522 - }, - "draft_model.layers.1.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.001875822437668495 - }, - "draft_model.layers.1.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0014387342277127767 - }, - "draft_model.layers.1.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0019749076950236724 - }, - "draft_model.layers.1.attention_conv.base_kernel": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0014293144980451072 - }, - "draft_model.layers.1.attention_conv.kernel_projection.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0014537681874349571 - }, - "draft_model.layers.1.mlp_conv.base_kernel": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0020733078021590995 - }, - "draft_model.layers.1.mlp_conv.kernel_projection.weight": { - "pass": true, - "max_abs_diff": 6.103515625e-05, - "relative_l2_diff": 0.002429591455188707 - }, - "draft_model.norm.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0014566753276068754 - }, - "draft_model.fc.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.003518625850596219 - }, - "draft_model.hidden_norm.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0033256846885824734 - }, - "draft_model.candidate_selector.predecessor_codebook": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.candidate_selector.successor_codebook": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.candidate_selector.hidden_projection.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - } - }, - "pass": true - }, - "padded": { - "mean_step_ms": 36.104287300258875, - "p50_step_ms": 35.84901336580515, - "useful_tokens_per_second": 81541.56251628573, - "p50_useful_tokens_per_second": 82122.20431171355, - "peak_allocated_gib": 5.079184532165527, - "baseline_allocated_gib": 2.13600492477417, - "warmup_seconds_including_compile": 0.18093057349324226, - "step_ms": [ - 36.35081835091114, - 36.468904465436935, - 36.12946905195713, - 36.11389920115471, - 36.354729905724525, - 35.956135019659996, - 35.82063689827919, - 35.81521473824978, - 35.775020718574524, - 35.78690066933632, - 36.06811165809631, - 38.790395483374596, - 35.74402630329132, - 35.74172966182232, - 35.803671926259995, - 35.80854274332523, - 36.143653094768524, - 35.87738983333111, - 35.764576867222786, - 35.771919414401054 - ], - "final_loss": 3.7936742305755615 - }, - "packed": { - "mean_step_ms": 30.86625747382641, - "p50_step_ms": 30.81074357032776, - "useful_tokens_per_second": 95379.23418465673, - "p50_useful_tokens_per_second": 95551.08572047624, - "peak_allocated_gib": 4.800002574920654, - "baseline_allocated_gib": 2.0974764823913574, - "warmup_seconds_including_compile": 0.16166752576828003, - "step_ms": [ - 31.302351504564285, - 31.165508553385735, - 31.053470447659492, - 30.985631048679352, - 30.950207263231277, - 30.96415288746357, - 30.836161226034164, - 30.886171385645866, - 30.823593959212303, - 30.811427161097527, - 30.788477510213852, - 30.81005997955799, - 30.732905492186546, - 30.733147636055946, - 30.7193323969841, - 30.75392358005047, - 30.746370553970337, - 30.771011486649513, - 30.745219439268112, - 30.74602596461773 - ], - "final_loss": 3.716245174407959 - }, - "p50_speedup": 1.1635231484750497, - "mean_speedup": 1.1697008401771465 - } - ] -} diff --git a/docs/sections/benchmarks/sequence-packing-results/dflash-family-512anchors-bf16.json b/docs/sections/benchmarks/sequence-packing-results/dflash-family-512anchors-bf16.json deleted file mode 100644 index 151037d09..000000000 --- a/docs/sections/benchmarks/sequence-packing-results/dflash-family-512anchors-bf16.json +++ /dev/null @@ -1,563 +0,0 @@ -{ - "timestamp_utc": "2026-10-02T04:44:09.597425+00:00", - "source": { - "files_sha256": { - "scripts/benchmark_dflash_sequence_packing.py": "0372ddf408fddc99c95138ab448025e37600c51004bceb6fd51d0432afb82e46", - "specforge/benchmarks/benchmark_dflash_sequence_packing.py": "00b9ff628d145d3f986eeb8c43f62671183a197f4c4575faf3f432f16ef0641c", - "specforge/algorithms/common/dflash_family_model.py": "86be39addbc5c50036f37c5d5282c53aaa8f35ab5ae1f8cfabf33078494e1421", - "specforge/algorithms/common/hidden_states_data.py": "fa09c5e5037c31b1165c2f8d1774e2cd4a40fde8b08f1f6deddf87f4f4af7f24", - "specforge/modeling/draft/dflash.py": "97af112a6ecf66d4b397aae77f0f97c08a49759c6305a4c70577a4f78b45ca90", - "specforge/modeling/draft/dflash2.py": "819c6b3d8d6d8ffc31a843ed40d539ddf635f58b148a01c8c43d0bff1dd02b09", - "specforge/modeling/packed_dflash.py": "d763e0ad3c609602323b9e3413968860cc680a8be8c986cc7612a956556ab96e", - "specforge/training/strategies/base.py": "276932ad331e95d5b91eddc67cf3acb0b5447ee5973f9bbc757e54093a82939f" - }, - "head": "unavailable (copied snapshot is identified by file hashes)" - }, - "settings": { - "preset": "medium", - "algorithm": "both", - "hidden_size": 2048, - "intermediate_size": 8192, - "layers": 2, - "heads": 16, - "kv_heads": 4, - "vocab_size": 32000, - "block_size": 16, - "anchors": 512, - "capture_layers": 2, - "conv_group_size": 32, - "conv_kernel_size": 4, - "selector_rank": 16, - "selector_top_k": 16, - "lengths": [ - [ - 128, - 256, - 512, - 2048 - ] - ], - "dtype": "bfloat16", - "sliding_window": null, - "warmup": 5, - "steps": 20, - "seed": 1729, - "learning_rate": 0.0001, - "objective_chunk_blocks": 128, - "correctness_only": false, - "skip_correctness": false, - "atol": 0.002, - "rtol": 0.02, - "output": "artifacts/sequence-packing/dflash-family-medium-512anchors-bf16-optimized.json" - }, - "environment": { - "torch": "2.13.0+cu130", - "cuda": "13.0", - "gpu": "NVIDIA H200", - "cuda_visible_devices": "0" - }, - "scope": "Single GPU; synthetic frozen target components and captured features; GPU-resident hidden features plus CPU integer features; production strategy forward/backward/BF16Optimizer; excludes capture, hidden-feature I/O/H2D, distributed training and serving.", - "cases": [ - { - "algorithm": "dflash", - "lengths": [ - 128, - 256, - 512, - 2048 - ], - "padding_fraction": 0.640625, - "useful_tokens": 2944, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 5.91278076171875e-05, - "relative_l2_diff": 5.61251249915586e-06 - }, - "loss_terms": { - "pass": true, - "max_abs_diff": 0.453125, - "relative_l2_diff": 5.5553522850025985e-06 - }, - "loss_values": [ - 10.534997940063477, - 10.535057067871094 - ], - "identical_sampled_anchors": true, - "sampled_anchor_count": 1178, - "gradients": { - "draft_model.layers.0.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0021281991404804583 - }, - "draft_model.layers.0.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0030810157665098494 - }, - "draft_model.layers.0.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0026500859492456135 - }, - "draft_model.layers.0.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0017068411541985204 - }, - "draft_model.layers.0.self_attn.q_norm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.0019287543064844075 - }, - "draft_model.layers.0.self_attn.k_norm.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0022529095517634795 - }, - "draft_model.layers.0.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.0021051663862464735 - }, - "draft_model.layers.0.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0020911271359065784 - }, - "draft_model.layers.0.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0018525617885881578 - }, - "draft_model.layers.0.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.002017398682331893 - }, - "draft_model.layers.0.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0024711602964318856 - }, - "draft_model.layers.1.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0017866582661795515 - }, - "draft_model.layers.1.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0028495004711782653 - }, - "draft_model.layers.1.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.002261717485471004 - }, - "draft_model.layers.1.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.001464825932831344 - }, - "draft_model.layers.1.self_attn.q_norm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.0017715359662342583 - }, - "draft_model.layers.1.self_attn.k_norm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.003161722116251433 - }, - "draft_model.layers.1.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0017373591305729018 - }, - "draft_model.layers.1.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0016626714244595531 - }, - "draft_model.layers.1.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0016341453129451533 - }, - "draft_model.layers.1.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.0018407111772094009 - }, - "draft_model.layers.1.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.002002886579895657 - }, - "draft_model.norm.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0013523338859282566 - }, - "draft_model.fc.weight": { - "pass": true, - "max_abs_diff": 1.0728836059570312e-06, - "relative_l2_diff": 0.004442411498462378 - }, - "draft_model.hidden_norm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.004556106678282558 - } - }, - "pass": true - }, - "padded": { - "mean_step_ms": 102.73919776082039, - "p50_step_ms": 102.67521720379591, - "useful_tokens_per_second": 28655.080671874734, - "p50_useful_tokens_per_second": 28672.936665491274, - "peak_allocated_gib": 11.8163743019104, - "baseline_allocated_gib": 2.180513858795166, - "warmup_seconds_including_compile": 0.5427277218550444, - "step_ms": [ - 103.04645448923111, - 102.31764800846577, - 102.57496125996113, - 102.48782113194466, - 102.40579582750797, - 102.40877233445644, - 102.65973210334778, - 102.4135909974575, - 102.76232659816742, - 103.21440547704697, - 102.89289988577366, - 102.41799987852573, - 102.82001830637455, - 102.51341387629509, - 102.5018971413374, - 102.69070230424404, - 103.15846651792526, - 103.27554307878017, - 103.38030196726322, - 102.8412040323019 - ], - "final_loss": 5.586607933044434 - }, - "packed": { - "mean_step_ms": 73.74237161129713, - "p50_step_ms": 73.68302810937166, - "useful_tokens_per_second": 39922.773510975436, - "p50_useful_tokens_per_second": 39954.9268744773, - "peak_allocated_gib": 8.335425853729248, - "baseline_allocated_gib": 2.026548385620117, - "warmup_seconds_including_compile": 0.3671108912676573, - "step_ms": [ - 73.44117760658264, - 74.23617132008076, - 73.869489133358, - 73.47449846565723, - 73.62687028944492, - 73.83274286985397, - 73.57754744589329, - 73.6483745276928, - 73.83186556398869, - 73.71129095554352, - 73.57209734618664, - 73.83942790329456, - 73.66623729467392, - 73.60509596765041, - 74.06741566956043, - 73.6998189240694, - 73.65257479250431, - 73.98152723908424, - 73.88443686068058, - 73.62877205014229 - ], - "final_loss": 5.585000514984131 - }, - "p50_speedup": 1.3934717374995718, - "mean_speedup": 1.3932179765300772 - }, - { - "algorithm": "dflash2", - "lengths": [ - 128, - 256, - 512, - 2048 - ], - "padding_fraction": 0.640625, - "useful_tokens": 2944, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 6.67572021484375e-06, - "relative_l2_diff": 6.336308441356832e-07 - }, - "loss_terms": { - "pass": true, - "max_abs_diff": 0.0546875, - "relative_l2_diff": 6.704316832987997e-07 - }, - "loss_values": [ - 10.535661697387695, - 10.53565502166748 - ], - "identical_sampled_anchors": true, - "sampled_anchor_count": 1178, - "gradients": { - "draft_model.layers.0.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.002155811162419652 - }, - "draft_model.layers.0.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.003115263876083677 - }, - "draft_model.layers.0.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0024893414012672446 - }, - "draft_model.layers.0.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0016564023704723882 - }, - "draft_model.layers.0.self_attn.q_norm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.0018726661822179513 - }, - "draft_model.layers.0.self_attn.k_norm.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0029641507160095156 - }, - "draft_model.layers.0.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.0020104000601537755 - }, - "draft_model.layers.0.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.001965311158535 - }, - "draft_model.layers.0.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 3.0517578125e-05, - "relative_l2_diff": 0.0017894700855622925 - }, - "draft_model.layers.0.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.001847257034798985 - }, - "draft_model.layers.0.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.002091376779893997 - }, - "draft_model.layers.0.attention_conv.base_kernel": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0018613340828882604 - }, - "draft_model.layers.0.attention_conv.kernel_projection.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0019815838381331673 - }, - "draft_model.layers.0.mlp_conv.base_kernel": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.002203816458103695 - }, - "draft_model.layers.0.mlp_conv.kernel_projection.weight": { - "pass": true, - "max_abs_diff": 6.103515625e-05, - "relative_l2_diff": 0.0021897462902934943 - }, - "draft_model.layers.1.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.001935534452710976 - }, - "draft_model.layers.1.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0029778696938854822 - }, - "draft_model.layers.1.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.002231563848236419 - }, - "draft_model.layers.1.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.0013835182743699866 - }, - "draft_model.layers.1.self_attn.q_norm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.002283248567519543 - }, - "draft_model.layers.1.self_attn.k_norm.weight": { - "pass": true, - "max_abs_diff": 1.9073486328125e-06, - "relative_l2_diff": 0.002612348363653426 - }, - "draft_model.layers.1.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.001615087994551397 - }, - "draft_model.layers.1.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0015551054945898413 - }, - "draft_model.layers.1.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 1.52587890625e-05, - "relative_l2_diff": 0.0015438830204557216 - }, - "draft_model.layers.1.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.0019442038075880707 - }, - "draft_model.layers.1.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0020446608565408367 - }, - "draft_model.layers.1.attention_conv.base_kernel": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.0017227739116648922 - }, - "draft_model.layers.1.attention_conv.kernel_projection.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.001838657839446509 - }, - "draft_model.layers.1.mlp_conv.base_kernel": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.002047133300927381 - }, - "draft_model.layers.1.mlp_conv.kernel_projection.weight": { - "pass": true, - "max_abs_diff": 6.103515625e-05, - "relative_l2_diff": 0.0023054439329669245 - }, - "draft_model.norm.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.001168930954718689 - }, - "draft_model.fc.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.004321447013206923 - }, - "draft_model.hidden_norm.weight": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 0.004421665559420483 - }, - "draft_model.candidate_selector.predecessor_codebook": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.candidate_selector.successor_codebook": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "draft_model.candidate_selector.hidden_projection.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - } - }, - "pass": true - }, - "padded": { - "mean_step_ms": 105.65225137397647, - "p50_step_ms": 105.6781467050314, - "useful_tokens_per_second": 27865.00014636835, - "p50_useful_tokens_per_second": 27858.172117810565, - "peak_allocated_gib": 13.28050422668457, - "baseline_allocated_gib": 2.1340060234069824, - "warmup_seconds_including_compile": 0.540035929530859, - "step_ms": [ - 106.40191659331322, - 106.54045268893242, - 106.38073086738586, - 106.02187551558018, - 105.94053752720356, - 105.82047514617443, - 105.60824908316135, - 106.06533102691174, - 106.33190535008907, - 105.74804432690144, - 105.80088756978512, - 105.59522919356823, - 105.08263297379017, - 105.43076135218143, - 104.99388724565506, - 105.23213259875774, - 104.83885742723942, - 105.21486029028893, - 104.66686449944973, - 105.32939620316029 - ], - "final_loss": 4.916550636291504 - }, - "packed": { - "mean_step_ms": 71.99693070724607, - "p50_step_ms": 72.01073691248894, - "useful_tokens_per_second": 40890.63201834108, - "p50_useful_tokens_per_second": 40882.79229217855, - "peak_allocated_gib": 9.44128131866455, - "baseline_allocated_gib": 2.0951266288757324, - "warmup_seconds_including_compile": 0.36581393890082836, - "step_ms": [ - 71.98422029614449, - 72.08905182778835, - 72.05628231167793, - 72.11560942232609, - 72.03729264438152, - 72.03725352883339, - 72.13184051215649, - 72.09071330726147, - 71.98000326752663, - 72.12523184716702, - 72.10700213909149, - 72.47685827314854, - 71.98372855782509, - 71.88108563423157, - 71.81445695459843, - 71.6748759150505, - 71.81508652865887, - 71.88605889678001, - 71.80662639439106, - 71.84533588588238 - ], - "final_loss": 4.9150261878967285 - }, - "p50_speedup": 1.467533193466091, - "mean_speedup": 1.467454936427494 - } - ] -} diff --git a/docs/sections/benchmarks/sequence-packing-results/dflash-family-tiny-fp32.json b/docs/sections/benchmarks/sequence-packing-results/dflash-family-tiny-fp32.json deleted file mode 100644 index 32ce6ae4a..000000000 --- a/docs/sections/benchmarks/sequence-packing-results/dflash-family-tiny-fp32.json +++ /dev/null @@ -1,214 +0,0 @@ -{ - "timestamp_utc": "2026-10-02T04:43:26.200384+00:00", - "source": { - "files_sha256": { - "scripts/benchmark_dflash_sequence_packing.py": "0372ddf408fddc99c95138ab448025e37600c51004bceb6fd51d0432afb82e46", - "specforge/benchmarks/benchmark_dflash_sequence_packing.py": "00b9ff628d145d3f986eeb8c43f62671183a197f4c4575faf3f432f16ef0641c", - "specforge/algorithms/common/dflash_family_model.py": "86be39addbc5c50036f37c5d5282c53aaa8f35ab5ae1f8cfabf33078494e1421", - "specforge/algorithms/common/hidden_states_data.py": "fa09c5e5037c31b1165c2f8d1774e2cd4a40fde8b08f1f6deddf87f4f4af7f24", - "specforge/modeling/draft/dflash.py": "97af112a6ecf66d4b397aae77f0f97c08a49759c6305a4c70577a4f78b45ca90", - "specforge/modeling/draft/dflash2.py": "819c6b3d8d6d8ffc31a843ed40d539ddf635f58b148a01c8c43d0bff1dd02b09", - "specforge/modeling/packed_dflash.py": "d763e0ad3c609602323b9e3413968860cc680a8be8c986cc7612a956556ab96e", - "specforge/training/strategies/base.py": "276932ad331e95d5b91eddc67cf3acb0b5447ee5973f9bbc757e54093a82939f" - }, - "head": "unavailable (copied snapshot is identified by file hashes)" - }, - "settings": { - "preset": "tiny", - "algorithm": "both", - "hidden_size": 64, - "intermediate_size": 128, - "layers": 2, - "heads": 4, - "kv_heads": 2, - "vocab_size": 128, - "block_size": 4, - "anchors": 8, - "capture_layers": 2, - "conv_group_size": 4, - "conv_kernel_size": 2, - "selector_rank": 4, - "selector_top_k": 8, - "lengths": [ - [ - 9, - 17, - 31, - 64 - ], - [ - 32, - 32, - 32, - 32 - ] - ], - "dtype": "float32", - "sliding_window": null, - "warmup": 5, - "steps": 20, - "seed": 1729, - "learning_rate": 0.0001, - "objective_chunk_blocks": 128, - "correctness_only": true, - "skip_correctness": false, - "atol": 2e-05, - "rtol": 0.0002, - "output": "artifacts/sequence-packing/dflash-family-tiny-fp32-optimized.json" - }, - "environment": { - "torch": "2.13.0+cu130", - "cuda": "13.0", - "gpu": "NVIDIA H200", - "cuda_visible_devices": "0" - }, - "scope": "Single GPU; synthetic frozen target components and captured features; GPU-resident hidden features plus CPU integer features; production strategy forward/backward/BF16Optimizer; excludes capture, hidden-feature I/O/H2D, distributed training and serving.", - "cases": [ - { - "algorithm": "dflash", - "lengths": [ - 9, - 17, - 31, - 64 - ], - "padding_fraction": 0.52734375, - "useful_tokens": 121, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "loss_terms": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "loss_values": [ - 4.920759201049805, - 4.920759201049805 - ], - "identical_sampled_anchors": true, - "sampled_anchor_count": 29, - "pass": true, - "gradient_summary": { - "parameter_tensors": 25, - "all_pass": true, - "max_abs_diff": 3.725290298461914e-09, - "max_relative_l2_diff": 2.936993432616088e-07 - } - } - }, - { - "algorithm": "dflash", - "lengths": [ - 32, - 32, - 32, - 32 - ], - "padding_fraction": 0.0, - "useful_tokens": 128, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "loss_terms": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "loss_values": [ - 4.951769828796387, - 4.951769828796387 - ], - "identical_sampled_anchors": true, - "sampled_anchor_count": 32, - "pass": true, - "gradient_summary": { - "parameter_tensors": 25, - "all_pass": true, - "max_abs_diff": 1.862645149230957e-09, - "max_relative_l2_diff": 2.8535136873839174e-07 - } - } - }, - { - "algorithm": "dflash2", - "lengths": [ - 9, - 17, - 31, - 64 - ], - "padding_fraction": 0.52734375, - "useful_tokens": 121, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "loss_terms": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "loss_values": [ - 5.123039722442627, - 5.123039722442627 - ], - "identical_sampled_anchors": true, - "sampled_anchor_count": 29, - "pass": true, - "gradient_summary": { - "parameter_tensors": 36, - "all_pass": true, - "max_abs_diff": 3.725290298461914e-09, - "max_relative_l2_diff": 2.952709571251088e-07 - } - } - }, - { - "algorithm": "dflash2", - "lengths": [ - 32, - 32, - 32, - 32 - ], - "padding_fraction": 0.0, - "useful_tokens": 128, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "loss_terms": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0 - }, - "loss_values": [ - 5.035518169403076, - 5.035518169403076 - ], - "identical_sampled_anchors": true, - "sampled_anchor_count": 32, - "pass": true, - "gradient_summary": { - "parameter_tensors": 36, - "all_pass": true, - "max_abs_diff": 9.313225746154785e-10, - "max_relative_l2_diff": 2.501790571376381e-07 - } - } - } - ], - "evidence_kind": "Derived summary: per-parameter gradient entries reduced to count, pass flag and maxima; all other fields preserved.", - "source_report_sha256": "18a6e3096731c0fbf87e7965ab5473770c6713b0a763821a299cca69b82a9e8f" -} diff --git a/docs/sections/benchmarks/sequence-packing-results/eagle3-large-bf16.json b/docs/sections/benchmarks/sequence-packing-results/eagle3-large-bf16.json deleted file mode 100644 index 3140021bd..000000000 --- a/docs/sections/benchmarks/sequence-packing-results/eagle3-large-bf16.json +++ /dev/null @@ -1,235 +0,0 @@ -{ - "timestamp_utc": "2026-10-02T04:10:49.449132+00:00", - "source": { - "files_sha256": { - "scripts/benchmark_sequence_packing.py": "0d11416603f81d5b97726abd299244029aa65b12beccb6ff38f9fb82c89dfe47", - "specforge/benchmarks/benchmark_sequence_packing.py": "39bf66ed00bee5cd272b6c8f6a562eae54537d98ea7cb41d066ad271eb816d65", - "specforge/algorithms/eagle3/data.py": "84d03344c68ba2e3f3a452db12c8b9cef95c931445ab8645b0444ec2580a8fe2", - "specforge/algorithms/eagle3/model.py": "6988a06491728c04929ee3d1471c91adcc44bc173de8faec7543810420d401d0", - "specforge/modeling/draft/llama3_eagle.py": "1488e38e36f1ff3bb11c0355da776122ef9c779105d38dee404b8e109aa06f4a", - "specforge/modeling/packed_sequence.py": "fd5207429606166aa4c8f124e5d4a68cd0eacbb1514d879921a5c53d15a938b4", - "specforge/training/strategies/base.py": "1f3eb9916fb399d9d5dc89214a7fad687db768bd74a27a18cba940c1216e1c7e" - }, - "head": "unavailable (file hashes identify copied snapshot)" - }, - "environment": { - "python": "3.12.3", - "torch": "2.13.0+cu130", - "cuda": "13.0", - "gpu": "NVIDIA H200", - "cuda_visible_devices": "0", - "transformers": "5.12.1" - }, - "settings": { - "preset": "large", - "hidden_size": 4096, - "intermediate_size": 14336, - "num_heads": 32, - "num_kv_heads": 8, - "vocab_size": 32000, - "draft_vocab_size": 32000, - "target_hidden_size": 4096, - "lengths": [ - [ - 128, - 256, - 512, - 2048 - ] - ], - "ttt_length": 7, - "dtype": "bfloat16", - "warmup": 5, - "steps": 20, - "seed": 1729, - "learning_rate": 0.0001, - "prompt_fraction": 0.25, - "correctness_only": false, - "skip_correctness": false, - "atol": 0.002, - "rtol": 0.02, - "output": "artifacts/sequence-packing/large-bf16-final.json" - }, - "scope": "single-GPU synthetic offline features; resident inputs; production forward/backward/BF16Optimizer; excludes capture, I/O, transfer, distributed communication, and serving", - "cases": [ - { - "lengths": [ - 128, - 256, - 512, - 2048 - ], - "useful_tokens": 2944, - "padded_tokens": 8192, - "padding_fraction": 0.640625, - "raw_supervised_tokens": 2204, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 8.015076037423352e-08, - "reference_l2": 11.898506164550781 - }, - "plosses": { - "pass": true, - "max_abs_diff": 4.76837158203125e-07, - "relative_l2_diff": 1.0790098863577887e-07, - "reference_l2": 7.96684455871582 - }, - "loss_padded": 11.898506164550781, - "loss_packed": 11.898505210876465, - "gradients": { - "draft_model.midlayer.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 1.7881393432617188e-07, - "relative_l2_diff": 0.008098428375980581, - "reference_l2": 0.018883956596255302 - }, - "draft_model.midlayer.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 3.5762786865234375e-07, - "relative_l2_diff": 0.008347887328420777, - "reference_l2": 0.019043944776058197 - }, - "draft_model.midlayer.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 1.7881393432617188e-07, - "relative_l2_diff": 0.007748204675858391, - "reference_l2": 0.009764185175299644 - }, - "draft_model.midlayer.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 1.1920928955078125e-07, - "relative_l2_diff": 0.0071544088294053345, - "reference_l2": 0.009779798798263073 - }, - "draft_model.midlayer.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 1.1920928955078125e-07, - "relative_l2_diff": 0.006828965463057812, - "reference_l2": 0.0157613605260849 - }, - "draft_model.midlayer.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 1.1920928955078125e-07, - "relative_l2_diff": 0.00662542103430625, - "reference_l2": 0.015876401215791702 - }, - "draft_model.midlayer.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 1.1920928955078125e-07, - "relative_l2_diff": 0.006333277633854752, - "reference_l2": 0.016165578737854958 - }, - "draft_model.midlayer.hidden_norm.weight": { - "pass": true, - "max_abs_diff": 2.384185791015625e-07, - "relative_l2_diff": 0.009126329141738658, - "reference_l2": 0.00040416946285404265 - }, - "draft_model.midlayer.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 1.7881393432617188e-07, - "relative_l2_diff": 0.008737133519347783, - "reference_l2": 0.00040028218063525856 - }, - "draft_model.midlayer.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 2.384185791015625e-07, - "relative_l2_diff": 0.00702031283790539, - "reference_l2": 0.00044801118201576173 - }, - "draft_model.fc.weight": { - "pass": true, - "max_abs_diff": 1.862645149230957e-07, - "relative_l2_diff": 0.008248254998890576, - "reference_l2": 0.027148302644491196 - }, - "draft_model.norm.weight": { - "pass": true, - "max_abs_diff": 3.814697265625e-06, - "relative_l2_diff": 0.0007218484337602093, - "reference_l2": 0.02720426581799984 - }, - "draft_model.lm_head.weight": { - "pass": true, - "max_abs_diff": 1.7881393432617188e-07, - "relative_l2_diff": 0.004232173488441028, - "reference_l2": 0.01552735734730959 - } - }, - "pass": true - }, - "padded": { - "mean_step_ms": 271.955263055861, - "p50_step_ms": 249.50438179075718, - "stdev_step_ms": 58.620318163146024, - "useful_tokens_per_second": 10825.309894426597, - "useful_ttt_positions_per_second": 75777.16926098618, - "peak_allocated_gib": 33.02751874923706, - "baseline_allocated_gib": 6.385047912597656, - "peak_increment_gib": 26.642470836639404, - "warmup_seconds_including_compile": 1.318971425294876, - "step_ms": [ - 248.97141940891743, - 248.47774393856525, - 248.4574057161808, - 249.15488995611668, - 248.9372342824936, - 249.21941943466663, - 258.02627205848694, - 511.20829954743385, - 248.72448295354843, - 249.78934414684772, - 247.9896154254675, - 247.11078964173794, - 247.17947468161583, - 253.61876748502254, - 264.03690315783024, - 293.2693623006344, - 289.70682993531227, - 261.5733686834574, - 293.9461972564459, - 279.70744110643864 - ], - "final_loss": 11.393714904785156 - }, - "packed": { - "mean_step_ms": 115.71002416312695, - "p50_step_ms": 113.14941477030516, - "stdev_step_ms": 7.805935035036344, - "useful_tokens_per_second": 25442.912325811765, - "useful_ttt_positions_per_second": 178100.38628068237, - "peak_allocated_gib": 12.972045421600342, - "baseline_allocated_gib": 6.1793012619018555, - "peak_increment_gib": 6.792744159698486, - "warmup_seconds_including_compile": 0.5725000947713852, - "step_ms": [ - 112.05416917800903, - 112.56376467645168, - 114.90379646420479, - 113.56857605278492, - 112.49293573200703, - 112.36764304339886, - 113.49863186478615, - 112.70488612353802, - 112.50686645507812, - 146.31127193570137, - 123.74899163842201, - 121.32737971842289, - 112.57140710949898, - 112.80683055520058, - 113.80611918866634, - 112.85982467234135, - 112.63692378997803, - 113.43900486826897, - 113.50067704916, - 114.5307831466198 - ], - "final_loss": 11.393773078918457 - }, - "speedup": 2.3503172263836096, - "peak_memory_reduction_fraction": 0.6072352416149942 - } - ] -} diff --git a/docs/sections/benchmarks/sequence-packing-results/eagle3-medium-bf16.json b/docs/sections/benchmarks/sequence-packing-results/eagle3-medium-bf16.json deleted file mode 100644 index e18d53fdf..000000000 --- a/docs/sections/benchmarks/sequence-packing-results/eagle3-medium-bf16.json +++ /dev/null @@ -1,605 +0,0 @@ -{ - "timestamp_utc": "2026-10-02T04:13:13.890368+00:00", - "source": { - "files_sha256": { - "scripts/benchmark_sequence_packing.py": "0d11416603f81d5b97726abd299244029aa65b12beccb6ff38f9fb82c89dfe47", - "specforge/benchmarks/benchmark_sequence_packing.py": "39bf66ed00bee5cd272b6c8f6a562eae54537d98ea7cb41d066ad271eb816d65", - "specforge/algorithms/eagle3/data.py": "84d03344c68ba2e3f3a452db12c8b9cef95c931445ab8645b0444ec2580a8fe2", - "specforge/algorithms/eagle3/model.py": "aadd395819e3292ab3ccde0f0200fdd6df90f0de1de8a03fb91d6640394ca022", - "specforge/modeling/draft/llama3_eagle.py": "1488e38e36f1ff3bb11c0355da776122ef9c779105d38dee404b8e109aa06f4a", - "specforge/modeling/packed_sequence.py": "ad057f15c08fb1c5cc80994ac799106536ec0a41c28f88f33d5f1bce34c75670", - "specforge/training/strategies/base.py": "a6e3aa2961d9d55f96c816fa2f6ce764672699e02005d89688eddec357120894" - }, - "head": "unavailable (file hashes identify copied snapshot)" - }, - "environment": { - "python": "3.12.3", - "torch": "2.13.0+cu130", - "cuda": "13.0", - "gpu": "NVIDIA H200", - "cuda_visible_devices": "0", - "transformers": "5.12.1" - }, - "settings": { - "preset": "medium", - "hidden_size": 2048, - "intermediate_size": 8192, - "num_heads": 16, - "num_kv_heads": 4, - "vocab_size": 32000, - "draft_vocab_size": 32000, - "target_hidden_size": 2048, - "lengths": [ - [ - 1024, - 1024, - 1024, - 1024 - ], - [ - 512, - 768, - 1024, - 2048 - ], - [ - 128, - 256, - 512, - 2048 - ] - ], - "ttt_length": 7, - "dtype": "bfloat16", - "warmup": 5, - "steps": 20, - "seed": 1729, - "learning_rate": 0.0001, - "prompt_fraction": 0.25, - "correctness_only": false, - "skip_correctness": false, - "atol": 0.002, - "rtol": 0.02, - "output": "artifacts/sequence-packing/medium-bf16-current.json" - }, - "scope": "single-GPU synthetic offline features; resident inputs; production forward/backward/BF16Optimizer; excludes capture, I/O, transfer, distributed communication, and serving", - "cases": [ - { - "lengths": [ - 1024, - 1024, - 1024, - 1024 - ], - "useful_tokens": 4096, - "padded_tokens": 4096, - "padding_fraction": 0.0, - "raw_supervised_tokens": 3068, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 31.914281845092773 - }, - "plosses": { - "pass": true, - "max_abs_diff": 9.5367431640625e-07, - "relative_l2_diff": 6.311522082349717e-08, - "reference_l2": 21.36884117126465 - }, - "loss_padded": 31.914281845092773, - "loss_packed": 31.914281845092773, - "gradients": { - "draft_model.midlayer.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 5.960464477539063e-08, - "relative_l2_diff": 0.003765711493526104, - "reference_l2": 0.0036451490595936775 - }, - "draft_model.midlayer.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 1.1920928955078125e-07, - "relative_l2_diff": 0.004440347213036328, - "reference_l2": 0.003682289272546768 - }, - "draft_model.midlayer.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 8.940696716308594e-08, - "relative_l2_diff": 0.00396974587705717, - "reference_l2": 0.002882526256144047 - }, - "draft_model.midlayer.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 5.960464477539063e-08, - "relative_l2_diff": 0.0033154438597801433, - "reference_l2": 0.0029025187250226736 - }, - "draft_model.midlayer.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 1.1920928955078125e-07, - "relative_l2_diff": 0.00329089278355116, - "reference_l2": 0.012741141952574253 - }, - "draft_model.midlayer.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 1.1920928955078125e-07, - "relative_l2_diff": 0.0031612475656136022, - "reference_l2": 0.01308818906545639 - }, - "draft_model.midlayer.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 1.1920928955078125e-07, - "relative_l2_diff": 0.0030215318893036898, - "reference_l2": 0.013826857320964336 - }, - "draft_model.midlayer.hidden_norm.weight": { - "pass": true, - "max_abs_diff": 4.470348358154297e-08, - "relative_l2_diff": 0.004985117251834739, - "reference_l2": 8.83292086655274e-05 - }, - "draft_model.midlayer.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 5.960464477539063e-08, - "relative_l2_diff": 0.005409839549368207, - "reference_l2": 7.969953730935231e-05 - }, - "draft_model.midlayer.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 2.384185791015625e-07, - "relative_l2_diff": 0.0033067118185063616, - "reference_l2": 0.00036759089562110603 - }, - "draft_model.fc.weight": { - "pass": true, - "max_abs_diff": 1.1920928955078125e-07, - "relative_l2_diff": 0.003179768823534524, - "reference_l2": 0.019182570278644562 - }, - "draft_model.norm.weight": { - "pass": true, - "max_abs_diff": 7.62939453125e-06, - "relative_l2_diff": 0.0004027977723562092, - "reference_l2": 0.05357325077056885 - }, - "draft_model.lm_head.weight": { - "pass": true, - "max_abs_diff": 1.1920928955078125e-07, - "relative_l2_diff": 0.0021179105111969712, - "reference_l2": 0.021415317431092262 - } - }, - "pass": true - }, - "padded": { - "mean_step_ms": 87.81536612659693, - "p50_step_ms": 80.03105316311121, - "stdev_step_ms": 26.06948085014276, - "useful_tokens_per_second": 46643.31745875886, - "useful_ttt_positions_per_second": 326503.22221131204, - "peak_allocated_gib": 13.090556621551514, - "baseline_allocated_gib": 2.298463821411133, - "peak_increment_gib": 10.79209280014038, - "warmup_seconds_including_compile": 0.48302813060581684, - "step_ms": [ - 79.99945804476738, - 82.36873708665371, - 81.48440718650818, - 79.86084371805191, - 80.0857711583376, - 80.15882037580013, - 80.10819368064404, - 80.06264828145504, - 79.99635487794876, - 79.9578819423914, - 79.92365770041943, - 80.3301278501749, - 80.08425869047642, - 191.34998694062233, - 121.55988439917564, - 79.96164448559284, - 79.73956502974033, - 79.78175953030586, - 79.893684014678, - 79.59963753819466 - ], - "final_loss": 30.858375549316406 - }, - "packed": { - "mean_step_ms": 76.67620368301868, - "p50_step_ms": 76.61013770848513, - "stdev_step_ms": 0.3185837712334884, - "useful_tokens_per_second": 53419.44179882673, - "useful_ttt_positions_per_second": 373936.0925917871, - "peak_allocated_gib": 9.165019989013672, - "baseline_allocated_gib": 2.268096923828125, - "peak_increment_gib": 6.896923065185547, - "warmup_seconds_including_compile": 0.4739191196858883, - "step_ms": [ - 76.65209099650383, - 76.59146375954151, - 76.6055267304182, - 76.77727565169334, - 76.62097364664078, - 76.59219950437546, - 76.76399871706963, - 76.5523910522461, - 76.61474868655205, - 76.77584514021873, - 76.5397660434246, - 76.64117217063904, - 76.64411887526512, - 76.56820304691792, - 76.97880268096924, - 76.55680924654007, - 76.49008370935917, - 77.8732467442751, - 76.46557316184044, - 76.21978409588337 - ], - "final_loss": 30.858409881591797 - }, - "speedup": 1.1452753515240246, - "peak_memory_reduction_fraction": 0.2998754557216514 - }, - { - "lengths": [ - 512, - 768, - 1024, - 2048 - ], - "useful_tokens": 4352, - "padded_tokens": 8192, - "padding_fraction": 0.46875, - "raw_supervised_tokens": 3260, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 16.955839157104492 - }, - "plosses": { - "pass": true, - "max_abs_diff": 4.76837158203125e-07, - "relative_l2_diff": 9.391610213704716e-08, - "reference_l2": 11.35311508178711 - }, - "loss_padded": 16.955839157104492, - "loss_packed": 16.955839157104492, - "gradients": { - "draft_model.midlayer.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 0.001796242082491517 - }, - "draft_model.midlayer.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 0.001820462173782289 - }, - "draft_model.midlayer.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 0.0014368355041369796 - }, - "draft_model.midlayer.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 0.0014455055352300406 - }, - "draft_model.midlayer.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 0.006567297503352165 - }, - "draft_model.midlayer.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 0.006763069424778223 - }, - "draft_model.midlayer.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 0.007175843231379986 - }, - "draft_model.midlayer.hidden_norm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 4.3062595068477094e-05 - }, - "draft_model.midlayer.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 4.0413109672954306e-05 - }, - "draft_model.midlayer.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 0.00018638117762748152 - }, - "draft_model.fc.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 0.009890600107610226 - }, - "draft_model.norm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 0.02846393920481205 - }, - "draft_model.lm_head.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 0.011251452378928661 - } - }, - "pass": true - }, - "padded": { - "mean_step_ms": 147.45035851374269, - "p50_step_ms": 135.93942299485207, - "stdev_step_ms": 42.45054906683103, - "useful_tokens_per_second": 29515.018097391632, - "useful_ttt_positions_per_second": 206605.12668174144, - "peak_allocated_gib": 23.980101108551025, - "baseline_allocated_gib": 2.403548240661621, - "peak_increment_gib": 21.576552867889404, - "warmup_seconds_including_compile": 0.8475298807024956, - "step_ms": [ - 138.96236196160316, - 135.83327271044254, - 136.9408555328846, - 136.9077805429697, - 135.7897948473692, - 135.7873361557722, - 135.86034625768661, - 135.84019988775253, - 135.85438393056393, - 324.67854768037796, - 171.35234735906124, - 135.78341156244278, - 135.83547621965408, - 136.48096099495888, - 136.54042035341263, - 136.01849973201752, - 135.85083559155464, - 135.7721146196127, - 136.49668917059898, - 136.4215351641178 - ], - "final_loss": 16.449453353881836 - }, - "packed": { - "mean_step_ms": 102.26912191137671, - "p50_step_ms": 90.35401325672865, - "stdev_step_ms": 30.748070235129727, - "useful_tokens_per_second": 42554.38903417309, - "useful_ttt_positions_per_second": 297880.7232392116, - "peak_allocated_gib": 9.601574420928955, - "baseline_allocated_gib": 2.2720112800598145, - "peak_increment_gib": 7.329563140869141, - "warmup_seconds_including_compile": 0.4550774786621332, - "step_ms": [ - 90.31735174357891, - 90.39067476987839, - 91.11217595636845, - 90.1272390037775, - 89.55635502934456, - 90.43884836137295, - 90.76938405632973, - 90.1151355355978, - 90.11649154126644, - 89.33848328888416, - 90.56887403130531, - 90.10537527501583, - 90.23720771074295, - 90.10511636734009, - 181.663291528821, - 193.6410814523697, - 132.5615793466568, - 93.92490051686764, - 90.40801785886288, - 89.88485485315323 - ], - "final_loss": 16.449453353881836 - }, - "speedup": 1.4417876653083874, - "peak_memory_reduction_fraction": 0.5996024212964997 - }, - { - "lengths": [ - 128, - 256, - 512, - 2048 - ], - "useful_tokens": 2944, - "padded_tokens": 8192, - "padding_fraction": 0.640625, - "raw_supervised_tokens": 2204, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 11.463168144226074 - }, - "plosses": { - "pass": true, - "max_abs_diff": 2.384185791015625e-07, - "relative_l2_diff": 3.1062749860994195e-08, - "reference_l2": 7.675385475158691 - }, - "loss_padded": 11.463168144226074, - "loss_packed": 11.463168144226074, - "gradients": { - "draft_model.midlayer.self_attn.q_proj.weight": { - "pass": true, - "max_abs_diff": 2.9802322387695312e-08, - "relative_l2_diff": 0.005071662953144663, - "reference_l2": 0.0015770653262734413 - }, - "draft_model.midlayer.self_attn.k_proj.weight": { - "pass": true, - "max_abs_diff": 5.960464477539063e-08, - "relative_l2_diff": 0.0055092919147389, - "reference_l2": 0.0016009289538487792 - }, - "draft_model.midlayer.self_attn.v_proj.weight": { - "pass": true, - "max_abs_diff": 2.9802322387695312e-08, - "relative_l2_diff": 0.004675468360171435, - "reference_l2": 0.0013346055056899786 - }, - "draft_model.midlayer.self_attn.o_proj.weight": { - "pass": true, - "max_abs_diff": 2.9802322387695312e-08, - "relative_l2_diff": 0.003780963952307209, - "reference_l2": 0.0013365180930122733 - }, - "draft_model.midlayer.mlp.gate_proj.weight": { - "pass": true, - "max_abs_diff": 5.960464477539063e-08, - "relative_l2_diff": 0.0035285330381317264, - "reference_l2": 0.005381997209042311 - }, - "draft_model.midlayer.mlp.up_proj.weight": { - "pass": true, - "max_abs_diff": 5.960464477539063e-08, - "relative_l2_diff": 0.0033779653002713973, - "reference_l2": 0.00546617154031992 - }, - "draft_model.midlayer.mlp.down_proj.weight": { - "pass": true, - "max_abs_diff": 5.960464477539063e-08, - "relative_l2_diff": 0.002940032226982802, - "reference_l2": 0.005697569809854031 - }, - "draft_model.midlayer.hidden_norm.weight": { - "pass": true, - "max_abs_diff": 2.9802322387695312e-08, - "relative_l2_diff": 0.00615527538571967, - "reference_l2": 3.794665462919511e-05 - }, - "draft_model.midlayer.input_layernorm.weight": { - "pass": true, - "max_abs_diff": 1.862645149230957e-08, - "relative_l2_diff": 0.006584594031334515, - "reference_l2": 3.436251063249074e-05 - }, - "draft_model.midlayer.post_attention_layernorm.weight": { - "pass": true, - "max_abs_diff": 5.960464477539063e-08, - "relative_l2_diff": 0.0038776387553202765, - "reference_l2": 0.00015499240544158965 - }, - "draft_model.fc.weight": { - "pass": true, - "max_abs_diff": 5.960464477539063e-08, - "relative_l2_diff": 0.004189682020519722, - "reference_l2": 0.008074476383626461 - }, - "draft_model.norm.weight": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 0.019233308732509613 - }, - "draft_model.lm_head.weight": { - "pass": true, - "max_abs_diff": 2.9802322387695312e-08, - "relative_l2_diff": 3.4444987028199235e-05, - "reference_l2": 0.008254210464656353 - } - }, - "pass": true - }, - "padded": { - "mean_step_ms": 142.04363320022821, - "p50_step_ms": 136.79443392902613, - "stdev_step_ms": 17.873005002610697, - "useful_tokens_per_second": 20726.025754706407, - "useful_ttt_positions_per_second": 145082.18028294484, - "peak_allocated_gib": 23.906234741210938, - "baseline_allocated_gib": 2.329681873321533, - "peak_increment_gib": 21.576552867889404, - "warmup_seconds_including_compile": 0.6789059638977051, - "step_ms": [ - 135.30797697603703, - 135.27469523251057, - 136.8685495108366, - 155.8755338191986, - 137.39495538175106, - 141.58212766051292, - 215.13050608336926, - 135.47670654952526, - 136.57748885452747, - 137.78152875602245, - 135.2911926805973, - 135.3690456598997, - 136.72031834721565, - 135.68365201354027, - 142.4194872379303, - 137.2075453400612, - 143.5843240469694, - 137.1347662061453, - 135.08125953376293, - 135.111004114151 - ], - "final_loss": 11.069210052490234 - }, - "packed": { - "mean_step_ms": 88.26659778133035, - "p50_step_ms": 69.67359222471714, - "stdev_step_ms": 32.720125563363545, - "useful_tokens_per_second": 33353.500350080314, - "useful_ttt_positions_per_second": 233474.5024505622, - "peak_allocated_gib": 7.208748817443848, - "baseline_allocated_gib": 2.2495083808898926, - "peak_increment_gib": 4.959240436553955, - "warmup_seconds_including_compile": 0.33644894510507584, - "step_ms": [ - 84.10226367413998, - 67.03308410942554, - 74.08362068235874, - 116.3476463407278, - 154.4363684952259, - 154.60862964391708, - 152.0404890179634, - 128.29125113785267, - 73.5000278800726, - 67.39425659179688, - 67.12955050170422, - 67.01149977743626, - 82.40591175854206, - 70.29546052217484, - 68.965008482337, - 69.05172392725945, - 68.15902143716812, - 66.88438914716244, - 66.8429471552372, - 66.74880534410477 - ], - "final_loss": 11.069201469421387 - }, - "speedup": 1.6092569190456831, - "peak_memory_reduction_fraction": 0.6984573733388055 - } - ] -} diff --git a/docs/sections/benchmarks/sequence-packing-results/eagle3-tiny-fp32.json b/docs/sections/benchmarks/sequence-packing-results/eagle3-tiny-fp32.json deleted file mode 100644 index 1cb48a255..000000000 --- a/docs/sections/benchmarks/sequence-packing-results/eagle3-tiny-fp32.json +++ /dev/null @@ -1,175 +0,0 @@ -{ - "timestamp_utc": "2026-10-02T04:12:33.131611+00:00", - "source": { - "files_sha256": { - "scripts/benchmark_sequence_packing.py": "0d11416603f81d5b97726abd299244029aa65b12beccb6ff38f9fb82c89dfe47", - "specforge/benchmarks/benchmark_sequence_packing.py": "39bf66ed00bee5cd272b6c8f6a562eae54537d98ea7cb41d066ad271eb816d65", - "specforge/algorithms/eagle3/data.py": "84d03344c68ba2e3f3a452db12c8b9cef95c931445ab8645b0444ec2580a8fe2", - "specforge/algorithms/eagle3/model.py": "aadd395819e3292ab3ccde0f0200fdd6df90f0de1de8a03fb91d6640394ca022", - "specforge/modeling/draft/llama3_eagle.py": "1488e38e36f1ff3bb11c0355da776122ef9c779105d38dee404b8e109aa06f4a", - "specforge/modeling/packed_sequence.py": "ad057f15c08fb1c5cc80994ac799106536ec0a41c28f88f33d5f1bce34c75670", - "specforge/training/strategies/base.py": "a6e3aa2961d9d55f96c816fa2f6ce764672699e02005d89688eddec357120894" - }, - "head": "unavailable (file hashes identify copied snapshot)" - }, - "environment": { - "python": "3.12.3", - "torch": "2.13.0+cu130", - "cuda": "13.0", - "gpu": "NVIDIA H200", - "cuda_visible_devices": "0", - "transformers": "5.12.1" - }, - "settings": { - "preset": "tiny", - "hidden_size": 128, - "intermediate_size": 256, - "num_heads": 4, - "num_kv_heads": 2, - "vocab_size": 512, - "draft_vocab_size": 256, - "target_hidden_size": 128, - "lengths": [ - [ - 8, - 17, - 31, - 64 - ], - [ - 2, - 3, - 5, - 17 - ], - [ - 32, - 32, - 32, - 32 - ] - ], - "ttt_length": 7, - "dtype": "float32", - "warmup": 5, - "steps": 20, - "seed": 1729, - "learning_rate": 0.0001, - "prompt_fraction": 0.25, - "correctness_only": true, - "skip_correctness": false, - "atol": 2e-05, - "rtol": 0.0002, - "output": "artifacts/sequence-packing/tiny-fp32-latest-metadata.json" - }, - "scope": "single-GPU synthetic offline features; resident inputs; production forward/backward/BF16Optimizer; excludes capture, I/O, transfer, distributed communication, and serving", - "cases": [ - { - "lengths": [ - 8, - 17, - 31, - 64 - ], - "useful_tokens": 120, - "padded_tokens": 256, - "padding_fraction": 0.53125, - "raw_supervised_tokens": 87, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 3.5112829208374023 - }, - "plosses": { - "pass": true, - "max_abs_diff": 5.960464477539063e-08, - "relative_l2_diff": 4.462841585418779e-08, - "reference_l2": 2.3132855892181396 - }, - "loss_padded": 3.5112829208374023, - "loss_packed": 3.5112829208374023, - "pass": true, - "gradient_summary": { - "parameter_tensors": 13, - "all_pass": true, - "max_abs_diff": 2.9103830456733704e-10, - "max_relative_l2_diff": 3.381816025687098e-07 - } - } - }, - { - "lengths": [ - 2, - 3, - 5, - 17 - ], - "useful_tokens": 27, - "padded_tokens": 68, - "padding_fraction": 0.6029411764705883, - "raw_supervised_tokens": 18, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 1.8310911655426025 - }, - "plosses": { - "pass": true, - "max_abs_diff": 2.9802322387695312e-08, - "relative_l2_diff": 4.682924771319063e-08, - "reference_l2": 1.1472936868667603 - }, - "loss_padded": 1.8310911655426025, - "loss_packed": 1.8310911655426025, - "pass": true, - "gradient_summary": { - "parameter_tensors": 13, - "all_pass": true, - "max_abs_diff": 4.0745362639427185e-10, - "max_relative_l2_diff": 4.456337620781133e-07 - } - } - }, - { - "lengths": [ - 32, - 32, - 32, - 32 - ], - "useful_tokens": 128, - "padded_tokens": 128, - "padding_fraction": 0.0, - "raw_supervised_tokens": 92, - "correctness": { - "loss": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 6.87973690032959 - }, - "plosses": { - "pass": true, - "max_abs_diff": 0.0, - "relative_l2_diff": 0.0, - "reference_l2": 4.606232166290283 - }, - "loss_padded": 6.87973690032959, - "loss_packed": 6.87973690032959, - "pass": true, - "gradient_summary": { - "parameter_tensors": 13, - "all_pass": true, - "max_abs_diff": 1.1641532182693481e-10, - "max_relative_l2_diff": 1.7126517309899537e-07 - } - } - } - ], - "evidence_kind": "Derived summary: per-parameter gradient entries reduced to count, pass flag and maxima; all other fields preserved.", - "source_report_sha256": "86a3ddd7e307768881d6a0ec594a2eb1f18405dae7e387b12fce9edd88d5c702" -} diff --git a/docs/sections/benchmarks/sequence-packing-results/full-model-analysis.json b/docs/sections/benchmarks/sequence-packing-results/full-model-analysis.json deleted file mode 100644 index 70816ed2e..000000000 --- a/docs/sections/benchmarks/sequence-packing-results/full-model-analysis.json +++ /dev/null @@ -1,899 +0,0 @@ -{ - "summary": { - "dflash": { - "padded": { - "runs": 4, - "metrics": { - "pipeline_seconds": { - "median": 135.77420610096306, - "min": 135.39080221019685, - "max": 136.07322796620429, - "values": [ - 136.07322796620429, - 136.05652987398207, - 135.49188232794404, - 135.39080221019685 - ] - }, - "trainer_fit_seconds_from_first_capture": { - "median": 145.5256609506905, - "min": 144.90472558513284, - "max": 145.755035309121, - "values": [ - 145.70392679609358, - 145.755035309121, - 145.34739510528743, - 144.90472558513284 - ] - }, - "checkpoint_seconds": { - "median": 19.004825842566788, - "min": 18.59124900586903, - "max": 19.063026294112206, - "values": [ - 18.94984213076532, - 19.059809554368258, - 19.063026294112206, - 18.59124900586903 - ] - }, - "useful_tokens_per_second": { - "median": 9647.448519004163, - "min": 9626.206562288098, - "max": 9674.726632954009, - "values": [ - 9626.206562288098, - 9627.387977726783, - 9667.509060281545, - 9674.726632954009 - ] - }, - "final_loss": { - "median": 7.263832092285156, - "min": 7.263832092285156, - "max": 7.263832092285156, - "values": [ - 7.263832092285156, - 7.263832092285156, - 7.263832092285156, - 7.263832092285156 - ] - }, - "perf/train_compute_time_s": { - "median": 0.4527946563698352, - "min": 0.4526336301639676, - "max": 0.45302082094550133, - "values": [ - 0.4526336301639676, - 0.4528050641492009, - 0.4527842485904694, - 0.45302082094550133 - ] - }, - "perf/data_wait_time_s": { - "median": 0.039065446685999636, - "min": 0.038275998421013355, - "max": 0.04061508445441723, - "values": [ - 0.04061508445441723, - 0.03976557278633118, - 0.038365320585668085, - 0.038275998421013355 - ] - }, - "perf/durable_ack_time_s": { - "median": 0.002404910206794739, - "min": 0.002309505730867386, - "max": 0.002473393775522709, - "values": [ - 0.002309505730867386, - 0.002416390419006348, - 0.002473393775522709, - 0.0023934299945831297 - ] - } - }, - "checks": [ - { - "run_id": "bd0233465065-dflash-repeat00-arm0-padded", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.320475473999977 - }, - { - "step": 256, - "seconds": 9.629366656765342 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - }, - { - "run_id": "1e720cd3e271-dflash-repeat00-arm3-padded", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.36259664222598 - }, - { - "step": 256, - "seconds": 9.697212912142277 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - }, - { - "run_id": "1b5f8203cf73-dflash-repeat01-arm0-padded", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.2087761182338 - }, - { - "step": 256, - "seconds": 9.854250175878406 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - }, - { - "run_id": "31ea763f3d18-dflash-repeat01-arm3-padded", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.078649101778865 - }, - { - "step": 256, - "seconds": 9.512599904090166 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - } - ] - }, - "packed": { - "runs": 4, - "metrics": { - "pipeline_seconds": { - "median": 130.65109005570412, - "min": 128.88156617432833, - "max": 130.91351471282542, - "values": [ - 128.88156617432833, - 130.78667958825827, - 130.51550052314997, - 130.91351471282542 - ] - }, - "trainer_fit_seconds_from_first_capture": { - "median": 140.37849941663444, - "min": 138.3309756219387, - "max": 140.5097789634019, - "values": [ - 138.3309756219387, - 140.4588652085513, - 140.5097789634019, - 140.2981336247176 - ] - }, - "checkpoint_seconds": { - "median": 18.75084599200636, - "min": 18.463733648881316, - "max": 19.00793844088912, - "values": [ - 18.463733648881316, - 19.00793844088912, - 18.919086307287216, - 18.582605676725507 - ] - }, - "useful_tokens_per_second": { - "median": 10025.713602590113, - "min": 10005.605631117272, - "max": 10163.35414661426, - "values": [ - 10163.35414661426, - 10015.308929959234, - 10036.11827522099, - 10005.605631117272 - ] - }, - "final_loss": { - "median": 7.263965606689453, - "min": 7.263965606689453, - "max": 7.263965606689453, - "values": [ - 7.263965606689453, - 7.263965606689453, - 7.263965606689453, - 7.263965606689453 - ] - }, - "perf/train_compute_time_s": { - "median": 0.43363652409240605, - "min": 0.43351573960483075, - "max": 0.4339526748508215, - "values": [ - 0.433688922919333, - 0.4339526748508215, - 0.43351573960483075, - 0.43358412526547907 - ] - }, - "perf/data_wait_time_s": { - "median": 0.03900999794527889, - "min": 0.03261461492627859, - "max": 0.0398359164968133, - "values": [ - 0.03261461492627859, - 0.038341693453490734, - 0.039678302437067034, - 0.0398359164968133 - ] - }, - "perf/durable_ack_time_s": { - "median": 0.002406857404857874, - "min": 0.002359313905239105, - "max": 0.0024157476499676706, - "values": [ - 0.002359313905239105, - 0.0024157476499676706, - 0.0024075431749224665, - 0.0024061716347932817 - ] - } - }, - "checks": [ - { - "run_id": "0db157a5b805-dflash-repeat00-arm1-packed", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.016127996146679 - }, - { - "step": 256, - "seconds": 9.447605652734637 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - }, - { - "run_id": "beb7bee09b70-dflash-repeat00-arm2-packed", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.337019385769963 - }, - { - "step": 256, - "seconds": 9.670919055119157 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - }, - { - "run_id": "d016ac38de6c-dflash-repeat01-arm1-packed", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 8.926233634352684 - }, - { - "step": 256, - "seconds": 9.992852672934532 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - }, - { - "run_id": "4dc06bfceb8e-dflash-repeat01-arm2-packed", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.199242942035198 - }, - { - "step": 256, - "seconds": 9.383362734690309 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - } - ] - }, - "speedups": { - "pipeline_seconds": 1.0392121951915951, - "trainer_fit_seconds_from_first_capture": 1.0366663096944755, - "perf/train_compute_time_s": 1.0441801629083869 - } - }, - "dflash2": { - "padded": { - "runs": 4, - "metrics": { - "pipeline_seconds": { - "median": 127.55567608494312, - "min": 127.01265624165535, - "max": 127.72959364019334, - "values": [ - 127.72959364019334, - 127.55552425421774, - 127.55582791566849, - 127.01265624165535 - ] - }, - "trainer_fit_seconds_from_first_capture": { - "median": 137.32001544442028, - "min": 136.6376235689968, - "max": 137.59124981798232, - "values": [ - 137.59124981798232, - 137.353132288903, - 137.28689859993756, - 136.6376235689968 - ] - }, - "checkpoint_seconds": { - "median": 19.090700599364936, - "min": 18.831517465412617, - "max": 19.25768494606018, - "values": [ - 19.159075815230608, - 19.25768494606018, - 19.022325383499265, - 18.831517465412617 - ] - }, - "useful_tokens_per_second": { - "median": 10268.99813638693, - "min": 10255.015792893093, - "max": 10312.901397068905, - "values": [ - 10255.015792893093, - 10269.010359672353, - 10268.98591310151, - 10312.901397068905 - ] - }, - "final_loss": { - "median": 7.968346118927002, - "min": 7.968346118927002, - "max": 7.968346118927002, - "values": [ - 7.968346118927002, - 7.968346118927002, - 7.968346118927002, - 7.968346118927002 - ] - }, - "perf/train_compute_time_s": { - "median": 0.42698242781683804, - "min": 0.4268245469406247, - "max": 0.42719867677241563, - "values": [ - 0.4269536115154624, - 0.4268245469406247, - 0.42719867677241563, - 0.4270112441182136 - ] - }, - "perf/data_wait_time_s": { - "median": 0.03236039846017957, - "min": 0.031001201815903188, - "max": 0.03348202735185623, - "values": [ - 0.03348202735185623, - 0.03243206156045198, - 0.03228873535990715, - 0.031001201815903188 - ] - }, - "perf/durable_ack_time_s": { - "median": 0.0024311989322304724, - "min": 0.002365817494690418, - "max": 0.0025716573223471643, - "values": [ - 0.0024104075357317925, - 0.0024519903287291527, - 0.0025716573223471643, - 0.002365817494690418 - ] - } - }, - "checks": [ - { - "run_id": "ca60b5621436-dflash2-repeat00-arm0-padded", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.298717353492975 - }, - { - "step": 256, - "seconds": 9.860358461737633 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - }, - { - "run_id": "e8afd561b8f6-dflash2-repeat00-arm3-padded", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.461436005309224 - }, - { - "step": 256, - "seconds": 9.796248940750957 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - }, - { - "run_id": "b026b309f6bf-dflash2-repeat01-arm0-padded", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.292883981019258 - }, - { - "step": 256, - "seconds": 9.729441402480006 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - }, - { - "run_id": "fd539583a214-dflash2-repeat01-arm3-padded", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.208016926422715 - }, - { - "step": 256, - "seconds": 9.623500538989902 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - } - ] - }, - "packed": { - "runs": 4, - "metrics": { - "pipeline_seconds": { - "median": 122.89778250828385, - "min": 122.41715203598142, - "max": 123.31176270730793, - "values": [ - 123.19319461472332, - 122.41715203598142, - 122.60237040184438, - 123.31176270730793 - ] - }, - "trainer_fit_seconds_from_first_capture": { - "median": 132.90509969182312, - "min": 132.2964224666357, - "max": 133.59795146621764, - "values": [ - 133.59795146621764, - 132.60274993814528, - 132.2964224666357, - 133.20744944550097 - ] - }, - "checkpoint_seconds": { - "median": 19.39351878874004, - "min": 19.042769499123096, - "max": 19.794485840946436, - "values": [ - 19.794485840946436, - 19.42600578442216, - 19.042769499123096, - 19.36103179305792 - ] - }, - "useful_tokens_per_second": { - "median": 10658.260397991853, - "min": 10622.41728803356, - "max": 10700.044709543621, - "values": [ - 10632.640902742303, - 10700.044709543621, - 10683.879893241403, - 10622.41728803356 - ] - }, - "final_loss": { - "median": 7.968497276306152, - "min": 7.968497276306152, - "max": 7.968497276306152, - "values": [ - 7.968497276306152, - 7.968497276306152, - 7.968497276306152, - 7.968497276306152 - ] - }, - "perf/train_compute_time_s": { - "median": 0.4057552154362202, - "min": 0.40529941400140523, - "max": 0.4060684394985437, - "values": [ - 0.40529941400140523, - 0.40570894527435303, - 0.4058014855980873, - 0.4060684394985437 - ] - }, - "perf/data_wait_time_s": { - "median": 0.03559140999987721, - "min": 0.03429850439727306, - "max": 0.03726461844146252, - "values": [ - 0.03726461844146252, - 0.03429850439727306, - 0.03459509003907442, - 0.03658772996068001 - ] - }, - "perf/durable_ack_time_s": { - "median": 0.0024793629013001917, - "min": 0.00236849345266819, - "max": 0.0025612031146883965, - "values": [ - 0.0025612031146883965, - 0.00236849345266819, - 0.002535621479153633, - 0.0024231043234467504 - ] - } - }, - "checks": [ - { - "run_id": "bf1e0bae505a-dflash2-repeat00-arm1-packed", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.391200210899115 - }, - { - "step": 256, - "seconds": 10.403285630047321 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - }, - { - "run_id": "33c10f6d8e0a-dflash2-repeat00-arm2-packed", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.241760902106762 - }, - { - "step": 256, - "seconds": 10.184244882315397 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - }, - { - "run_id": "1874cfdf56c1-dflash2-repeat01-arm1-packed", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.35001527518034 - }, - { - "step": 256, - "seconds": 9.692754223942757 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - }, - { - "run_id": "e269784cb040-dflash2-repeat01-arm2-packed", - "samples": 1024, - "steps": 256, - "compiler_counter_delta": { - "stats": {}, - "frames": {}, - "unimplemented": {}, - "graph_break": {}, - "inductor": {}, - "aot_autograd": {} - }, - "checkpoint_events": [ - { - "step": 128, - "seconds": 9.466601584106684 - }, - { - "step": 256, - "seconds": 9.894430208951235 - } - ], - "logged_steps": [ - 50, - 100, - 150, - 200, - 250 - ], - "sustained_overlap": true - } - ] - }, - "speedups": { - "pipeline_seconds": 1.037900550210052, - "trainer_fit_seconds_from_first_capture": 1.0332185579246722, - "perf/train_compute_time_s": 1.0523153161637044 - } - } - }, - "notes": [ - "Training-thread compute diagnostics cover five 50-step windows (250 steps), use host wall time, and include detailed metrics at window ends.", - "Pipeline includes intermediate checkpoint at step 128; full completion includes the final step-256 checkpoint.", - "Ratios are medians of four runs per arm when complete; no claim of convergence or serving quality." - ] -} diff --git a/docs/sections/benchmarks/sequence-packing-results/full-model-audit.json b/docs/sections/benchmarks/sequence-packing-results/full-model-audit.json deleted file mode 100644 index 22957abf6..000000000 --- a/docs/sections/benchmarks/sequence-packing-results/full-model-audit.json +++ /dev/null @@ -1,1107 +0,0 @@ -{ - "audit_utc": "2026-10-02T09:08:41.738677+00:00", - "source_report_remote": "/scratch/specforge-packing-full-model-20261002/long_v2.json", - "source_report_sha256": "f8895e415d1baaef7ddaa74a7f8e322d83d13e008d2f7606b0378fb0b4861179", - "dataset_sha256": "f90478e0d3e32f0d3c45c515381b9bc42951cdd6931830ac1a7380f4459f24b1", - "all_run_assertions_passed": true, - "measured_runs": 16, - "warmup_runs": 4, - "runs_per_architecture_per_mode": 4, - "target_architecture": { - "model": "Qwen3-4B", - "layers": 36, - "hidden_size": 2560, - "vocabulary": 151936, - "role": "frozen real target forward in separate SGLang producer" - }, - "validated_trainable_counts": { - "dflash": { - "elements": 537427200, - "tensors": 58, - "decoder_layers": 5, - "optimizer_updates_per_tensor_per_run": 256 - }, - "dflash2": { - "elements": 558918912, - "tensors": 81, - "decoder_layers": 5, - "optimizer_updates_per_tensor_per_run": 256 - } - }, - "frozen_tied_target_embedding_head_unique_elements_in_consumer": 388956160, - "checks": { - "exact_preprocessed_dataset_hash": true, - "same_actual_http_inputs_and_masks_per_architecture": true, - "exact_capture_publication_consumption_and_ack_order": true, - "1024_samples_and_256_optimizer_steps_per_run": true, - "all_optimizer_original_parameters_and_fp32_masters_covered": true, - "all_trainable_tensors_have_256_adam_updates": true, - "all_first_warmup_gradients_present_and_finite_global_norm": true, - "no_gradient_observation_hook_in_measured_runs": true, - "all_five_decoder_layers_have_nonempty_trainable_names_and_sampled_updates": true, - "all_trainable_elements_finite_after_training": true, - "frozen_target_parameter_samples_unchanged": true, - "logging_at_steps_50_100_150_200_250": true, - "checkpoint_events_at_128_and_256": true, - "final_checkpoint_step_and_sample_count_match_acks": true, - "all_measured_compiler_counter_deltas_empty": true, - "continued_live_capture_after_first_optimizer_ack": true, - "final_ack_equals_pipeline_endpoint": true, - "all_final_losses_finite": true - }, - "summary": { - "dflash": { - "arms": { - "padded": { - "runs": 4, - "metrics": { - "pipeline_seconds": { - "median": 135.77420610096306, - "min": 135.39080221019685, - "max": 136.07322796620429, - "values": [ - 136.07322796620429, - 136.05652987398207, - 135.49188232794404, - 135.39080221019685 - ] - }, - "fit_with_final_checkpoint_seconds": { - "median": 145.5256609506905, - "min": 144.90472558513284, - "max": 145.755035309121, - "values": [ - 145.70392679609358, - 145.755035309121, - 145.34739510528743, - 144.90472558513284 - ] - }, - "checkpoint_total_seconds": { - "median": 19.004825842566788, - "min": 18.59124900586903, - "max": 19.063026294112206, - "values": [ - 18.94984213076532, - 19.059809554368258, - 19.063026294112206, - 18.59124900586903 - ] - }, - "host_train_compute_seconds_per_step": { - "median": 0.4527946563698352, - "min": 0.4526336301639676, - "max": 0.45302082094550133, - "values": [ - 0.4526336301639676, - 0.4528050641492009, - 0.4527842485904694, - 0.45302082094550133 - ] - }, - "host_data_wait_seconds_per_step": { - "median": 0.039065446685999636, - "min": 0.038275998421013355, - "max": 0.04061508445441723, - "values": [ - 0.04061508445441723, - 0.03976557278633118, - 0.038365320585668085, - 0.038275998421013355 - ] - }, - "host_durable_ack_seconds_per_step": { - "median": 0.002404910206794739, - "min": 0.002309505730867386, - "max": 0.002473393775522709, - "values": [ - 0.002309505730867386, - 0.002416390419006348, - 0.002473393775522709, - 0.0023934299945831297 - ] - }, - "loader_wait_producer_seconds": { - "median": 0.20019448921084404, - "min": 0.20013251528143883, - "max": 0.2004451733082533, - "values": [ - 0.20013251528143883, - 0.20024952851235867, - 0.2004451733082533, - 0.20013944990932941 - ] - }, - "loader_wait_fetch_seconds": { - "median": 9.73675948008895, - "min": 9.515376891940832, - "max": 10.09336070343852, - "values": [ - 10.09336070343852, - 9.953005198389292, - 9.515376891940832, - 9.520513761788607 - ] - }, - "final_loss": { - "median": 7.263832092285156, - "min": 7.263832092285156, - "max": 7.263832092285156, - "values": [ - 7.263832092285156, - 7.263832092285156, - 7.263832092285156, - 7.263832092285156 - ] - } - } - }, - "packed": { - "runs": 4, - "metrics": { - "pipeline_seconds": { - "median": 130.65109005570412, - "min": 128.88156617432833, - "max": 130.91351471282542, - "values": [ - 128.88156617432833, - 130.78667958825827, - 130.51550052314997, - 130.91351471282542 - ] - }, - "fit_with_final_checkpoint_seconds": { - "median": 140.37849941663444, - "min": 138.3309756219387, - "max": 140.5097789634019, - "values": [ - 138.3309756219387, - 140.4588652085513, - 140.5097789634019, - 140.2981336247176 - ] - }, - "checkpoint_total_seconds": { - "median": 18.75084599200636, - "min": 18.463733648881316, - "max": 19.00793844088912, - "values": [ - 18.463733648881316, - 19.00793844088912, - 18.919086307287216, - 18.582605676725507 - ] - }, - "host_train_compute_seconds_per_step": { - "median": 0.43363652409240605, - "min": 0.43351573960483075, - "max": 0.4339526748508215, - "values": [ - 0.433688922919333, - 0.4339526748508215, - 0.43351573960483075, - 0.43358412526547907 - ] - }, - "host_data_wait_seconds_per_step": { - "median": 0.03900999794527889, - "min": 0.03261461492627859, - "max": 0.0398359164968133, - "values": [ - 0.03261461492627859, - 0.038341693453490734, - 0.039678302437067034, - 0.0398359164968133 - ] - }, - "host_durable_ack_seconds_per_step": { - "median": 0.002406857404857874, - "min": 0.002359313905239105, - "max": 0.0024157476499676706, - "values": [ - 0.002359313905239105, - 0.0024157476499676706, - 0.0024075431749224665, - 0.0024061716347932817 - ] - }, - "loader_wait_producer_seconds": { - "median": 0.2003897288814187, - "min": 0.20013267919421196, - "max": 0.30021679401397705, - "values": [ - 0.20013267919421196, - 0.30021679401397705, - 0.20041996613144875, - 0.20035949163138866 - ] - }, - "loader_wait_fetch_seconds": { - "median": 9.543545130640268, - "min": 8.035708643496037, - "max": 9.886001640930772, - "values": [ - 8.035708643496037, - 9.43034845776856, - 9.656741803511977, - 9.886001640930772 - ] - }, - "final_loss": { - "median": 7.263965606689453, - "min": 7.263965606689453, - "max": 7.263965606689453, - "values": [ - 7.263965606689453, - 7.263965606689453, - 7.263965606689453, - 7.263965606689453 - ] - } - } - } - }, - "median_speedups": { - "pipeline_seconds": 1.0392121951915951, - "fit_with_final_checkpoint_seconds": 1.0366663096944755, - "host_train_compute_seconds_per_step": 1.0441801629083869 - }, - "matched_chronological_pair_ratios": [ - { - "repeat": 0, - "padded_arm": 0, - "packed_arm": 1, - "ratios": { - "pipeline_seconds": 1.055800546232875, - "fit_with_final_checkpoint_seconds": 1.0532993506407797, - "host_train_compute_seconds_per_step": 1.0436827095262435 - } - }, - { - "repeat": 0, - "padded_arm": 3, - "packed_arm": 2, - "ratios": { - "pipeline_seconds": 1.040293478680813, - "fit_with_final_checkpoint_seconds": 1.0377062002651527, - "host_train_compute_seconds_per_step": 1.0434434222691684 - } - }, - { - "repeat": 1, - "padded_arm": 0, - "packed_arm": 1, - "ratios": { - "pipeline_seconds": 1.0381286650615986, - "fit_with_final_checkpoint_seconds": 1.0344290353139447, - "host_train_compute_seconds_per_step": 1.0444470805216963 - } - }, - { - "repeat": 1, - "padded_arm": 3, - "packed_arm": 2, - "ratios": { - "pipeline_seconds": 1.0342003459856144, - "fit_with_final_checkpoint_seconds": 1.0328343067822796, - "host_train_compute_seconds_per_step": 1.044827968893283 - } - } - ] - }, - "dflash2": { - "arms": { - "padded": { - "runs": 4, - "metrics": { - "pipeline_seconds": { - "median": 127.55567608494312, - "min": 127.01265624165535, - "max": 127.72959364019334, - "values": [ - 127.72959364019334, - 127.55552425421774, - 127.55582791566849, - 127.01265624165535 - ] - }, - "fit_with_final_checkpoint_seconds": { - "median": 137.32001544442028, - "min": 136.6376235689968, - "max": 137.59124981798232, - "values": [ - 137.59124981798232, - 137.353132288903, - 137.28689859993756, - 136.6376235689968 - ] - }, - "checkpoint_total_seconds": { - "median": 19.090700599364936, - "min": 18.831517465412617, - "max": 19.25768494606018, - "values": [ - 19.159075815230608, - 19.25768494606018, - 19.022325383499265, - 18.831517465412617 - ] - }, - "host_train_compute_seconds_per_step": { - "median": 0.42698242781683804, - "min": 0.4268245469406247, - "max": 0.42719867677241563, - "values": [ - 0.4269536115154624, - 0.4268245469406247, - 0.42719867677241563, - 0.4270112441182136 - ] - }, - "host_data_wait_seconds_per_step": { - "median": 0.03236039846017957, - "min": 0.031001201815903188, - "max": 0.03348202735185623, - "values": [ - 0.03348202735185623, - 0.03243206156045198, - 0.03228873535990715, - 0.031001201815903188 - ] - }, - "host_durable_ack_seconds_per_step": { - "median": 0.0024311989322304724, - "min": 0.002365817494690418, - "max": 0.0025716573223471643, - "values": [ - 0.0024104075357317925, - 0.0024519903287291527, - 0.0025716573223471643, - 0.002365817494690418 - ] - }, - "loader_wait_producer_seconds": { - "median": 0.20038656052201986, - "min": 0.20015659555792809, - "max": 0.3005787916481495, - "values": [ - 0.20015659555792809, - 0.2004134152084589, - 0.20035970583558083, - 0.3005787916481495 - ] - }, - "loader_wait_fetch_seconds": { - "median": 8.02211464382708, - "min": 7.595450304448605, - "max": 8.327452383935452, - "values": [ - 8.327452383935452, - 7.998322926461697, - 8.045906361192465, - 7.595450304448605 - ] - }, - "final_loss": { - "median": 7.968346118927002, - "min": 7.968346118927002, - "max": 7.968346118927002, - "values": [ - 7.968346118927002, - 7.968346118927002, - 7.968346118927002, - 7.968346118927002 - ] - } - } - }, - "packed": { - "runs": 4, - "metrics": { - "pipeline_seconds": { - "median": 122.89778250828385, - "min": 122.41715203598142, - "max": 123.31176270730793, - "values": [ - 123.19319461472332, - 122.41715203598142, - 122.60237040184438, - 123.31176270730793 - ] - }, - "fit_with_final_checkpoint_seconds": { - "median": 132.90509969182312, - "min": 132.2964224666357, - "max": 133.59795146621764, - "values": [ - 133.59795146621764, - 132.60274993814528, - 132.2964224666357, - 133.20744944550097 - ] - }, - "checkpoint_total_seconds": { - "median": 19.39351878874004, - "min": 19.042769499123096, - "max": 19.794485840946436, - "values": [ - 19.794485840946436, - 19.42600578442216, - 19.042769499123096, - 19.36103179305792 - ] - }, - "host_train_compute_seconds_per_step": { - "median": 0.4057552154362202, - "min": 0.40529941400140523, - "max": 0.4060684394985437, - "values": [ - 0.40529941400140523, - 0.40570894527435303, - 0.4058014855980873, - 0.4060684394985437 - ] - }, - "host_data_wait_seconds_per_step": { - "median": 0.03559140999987721, - "min": 0.03429850439727306, - "max": 0.03726461844146252, - "values": [ - 0.03726461844146252, - 0.03429850439727306, - 0.03459509003907442, - 0.03658772996068001 - ] - }, - "host_durable_ack_seconds_per_step": { - "median": 0.0024793629013001917, - "min": 0.00236849345266819, - "max": 0.0025612031146883965, - "values": [ - 0.0025612031146883965, - 0.00236849345266819, - 0.002535621479153633, - 0.0024231043234467504 - ] - }, - "loader_wait_producer_seconds": { - "median": 0.3002029359340668, - "min": 0.20014070346951485, - "max": 0.30046532303094864, - "values": [ - 0.20014070346951485, - 0.300190145149827, - 0.30046532303094864, - 0.30021572671830654 - ] - }, - "loader_wait_fetch_seconds": { - "median": 8.65784180443734, - "min": 8.398219551891088, - "max": 9.154761364683509, - "values": [ - 9.154761364683509, - 8.398219551891088, - 8.409457132220268, - 8.90622647665441 - ] - }, - "final_loss": { - "median": 7.968497276306152, - "min": 7.968497276306152, - "max": 7.968497276306152, - "values": [ - 7.968497276306152, - 7.968497276306152, - 7.968497276306152, - 7.968497276306152 - ] - } - } - } - }, - "median_speedups": { - "pipeline_seconds": 1.037900550210052, - "fit_with_final_checkpoint_seconds": 1.0332185579246722, - "host_train_compute_seconds_per_step": 1.0523153161637044 - }, - "matched_chronological_pair_ratios": [ - { - "repeat": 0, - "padded_arm": 0, - "packed_arm": 1, - "ratios": { - "pipeline_seconds": 1.0368234547343076, - "fit_with_final_checkpoint_seconds": 1.0298904160426026, - "host_train_compute_seconds_per_step": 1.0534276556195121 - } - }, - { - "repeat": 0, - "padded_arm": 3, - "packed_arm": 2, - "ratios": { - "pipeline_seconds": 1.0419742832828363, - "fit_with_final_checkpoint_seconds": 1.035824161663115, - "host_train_compute_seconds_per_step": 1.0520461821515734 - } - }, - { - "repeat": 1, - "padded_arm": 0, - "packed_arm": 1, - "ratios": { - "pipeline_seconds": 1.0404026243341669, - "fit_with_final_checkpoint_seconds": 1.0377219280782928, - "host_train_compute_seconds_per_step": 1.0527282228718118 - } - }, - { - "repeat": 1, - "padded_arm": 3, - "packed_arm": 2, - "ratios": { - "pipeline_seconds": 1.0300124939672775, - "fit_with_final_checkpoint_seconds": 1.0257506178353726, - "host_train_compute_seconds_per_step": 1.0515745686750053 - } - } - ] - } - }, - "per_run_checks": [ - { - "arm": "dflash/warmup/False", - "seconds": 157.2685, - "fit_seconds": 166.5949, - "host_train_compute_mean_s": 0.537832, - "checkpoint_seconds": 18.6059, - "layers": 5, - "trainable_parameters": 537427200, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": { - "stats": { - "calls_captured": 8, - "unique_graphs": 4 - }, - "frames": { - "ok": 4, - "total": 4 - }, - "aot_autograd": { - "autograd_cache_saved": 4, - "ok": 4, - "total": 4, - "autograd_cache_miss": 4 - }, - "inductor": { - "triton_bundler_save_kernel": 112, - "benchmarking.InductorBenchmarker.benchmark_gpu": 3, - "benchmarking.InductorBenchmarker.benchmark": 3, - "fxgraph_cache_miss": 8, - "async_compile_cache_hit": 3, - "async_compile_cache_miss": 21, - "triton_bundler_save_static_autotuner": 8 - } - }, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 1.0006, - "wait_fetch_s": 9.2088 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash/warmup/True", - "seconds": 136.9922, - "fit_seconds": 146.929, - "host_train_compute_mean_s": 0.46024, - "checkpoint_seconds": 19.0491, - "layers": 5, - "trainable_parameters": 537427200, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": { - "stats": { - "calls_captured": 4, - "unique_graphs": 2 - }, - "frames": { - "ok": 2, - "total": 2 - }, - "inductor": { - "async_compile_cache_miss": 12, - "triton_bundler_save_static_autotuner": 4, - "triton_bundler_save_kernel": 48, - "fxgraph_cache_miss": 4, - "async_compile_cache_hit": 2 - }, - "aot_autograd": { - "autograd_cache_saved": 2, - "ok": 2, - "total": 2, - "autograd_cache_miss": 2 - } - }, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.2003, - "wait_fetch_s": 9.4391 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash2/warmup/False", - "seconds": 129.4135, - "fit_seconds": 139.8324, - "host_train_compute_mean_s": 0.434597, - "checkpoint_seconds": 19.7909, - "layers": 5, - "trainable_parameters": 558918912, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.2006, - "wait_fetch_s": 8.0094 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash2/warmup/True", - "seconds": 123.3042, - "fit_seconds": 133.2789, - "host_train_compute_mean_s": 0.405485, - "checkpoint_seconds": 20.058, - "layers": 5, - "trainable_parameters": 558918912, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.2007, - "wait_fetch_s": 8.4689 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash/repeat00-arm0/False", - "seconds": 136.0732, - "fit_seconds": 145.7039, - "host_train_compute_mean_s": 0.452634, - "checkpoint_seconds": 18.9498, - "layers": 5, - "trainable_parameters": 537427200, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.2001, - "wait_fetch_s": 10.0934 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash/repeat00-arm1/True", - "seconds": 128.8816, - "fit_seconds": 138.331, - "host_train_compute_mean_s": 0.433689, - "checkpoint_seconds": 18.4637, - "layers": 5, - "trainable_parameters": 537427200, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.2001, - "wait_fetch_s": 8.0357 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash/repeat00-arm2/True", - "seconds": 130.7867, - "fit_seconds": 140.4589, - "host_train_compute_mean_s": 0.433953, - "checkpoint_seconds": 19.0079, - "layers": 5, - "trainable_parameters": 537427200, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.3002, - "wait_fetch_s": 9.4303 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash/repeat00-arm3/False", - "seconds": 136.0565, - "fit_seconds": 145.755, - "host_train_compute_mean_s": 0.452805, - "checkpoint_seconds": 19.0598, - "layers": 5, - "trainable_parameters": 537427200, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.2002, - "wait_fetch_s": 9.953 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash/repeat01-arm0/False", - "seconds": 135.4919, - "fit_seconds": 145.3474, - "host_train_compute_mean_s": 0.452784, - "checkpoint_seconds": 19.063, - "layers": 5, - "trainable_parameters": 537427200, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.2004, - "wait_fetch_s": 9.5154 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash/repeat01-arm1/True", - "seconds": 130.5155, - "fit_seconds": 140.5098, - "host_train_compute_mean_s": 0.433516, - "checkpoint_seconds": 18.9191, - "layers": 5, - "trainable_parameters": 537427200, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.2004, - "wait_fetch_s": 9.6567 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash/repeat01-arm2/True", - "seconds": 130.9135, - "fit_seconds": 140.2981, - "host_train_compute_mean_s": 0.433584, - "checkpoint_seconds": 18.5826, - "layers": 5, - "trainable_parameters": 537427200, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.2004, - "wait_fetch_s": 9.886 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash/repeat01-arm3/False", - "seconds": 135.3908, - "fit_seconds": 144.9047, - "host_train_compute_mean_s": 0.453021, - "checkpoint_seconds": 18.5912, - "layers": 5, - "trainable_parameters": 537427200, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.2001, - "wait_fetch_s": 9.5205 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash2/repeat00-arm0/False", - "seconds": 127.7296, - "fit_seconds": 137.5912, - "host_train_compute_mean_s": 0.426954, - "checkpoint_seconds": 19.1591, - "layers": 5, - "trainable_parameters": 558918912, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.2002, - "wait_fetch_s": 8.3275 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash2/repeat00-arm1/True", - "seconds": 123.1932, - "fit_seconds": 133.598, - "host_train_compute_mean_s": 0.405299, - "checkpoint_seconds": 19.7945, - "layers": 5, - "trainable_parameters": 558918912, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.2001, - "wait_fetch_s": 9.1548 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash2/repeat00-arm2/True", - "seconds": 122.4172, - "fit_seconds": 132.6027, - "host_train_compute_mean_s": 0.405709, - "checkpoint_seconds": 19.426, - "layers": 5, - "trainable_parameters": 558918912, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.3002, - "wait_fetch_s": 8.3982 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash2/repeat00-arm3/False", - "seconds": 127.5555, - "fit_seconds": 137.3531, - "host_train_compute_mean_s": 0.426825, - "checkpoint_seconds": 19.2577, - "layers": 5, - "trainable_parameters": 558918912, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.2004, - "wait_fetch_s": 7.9983 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash2/repeat01-arm0/False", - "seconds": 127.5558, - "fit_seconds": 137.2869, - "host_train_compute_mean_s": 0.427199, - "checkpoint_seconds": 19.0223, - "layers": 5, - "trainable_parameters": 558918912, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.2004, - "wait_fetch_s": 8.0459 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash2/repeat01-arm1/True", - "seconds": 122.6024, - "fit_seconds": 132.2964, - "host_train_compute_mean_s": 0.405801, - "checkpoint_seconds": 19.0428, - "layers": 5, - "trainable_parameters": 558918912, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.3005, - "wait_fetch_s": 8.4095 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash2/repeat01-arm2/True", - "seconds": 123.3118, - "fit_seconds": 133.2074, - "host_train_compute_mean_s": 0.406068, - "checkpoint_seconds": 19.361, - "layers": 5, - "trainable_parameters": 558918912, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.3002, - "wait_fetch_s": 8.9062 - }, - "full_model_and_pipeline_checks": "PASS" - }, - { - "arm": "dflash2/repeat01-arm3/False", - "seconds": 127.0127, - "fit_seconds": 136.6376, - "host_train_compute_mean_s": 0.427011, - "checkpoint_seconds": 18.8315, - "layers": 5, - "trainable_parameters": 558918912, - "all_tensor_adam_steps": 256, - "logs": [ - 50, - 100, - 150, - 200, - 250 - ], - "compiler_delta": {}, - "capture_after_first_ack": 254, - "loader_wait": { - "wait_producer_s": 0.3006, - "wait_fetch_s": 7.5955 - }, - "full_model_and_pipeline_checks": "PASS" - } - ], - "limits": [ - "Primary pipeline interval is first actual HTTP capture dispatch through final durable ack and CUDA completion; it includes the intermediate step-128 checkpoint.", - "Full fit includes the step-256 checkpoint and cleanup. Server/model initialization and data preparation are outside these intervals.", - "Host train_compute diagnostic means cover five 50-step windows (250 of 256 steps), including detailed metrics at window ends. They are not CUDA kernel-only timings.", - "Frozen target samples and per-layer update hashes are sampled; full finiteness scans and optimizer participation checks cover every trainable tensor.", - "Observed timing results from four runs per mode; no claim of convergence, model quality or serving throughput." - ] -} diff --git a/docs/sections/benchmarks/sequence-packing-results/online-32step.json b/docs/sections/benchmarks/sequence-packing-results/online-32step.json deleted file mode 100644 index 0ab362017..000000000 --- a/docs/sections/benchmarks/sequence-packing-results/online-32step.json +++ /dev/null @@ -1,568 +0,0 @@ -{ - "created_utc": "2026-10-02T06:02:31.254497+00:00", - "torch_version": "2.13.0+cu130", - "gpu": "NVIDIA H200", - "timing_contract": "first capture dispatch -> final synchronous durable ack + CUDA sync; final checkpoint separately timed", - "scope": "actual target weights and target embeddings/head; freshly initialized draft; single-rank FSDP NO_SHARD", - "prompt_source": "/scratch/specforge-packing-e2e-20261002/sharegpt-prompts.jsonl", - "prompt_file_sha256": "d99771e31b4d6bc79bdeae630b689e95abbfbc5f858db7b6f39383cbd2461e20", - "capture_cache_policy": "canonical adapter generates a fresh extra_key per request attempt, forcing full prefill", - "producer": "canonical drive_producer in parallel thread, one worker/concurrency=1, bounded ref backlog", - "async_ack": false, - "summary": { - "dflash": { - "padded": { - "runs": 2, - "median_pipeline_seconds": 15.335569551214576, - "median_useful_tokens_per_second": 10810.815020745238, - "median_fit_seconds_with_checkpoint": 25.21502728294581 - }, - "packed": { - "runs": 2, - "median_pipeline_seconds": 14.732898173853755, - "median_useful_tokens_per_second": 11253.13099697949, - "median_fit_seconds_with_checkpoint": 24.90839832369238 - }, - "pipeline_speedup": 1.0409065053086686, - "fit_speedup_with_checkpoint": 1.0123102640028754 - }, - "dflash2": { - "padded": { - "runs": 2, - "median_pipeline_seconds": 14.386700802482665, - "median_useful_tokens_per_second": 11523.70540099725, - "median_fit_seconds_with_checkpoint": 25.107247687876225 - }, - "packed": { - "runs": 2, - "median_pipeline_seconds": 13.76857764646411, - "median_useful_tokens_per_second": 12041.194447433445, - "median_fit_seconds_with_checkpoint": 24.000815202482045 - }, - "pipeline_speedup": 1.0448937553239055, - "fit_speedup_with_checkpoint": 1.0460997876972011 - } - }, - "evidence_kind": "Derived 32-step report; preserves all per-run timing samples, including warmups, plus settings and lifecycle checks. Detailed per-capture events and original raw report are retained locally.", - "source_report_sha256": "d0284b6ab87ad26bba3c6624c4d4b203cb4f18fef08609476729d650b55f9be4", - "settings": { - "server_url": "http://127.0.0.1:31012", - "target_model": "/cluster-storage/models/Qwen3-4B", - "draft_config": "configs/qwen3-4b-dflash.json", - "prompts_path": "/scratch/specforge-packing-e2e-20261002/sharegpt-prompts.jsonl", - "algorithm": "both", - "capture_layers": null, - "draft_layers": 2, - "lengths": [ - 128, - 256, - 512, - 2048 - ], - "batch_size": 4, - "accumulation_steps": 1, - "anchors": 512, - "steps": 32, - "warmup_steps": 32, - "repeats": 1, - "seed": 1729, - "prompt_fraction": 0.25, - "learning_rate": 0.0001, - "objective_chunk_blocks": 128, - "capture_batch_size": 4, - "backlog": 8, - "log_interval": 50, - "save_interval": 0, - "teacher_metrics": true, - "dataloader_workers": 4, - "receive_buffers": "pinned", - "segment_mib": 1024, - "local_buffer_mib": 256, - "request_timeout": 300, - "dist_port": 29712, - "keep_checkpoints": false, - "work_dir": "/scratch/specforge-packing-e2e-20261002/full_v1", - "output": "/scratch/specforge-packing-e2e-20261002/full_v1.json" - }, - "settings_resolution_note": "draft_config overrides unused draft_layers=2; every resolved draft has five layers. prompts_path overrides synthetic lengths/prompt_fraction defaults and uses actual assistant masks.", - "target_dimensions": { - "model_type": "qwen3", - "num_hidden_layers": 36, - "hidden_size": 2560, - "intermediate_size": 9728, - "num_attention_heads": 32, - "num_key_value_heads": 8, - "vocab_size": 151936 - }, - "warmup_runs": [ - { - "architecture": "dflash", - "packing": false, - "label": "warmup", - "optimizer_steps": 32, - "microsteps": 32, - "samples": 128, - "durable_acked_samples": 128, - "useful_tokens": 165788, - "supervised_tokens": 131810, - "pipeline_seconds": 32.0647142175585, - "useful_tokens_per_second": 5170.418762354513, - "samples_per_second": 3.9919270488900143, - "capture_finished_seconds": 31.251980060711503, - "producer_finished_seconds": 31.253749769181013, - "trainer_fit_seconds_from_first_capture": 42.36139240115881, - "fit_wall_seconds": 42.36776242032647, - "model_and_runtime_setup_seconds": 9.31502272374928, - "checkpoint_seconds": 10.295483123511076, - "final_loss": 8.513822555541992, - "capture_calls_after_first_optimizer": 30, - "sustained_live_overlap_established": true, - "compiler_counter_delta": { - "inductor": { - "async_compile_cache_miss": 23, - "triton_bundler_save_static_autotuner": 6, - "triton_bundler_save_kernel": 72, - "fxgraph_cache_hit": 2, - "triton_bundler_load_static_autotuner": 3, - "fxgraph_cache_miss": 6, - "async_compile_cache_hit": 6 - }, - "unimplemented": {}, - "frames": { - "ok": 4, - "total": 4 - }, - "graph_break": {}, - "aot_autograd": { - "total": 4, - "autograd_cache_miss": 3, - "ok": 4, - "autograd_cache_saved": 3, - "autograd_cache_hit": 1 - }, - "stats": { - "calls_captured": 8, - "unique_graphs": 4 - } - }, - "full_warmup_replay": true, - "resolved_draft_layers": 5, - "publication_consumption_order_identical": true, - "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", - "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" - }, - { - "architecture": "dflash", - "packing": true, - "label": "warmup", - "optimizer_steps": 32, - "microsteps": 32, - "samples": 128, - "durable_acked_samples": 128, - "useful_tokens": 165788, - "supervised_tokens": 131810, - "pipeline_seconds": 15.39916873909533, - "useful_tokens_per_second": 10766.03567432171, - "samples_per_second": 8.312136984059032, - "capture_finished_seconds": 14.608486795797944, - "producer_finished_seconds": 14.610368017107248, - "trainer_fit_seconds_from_first_capture": 26.10293635353446, - "fit_wall_seconds": 26.1091186106205, - "model_and_runtime_setup_seconds": 9.027630560100079, - "checkpoint_seconds": 10.695947425439954, - "final_loss": 8.513858795166016, - "capture_calls_after_first_optimizer": 30, - "sustained_live_overlap_established": true, - "compiler_counter_delta": { - "stats": { - "calls_captured": 4, - "unique_graphs": 2 - }, - "inductor": { - "triton_bundler_load_static_autotuner": 6, - "async_compile_cache_miss": 18, - "fxgraph_cache_hit": 4, - "async_compile_cache_hit": 8 - }, - "unimplemented": {}, - "frames": { - "ok": 2, - "total": 2 - }, - "graph_break": {}, - "aot_autograd": { - "autograd_cache_hit": 2, - "total": 2, - "ok": 2 - } - }, - "full_warmup_replay": true, - "resolved_draft_layers": 5, - "publication_consumption_order_identical": true, - "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", - "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" - }, - { - "architecture": "dflash2", - "packing": false, - "label": "warmup", - "optimizer_steps": 32, - "microsteps": 32, - "samples": 128, - "durable_acked_samples": 128, - "useful_tokens": 165788, - "supervised_tokens": 131810, - "pipeline_seconds": 14.374922448769212, - "useful_tokens_per_second": 11533.140480642722, - "samples_per_second": 8.9043958641293, - "capture_finished_seconds": 13.643273144960403, - "producer_finished_seconds": 13.644904932007194, - "trainer_fit_seconds_from_first_capture": 24.717275140807033, - "fit_wall_seconds": 24.723629055544734, - "model_and_runtime_setup_seconds": 8.944378308951855, - "checkpoint_seconds": 10.341318672522902, - "final_loss": 9.105286598205566, - "capture_calls_after_first_optimizer": 30, - "sustained_live_overlap_established": true, - "compiler_counter_delta": { - "stats": {}, - "inductor": {}, - "unimplemented": {}, - "frames": {}, - "graph_break": {}, - "aot_autograd": {} - }, - "full_warmup_replay": true, - "resolved_draft_layers": 5, - "publication_consumption_order_identical": true, - "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", - "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" - }, - { - "architecture": "dflash2", - "packing": true, - "label": "warmup", - "optimizer_steps": 32, - "microsteps": 32, - "samples": 128, - "durable_acked_samples": 128, - "useful_tokens": 165788, - "supervised_tokens": 131810, - "pipeline_seconds": 13.88367784023285, - "useful_tokens_per_second": 11941.21629065541, - "samples_per_second": 9.219459099596426, - "capture_finished_seconds": 13.120847288519144, - "producer_finished_seconds": 13.122505459934473, - "trainer_fit_seconds_from_first_capture": 24.113164821639657, - "fit_wall_seconds": 24.11945860646665, - "model_and_runtime_setup_seconds": 8.892530964687467, - "checkpoint_seconds": 10.228421663865447, - "final_loss": 9.104482650756836, - "capture_calls_after_first_optimizer": 30, - "sustained_live_overlap_established": true, - "compiler_counter_delta": { - "stats": {}, - "inductor": {}, - "unimplemented": {}, - "frames": {}, - "graph_break": {}, - "aot_autograd": {} - }, - "full_warmup_replay": true, - "resolved_draft_layers": 5, - "publication_consumption_order_identical": true, - "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", - "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" - } - ], - "measured_runs": [ - { - "architecture": "dflash", - "packing": false, - "label": "repeat00-arm0", - "optimizer_steps": 32, - "microsteps": 32, - "samples": 128, - "durable_acked_samples": 128, - "useful_tokens": 165788, - "supervised_tokens": 131810, - "pipeline_seconds": 15.28223004937172, - "useful_tokens_per_second": 10848.416720883995, - "samples_per_second": 8.375740947916324, - "capture_finished_seconds": 14.491078296676278, - "producer_finished_seconds": 14.492836989462376, - "trainer_fit_seconds_from_first_capture": 25.171482319012284, - "fit_wall_seconds": 25.17781513929367, - "model_and_runtime_setup_seconds": 8.702558502554893, - "checkpoint_seconds": 9.880388440564275, - "final_loss": 8.514045715332031, - "capture_calls_after_first_optimizer": 30, - "sustained_live_overlap_established": true, - "compiler_counter_delta": { - "stats": {}, - "inductor": {}, - "unimplemented": {}, - "frames": {}, - "graph_break": {}, - "aot_autograd": {} - }, - "full_warmup_replay": true, - "resolved_draft_layers": 5, - "publication_consumption_order_identical": true, - "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", - "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" - }, - { - "architecture": "dflash", - "packing": true, - "label": "repeat00-arm1", - "optimizer_steps": 32, - "microsteps": 32, - "samples": 128, - "durable_acked_samples": 128, - "useful_tokens": 165788, - "supervised_tokens": 131810, - "pipeline_seconds": 14.797958767041564, - "useful_tokens_per_second": 11203.437082771698, - "samples_per_second": 8.649841644719626, - "capture_finished_seconds": 14.030732162296772, - "producer_finished_seconds": 14.03259969688952, - "trainer_fit_seconds_from_first_capture": 24.82643250748515, - "fit_wall_seconds": 24.83258822746575, - "model_and_runtime_setup_seconds": 8.715459797531366, - "checkpoint_seconds": 10.020536547526717, - "final_loss": 8.513858795166016, - "capture_calls_after_first_optimizer": 30, - "sustained_live_overlap_established": true, - "compiler_counter_delta": { - "stats": {}, - "inductor": {}, - "unimplemented": {}, - "frames": {}, - "graph_break": {}, - "aot_autograd": {} - }, - "full_warmup_replay": true, - "resolved_draft_layers": 5, - "publication_consumption_order_identical": true, - "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", - "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" - }, - { - "architecture": "dflash", - "packing": true, - "label": "repeat00-arm2", - "optimizer_steps": 32, - "microsteps": 32, - "samples": 128, - "durable_acked_samples": 128, - "useful_tokens": 165788, - "supervised_tokens": 131810, - "pipeline_seconds": 14.667837580665946, - "useful_tokens_per_second": 11302.82491118728, - "samples_per_second": 8.726576040678285, - "capture_finished_seconds": 13.881927194073796, - "producer_finished_seconds": 13.883811619132757, - "trainer_fit_seconds_from_first_capture": 24.99036413989961, - "fit_wall_seconds": 24.997150415554643, - "model_and_runtime_setup_seconds": 8.881739912554622, - "checkpoint_seconds": 10.314807986840606, - "final_loss": 8.513858795166016, - "capture_calls_after_first_optimizer": 30, - "sustained_live_overlap_established": true, - "compiler_counter_delta": { - "stats": {}, - "inductor": {}, - "unimplemented": {}, - "frames": {}, - "graph_break": {}, - "aot_autograd": {} - }, - "full_warmup_replay": true, - "resolved_draft_layers": 5, - "publication_consumption_order_identical": true, - "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", - "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" - }, - { - "architecture": "dflash", - "packing": false, - "label": "repeat00-arm3", - "optimizer_steps": 32, - "microsteps": 32, - "samples": 128, - "durable_acked_samples": 128, - "useful_tokens": 165788, - "supervised_tokens": 131810, - "pipeline_seconds": 15.388909053057432, - "useful_tokens_per_second": 10773.213320606481, - "samples_per_second": 8.317678631973543, - "capture_finished_seconds": 14.578291799873114, - "producer_finished_seconds": 14.579897599294782, - "trainer_fit_seconds_from_first_capture": 25.25857224687934, - "fit_wall_seconds": 25.26484971679747, - "model_and_runtime_setup_seconds": 8.783952347934246, - "checkpoint_seconds": 9.86840933188796, - "final_loss": 8.514045715332031, - "capture_calls_after_first_optimizer": 30, - "sustained_live_overlap_established": true, - "compiler_counter_delta": { - "stats": {}, - "inductor": {}, - "unimplemented": {}, - "frames": {}, - "graph_break": {}, - "aot_autograd": {} - }, - "full_warmup_replay": true, - "resolved_draft_layers": 5, - "publication_consumption_order_identical": true, - "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", - "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" - }, - { - "architecture": "dflash2", - "packing": false, - "label": "repeat00-arm0", - "optimizer_steps": 32, - "microsteps": 32, - "samples": 128, - "durable_acked_samples": 128, - "useful_tokens": 165788, - "supervised_tokens": 131810, - "pipeline_seconds": 14.375430628657341, - "useful_tokens_per_second": 11532.732777375208, - "samples_per_second": 8.904081088522851, - "capture_finished_seconds": 13.629519296810031, - "producer_finished_seconds": 13.631206966936588, - "trainer_fit_seconds_from_first_capture": 25.25617554783821, - "fit_wall_seconds": 25.263109516352415, - "model_and_runtime_setup_seconds": 8.989978183060884, - "checkpoint_seconds": 10.879672184586525, - "final_loss": 9.105286598205566, - "capture_calls_after_first_optimizer": 30, - "sustained_live_overlap_established": true, - "compiler_counter_delta": { - "stats": {}, - "inductor": {}, - "unimplemented": {}, - "frames": {}, - "graph_break": {}, - "aot_autograd": {} - }, - "full_warmup_replay": true, - "resolved_draft_layers": 5, - "publication_consumption_order_identical": true, - "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", - "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" - }, - { - "architecture": "dflash2", - "packing": true, - "label": "repeat00-arm1", - "optimizer_steps": 32, - "microsteps": 32, - "samples": 128, - "durable_acked_samples": 128, - "useful_tokens": 165788, - "supervised_tokens": 131810, - "pipeline_seconds": 13.817821264266968, - "useful_tokens_per_second": 11998.128853260645, - "samples_per_second": 9.26339960200595, - "capture_finished_seconds": 13.074846714735031, - "producer_finished_seconds": 13.076601503416896, - "trainer_fit_seconds_from_first_capture": 24.060468524694443, - "fit_wall_seconds": 24.066856909543276, - "model_and_runtime_setup_seconds": 9.183672599494457, - "checkpoint_seconds": 10.24164686910808, - "final_loss": 9.104482650756836, - "capture_calls_after_first_optimizer": 30, - "sustained_live_overlap_established": true, - "compiler_counter_delta": { - "stats": {}, - "inductor": {}, - "unimplemented": {}, - "frames": {}, - "graph_break": {}, - "aot_autograd": {} - }, - "full_warmup_replay": true, - "resolved_draft_layers": 5, - "publication_consumption_order_identical": true, - "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", - "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" - }, - { - "architecture": "dflash2", - "packing": true, - "label": "repeat00-arm2", - "optimizer_steps": 32, - "microsteps": 32, - "samples": 128, - "durable_acked_samples": 128, - "useful_tokens": 165788, - "supervised_tokens": 131810, - "pipeline_seconds": 13.719334028661251, - "useful_tokens_per_second": 12084.260041606247, - "samples_per_second": 9.329898939160854, - "capture_finished_seconds": 12.968088772147894, - "producer_finished_seconds": 12.969761220738292, - "trainer_fit_seconds_from_first_capture": 23.941161880269647, - "fit_wall_seconds": 23.947430515661836, - "model_and_runtime_setup_seconds": 8.923571102321148, - "checkpoint_seconds": 10.220690120011568, - "final_loss": 9.104482650756836, - "capture_calls_after_first_optimizer": 30, - "sustained_live_overlap_established": true, - "compiler_counter_delta": { - "stats": {}, - "inductor": {}, - "unimplemented": {}, - "frames": {}, - "graph_break": {}, - "aot_autograd": {} - }, - "full_warmup_replay": true, - "resolved_draft_layers": 5, - "publication_consumption_order_identical": true, - "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", - "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" - }, - { - "architecture": "dflash2", - "packing": false, - "label": "repeat00-arm3", - "optimizer_steps": 32, - "microsteps": 32, - "samples": 128, - "durable_acked_samples": 128, - "useful_tokens": 165788, - "supervised_tokens": 131810, - "pipeline_seconds": 14.397970976307988, - "useful_tokens_per_second": 11514.678024619294, - "samples_per_second": 8.890141549154762, - "capture_finished_seconds": 13.634843789041042, - "producer_finished_seconds": 13.637094628065825, - "trainer_fit_seconds_from_first_capture": 24.958319827914238, - "fit_wall_seconds": 24.96457778289914, - "model_and_runtime_setup_seconds": 8.878480760380626, - "checkpoint_seconds": 10.559298420324922, - "final_loss": 9.105286598205566, - "capture_calls_after_first_optimizer": 30, - "sustained_live_overlap_established": true, - "compiler_counter_delta": { - "stats": {}, - "inductor": {}, - "unimplemented": {}, - "frames": {}, - "graph_break": {}, - "aot_autograd": {} - }, - "full_warmup_replay": true, - "resolved_draft_layers": 5, - "publication_consumption_order_identical": true, - "prompt_sha256": "f8719fc7ee18fab8e17cd63e578c45a30db6459aa018fef80ab75e949e454e0d", - "actual_request_sha256": "b4f764be42c1fac906f4a79b980a8a559f0c362fac78ffb111489afb2b01f820" - } - ] -} diff --git a/docs/sections/benchmarks/sequence-packing-results/target-parameter-inventory.json b/docs/sections/benchmarks/sequence-packing-results/target-parameter-inventory.json deleted file mode 100644 index 0c8be9f3a..000000000 --- a/docs/sections/benchmarks/sequence-packing-results/target-parameter-inventory.json +++ /dev/null @@ -1,3086 +0,0 @@ -{ - "target_model": "/cluster-storage/models/Qwen3-4B", - "unique_saved_parameters": 4022468096, - "parameter_tensors": 398, - "layer_indices": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7, - 8, - 9, - 10, - 11, - 12, - 13, - 14, - 15, - 16, - 17, - 18, - 19, - 20, - 21, - 22, - 23, - 24, - 25, - 26, - 27, - 28, - 29, - 30, - 31, - 32, - 33, - 34, - 35 - ], - "all_36_layers_present": true, - "config_sha256": "8ba006f74fecfaaeb392872a60f4a480e7ec9860153d2e1b769ec81f9a147f8a", - "parameters": [ - { - "name": "model.embed_tokens.weight", - "shape": [ - 151936, - 2560 - ], - "numel": 388956160 - }, - { - "name": "model.layers.0.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.0.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.0.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.0.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.0.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.0.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.0.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.0.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.0.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.0.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.0.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.1.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.1.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.1.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.1.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.1.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.1.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.1.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.1.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.1.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.1.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.1.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.10.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.10.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.10.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.10.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.10.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.10.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.10.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.10.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.10.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.10.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.10.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.11.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.11.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.11.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.11.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.11.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.11.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.11.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.11.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.11.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.11.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.11.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.12.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.12.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.12.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.12.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.12.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.12.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.12.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.12.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.12.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.12.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.12.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.13.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.13.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.13.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.13.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.13.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.13.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.13.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.13.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.13.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.13.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.13.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.14.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.14.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.14.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.14.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.14.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.14.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.14.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.14.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.14.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.14.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.14.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.15.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.15.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.15.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.15.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.15.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.15.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.15.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.15.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.2.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.2.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.2.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.2.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.2.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.2.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.2.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.2.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.2.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.2.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.2.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.3.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.3.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.3.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.3.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.3.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.3.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.3.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.3.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.3.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.3.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.3.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.4.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.4.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.4.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.4.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.4.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.4.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.4.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.4.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.4.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.4.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.4.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.5.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.5.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.5.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.5.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.5.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.5.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.5.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.5.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.5.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.5.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.5.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.6.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.6.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.6.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.6.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.6.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.6.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.6.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.6.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.6.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.6.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.6.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.7.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.7.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.7.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.7.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.7.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.7.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.7.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.7.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.7.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.7.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.7.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.8.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.8.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.8.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.8.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.8.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.8.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.8.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.8.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.8.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.8.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.8.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.9.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.9.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.9.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.9.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.9.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.9.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.9.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.9.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.9.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.9.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.9.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.15.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.15.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.15.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.16.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.16.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.16.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.16.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.16.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.16.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.16.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.16.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.16.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.16.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.16.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.17.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.17.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.17.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.17.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.17.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.17.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.17.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.17.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.17.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.17.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.17.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.18.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.18.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.18.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.18.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.18.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.18.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.18.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.18.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.18.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.18.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.18.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.19.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.19.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.19.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.19.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.19.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.19.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.19.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.19.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.19.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.19.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.19.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.20.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.20.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.20.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.20.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.20.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.20.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.20.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.20.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.20.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.20.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.20.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.21.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.21.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.21.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.21.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.21.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.21.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.21.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.21.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.21.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.21.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.21.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.22.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.22.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.22.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.22.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.22.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.22.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.22.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.22.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.22.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.22.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.22.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.23.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.23.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.23.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.23.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.23.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.23.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.23.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.23.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.23.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.23.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.23.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.24.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.24.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.24.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.24.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.24.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.24.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.24.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.24.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.24.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.24.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.24.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.25.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.25.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.25.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.25.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.25.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.25.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.25.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.25.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.25.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.25.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.25.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.26.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.26.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.26.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.26.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.26.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.26.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.26.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.26.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.26.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.26.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.26.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.27.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.27.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.27.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.27.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.27.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.27.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.27.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.27.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.27.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.27.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.27.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.28.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.28.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.28.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.28.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.28.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.28.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.28.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.28.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.28.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.28.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.28.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.29.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.29.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.29.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.29.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.29.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.29.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.29.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.29.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.29.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.29.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.29.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.30.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.30.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.30.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.30.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.30.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.30.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.30.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.30.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.30.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.30.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.30.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.31.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.31.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.31.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.31.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.31.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.31.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.31.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.31.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.31.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.31.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.31.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.32.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.32.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.32.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.32.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.32.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.32.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.32.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.32.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.32.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.32.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.32.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.33.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.33.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.33.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.33.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.33.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.33.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.33.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.33.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.33.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.33.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.33.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.34.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.34.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.34.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.34.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.34.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.34.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.34.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.34.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.34.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.34.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.34.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.35.mlp.gate_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.35.self_attn.k_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.35.self_attn.k_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.35.self_attn.o_proj.weight", - "shape": [ - 2560, - 4096 - ], - "numel": 10485760 - }, - { - "name": "model.layers.35.self_attn.q_norm.weight", - "shape": [ - 128 - ], - "numel": 128 - }, - { - "name": "model.layers.35.self_attn.q_proj.weight", - "shape": [ - 4096, - 2560 - ], - "numel": 10485760 - }, - { - "name": "model.layers.35.self_attn.v_proj.weight", - "shape": [ - 1024, - 2560 - ], - "numel": 2621440 - }, - { - "name": "model.layers.35.input_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.layers.35.mlp.down_proj.weight", - "shape": [ - 2560, - 9728 - ], - "numel": 24903680 - }, - { - "name": "model.layers.35.mlp.up_proj.weight", - "shape": [ - 9728, - 2560 - ], - "numel": 24903680 - }, - { - "name": "model.layers.35.post_attention_layernorm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - }, - { - "name": "model.norm.weight", - "shape": [ - 2560 - ], - "numel": 2560 - } - ] -} diff --git a/scripts/benchmark_dflash_sequence_packing.py b/scripts/benchmark_dflash_sequence_packing.py deleted file mode 100755 index 421fe9bee..000000000 --- a/scripts/benchmark_dflash_sequence_packing.py +++ /dev/null @@ -1,7 +0,0 @@ -#!/usr/bin/env python3 -"""Entrypoint for the DFlash/DFlash2 sequence-packing benchmark.""" - -from specforge.benchmarks.benchmark_dflash_sequence_packing import main - -if __name__ == "__main__": - main() diff --git a/scripts/benchmark_online_sequence_packing.py b/scripts/benchmark_online_sequence_packing.py deleted file mode 100755 index 94457aa2e..000000000 --- a/scripts/benchmark_online_sequence_packing.py +++ /dev/null @@ -1,7 +0,0 @@ -#!/usr/bin/env python3 -"""Entrypoint for the real online sequence-packing pipeline benchmark.""" - -from specforge.benchmarks.benchmark_online_sequence_packing import main - -if __name__ == "__main__": - main() diff --git a/scripts/benchmark_sequence_packing.py b/scripts/benchmark_sequence_packing.py deleted file mode 100755 index 0b33ccd5f..000000000 --- a/scripts/benchmark_sequence_packing.py +++ /dev/null @@ -1,7 +0,0 @@ -#!/usr/bin/env python3 -"""Entrypoint for the single-GPU EAGLE3 sequence-packing benchmark.""" - -from specforge.benchmarks.benchmark_sequence_packing import main - -if __name__ == "__main__": - main() diff --git a/specforge/benchmarks/benchmark_dflash_sequence_packing.py b/specforge/benchmarks/benchmark_dflash_sequence_packing.py deleted file mode 100644 index b1fa26ce5..000000000 --- a/specforge/benchmarks/benchmark_dflash_sequence_packing.py +++ /dev/null @@ -1,465 +0,0 @@ -"""Measure DFlash and DFlash2 packing separately, with identical sampled anchors. - -Example (CUDA, no model downloads):: - - PYTHONPATH=. python scripts/benchmark_dflash_sequence_packing.py \ - --preset tiny --dtype float32 --correctness-only --output tiny.json - PYTHONPATH=. python scripts/benchmark_dflash_sequence_packing.py \ - --preset medium --steps 20 --output perf.json - -Production model, collators, strategy, and BF16Optimizer are used. Frozen target -embeddings/head and captured features are synthetic. The timed region includes -the strategy's CPU integer-feature processing/transfers, forward, backward, and -optimizer update. Hidden features are GPU resident. Capture, hidden-feature I/O -and H2D, distributed communication, and serving are outside this benchmark. -""" - -from __future__ import annotations - -import argparse -import gc -import hashlib -import json -import os -import statistics -import subprocess -import time -from datetime import datetime, timezone -from pathlib import Path -from unittest import mock - -import torch -from torch import nn -from transformers import Qwen3Config - -from specforge.algorithms.common.dflash_family_model import OnlineDFlashModel -from specforge.algorithms.common.hidden_states_data import ( - build_collator, - build_packed_collator, -) -from specforge.modeling.draft.dflash import DFlashDraftModel -from specforge.modeling.draft.dflash2 import DFlash2DraftModel -from specforge.optimizer import BF16Optimizer -from specforge.runtime.contracts import TrainBatch -from specforge.training.strategies.base import DFlashTrainStrategy, StepContext - -PRESETS = { - "tiny": dict( - hidden_size=64, - intermediate_size=128, - layers=2, - heads=4, - kv_heads=2, - vocab_size=128, - block_size=4, - anchors=8, - capture_layers=2, - conv_group_size=4, - conv_kernel_size=2, - selector_rank=4, - selector_top_k=8, - lengths=[[9, 17, 31, 64], [32, 32, 32, 32]], - ), - "medium": dict( - hidden_size=2048, - intermediate_size=8192, - layers=2, - heads=16, - kv_heads=4, - vocab_size=32000, - block_size=16, - anchors=128, - capture_layers=2, - conv_group_size=32, - conv_kernel_size=4, - selector_rank=16, - selector_top_k=16, - lengths=[[1024, 1024, 1024, 1024], [128, 256, 512, 2048]], - ), -} -CONTEXT = StepContext(global_step=10, total_steps=100, collect_detailed_metrics=False) - - -def parse_args(): - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--preset", choices=PRESETS, default="medium") - parser.add_argument( - "--algorithm", choices=["dflash", "dflash2", "both"], default="both" - ) - for key in PRESETS["tiny"]: - if key != "lengths": - parser.add_argument("--" + key.replace("_", "-"), type=int) - parser.add_argument("--lengths", action="append") - parser.add_argument("--dtype", choices=["float32", "bfloat16"], default="bfloat16") - parser.add_argument("--sliding-window", type=int) - parser.add_argument("--warmup", type=int, default=5) - parser.add_argument("--steps", type=int, default=20) - parser.add_argument("--seed", type=int, default=1729) - parser.add_argument("--learning-rate", type=float, default=1e-4) - parser.add_argument("--objective-chunk-blocks", type=int, default=128) - parser.add_argument("--correctness-only", action="store_true") - parser.add_argument("--skip-correctness", action="store_true") - parser.add_argument("--atol", type=float) - parser.add_argument("--rtol", type=float) - parser.add_argument("--output", type=Path, required=True) - args = parser.parse_args() - for key, value in PRESETS[args.preset].items(): - if getattr(args, key) is None: - setattr(args, key, value) - if isinstance(args.lengths[0], str): - args.lengths = [ - [int(value) for value in row.split(",")] for row in args.lengths - ] - args.atol = ( - args.atol - if args.atol is not None - else (2e-5 if args.dtype == "float32" else 2e-3) - ) - args.rtol = ( - args.rtol - if args.rtol is not None - else (2e-4 if args.dtype == "float32" else 2e-2) - ) - if any(not row or min(row) < 4 for row in args.lengths): - parser.error("each batch must contain sequence lengths >= 4") - if args.steps < 1 or args.warmup < 1 or args.anchors < 1: - parser.error("steps, warmup, and anchors must be positive") - if args.hidden_size % args.heads or args.heads % args.kv_heads: - parser.error("hidden_size/heads and heads/kv_heads must be integral") - if args.correctness_only and args.skip_correctness: - parser.error("correctness-only and skip-correctness cannot be combined") - return args - - -def build_strategy(args, algorithm): - torch.manual_seed(args.seed) - config = Qwen3Config( - architectures=[ - "DFlash2DraftModel" if algorithm == "dflash2" else "DFlashDraftModel" - ], - hidden_size=args.hidden_size, - intermediate_size=args.intermediate_size, - num_hidden_layers=args.layers, - num_target_layers=args.capture_layers + 4, - num_attention_heads=args.heads, - num_key_value_heads=args.kv_heads, - head_dim=args.hidden_size // args.heads, - vocab_size=args.vocab_size, - attention_dropout=0.0, - max_position_embeddings=max(map(max, args.lengths)) + args.block_size, - layer_types=["sliding_attention" if args.sliding_window else "full_attention"] - * args.layers, - use_sliding_window=bool(args.sliding_window), - sliding_window=args.sliding_window, - dflash_config={ - "block_size": args.block_size, - "mask_token_id": args.vocab_size - 1, - "target_layer_ids": list(range(1, args.capture_layers + 1)), - "conv_group_size": args.conv_group_size, - "conv_kernel_size": args.conv_kernel_size, - "selector_rank": args.selector_rank, - "selector_top_k": args.selector_top_k, - }, - ) - config._attn_implementation = "flex_attention" - draft_type = DFlash2DraftModel if algorithm == "dflash2" else DFlashDraftModel - draft = draft_type(config) - model = OnlineDFlashModel( - draft_model=draft, - target_lm_head=nn.Linear( - args.hidden_size, args.vocab_size, bias=False - ).requires_grad_(False), - target_embed_tokens=nn.Embedding( - args.vocab_size, args.hidden_size - ).requires_grad_(False), - mask_token_id=args.vocab_size - 1, - block_size=args.block_size, - attention_backend="flex_attention", - num_anchors=args.anchors, - objective_chunk_blocks=args.objective_chunk_blocks, - loss_decay_gamma=7.0, - selector_loss_alpha=1.0 if algorithm == "dflash2" else 0.0, - teacher_metrics=False, - ).to("cuda", getattr(torch, args.dtype)) - return DFlashTrainStrategy(model.train()) - - -def make_features(args, lengths): - generator = torch.Generator().manual_seed(args.seed + 1) - features = [] - for length in lengths: - loss_mask = torch.ones(1, length, dtype=torch.long) - loss_mask[:, : length // 4] = 0 - loss_mask[:, -1] = 0 - features.append( - { - "input_ids": torch.randint( - 0, args.vocab_size - 1, (1, length), generator=generator - ), - "loss_mask": loss_mask, - "hidden_states": torch.randn( - 1, - length, - args.capture_layers * args.hidden_size, - generator=generator, - ).to(getattr(torch, args.dtype)), - } - ) - return features - - -def make_batch(features, mode, algorithm): - collator = build_collator() if mode == "padded" else build_packed_collator() - tensors = collator(features) - tensors["hidden_states"] = tensors["hidden_states"].cuda() - return TrainBatch( - sample_ids=[str(i) for i in range(len(features))], - strategy=algorithm, - tensors=tensors, - metadata={}, - ) - - -def compare(reference, actual, args): - reference, actual = reference.float(), actual.float() - delta = actual - reference - return { - "pass": bool(torch.allclose(reference, actual, atol=args.atol, rtol=args.rtol)), - "max_abs_diff": float(delta.abs().max()), - "relative_l2_diff": float(delta.norm()) / max(float(reference.norm()), 1e-30), - } - - -def correctness(strategy, features, args, algorithm): - model = strategy.trainable_module() - results = {} - for mode in ["padded", "packed"]: - model.zero_grad(set_to_none=True) - batch = make_batch(features, mode, algorithm) - recorded_anchors = [] - sampler = model._sample_anchor_positions - - def record(*pos, **kwargs): - anchors, keep = sampler(*pos, **kwargs) - recorded_anchors.append((anchors.cpu(), keep.cpu())) - return anchors, keep - - torch.cuda.manual_seed(args.seed + 100) - with mock.patch.object(model, "_sample_anchor_positions", side_effect=record): - output = strategy.forward_loss(batch, CONTEXT) - output.loss.backward() - results[mode] = { - "loss": output.loss.detach().float().cpu(), - "loss_terms": torch.stack(output.loss_terms).detach().float().cpu(), - "anchors": recorded_anchors, - "grads": { - name: None if p.grad is None else p.grad.detach().float().cpu() - for name, p in model.named_parameters() - if p.requires_grad - }, - } - del output, batch - baseline, packed = results["padded"], results["packed"] - checks = { - key: compare(baseline[key], packed[key], args) for key in ["loss", "loss_terms"] - } - checks["loss_values"] = [float(baseline["loss"]), float(packed["loss"])] - checks["identical_sampled_anchors"] = len(baseline["anchors"]) == len( - packed["anchors"] - ) and all( - torch.equal(a, b) and torch.equal(ka, kb) - for (a, ka), (b, kb) in zip(baseline["anchors"], packed["anchors"]) - ) - checks["sampled_anchor_count"] = sum( - int(keep.sum()) for _, keep in baseline["anchors"] - ) - checks["gradients"] = {} - for name, ref in baseline["grads"].items(): - value = packed["grads"][name] - checks["gradients"][name] = ( - {"pass": ref is None and value is None, "missing_gradient": True} - if ref is None or value is None - else compare(ref, value, args) - ) - checks["pass"] = ( - checks["identical_sampled_anchors"] - and all(checks[key]["pass"] for key in ["loss", "loss_terms"]) - and all(row["pass"] for row in checks["gradients"].values()) - ) - model.zero_grad(set_to_none=True) - return checks - - -def measure(strategy, initial_state, features, lengths, args, algorithm, mode): - model = strategy.trainable_module() - model.load_state_dict(initial_state) - model.zero_grad(set_to_none=True) - batch = make_batch(features, mode, algorithm) - optimizer = BF16Optimizer( - model, - lr=args.learning_rate, - warmup_ratio=0.0, - total_steps=args.warmup + args.steps + 1, - lr_scheduler="constant", - ) - torch.cuda.manual_seed(args.seed + 100) - - def step(): - output = strategy.forward_loss(batch, CONTEXT) - output.loss.backward() - optimizer.step() - return output.loss.detach() - - torch.cuda.synchronize() - start = time.perf_counter() - for _ in range(args.warmup): - step() - torch.cuda.synchronize() - warmup_seconds = time.perf_counter() - start - torch.cuda.reset_peak_memory_stats() - baseline_bytes = torch.cuda.memory_allocated() - times = [] - for _ in range(args.steps): - torch.cuda.synchronize() - start = time.perf_counter() - loss = step() - torch.cuda.synchronize() - times.append(time.perf_counter() - start) - mean, median = statistics.mean(times), statistics.median(times) - result = { - "mean_step_ms": mean * 1000, - "p50_step_ms": median * 1000, - "useful_tokens_per_second": sum(lengths) / mean, - "p50_useful_tokens_per_second": sum(lengths) / median, - "peak_allocated_gib": torch.cuda.max_memory_allocated() / 2**30, - "baseline_allocated_gib": baseline_bytes / 2**30, - "warmup_seconds_including_compile": warmup_seconds, - "step_ms": [duration * 1000 for duration in times], - "final_loss": float(loss), - } - optimizer = batch = None - model.zero_grad(set_to_none=True) - gc.collect() - torch.cuda.empty_cache() - return result - - -def provenance(): - root = Path(__file__).resolve().parents[2] - files = [ - "scripts/benchmark_dflash_sequence_packing.py", - "specforge/benchmarks/benchmark_dflash_sequence_packing.py", - "specforge/algorithms/common/dflash_family_model.py", - "specforge/algorithms/common/hidden_states_data.py", - "specforge/modeling/draft/dflash.py", - "specforge/modeling/draft/dflash2.py", - "specforge/modeling/packed_dflash.py", - "specforge/training/strategies/base.py", - ] - result = { - "files_sha256": { - name: hashlib.sha256((root / name).read_bytes()).hexdigest() - for name in files - } - } - try: - result["head"] = subprocess.check_output( - ["git", "rev-parse", "HEAD"], cwd=root, text=True, stderr=subprocess.DEVNULL - ).strip() - except (OSError, subprocess.CalledProcessError): - result["head"] = "unavailable (copied snapshot is identified by file hashes)" - return result - - -def save(report, path): - path.parent.mkdir(parents=True, exist_ok=True) - path.write_text(json.dumps(report, indent=2) + "\n") - - -def main(): - args = parse_args() - if not torch.cuda.is_available(): - raise RuntimeError("CUDA GPU required") - torch.set_num_threads(4) - torch.backends.cuda.matmul.allow_tf32 = False - torch.backends.cudnn.allow_tf32 = False - settings = {**vars(args), "output": str(args.output)} - report = { - "timestamp_utc": datetime.now(timezone.utc).isoformat(), - "source": provenance(), - "settings": settings, - "environment": { - "torch": torch.__version__, - "cuda": torch.version.cuda, - "gpu": torch.cuda.get_device_name(), - "cuda_visible_devices": os.environ.get("CUDA_VISIBLE_DEVICES"), - }, - "scope": "Single GPU; synthetic frozen target components and captured features; GPU-resident hidden features plus CPU integer features; production strategy forward/backward/BF16Optimizer; excludes capture, hidden-feature I/O/H2D, distributed training and serving.", - "cases": [], - } - algorithms = ["dflash", "dflash2"] if args.algorithm == "both" else [args.algorithm] - for algorithm in algorithms: - strategy = build_strategy(args, algorithm) - initial_state = { - name: value.detach().cpu().clone() - for name, value in strategy.trainable_module().state_dict().items() - } - for lengths in args.lengths: - strategy.trainable_module().load_state_dict(initial_state) - features = make_features(args, lengths) - case = { - "algorithm": algorithm, - "lengths": lengths, - "padding_fraction": 1 - sum(lengths) / (len(lengths) * max(lengths)), - "useful_tokens": sum(lengths), - } - report["cases"].append(case) - print(json.dumps({"case_start": algorithm, "lengths": lengths}), flush=True) - if not args.skip_correctness: - case["correctness"] = correctness(strategy, features, args, algorithm) - save(report, args.output) - print( - json.dumps( - { - "correctness": case["correctness"]["pass"], - "anchors": case["correctness"]["sampled_anchor_count"], - } - ), - flush=True, - ) - if not case["correctness"]["pass"]: - raise AssertionError(f"Packed parity failed; inspect {args.output}") - if not args.correctness_only: - for mode in ["padded", "packed"]: - case[mode] = measure( - strategy, - initial_state, - features, - lengths, - args, - algorithm, - mode, - ) - save(report, args.output) - print( - json.dumps( - {"algorithm": algorithm, "mode": mode, **case[mode]} - ), - flush=True, - ) - case["p50_speedup"] = ( - case["padded"]["p50_step_ms"] / case["packed"]["p50_step_ms"] - ) - case["mean_speedup"] = ( - case["padded"]["mean_step_ms"] / case["packed"]["mean_step_ms"] - ) - save(report, args.output) - del features - del strategy, initial_state - gc.collect() - torch.cuda.empty_cache() - print(f"Report written to {args.output}", flush=True) - - -if __name__ == "__main__": - main() diff --git a/specforge/benchmarks/benchmark_online_sequence_packing.py b/specforge/benchmarks/benchmark_online_sequence_packing.py deleted file mode 100644 index 36d681cca..000000000 --- a/specforge/benchmarks/benchmark_online_sequence_packing.py +++ /dev/null @@ -1,1166 +0,0 @@ -"""Measure an overlapping SGLang -> Mooncake -> online trainer pipeline. - -The server and Mooncake master must already be running. A fresh bounded producer -and consumer run concurrently for every arm; features are never precaptured. -The primary interval starts at the first HTTP capture dispatch and ends after -the final optimizer update, synchronous durable acknowledgement and CUDA sync. -Configured periodic checkpoints occur inside the primary interval; the final -checkpoint/cleanup is outside it and included in fit wall time. All checkpoint -time is also reported separately. Warmup runs precede measured ABBA arms. -""" - -from __future__ import annotations - -import argparse -import gc -import hashlib -import json -import math -import os -import random -import shutil -import sqlite3 -import statistics -import threading -import time -import uuid -from datetime import datetime, timezone -from pathlib import Path - -import torch -from transformers import AutoConfig, Qwen3Config - -from specforge.algorithms.builtin import builtin_algorithm_registry -from specforge.algorithms.common.dflash_family_model import OnlineDFlashModel -from specforge.inference.adapters.server_capture import ( - ServerCaptureSchema, - SGLangServerCaptureAdapter, -) -from specforge.launch import build_disagg_online_consumer, build_disagg_online_producer -from specforge.modeling.draft.dflash import DFlashDraftModel -from specforge.modeling.draft.dflash2 import DFlash2DraftModel -from specforge.modeling.target.target_utils import TargetEmbeddingsAndHead -from specforge.optimizer import BF16Optimizer -from specforge.runtime.data_plane.mooncake_store import MooncakeFeatureStore -from specforge.runtime.data_plane.streaming_ref_channel import StreamingRefChannel -from specforge.training.checkpoint import STATE_FILE - - -def parse_args(argv=None): - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--server-url", required=True) - parser.add_argument("--target-model", required=True) - parser.add_argument("--draft-config", type=Path) - parser.add_argument( - "--prompts-path", - type=Path, - help="Pretokenized JSONL input_ids/loss_mask in the exact desired order", - ) - parser.add_argument( - "--algorithm", choices=("dflash", "dflash2", "both"), default="both" - ) - parser.add_argument( - "--capture-layers", - help="Comma-separated target layer IDs; defaults to draft config", - ) - parser.add_argument( - "--draft-layers", type=int, default=2, help="Used only without --draft-config" - ) - parser.add_argument("--lengths", default="128,256,512,2048") - parser.add_argument("--batch-size", type=int, default=4) - parser.add_argument("--accumulation-steps", type=int, default=1) - parser.add_argument("--anchors", type=int, default=512) - parser.add_argument("--steps", type=int, default=32) - parser.add_argument( - "--warmup-steps", - type=int, - help="Defaults to a full untimed replay of all measured steps", - ) - parser.add_argument( - "--repeats", type=int, default=3, help="Number of ABBA blocks (4 runs each)" - ) - parser.add_argument("--seed", type=int, default=1729) - parser.add_argument("--prompt-fraction", type=float, default=0.25) - parser.add_argument("--learning-rate", type=float, default=1e-4) - parser.add_argument("--objective-chunk-blocks", type=int, default=128) - parser.add_argument("--capture-batch-size", type=int) - parser.add_argument( - "--backlog", - type=int, - default=8, - help="High watermark refs; one capture batch may overshoot", - ) - parser.add_argument("--log-interval", type=int, default=50) - parser.add_argument("--save-interval", type=int, default=0) - parser.add_argument( - "--teacher-metrics", action=argparse.BooleanOptionalAction, default=True - ) - parser.add_argument("--dataloader-workers", type=int, default=4) - parser.add_argument( - "--receive-buffers", choices=("pageable", "pinned"), default="pinned" - ) - parser.add_argument("--segment-mib", type=int, default=1024) - parser.add_argument("--local-buffer-mib", type=int, default=256) - parser.add_argument("--request-timeout", type=float, default=300) - parser.add_argument("--dist-port", type=int, default=29712) - parser.add_argument("--keep-checkpoints", action="store_true") - parser.add_argument("--work-dir", type=Path, required=True) - parser.add_argument("--output", type=Path, required=True) - args = parser.parse_args(argv) - args.warmup_steps = args.steps if args.warmup_steps is None else args.warmup_steps - args.lengths = [int(value) for value in args.lengths.split(",")] - if args.capture_layers: - args.capture_layers = [int(value) for value in args.capture_layers.split(",")] - args.capture_batch_size = args.capture_batch_size or args.batch_size - positive = ( - args.batch_size, - args.accumulation_steps, - args.anchors, - args.steps, - args.warmup_steps, - args.repeats, - args.capture_batch_size, - ) - if min(positive) < 1 or not args.lengths or min(args.lengths) < 4: - parser.error("batch/step/anchor counts must be positive and lengths >= 4") - if not 0 <= args.prompt_fraction < 1: - parser.error("prompt-fraction must be in [0,1)") - quantum = args.batch_size * args.accumulation_steps - if args.backlog < 2 * quantum: - parser.error("backlog must be at least twice the optimizer sample quantum") - if args.log_interval < 1 or args.save_interval < 0: - parser.error("log-interval must be positive and save-interval nonnegative") - return args - - -def _draft_config(args, target_config, architecture): - if args.draft_config: - payload = json.loads(args.draft_config.read_text()) - else: - payload = target_config.to_dict() - payload.update( - num_hidden_layers=args.draft_layers, - num_target_layers=target_config.num_hidden_layers, - layer_types=["full_attention"] * args.draft_layers, - block_size=16, - ) - method = dict(payload.get("dflash_config") or {}) - layers = args.capture_layers or method.get("target_layer_ids") - if not layers: - raise ValueError( - "supply --capture-layers or a draft config with target_layer_ids" - ) - if min(layers) < 0 or max(layers) >= target_config.num_hidden_layers: - raise ValueError("capture layers are outside the target model") - if int(payload["hidden_size"]) != int(target_config.hidden_size) or int( - payload["vocab_size"] - ) != int(target_config.vocab_size): - raise ValueError("draft hidden size and vocabulary must match the target") - method["target_layer_ids"] = list(layers) - method.setdefault("mask_token_id", int(target_config.vocab_size) - 1) - if architecture == "dflash2": - method.setdefault("conv_group_size", 32) - method.setdefault("conv_kernel_size", 4) - method.setdefault("selector_rank", 16) - method.setdefault("selector_top_k", 16) - payload["architectures"] = [ - "DFlash2DraftModel" if architecture == "dflash2" else "DFlashDraftModel" - ] - payload["dflash_config"] = method - payload["num_target_layers"] = int(target_config.num_hidden_layers) - payload["attention_dropout"] = 0.0 - config = Qwen3Config(**payload) - config._attn_implementation = "flex_attention" - return config - - -def _fingerprint(model): - """Check every parameter's shape and deterministic boundary values cheaply.""" - digest = hashlib.sha256() - for name, value in model.state_dict().items(): - digest.update(f"{name}:{tuple(value.shape)}:{value.dtype}".encode()) - flat = value.detach().reshape(-1) - digest.update( - torch.cat((flat[:64], flat[-64:])).float().cpu().numpy().tobytes() - ) - return digest.hexdigest() - - -def _parameter_inventory(model): - """Inventory unique Parameters before wrapping; component counts may overlap.""" - named = dict(model.named_parameters()) - - def counts(parameters): - unique = {id(parameter): parameter for parameter in parameters}.values() - unique = list(unique) - total = sum(parameter.numel() for parameter in unique) - trainable = sum( - parameter.numel() for parameter in unique if parameter.requires_grad - ) - return { - "tensors": len(unique), - "total": total, - "trainable": trainable, - "frozen": total - trainable, - } - - aliases = {} - for name, parameter in model.named_parameters(remove_duplicate=False): - aliases.setdefault(id(parameter), []).append(name) - components = { - "draft": model.draft_model, - "target_embedding": model.embed_tokens, - "target_lm_head": model.lm_head, - } - return named, { - "whole_model_unique": counts(named.values()), - "components": { - name: counts(module.parameters()) for name, module in components.items() - }, - "component_count_note": "Each component deduplicates Parameters; tied embedding/head weights appear in both component counts, but only once in whole_model_unique.", - "draft_layer_count": len(model.draft_model.layers), - "parameters": [ - { - "name": name, - "aliases": aliases[id(parameter)], - "shape": list(parameter.shape), - "numel": parameter.numel(), - "dtype": str(parameter.dtype), - "trainable": parameter.requires_grad, - } - for name, parameter in named.items() - ], - } - - -def _optimizer_coverage(model, named, optimizer): - """Check original-parameter and FP32-master identities, not just counts.""" - draft_ids = { - id(parameter) - for parameter in model.draft_model.parameters() - if parameter.requires_grad - } - target_ids = { - id(parameter) - for component in (model.embed_tokens, model.lm_head) - for parameter in component.parameters() - } - trainable_ids = { - id(parameter) for parameter in named.values() if parameter.requires_grad - } - optimizer_ids = [id(parameter) for parameter in optimizer.model_params] - master_ids = [id(parameter) for parameter in optimizer.fp32_params] - adam_ids = [ - id(parameter) - for group in optimizer.optimizer.param_groups - for parameter in group["params"] - ] - if ( - set(optimizer_ids) != draft_ids - or draft_ids != trainable_ids - or len(optimizer_ids) != len(draft_ids) - or target_ids & set(optimizer_ids) - ): - raise AssertionError( - "optimizer must include every trainable draft Parameter exactly once and no target Parameter" - ) - if ( - len(master_ids) != len(optimizer_ids) - or len(adam_ids) != len(master_ids) - or len(set(master_ids)) != len(master_ids) - or set(adam_ids) != set(master_ids) - or any( - parameter.shape != master.shape or master.dtype != torch.float32 - for parameter, master in zip(optimizer.model_params, optimizer.fp32_params) - ) - ): - raise AssertionError( - "AdamW must optimize one matching FP32 master per draft Parameter" - ) - return { - "all_trainable_draft_parameters_included": True, - "target_parameters_excluded": True, - "one_fp32_adamw_master_per_trainable_parameter": True, - "optimized_tensors": len(optimizer_ids), - "optimized_elements": sum( - parameter.numel() for parameter in optimizer.model_params - ), - "parameter_names": [ - name for name, parameter in named.items() if id(parameter) in draft_ids - ], - } - - -def _parameter_sample_indices(numel, *, device="cpu"): - """Integer interpolation keeps large tensor endpoints exact and in bounds.""" - count = min(128, numel) - return ( - torch.arange(count, dtype=torch.int64, device=device) - * max(0, numel - 1) - // max(1, count - 1) - ) - - -def _parameter_samples(named): - """Sample up to 128 evenly spaced values per tensor outside pipeline timing.""" - snapshots = {} - for name, parameter in named.items(): - flat = parameter.detach().reshape(-1) - indices = _parameter_sample_indices(flat.numel(), device=flat.device) - values = flat[indices].float().cpu() - snapshots[name] = { - "sha256": hashlib.sha256(values.numpy().tobytes()).hexdigest(), - "sampled_elements": values.numel(), - "sampled_values_finite": bool(torch.isfinite(values).all()), - } - return snapshots - - -def _parameter_update_evidence(named, before, layer_count): - """Full finiteness scan plus sampled change evidence, after the timed fit.""" - after = _parameter_samples(named) - trainable = [ - (name, parameter) - for name, parameter in named.items() - if parameter.requires_grad - ] - finite = ( - torch.stack( - [torch.isfinite(parameter.detach()).all() for _, parameter in trainable] - ) - .cpu() - .tolist() - ) - if not all(finite): - raise AssertionError("training left nonfinite values in a trainable Parameter") - changed = {name: before[name]["sha256"] != after[name]["sha256"] for name in named} - frozen_changed = [ - name - for name, parameter in named.items() - if not parameter.requires_grad and changed[name] - ] - if frozen_changed: - raise AssertionError(f"frozen target/draft samples changed: {frozen_changed}") - layers = [] - for index in range(layer_count): - prefix = f"draft_model.layers.{index}." - layer_names = [name for name, _ in trainable if name.startswith(prefix)] - changed_names = [name for name in layer_names if changed[name]] - layers.append( - { - "layer": index, - "trainable_tensors": len(layer_names), - "changed_sampled_tensors": changed_names, - "sampled_update_observed": bool(changed_names), - } - ) - return { - "timing": "Snapshots before first capture and after fit; full finite scan after fit. No measured-step hooks or synchronizations.", - "semantics": "All trainable elements are checked finite. Change hashes cover at most 128 evenly spaced elements per tensor; an unchanged hash does not prove an entire tensor was unchanged. Zero initial gradients or sparse selector updates are legitimate.", - "all_trainable_elements_finite": True, - "all_frozen_parameter_samples_unchanged": True, - "all_decoder_layers_have_sampled_updates": all( - layer["sampled_update_observed"] for layer in layers - ), - "layers": layers, - "parameters": { - name: { - "before": before[name], - "after": after[name], - "sampled_update_observed": changed[name], - } - for name in named - }, - } - - -def _observe_first_warmup_step(optimizer, named): - """Observe presence without reading gradient tensors; never used in timed arms.""" - evidence = { - "performed": False, - "scope": "first optimizer step of untimed warmup only", - } - original_step = optimizer.step - names = {id(parameter): name for name, parameter in named.items()} - - def step(**kwargs): - evidence["gradient_present"] = { - names[id(parameter)]: parameter.grad is not None - for parameter in optimizer.model_params - } - result = original_step(**kwargs) - evidence["performed"] = True - evidence["_gradient_norm"] = optimizer.last_grad_norm.detach().clone() - optimizer.step = original_step - return result - - optimizer.step = step - return evidence - - -def _optimizer_step_evidence(optimizer, named, expected_steps): - names = {id(parameter): name for name, parameter in named.items()} - steps = {} - for parameter, master in zip(optimizer.model_params, optimizer.fp32_params): - value = optimizer.optimizer.state.get(master, {}).get("step", 0) - steps[names[id(parameter)]] = int( - value.item() if isinstance(value, torch.Tensor) else value - ) - return { - "expected_steps": expected_steps, - "adamw_steps_by_parameter": steps, - "all_trainable_parameters_received_every_optimizer_step": all( - value == expected_steps for value in steps.values() - ), - "semantics": "AdamW step counters prove optimizer participation, including legitimate zero gradients; they do not imply every element changed.", - } - - -def _model(args, target_config, architecture): - torch.manual_seed(args.seed) - config = _draft_config(args, target_config, architecture) - draft_type = DFlash2DraftModel if architecture == "dflash2" else DFlashDraftModel - draft = draft_type(config).to(dtype=torch.bfloat16) - fingerprint = _fingerprint(draft) - components = TargetEmbeddingsAndHead.from_pretrained( - args.target_model, - device="cuda", - dtype=torch.bfloat16, - ).requires_grad_(False) - model = OnlineDFlashModel( - draft_model=draft.cuda(), - target_lm_head=components.lm_head, - target_embed_tokens=components.embed_tokens, - mask_token_id=int(config.dflash_config["mask_token_id"]), - block_size=int(config.block_size), - attention_backend="flex_attention", - num_anchors=args.anchors, - objective_chunk_blocks=args.objective_chunk_blocks, - teacher_metrics=args.teacher_metrics, - ).cuda() - return model, config, fingerprint - - -def _prompts(args, steps, vocab_size): - count = steps * args.batch_size * args.accumulation_steps - generator = torch.Generator().manual_seed(args.seed + 1) - desired = [] - digest = hashlib.sha256() - useful_tokens = supervised_tokens = 0 - supplied = None - if args.prompts_path: - with args.prompts_path.open() as stream: - supplied = [json.loads(line) for line in stream if line.strip()] - if len(supplied) < count: - raise ValueError( - f"{args.prompts_path} has {len(supplied)} prompts; this run requires {count}" - ) - for index in range(count): - if supplied is None: - length = args.lengths[index % len(args.lengths)] - ids = torch.randint(0, vocab_size, (length,), generator=generator).tolist() - prompt_length = min(int(length * args.prompt_fraction), length - 2) - mask = [0] * prompt_length + [1] * (length - prompt_length) - else: - row = supplied[index].get("payload", supplied[index]) - ids, mask = list(row["input_ids"]), list(row["loss_mask"]) - length = len(ids) - if ( - len(mask) != length - or not ids - or not all( - isinstance(token, int) and 0 <= token < vocab_size for token in ids - ) - ): - raise ValueError(f"invalid token IDs or mask shape in prompt {index}") - if not any(a > 0.5 and b > 0.5 for a, b in zip(mask, mask[1:])): - raise ValueError( - f"prompt {index} has no two consecutive supervised tokens" - ) - payload = {"input_ids": ids, "loss_mask": mask} - digest.update(json.dumps(payload, separators=(",", ":")).encode()) - desired.append( - { - "task_id": f"prompt-{index:08d}", - "source_id": ( - "pretokenized-jsonl" - if supplied is not None - else "synthetic-length-pattern" - ), - "payload": payload, - "max_length": length, - } - ) - useful_tokens += length - supervised_tokens += sum(mask) - # Compensate for the canonical producer's deterministic shuffle so every - # warm/measured microbatch sees the same repeating length pattern. Capture - # order is checked below; an upstream ordering change fails this benchmark. - order = list(range(count)) - random.Random(args.seed).shuffle(order) - prompts = [None] * count - for position, shuffled_index in enumerate(order): - prompts[shuffled_index] = desired[position] - return prompts, digest.hexdigest(), useful_tokens, supervised_tokens - - -def _compiler_counters(): - from torch._dynamo.utils import counters - - return { - str(namespace): {str(key): int(value) for key, value in counts.items()} - for namespace, counts in counters.items() - } - - -def _counter_delta(before, after): - return { - namespace: { - key: after.get(namespace, {}).get(key, 0) - - before.get(namespace, {}).get(key, 0) - for key in set(before.get(namespace, {})) | set(after.get(namespace, {})) - if after.get(namespace, {}).get(key, 0) - != before.get(namespace, {}).get(key, 0) - } - for namespace in set(before) | set(after) - } - - -class _TimedSource: - def __init__(self, adapter): - self.adapter = adapter - self.calls = [] - self.task_ids = [] - self.first_dispatch = None - self.request_digest = hashlib.sha256() - original_post = adapter.post_fn - - def timed_post(url, *, json_body, timeout): - # Hash the actual HTTP inputs and masks after canonical request - # construction, excluding fresh transport/cache namespaces. - normalized = { - key: value - for key, value in json_body.items() - if key not in ("extra_key", "spec_capture") - } - normalized["spec_capture"] = [ - { - **{ - key: value - for key, value in capture.items() - if key not in ("store_id", "sample_id") - }, - "sample_id": capture["sample_id"].split(":", 1)[1], - } - for capture in json_body["spec_capture"] - ] - self.request_digest.update(json.dumps(normalized, sort_keys=True).encode()) - begin = time.perf_counter() - if self.first_dispatch is None: - self.first_dispatch = begin - result = original_post(url, json_body=json_body, timeout=timeout) - self.calls.append( - { - "start": begin, - "end": time.perf_counter(), - "samples": len(json_body["spec_capture"]), - } - ) - return result - - adapter.post_fn = timed_post - - def produce_refs(self, tasks, *, capture): - result = self.adapter.produce_refs(tasks, capture=capture) - self.task_ids.extend(task.task_id for task in tasks) - return result - - def __getattr__(self, name): - return getattr(self.adapter, name) - - -class _TimedChannel(StreamingRefChannel): - def __init__(self, path): - super().__init__(path) - self.publications = [] - self.max_backlog = 0 - - def begin_publish(self, refs): - transaction = super().begin_publish(refs) - channel = self - - class Transaction: - def commit(self): - result = transaction.commit() - backlog = channel.in_flight_remote() - channel.max_backlog = max(channel.max_backlog, backlog) - channel.publications.append( - { - "time": time.perf_counter(), - "samples": len(refs), - "backlog": backlog, - "task_ids": [ref.source_task_id for ref in refs], - } - ) - return result - - def __getattr__(self, name): - return getattr(transaction, name) - - return Transaction() - - -def _store(args, run_id): - required = ("MOONCAKE_MASTER_SERVER_ADDR", "MOONCAKE_METADATA_SERVER") - for name in required: - if not os.environ.get(name): - raise ValueError(f"set {name} for the existing server's Mooncake master") - return MooncakeFeatureStore( - store_id=run_id, - retain_on_release=True, - receive_buffers=args.receive_buffers, - setup_kwargs={ - "local_hostname": os.environ.get("MOONCAKE_LOCAL_HOSTNAME", "127.0.0.1"), - "metadata_server": os.environ["MOONCAKE_METADATA_SERVER"], - "master_server_addr": os.environ["MOONCAKE_MASTER_SERVER_ADDR"], - "global_segment_size": args.segment_mib << 20, - "local_buffer_size": args.local_buffer_mib << 20, - "protocol": os.environ.get("MOONCAKE_PROTOCOL", "tcp"), - "rdma_devices": os.environ.get("MOONCAKE_RDMA_DEVICES", ""), - }, - ) - - -def run_pipeline(args, target_config, architecture, packing, steps, label): - run_id = f"{uuid.uuid4().hex[:12]}-{architecture}-{label}-{'packed' if packing else 'padded'}" - work = args.work_dir / run_id - work.mkdir(parents=True, exist_ok=False) - assembled_at = time.perf_counter() - model, config, fingerprint = _model(args, target_config, architecture) - named_parameters, parameter_inventory = _parameter_inventory(model) - before_parameter_samples = _parameter_samples(named_parameters) - prompts, prompt_hash, useful_tokens, supervised_tokens = _prompts( - args, steps, config.vocab_size - ) - algorithm = builtin_algorithm_registry().resolve("dflash") - provider = algorithm.providers.server_streaming_for("text") - layout = provider.layout - schema = ServerCaptureSchema( - aux_feature=layout.aux_feature, - last_hidden_feature=( - layout.last_hidden_feature if args.teacher_metrics else None - ), - passthrough=layout.passthrough, - attention_mask_feature=layout.attention_mask_feature, - ) - store = _store(args, run_id) - channel = _TimedChannel(str(work / "refs.jsonl")) - source = _TimedSource( - SGLangServerCaptureAdapter( - args.server_url, - store, - run_id=run_id, - algorithm="dflash", - schema=schema, - timeout_s=args.request_timeout, - target_model_version=args.target_model, - ) - ) - producer_thread = None - trainer = None - stop = threading.Event() - producer_state = {} - acknowledgements = [] - checkpoints = [] - logged = [] - loader_counter_windows = [] - fit_started = False - try: - trainer = build_disagg_online_consumer( - algorithm=algorithm, - feature_store=store, - channel=channel, - draft_model=model, - optimizer_factory=lambda module: BF16Optimizer( - module, - lr=args.learning_rate, - max_grad_norm=0.5, - warmup_ratio=0.0, - total_steps=steps, - ), - run_id=run_id, - output_dir=str(work / "output"), - batch_size=args.batch_size, - accumulation_steps=args.accumulation_steps, - max_steps=steps, - sequence_packing=packing, - save_interval=args.save_interval, - log_interval=args.log_interval, - metadata_db_path=str(work / "consumer.sqlite"), - async_ack=False, - idle_timeout_s=args.request_timeout * 2, - dataloader_num_workers=args.dataloader_workers, - logger=lambda metrics, step: logged.append( - {"step": step, "metrics": dict(metrics)} - ), - ) - optimizer = trainer.backend.optimizer - optimizer_coverage = _optimizer_coverage(model, named_parameters, optimizer) - first_step_evidence = ( - _observe_first_warmup_step(optimizer, named_parameters) - if label == "warmup" - else { - "performed": False, - "scope": "measured arms have no gradient-observation hook; see matching warmup", - } - ) - backend_evidence = { - "wrapper_kind": trainer.backend._wrapper_kind, - "configured_sharding_strategy": trainer.backend.parallel_config.sharding_strategy, - "effective_sharding_strategy": str( - getattr(trainer.backend.module, "sharding_strategy", "not applicable") - ), - } - original_snapshot = trainer._loader.perf_counters_snapshot - - def tracked_snapshot(reset=False): - snapshot = original_snapshot(reset=reset) - if reset: - loader_counter_windows.append(snapshot) - return snapshot - - trainer._loader.perf_counters_snapshot = tracked_snapshot - original_ack = trainer._controller.ack_fn - - def timed_ack(ids, step): - original_ack(ids, step) - # Add a CUDA barrier only at the final timing endpoint. Intermediate - # event timestamps record canonical durable-ack completion. - if step == steps: - torch.cuda.synchronize() - acknowledgements.append( - {"step": step, "time": time.perf_counter(), "sample_ids": list(ids)} - ) - - trainer._controller.ack_fn = timed_ack - original_checkpoint = trainer._controller.save_checkpoint - - def timed_checkpoint(step): - start = time.perf_counter() - result = original_checkpoint(step) - torch.cuda.synchronize() - checkpoints.append({"step": step, "seconds": time.perf_counter() - start}) - return result - - trainer._controller.save_checkpoint = timed_checkpoint - _, drive = build_disagg_online_producer( - algorithm=algorithm, - prompts=prompts, - feature_store=store, - channel=channel, - run_id=run_id, - target_hidden_size=config.hidden_size, - target_vocab_size=config.vocab_size, - target_repr=provider.target_representation, - aux_hidden_state_layer_ids=config.dflash_config["target_layer_ids"], - feature_source=source, - lease=args.capture_batch_size, - producer_concurrency=1, - num_rollout_workers=1, - in_flight_high_watermark=args.backlog, - in_flight_low_watermark=max( - args.batch_size * args.accumulation_steps, args.backlog // 2 - ), - prompt_seed=args.seed, - prompt_ingest_batch_size=max(64, args.backlog), - backpressure_poll_s=0.01, - peer_wait_timeout_s=args.request_timeout * 2, - max_prompt_attempts=1, - max_worker_failures=1, - ) - - def produce(): - try: - producer_state["samples"] = drive( - should_stop=lambda: stop.is_set() or channel.consumer_stopped() - ) - except BaseException as exc: - producer_state["error"] = repr(exc) - channel.fail(repr(exc)) - finally: - producer_state["end"] = time.perf_counter() - - torch.manual_seed(args.seed + 2) - torch.cuda.synchronize() - compiler_before = _compiler_counters() - setup_seconds = time.perf_counter() - assembled_at - fit_begin = time.perf_counter() - producer_thread = threading.Thread( - target=produce, name=f"capture-{run_id}", daemon=True - ) - producer_thread.start() - fit_started = True - final_step = trainer.fit() - torch.cuda.synchronize() - fit_end = time.perf_counter() - producer_thread.join(timeout=args.request_timeout + 10) - if producer_thread.is_alive(): - raise RuntimeError("producer did not finish after the final optimizer step") - if "error" in producer_state: - raise RuntimeError(producer_state["error"]) - expected_count = steps * args.batch_size * args.accumulation_steps - if producer_state.get("samples") != expected_count: - raise AssertionError("producer did not publish the complete workload") - if not channel.consumer_stopped() or channel.consumer_failure() is not None: - raise AssertionError("consumer did not publish a clean completion") - expected_order = [f"prompt-{i:08d}" for i in range(expected_count)] - published_order = [ - task for event in channel.publications for task in event["task_ids"] - ] - if source.task_ids != expected_order or published_order != expected_order: - raise AssertionError( - "capture/publication ordering changed or capture retried; comparison invalid" - ) - with sqlite3.connect(work / "consumer.sqlite") as connection: - acked = connection.execute( - "SELECT sample_id FROM acked ORDER BY sample_id" - ).fetchall() - ack_ids = [ - sample for event in acknowledgements for sample in event["sample_ids"] - ] - if ( - final_step != steps - or len(ack_ids) != expected_count - or len(acked) != expected_count - ): - raise AssertionError( - "optimizer steps or durable sample acknowledgements do not match the workload" - ) - if ( - set(ack_ids) != {row[0] for row in acked} - or len(set(ack_ids)) != expected_count - ): - raise AssertionError( - "sample identities changed or duplicate acknowledgements occurred" - ) - ack_order = [sample.split(":", 1)[1] for sample in ack_ids] - if ack_order != expected_order: - raise AssertionError("consumed sample order or logical batching changed") - final_checkpoint = work / "output" / f"{run_id}-step{steps}" / STATE_FILE - if not final_checkpoint.is_file(): - raise AssertionError("canonical final checkpoint is missing") - state = torch.load( - final_checkpoint, map_location="cpu", weights_only=False, mmap=True - ) - checkpoint_proof = { - key: state[key] - for key in ("global_step", "epoch", "epoch_batch", "epoch_samples") - } - del state - if ( - checkpoint_proof["global_step"] != steps - or checkpoint_proof["epoch_samples"] != expected_count - ): - raise AssertionError( - "checkpoint step/sample progress disagrees with durable acks" - ) - loader_counter_windows.append(original_snapshot(reset=False)) - loader_counters = { - key: sum(window.get(key, 0) for window in loader_counter_windows) - for key in loader_counter_windows[-1] - } - if trainer.micro_step != steps * args.accumulation_steps: - raise AssertionError("logical microbatch count changed") - final_loss = trainer._controller._last_result.loss - final_loss = ( - float(final_loss.detach().cpu()) - if isinstance(final_loss, torch.Tensor) - else float(final_loss) - ) - if not math.isfinite(final_loss): - raise AssertionError("nonfinite final training loss") - start = source.first_dispatch - finish = acknowledgements[-1]["time"] - capture_end = max(call["end"] for call in source.calls) - if start is None or finish <= start: - raise AssertionError("invalid pipeline timing boundaries") - elapsed = finish - start - # These checks run after the final ack AND the checkpoint/fit endpoint; - # their tensor reads and finite scans affect neither reported interval. - parameter_updates = _parameter_update_evidence( - named_parameters, - before_parameter_samples, - parameter_inventory["draft_layer_count"], - ) - optimizer_steps = _optimizer_step_evidence(optimizer, named_parameters, steps) - if first_step_evidence["performed"]: - first_step_evidence["global_gradient_norm"] = float( - first_step_evidence.pop("_gradient_norm").cpu() - ) - first_step_evidence["all_trainable_gradients_present"] = all( - first_step_evidence["gradient_present"].values() - ) - first_step_evidence["global_gradient_norm_finite"] = math.isfinite( - first_step_evidence["global_gradient_norm"] - ) - first_step_evidence["semantics"] = ( - "Presence is observed before BF16Optimizer clears gradients. Its finite global norm check covers all present gradients. Zero gradients are valid at initialization, including DFlash2 bilinear selector factors." - ) - result = { - "run_id": run_id, - "architecture": architecture, - "packing": packing, - "label": label, - "optimizer_steps": steps, - "microsteps": trainer.micro_step, - "samples": expected_count, - "durable_acked_samples": len(acked), - "useful_tokens": useful_tokens, - "supervised_tokens": supervised_tokens, - "pipeline_seconds": elapsed, - "useful_tokens_per_second": useful_tokens / elapsed, - "samples_per_second": expected_count / elapsed, - "capture_finished_seconds": capture_end - start, - "producer_finished_seconds": producer_state["end"] - start, - "trainer_fit_seconds_from_first_capture": fit_end - start, - "fit_wall_seconds": fit_end - fit_begin, - "model_and_runtime_setup_seconds": setup_seconds, - "checkpoint_seconds": sum(item["seconds"] for item in checkpoints), - "checkpoint_events": checkpoints, - "checkpoint_proof": checkpoint_proof, - "backend": backend_evidence, - "parameter_inventory": parameter_inventory, - "optimizer_coverage": optimizer_coverage, - "first_warmup_step_gradient_evidence": first_step_evidence, - "parameter_update_evidence": parameter_updates, - "optimizer_step_evidence": optimizer_steps, - "loader_perf_counters": loader_counters, - "loader_perf_note": "wait_producer_s and wait_fetch_s are consumer blocking; fetch_s overlaps training and is not additive with wall time", - "final_loss": final_loss, - "max_observed_backlog_refs": channel.max_backlog, - "prompt_sha256": prompt_hash, - "initial_draft_boundary_fingerprint": fingerprint, - "actual_capture_task_order": source.task_ids, - "actual_request_sha256": source.request_digest.hexdigest(), - "actual_consumed_task_order": ack_order, - "capture_calls": [ - {**event, "start": event["start"] - start, "end": event["end"] - start} - for event in source.calls - ], - "publication_events": [ - {**event, "time": event["time"] - start} - for event in channel.publications - ], - "optimizer_ack_events": [ - { - "step": event["step"], - "time": event["time"] - start, - "sample_count": len(event["sample_ids"]), - "task_ids": [ - sample.split(":", 1)[1] for sample in event["sample_ids"] - ], - } - for event in acknowledgements - ], - "capture_calls_after_first_optimizer": sum( - call["start"] > acknowledgements[0]["time"] for call in source.calls - ), - "sustained_live_overlap_established": any( - call["start"] > acknowledgements[0]["time"] for call in source.calls - ), - "checkpoint_policy": { - "save_interval": args.save_interval, - "canonical_final_save": True, - }, - "resolved_draft_config": config.to_dict(), - "logged": logged, - "compiler_counters_before": compiler_before, - "compiler_counters_after": _compiler_counters(), - "compiler_counter_delta": _counter_delta( - compiler_before, _compiler_counters() - ), - "full_warmup_replay": args.warmup_steps >= args.steps, - } - (work / "result.json").write_text(json.dumps(result, indent=2, default=str)) - return result - finally: - stop.set() - if producer_thread is not None and producer_thread.is_alive(): - channel.mark_consumer_failed("benchmark consumer stopped") - producer_thread.join(timeout=args.request_timeout + 10) - if trainer is not None and not fit_started: - trainer._loader.close() - if trainer._on_fit_finally is not None: - trainer._on_fit_finally() - store.discard_external_attempts(reason="benchmark-finished") - store.close() - if not args.keep_checkpoints and (work / "output").exists(): - shutil.rmtree(work / "output") - del trainer, model - gc.collect() - torch.cuda.empty_cache() - - -def _summary(results): - summary = {} - for architecture in sorted({row["architecture"] for row in results}): - arms = {} - for packing in (False, True): - rows = [ - row - for row in results - if row["architecture"] == architecture and row["packing"] == packing - ] - if rows: - arms["packed" if packing else "padded"] = { - "runs": len(rows), - "median_pipeline_seconds": statistics.median( - row["pipeline_seconds"] for row in rows - ), - "median_useful_tokens_per_second": statistics.median( - row["useful_tokens_per_second"] for row in rows - ), - "median_fit_seconds_with_checkpoint": statistics.median( - row["trainer_fit_seconds_from_first_capture"] for row in rows - ), - } - if len(arms) == 2: - arms["pipeline_speedup"] = ( - arms["padded"]["median_pipeline_seconds"] - / arms["packed"]["median_pipeline_seconds"] - ) - arms["fit_speedup_with_checkpoint"] = ( - arms["padded"]["median_fit_seconds_with_checkpoint"] - / arms["packed"]["median_fit_seconds_with_checkpoint"] - ) - summary[architecture] = arms - return summary - - -def main(argv=None): - args = parse_args(argv) - if not torch.cuda.is_available(): - raise RuntimeError("online pipeline benchmark requires a CUDA consumer") - import requests - import torch.distributed as dist - - requests.get(args.server_url.rstrip("/") + "/health", timeout=10).raise_for_status() - args.work_dir.mkdir(parents=True, exist_ok=True) - args.output.parent.mkdir(parents=True, exist_ok=True) - own_group = not dist.is_initialized() - if own_group: - os.environ.update( - RANK="0", - WORLD_SIZE="1", - LOCAL_RANK="0", - MASTER_ADDR="127.0.0.1", - MASTER_PORT=str(args.dist_port), - ) - torch.cuda.set_device(0) - from specforge.distributed import init_distributed - - init_distributed(timeout=10, tp_size=1) - target_config = AutoConfig.from_pretrained(args.target_model) - architectures = ( - ("dflash", "dflash2") if args.algorithm == "both" else (args.algorithm,) - ) - report = { - "created_utc": datetime.now(timezone.utc).isoformat(), - "settings": { - key: str(value) if isinstance(value, Path) else value - for key, value in vars(args).items() - }, - "torch_version": torch.__version__, - "gpu": torch.cuda.get_device_name(), - "target_config": target_config.to_dict(), - "warmup_runs": [], - "measured_runs": [], - "timing_contract": "first capture dispatch -> final synchronous durable ack + CUDA sync, including configured periodic checkpoints before the final step; final checkpoint and cleanup excluded from pipeline_seconds but included in trainer_fit_seconds_from_first_capture; all checkpoint events separately timed", - "scope": "actual target weights and target embeddings/head; freshly initialized full draft; single-rank canonical trainer with actual wrapper/sharding recorded per run", - "prompt_source": ( - str(args.prompts_path) - if args.prompts_path - else "synthetic deterministic token prompts" - ), - "prompt_file_sha256": ( - hashlib.sha256(args.prompts_path.read_bytes()).hexdigest() - if args.prompts_path - else None - ), - "server_startup": "existing server: startup excluded, not measured by this process", - "capture_cache_policy": "canonical adapter generates a fresh extra_key per request attempt, forcing full prefill", - "producer": "canonical drive_producer in parallel thread, one worker/concurrency=1, bounded ref backlog", - "async_ack": False, - } - - def save(): - report["summary"] = _summary(report["measured_runs"]) - temporary = args.output.with_suffix(args.output.suffix + ".tmp") - temporary.write_text(json.dumps(report, indent=2, default=str)) - temporary.replace(args.output) - - try: - for architecture in architectures: - for packing in (False, True): - row = run_pipeline( - args, - target_config, - architecture, - packing, - args.warmup_steps, - "warmup", - ) - row["cold_first_use_for_mode"] = True - report["warmup_runs"].append(row) - save() - gc.collect() - torch.cuda.empty_cache() - expected = None - for repeat in range(args.repeats): - for index, packing in enumerate((False, True, True, False)): - label = f"repeat{repeat:02d}-arm{index}" - row = run_pipeline( - args, target_config, architecture, packing, args.steps, label - ) - identity = ( - row["prompt_sha256"], - row["initial_draft_boundary_fingerprint"], - row["actual_capture_task_order"], - row["actual_consumed_task_order"], - row["actual_request_sha256"], - ) - if expected is None: - expected = identity - elif identity != expected: - raise AssertionError( - "A/B prompt order or draft initialization differs" - ) - report["measured_runs"].append(row) - save() - gc.collect() - torch.cuda.empty_cache() - print( - json.dumps( - { - key: row[key] - for key in ( - "architecture", - "packing", - "label", - "pipeline_seconds", - "useful_tokens_per_second", - "checkpoint_seconds", - "capture_calls_after_first_optimizer", - ) - } - ), - flush=True, - ) - finally: - save() - if own_group and dist.is_initialized(): - dist.destroy_process_group() - - -if __name__ == "__main__": - main() diff --git a/specforge/benchmarks/benchmark_sequence_packing.py b/specforge/benchmarks/benchmark_sequence_packing.py deleted file mode 100644 index 8cac474be..000000000 --- a/specforge/benchmarks/benchmark_sequence_packing.py +++ /dev/null @@ -1,460 +0,0 @@ -"""Compare padded and packed EAGLE3 training on identical synthetic features. - -This exercises the production collators, Eagle3TrainStrategy, OnlineEagle3Model, -LlamaForCausalLMEagle3, TargetHead preprocessing/projection, and BF16Optimizer. -No model downloads are needed. It measures one GPU with resident input features; -target feature generation, disk loading, H2D transfers, and distributed training -are outside the measurement. Synthetic features do not establish model quality -or speculative serving speedups. - -Run a strict small FP32 gate before the representative BF16 benchmark:: - - PYTHONPATH=. python scripts/benchmark_sequence_packing.py --preset tiny \ - --dtype float32 --correctness-only --output /tmp/packing-correctness.json - PYTHONPATH=. python scripts/benchmark_sequence_packing.py --preset medium \ - --warmup 5 --steps 20 --output /tmp/packing-perf.json - -Each --lengths argument defines one batch (repeat it for several padding ratios). -For example: --lengths 1024,1024,1024,1024 --lengths 128,256,512,2048. -""" - -from __future__ import annotations - -import argparse -import gc -import hashlib -import importlib.metadata -import json -import os -import platform -import statistics -import subprocess -import time -from datetime import datetime, timezone -from pathlib import Path -from types import SimpleNamespace - -import torch -from transformers import LlamaConfig - -from specforge.algorithms.eagle3.data import DataCollatorWithPacking -from specforge.algorithms.eagle3.model import OnlineEagle3Model -from specforge.data.utils import DataCollatorWithPadding -from specforge.modeling.draft.llama3_eagle import LlamaForCausalLMEagle3 -from specforge.modeling.target.target_head import TargetHead -from specforge.optimizer import BF16Optimizer -from specforge.runtime.contracts import TrainBatch -from specforge.training.strategies.base import Eagle3TrainStrategy - - -class SyntheticTargetHead(TargetHead): - """Random frozen head with the production forward and preprocess methods.""" - - def __init__(self, hidden_size: int, vocab_size: int): - torch.nn.Module.__init__(self) - self.hidden_size = hidden_size - self.vocab_size = vocab_size - self.config = SimpleNamespace(hidden_size=hidden_size, vocab_size=vocab_size) - self.fc = torch.nn.Linear(hidden_size, vocab_size, bias=False) - self.freeze_weights() - - -PRESETS = { - "tiny": { - "hidden_size": 128, - "intermediate_size": 256, - "num_heads": 4, - "num_kv_heads": 2, - "vocab_size": 512, - "draft_vocab_size": 256, - "lengths": [[8, 17, 31, 64], [2, 3, 5, 17], [32, 32, 32, 32]], - }, - "medium": { - "hidden_size": 2048, - "intermediate_size": 8192, - "num_heads": 16, - "num_kv_heads": 4, - "vocab_size": 32000, - "draft_vocab_size": 32000, - "lengths": [ - [1024, 1024, 1024, 1024], - [512, 768, 1024, 2048], - [128, 256, 512, 2048], - ], - }, - "large": { - "hidden_size": 4096, - "intermediate_size": 14336, - "num_heads": 32, - "num_kv_heads": 8, - "vocab_size": 32000, - "draft_vocab_size": 32000, - "lengths": [[128, 256, 512, 2048]], - }, -} - - -def parse_args(): - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--preset", choices=PRESETS, default="medium") - for key in PRESETS["tiny"]: - if key != "lengths": - parser.add_argument("--" + key.replace("_", "-"), type=int) - parser.add_argument("--target-hidden-size", type=int) - parser.add_argument("--lengths", action="append", help="comma-separated lengths") - parser.add_argument("--ttt-length", type=int, default=7) - parser.add_argument("--dtype", choices=["float32", "bfloat16"], default="bfloat16") - parser.add_argument("--warmup", type=int, default=5) - parser.add_argument("--steps", type=int, default=20) - parser.add_argument("--seed", type=int, default=1729) - parser.add_argument("--learning-rate", type=float, default=1e-4) - parser.add_argument("--prompt-fraction", type=float, default=0.25) - parser.add_argument("--correctness-only", action="store_true") - parser.add_argument("--skip-correctness", action="store_true") - parser.add_argument("--atol", type=float) - parser.add_argument("--rtol", type=float) - parser.add_argument("--output", type=Path, required=True) - args = parser.parse_args() - preset = PRESETS[args.preset] - for key, value in preset.items(): - if getattr(args, key) is None: - setattr(args, key, value) - if isinstance(args.lengths[0], str): - args.lengths = [[int(item) for item in row.split(",")] for row in args.lengths] - args.target_hidden_size = args.target_hidden_size or args.hidden_size - args.atol = ( - args.atol - if args.atol is not None - else (2e-5 if args.dtype == "float32" else 2e-3) - ) - args.rtol = ( - args.rtol - if args.rtol is not None - else (2e-4 if args.dtype == "float32" else 2e-2) - ) - if any(not row or min(row) < 1 for row in args.lengths): - parser.error("each batch must contain positive sequence lengths") - if args.steps < 1 or args.warmup < 1 or args.ttt_length < 1: - parser.error("steps, warmup, and ttt-length must be positive") - if not 0 <= args.prompt_fraction < 1: - parser.error("prompt-fraction must be in [0, 1)") - if args.draft_vocab_size > args.vocab_size: - parser.error("draft-vocab-size cannot exceed vocab-size") - if args.hidden_size % args.num_heads or args.num_heads % args.num_kv_heads: - parser.error( - "hidden-size / num-heads and num-heads / num-kv-heads must be integral" - ) - if args.correctness_only and args.skip_correctness: - parser.error("correctness-only and skip-correctness are incompatible") - return args - - -def build_strategy(args): - torch.manual_seed(args.seed) - config = LlamaConfig( - hidden_size=args.hidden_size, - target_hidden_size=args.target_hidden_size, - intermediate_size=args.intermediate_size, - num_attention_heads=args.num_heads, - num_key_value_heads=args.num_kv_heads, - num_hidden_layers=1, - vocab_size=args.vocab_size, - draft_vocab_size=args.draft_vocab_size, - max_position_embeddings=max(map(max, args.lengths)) + args.ttt_length, - pad_token_id=0, - attention_dropout=0.0, - rms_norm_eps=1e-5, - tie_word_embeddings=False, - ) - draft = LlamaForCausalLMEagle3(config, attention_backend="flex_attention") - # Exercise the real target-to-draft mapping, including a reduced vocabulary. - selected = torch.randperm(args.vocab_size)[: args.draft_vocab_size].sort().values - draft.t2d.zero_() - draft.t2d[selected] = True - draft.d2t.copy_(selected - torch.arange(args.draft_vocab_size)) - draft.freeze_embedding() - dtype = getattr(torch, args.dtype) - model = OnlineEagle3Model( - draft, length=args.ttt_length, attention_backend="flex_attention" - ).to(device="cuda", dtype=dtype) - head = ( - SyntheticTargetHead(args.target_hidden_size, args.vocab_size) - .to(device="cuda", dtype=dtype) - .eval() - ) - model.train() - return Eagle3TrainStrategy(model, target_head=head) - - -def make_features(args, lengths): - generator = torch.Generator().manual_seed(args.seed + 1) - dtype = getattr(torch, args.dtype) - features = [] - for length in lengths: - loss_mask = torch.ones(1, length, dtype=torch.long) - loss_mask[:, : int(length * args.prompt_fraction)] = 0 - loss_mask[:, -1] = 0 # The production offline normalizer does this. - features.append( - { - "input_ids": torch.randint( - 1, args.vocab_size, (1, length), generator=generator - ), - "attention_mask": torch.ones(1, length, dtype=torch.long), - "loss_mask": loss_mask, - "hidden_state": torch.randn( - 1, length, 3 * args.target_hidden_size, generator=generator - ).to(dtype), - "target": torch.randn( - 1, length, args.target_hidden_size, generator=generator - ).to(dtype), - } - ) - return features - - -def make_batch(features, mode): - collator = ( - DataCollatorWithPadding() if mode == "padded" else DataCollatorWithPacking() - ) - tensors = collator(features) - # Keep packing control metadata on CPU, matching the production loader. - tensors = { - name: ( - value if name in {"sequence_lengths", "loss_denominator"} else value.cuda() - ) - for name, value in tensors.items() - } - return TrainBatch( - sample_ids=[str(i) for i in range(len(features))], - strategy="eagle3", - tensors=tensors, - metadata={"target_repr": "hidden_state"}, - ) - - -def tensor_comparison(reference, actual, args): - reference, actual = reference.float(), actual.float() - delta = actual - reference - reference_norm = float(reference.norm()) - return { - "pass": bool(torch.allclose(reference, actual, atol=args.atol, rtol=args.rtol)), - "max_abs_diff": float(delta.abs().max()), - "relative_l2_diff": float(delta.norm()) / max(reference_norm, 1e-30), - "reference_l2": reference_norm, - } - - -def check_correctness(strategy, features, args): - model = strategy.trainable_module() - outputs = {} - for mode in ("padded", "packed"): - model.zero_grad(set_to_none=True) - batch = make_batch(features, mode) - result = strategy.forward_loss(batch) - result.loss.backward() - outputs[mode] = { - "loss": result.loss.detach().float().cpu(), - "plosses": torch.stack(result.metrics["plosses"]).float().cpu(), - "grads": { - name: None if param.grad is None else param.grad.detach().float().cpu() - for name, param in model.named_parameters() - if param.requires_grad - }, - } - del batch, result - checks = { - key: tensor_comparison(outputs["padded"][key], outputs["packed"][key], args) - for key in ("loss", "plosses") - } - checks["loss_padded"] = float(outputs["padded"]["loss"]) - checks["loss_packed"] = float(outputs["packed"]["loss"]) - checks["gradients"] = {} - for name, reference in outputs["padded"]["grads"].items(): - actual = outputs["packed"]["grads"][name] - checks["gradients"][name] = ( - {"pass": reference is None and actual is None, "missing_gradient": True} - if reference is None or actual is None - else tensor_comparison(reference, actual, args) - ) - checks["pass"] = all(checks[key]["pass"] for key in ("loss", "plosses")) and all( - row["pass"] for row in checks["gradients"].values() - ) - model.zero_grad(set_to_none=True) - return checks - - -def time_mode(strategy, features, lengths, mode, args, initial_state): - model = strategy.trainable_module() - model.load_state_dict(initial_state) - model.zero_grad(set_to_none=True) - batch = make_batch(features, mode) - optimizer = BF16Optimizer( - model, - lr=args.learning_rate, - warmup_ratio=0.0, - total_steps=args.warmup + args.steps + 1, - lr_scheduler="constant", - ) - - def step(): - output = strategy.forward_loss(batch) - output.loss.backward() - optimizer.step() - return output.loss.detach() - - torch.cuda.synchronize() - warm_start = time.perf_counter() - for _ in range(args.warmup): - step() - torch.cuda.synchronize() - warmup_seconds = time.perf_counter() - warm_start - torch.cuda.reset_peak_memory_stats() - baseline_bytes = torch.cuda.memory_allocated() - durations = [] - for _ in range(args.steps): - torch.cuda.synchronize() - start = time.perf_counter() - final_loss = step() - torch.cuda.synchronize() - durations.append(time.perf_counter() - start) - peak_bytes = torch.cuda.max_memory_allocated() - mean_seconds = statistics.mean(durations) - result = { - "mean_step_ms": mean_seconds * 1000, - "p50_step_ms": statistics.median(durations) * 1000, - "stdev_step_ms": ( - statistics.stdev(durations) * 1000 if len(durations) > 1 else 0 - ), - "useful_tokens_per_second": sum(lengths) / mean_seconds, - "useful_ttt_positions_per_second": sum(lengths) - * args.ttt_length - / mean_seconds, - "peak_allocated_gib": peak_bytes / 2**30, - "baseline_allocated_gib": baseline_bytes / 2**30, - "peak_increment_gib": (peak_bytes - baseline_bytes) / 2**30, - "warmup_seconds_including_compile": warmup_seconds, - "step_ms": [duration * 1000 for duration in durations], - "final_loss": float(final_loss), - } - optimizer = batch = None - model.zero_grad(set_to_none=True) - gc.collect() - torch.cuda.empty_cache() - return result - - -def source_state(): - root = Path(__file__).resolve().parents[2] - source_paths = [ - "scripts/benchmark_sequence_packing.py", - "specforge/benchmarks/benchmark_sequence_packing.py", - "specforge/algorithms/eagle3/data.py", - "specforge/algorithms/eagle3/model.py", - "specforge/modeling/draft/llama3_eagle.py", - "specforge/modeling/packed_sequence.py", - "specforge/training/strategies/base.py", - ] - result = { - "files_sha256": { - name: hashlib.sha256((root / name).read_bytes()).hexdigest() - for name in source_paths - } - } - - def git(*command): - return subprocess.check_output( - ["git", *command], cwd=root, text=True, stderr=subprocess.DEVNULL - ).strip() - - try: - result.update(head=git("rev-parse", "HEAD"), dirty=git("status", "--short")) - except (OSError, subprocess.CalledProcessError): - result["head"] = "unavailable (file hashes identify copied snapshot)" - return result - - -def write_report(report, path): - path.parent.mkdir(parents=True, exist_ok=True) - path.write_text(json.dumps(report, indent=2) + "\n") - - -def main(): - args = parse_args() - if not torch.cuda.is_available(): - raise RuntimeError("This benchmark requires a CUDA GPU") - torch.backends.cuda.matmul.allow_tf32 = False - torch.backends.cudnn.allow_tf32 = False - torch.set_num_threads(4) - settings = vars(args).copy() - settings["output"] = str(args.output) - report = { - "timestamp_utc": datetime.now(timezone.utc).isoformat(), - "source": source_state(), - "environment": { - "python": platform.python_version(), - "torch": torch.__version__, - "cuda": torch.version.cuda, - "gpu": torch.cuda.get_device_name(), - "cuda_visible_devices": os.environ.get("CUDA_VISIBLE_DEVICES"), - "transformers": importlib.metadata.version("transformers"), - }, - "settings": settings, - "scope": "single-GPU synthetic offline features; resident inputs; production forward/backward/BF16Optimizer; excludes capture, I/O, transfer, distributed communication, and serving", - "cases": [], - } - strategy = build_strategy(args) - initial_state = { - key: value.detach().cpu().clone() - for key, value in strategy.trainable_module().state_dict().items() - } - for lengths in args.lengths: - strategy.trainable_module().load_state_dict(initial_state) - features = make_features(args, lengths) - case = { - "lengths": lengths, - "useful_tokens": sum(lengths), - "padded_tokens": len(lengths) * max(lengths), - "padding_fraction": 1 - sum(lengths) / (len(lengths) * max(lengths)), - "raw_supervised_tokens": sum( - int(item["loss_mask"].sum()) for item in features - ), - } - report["cases"].append(case) - print( - json.dumps( - {"case_start": lengths, "padding_fraction": case["padding_fraction"]} - ), - flush=True, - ) - if not args.skip_correctness: - case["correctness"] = check_correctness(strategy, features, args) - write_report(report, args.output) - print( - json.dumps({"correctness_pass": case["correctness"]["pass"]}), - flush=True, - ) - if not case["correctness"]["pass"]: - raise AssertionError( - f"Packed loss/gradient parity failed; inspect {args.output}" - ) - if not args.correctness_only: - for mode in ("padded", "packed"): - case[mode] = time_mode( - strategy, features, lengths, mode, args, initial_state - ) - write_report(report, args.output) - print(json.dumps({"mode": mode, **case[mode]}), flush=True) - case["speedup"] = ( - case["padded"]["mean_step_ms"] / case["packed"]["mean_step_ms"] - ) - case["peak_memory_reduction_fraction"] = 1 - ( - case["packed"]["peak_allocated_gib"] - / case["padded"]["peak_allocated_gib"] - ) - write_report(report, args.output) - del features - print(f"Report written to {args.output}", flush=True) - - -if __name__ == "__main__": - main() diff --git a/tests/test_runtime/test_online_packing_benchmark.py b/tests/test_runtime/test_online_packing_benchmark.py deleted file mode 100644 index 56c7eb64d..000000000 --- a/tests/test_runtime/test_online_packing_benchmark.py +++ /dev/null @@ -1,211 +0,0 @@ -"""Fairness checks for the overlapping online packing benchmark.""" - -import json -import tempfile -import unittest -from pathlib import Path -from types import SimpleNamespace - -import torch - -from specforge.benchmarks.benchmark_online_sequence_packing import ( - _observe_first_warmup_step, - _optimizer_coverage, - _optimizer_step_evidence, - _parameter_inventory, - _parameter_sample_indices, - _parameter_samples, - _parameter_update_evidence, - _prompts, - _TimedSource, - parse_args, -) -from specforge.launch import _iter_epoch_online_prompt_batches -from specforge.optimizer import BF16Optimizer - - -class OnlinePackingBenchmarkTests(unittest.TestCase): - def _model(self): - model = torch.nn.Module() - model.draft_model = torch.nn.Module() - model.draft_model.layers = torch.nn.ModuleList( - [torch.nn.Linear(2, 2, bias=False) for _ in range(2)] - ) - # This factor has a legitimate zero gradient and no first-step change. - model.draft_model.zero_factor = torch.nn.Parameter(torch.zeros(2)) - model.embed_tokens = torch.nn.Embedding(4, 2).requires_grad_(False) - model.lm_head = torch.nn.Linear(2, 4, bias=False) - model.lm_head.weight = model.embed_tokens.weight - return model - - def _args(self, *extra): - return parse_args( - [ - "--server-url", - "http://localhost:31012", - "--target-model", - "unused", - "--work-dir", - "unused", - "--output", - "unused.json", - *extra, - ] - ) - - def test_default_warmup_replays_full_corpus(self): - args = self._args("--steps", "7") - self.assertEqual(args.warmup_steps, 7) - self.assertTrue(args.teacher_metrics) - self.assertEqual(args.log_interval, 50) - self.assertEqual(args.objective_chunk_blocks, 128) - - def test_sample_indices_keep_large_embedding_endpoint_in_bounds(self): - # Only allocate <=128 indices, never the 389-million-element embedding. - for numel in (0, 1, 2, 127, 128, 129, 388956160, 2**40): - with self.subTest(numel=numel): - indices = _parameter_sample_indices(numel) - self.assertEqual(indices.dtype, torch.int64) - self.assertEqual(indices.numel(), min(128, numel)) - if numel: - self.assertEqual(indices[0].item(), 0) - self.assertEqual(indices[-1].item(), numel - 1) - self.assertTrue(bool(((indices >= 0) & (indices < numel)).all())) - self.assertTrue(bool((indices[1:] > indices[:-1]).all())) - - def test_inventory_counts_tied_target_once_and_checks_optimizer_identity(self): - model = self._model() - named, inventory = _parameter_inventory(model) - self.assertEqual(inventory["draft_layer_count"], 2) - self.assertEqual(inventory["whole_model_unique"]["total"], 18) - self.assertEqual(inventory["whole_model_unique"]["trainable"], 10) - self.assertEqual(inventory["whole_model_unique"]["frozen"], 8) - self.assertEqual(inventory["components"]["target_embedding"]["total"], 8) - self.assertEqual(inventory["components"]["target_lm_head"]["total"], 8) - tied = next(row for row in inventory["parameters"] if not row["trainable"]) - self.assertEqual( - set(tied["aliases"]), {"embed_tokens.weight", "lm_head.weight"} - ) - optimizer = BF16Optimizer(model.draft_model, lr=0.01, total_steps=2) - coverage = _optimizer_coverage(model, named, optimizer) - self.assertEqual(coverage["optimized_elements"], 10) - self.assertTrue(coverage["target_parameters_excluded"]) - optimizer.model_params[-1] = model.embed_tokens.weight - with self.assertRaisesRegex(AssertionError, "no target Parameter"): - _optimizer_coverage(model, named, optimizer) - - def test_full_update_evidence_accepts_legitimate_zero_gradient_parameters(self): - model = self._model() - named, _ = _parameter_inventory(model) - before = _parameter_samples(named) - optimizer = BF16Optimizer(model.draft_model, lr=0.01, total_steps=2) - gradient_evidence = _observe_first_warmup_step(optimizer, named) - sum( - parameter.square().sum() for parameter in model.draft_model.parameters() - ).backward() - optimizer.step() - self.assertTrue(gradient_evidence["performed"]) - self.assertTrue(all(gradient_evidence["gradient_present"].values())) - self.assertTrue(torch.isfinite(gradient_evidence["_gradient_norm"])) - self.assertTrue( - all(parameter.grad is None for parameter in optimizer.model_params) - ) - updates = _parameter_update_evidence(named, before, 2) - self.assertTrue(updates["all_decoder_layers_have_sampled_updates"]) - self.assertTrue(updates["all_trainable_elements_finite"]) - self.assertTrue(updates["all_frozen_parameter_samples_unchanged"]) - self.assertFalse( - updates["parameters"]["draft_model.zero_factor"]["sampled_update_observed"] - ) - steps = _optimizer_step_evidence(optimizer, named, 1) - self.assertTrue(steps["all_trainable_parameters_received_every_optimizer_step"]) - self.assertEqual( - steps["adamw_steps_by_parameter"]["draft_model.zero_factor"], 1 - ) - - def test_evidence_rejects_nonfinite_parameters_and_changed_frozen_samples(self): - model = self._model() - named, _ = _parameter_inventory(model) - before = _parameter_samples(named) - with torch.no_grad(): - model.draft_model.zero_factor[0] = float("inf") - with self.assertRaisesRegex(AssertionError, "nonfinite"): - _parameter_update_evidence(named, before, 2) - with torch.no_grad(): - model.draft_model.zero_factor[0] = 0 - model.embed_tokens.weight[0, 0] += 1 - with self.assertRaisesRegex(AssertionError, "frozen"): - _parameter_update_evidence(named, before, 2) - - def test_real_masks_and_batch_order_survive_canonical_shuffle(self): - rows = [ - { - "input_ids": [index] * (index + 4), - "loss_mask": [0] * (index + 2) + [1, 1], - } - for index in range(8) - ] - with tempfile.TemporaryDirectory() as directory: - path = Path(directory) / "prompts.jsonl" - path.write_text("\n".join(json.dumps(row) for row in rows)) - args = self._args("--prompts-path", str(path), "--steps", "2") - prompts, digest, tokens, supervised = _prompts(args, 2, 16) - ordered = [ - prompt - for batch in _iter_epoch_online_prompt_batches( - prompts, 0, 1, seed=args.seed, batch_size=3 - ) - for prompt in batch - ] - self.assertEqual([prompt["payload"] for prompt in ordered], rows) - self.assertEqual( - [prompt["task_id"] for prompt in ordered], - [f"prompt-{index:08d}" for index in range(8)], - ) - self.assertEqual(tokens, sum(len(row["input_ids"]) for row in rows)) - self.assertEqual(supervised, 16) - self.assertEqual(_prompts(args, 2, 16)[1], digest) - # Warmup prefixes use the same input order despite a different - # shuffle length, making short debugging runs deterministic too. - warm = _prompts(args, 1, 16)[0] - warm_ordered = [ - prompt - for batch in _iter_epoch_online_prompt_batches( - warm, 0, 1, seed=args.seed - ) - for prompt in batch - ] - self.assertEqual([prompt["payload"] for prompt in warm_ordered], rows[:4]) - - def test_request_hash_checks_masks_but_ignores_transport_namespaces(self): - def capture(run, mask): - sent = [] - adapter = SimpleNamespace(post_fn=lambda url, **kw: sent.append(kw) or []) - source = _TimedSource(adapter) - adapter.post_fn( - "http://localhost/generate", - timeout=1, - json_body={ - "input_ids": [[1, 2, 3]], - "extra_key": [run], - "sampling_params": {"max_new_tokens": 0}, - "spec_capture": [ - { - "store_id": run, - "sample_id": f"{run}:prompt-00000000", - "passthrough": [{"data": mask}], - } - ], - }, - ) - self.assertEqual(len(sent), 1) - self.assertEqual(source.calls[0]["samples"], 1) - self.assertLessEqual(source.first_dispatch, source.calls[0]["end"]) - return source.request_digest.hexdigest() - - self.assertEqual(capture("run-a", [0, 1, 1]), capture("run-b", [0, 1, 1])) - self.assertNotEqual(capture("run-a", [0, 1, 1]), capture("run-b", [1, 1, 1])) - - -if __name__ == "__main__": - unittest.main()