From f1d191a0e4863122ac7e331cefc2bbe571d3a41a Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Tue, 15 Sep 2026 19:00:09 +0100 Subject: [PATCH 01/40] SPOR-0001 llo/protocol: reject unknown aggregators at admission, skip them when aggregating --- llo/dev/v31/statetransition.go | 8 ++++++- llo/protocol/channel_definitions.go | 8 +++++++ llo/protocol/channel_definitions_test.go | 28 ++++++++++++++++++++++++ llo/v30/plugin_outcome.go | 5 ++++- 4 files changed, 47 insertions(+), 2 deletions(-) diff --git a/llo/dev/v31/statetransition.go b/llo/dev/v31/statetransition.go index 84b6566..728fb88 100644 --- a/llo/dev/v31/statetransition.go +++ b/llo/dev/v31/statetransition.go @@ -406,7 +406,13 @@ func (p *Plugin) aggregate( aggF := protocol.GetAggregatorFunc(agg) if aggF == nil { - return fmt.Errorf("no aggregator function defined for aggregator of type %v", agg) + // Unknown aggregator, e.g. one added by a newer version. Admission + // rejects these, but a committed definition must not halt the + // protocol: skip the pair and carry forward what it had. + if prevTSV != nil { + keep(sid, agg, prevTSV) + } + continue } result, aerr := aggF(streamObservations[sid], p.F) diff --git a/llo/protocol/channel_definitions.go b/llo/protocol/channel_definitions.go index 8b4991d..40b3bdc 100644 --- a/llo/protocol/channel_definitions.go +++ b/llo/protocol/channel_definitions.go @@ -120,6 +120,14 @@ func verifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, chan merr = errors.Join(merr, fmt.Errorf("ChannelDefinition with ID %d has stream %d with zero aggregator (this may indicate an uninitialized struct)", channelID, strm.StreamID)) continue } + // An aggregator this binary does not know has no aggregator + // function, so the pair can never produce an aggregate. Rejected at + // admission only: a committed definition carrying one is left alone + // (aggregation skips the pair) rather than failing verification on + // every oracle, every round. + if strm.Aggregator != llotypes.AggregatorCalculated && GetAggregatorFunc(strm.Aggregator) == nil { + admit(fmt.Errorf("ChannelDefinition with ID %d has stream %d with unknown aggregator %d", channelID, strm.StreamID, strm.Aggregator), channelID) + } uniqueStreamIDs[strm.StreamID] = struct{}{} // Calculated streams are derived from the opts that declare them and // are not stored on the definition, so anything listed here is diff --git a/llo/protocol/channel_definitions_test.go b/llo/protocol/channel_definitions_test.go index 3e794e0..edbc7bf 100644 --- a/llo/protocol/channel_definitions_test.go +++ b/llo/protocol/channel_definitions_test.go @@ -62,6 +62,34 @@ func Test_VerifyChannelDefinitions(t *testing.T) { require.EqualError(t, err, "ChannelDefinition with ID 1 has stream 0 with zero aggregator (this may indicate an uninitialized struct)") }) + t.Run("rejects unknown aggregator at admission but not once committed", func(t *testing.T) { + channelDefs := llotypes.ChannelDefinitions{ + 1: llotypes.ChannelDefinition{ + Streams: []llotypes.Stream{{StreamID: 7, Aggregator: llotypes.Aggregator(99)}}, + }, + } + err := verifyAdmittingAll(codecs, channelDefs) + require.EqualError(t, err, "ChannelDefinition with ID 1 has stream 7 with unknown aggregator 99") + + // Already committed: verification must pass, otherwise every oracle + // fails every round and the protocol halts with no recovery path. + require.NoError(t, VerifyChannelDefinitions(codecs, channelDefs)) + }) + + t.Run("accepts known and calculated aggregators at admission", func(t *testing.T) { + channelDefs := llotypes.ChannelDefinitions{ + 1: llotypes.ChannelDefinition{ + Streams: []llotypes.Stream{ + {StreamID: 1, Aggregator: llotypes.AggregatorMedian}, + {StreamID: 2, Aggregator: llotypes.AggregatorMode}, + {StreamID: 3, Aggregator: llotypes.AggregatorQuote}, + {StreamID: 4, Aggregator: llotypes.AggregatorCalculated}, + }, + }, + } + require.NoError(t, verifyAdmittingAll(codecs, channelDefs)) + }) + t.Run("fails if too many total unique stream IDs", func(t *testing.T) { streams := make([]llotypes.Stream, MaxObservationStreamValuesLength) for i := uint32(0); i < MaxObservationStreamValuesLength; i++ { diff --git a/llo/v30/plugin_outcome.go b/llo/v30/plugin_outcome.go index 1391b87..9ea3efe 100644 --- a/llo/v30/plugin_outcome.go +++ b/llo/v30/plugin_outcome.go @@ -283,7 +283,10 @@ func (p *Plugin) outcome(outctx ocr3types.OutcomeContext, query types.Query, aos // Perform the aggregation aggF := protocol.GetAggregatorFunc(agg) if aggF == nil { - return nil, fmt.Errorf("no aggregator function defined for aggregator of type %v", agg) + // Unknown aggregator, e.g. one added by a newer version. Admission + // rejects these, but a committed definition must not halt the + // protocol: skip the pair, keeping any carried-forward value. + continue } result, err := aggF(streamObservations[sid], p.F) From 854d28fad729c9ed2342b30c946a186d786852ec Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Wed, 16 Sep 2026 08:40:09 +0100 Subject: [PATCH 02/40] SPOR-0003 llo/v31: bound blob payload and persisted aggregates against declared limits The factory declares libocr's maximum for every limit, but several values the plugin produces were not bounded by any admission rule, so libocr could reject the write or message and fail the round for every oracle at once. --- llo/dev/v31/blobcompress.go | 28 +++- llo/dev/v31/concurrency_test.go | 2 +- llo/dev/v31/factory.go | 34 ++++- llo/dev/v31/history_bench_test.go | 45 ++++-- llo/dev/v31/history_flow_test.go | 2 +- llo/dev/v31/history_requirements_test.go | 2 +- llo/dev/v31/kv.go | 12 ++ llo/dev/v31/kv_test.go | 7 +- llo/dev/v31/limits_test.go | 180 +++++++++++++++++++++++ llo/dev/v31/plugin.go | 4 +- llo/dev/v31/statetransition.go | 8 +- llo/protocol/limits.go | 21 +++ llo/v30/plugin.go | 66 ++++----- 13 files changed, 346 insertions(+), 65 deletions(-) create mode 100644 llo/dev/v31/limits_test.go diff --git a/llo/dev/v31/blobcompress.go b/llo/dev/v31/blobcompress.go index 781954b..49b1cf3 100644 --- a/llo/dev/v31/blobcompress.go +++ b/llo/dev/v31/blobcompress.go @@ -6,6 +6,8 @@ import ( "github.com/klauspost/compress/zstd" + "github.com/smartcontractkit/libocr/offchainreporting2plus/ocr3_1types" + protocol "github.com/smartcontractkit/chainlink-data-streams/llo/protocol" ) @@ -24,6 +26,15 @@ const ( // a huge allocation (zstd bomb). const maxDecompressedBlobPayloadBytes = protocol.MaxDecompressedObservationLength +// MaxBlobPayloadBytes is the size of a broadcast blob payload libocr accepts, +// and is the value the factory declares as MaxBlobPayloadBytes. It bounds the +// payload as framed for broadcast (codec byte plus body), which is a different +// quantity from maxDecompressedBlobPayloadBytes: the latter is the anti-bomb +// bound applied to the decompressed bytes on the read side, and is deliberately +// looser. Enforcing this one on the write side is what stops the pump from +// broadcasting a blob every peer's libocr would reject. +const MaxBlobPayloadBytes = ocr3_1types.MaxMaxBlobPayloadBytes + // zstd Encoder/Decoder are safe for concurrent use via EncodeAll/DecodeAll and // are expensive to build, so a single pair is shared process-wide. Built lazily // so a construction failure surfaces at the call site rather than in init. @@ -74,12 +85,19 @@ func encodeBlobPayload(raw []byte) ([]byte, error) { return nil, err } compressed := c.encoder.EncodeAll(raw, []byte{blobCodecZstd}) - if len(compressed) < len(raw)+1 { - return compressed, nil + out := compressed + if len(compressed) >= len(raw)+1 { + out = make([]byte, 0, len(raw)+1) + out = append(out, blobCodecRaw) + out = append(out, raw...) + } + // The framed payload is what is broadcast, so it -- not the raw bytes -- is + // what has to fit the declared limit. A payload that compresses poorly can + // pass the check above and still land over it. + if len(out) > MaxBlobPayloadBytes { + return nil, fmt.Errorf("framed blob payload too large: %d > %d bytes", len(out), MaxBlobPayloadBytes) } - out := make([]byte, 0, len(raw)+1) - out = append(out, blobCodecRaw) - return append(out, raw...), nil + return out, nil } // decodeBlobPayload reverses encodeBlobPayload. The payload is untrusted, so diff --git a/llo/dev/v31/concurrency_test.go b/llo/dev/v31/concurrency_test.go index 63b93d0..91b11fc 100644 --- a/llo/dev/v31/concurrency_test.go +++ b/llo/dev/v31/concurrency_test.go @@ -30,7 +30,7 @@ func (optsEchoCodec) Encode(r protocol.Report, _ llotypes.ChannelDefinition, opt if err != nil { return nil, err } - return []byte(fmt.Sprintf("v=%d", o.V)), nil + return fmt.Appendf(nil, "v=%d", o.V), nil } func (optsEchoCodec) Verify(llotypes.ChannelDefinition) error { return nil } diff --git a/llo/dev/v31/factory.go b/llo/dev/v31/factory.go index 09dd12a..5bc94a4 100644 --- a/llo/dev/v31/factory.go +++ b/llo/dev/v31/factory.go @@ -122,6 +122,36 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re p.pump.Start() unexpiredBlobCount := perOracleUnexpiredBlobCount(blobLifetimeRounds) + // Declared limits. Each is the libocr maximum, which is only honest if the + // plugin's own admission rules keep what it produces underneath it -- libocr + // rejects an oversized message or write set, which fails the round for every + // oracle. The derivations, and which of them are currently enforced, are: + // + // MaxObservationBytes bounded post-decompression by + // protocol.MaxDecompressedObservationLength, + // itself derived from + // MaxObservationStreamValuesLength stream + // values plus + // MaxObservationUpdateChannelDefinitionsLength + // definitions. + // MaxReportsPlusPrecursorBytes the precursor embeds every definition plus + // every stream aggregate. The aggregate count + // is bounded by protocol.MaxPersistedAggregates; + // the definition set is bounded in channel + // count (MaxOutcomeChannelDefinitionsLength) + // and per-channel stream count + // (MaxStreamsPerChannel) but NOT yet in total + // stream entries or channel opts bytes, and a + // single stream value is not yet bounded in + // decimal coefficient length. Those three + // bounds are what make this number true rather + // than aspirational. + // MaxKeyValueModifiedKeys* the per-round write set is c/defs plus r/agg + // plus the history windows, and history is + // held to protocol.MaxHistoryTotalBytes so it + // cannot consume the whole budget on its own. + // MaxBlobPayloadBytes enforced on the write side by + // encodeBlobPayload. info := ocr3_1types.ReportingPluginInfo1{ Name: "LLO-3.1", Limits: ocr3_1types.ReportingPluginLimits{ @@ -134,13 +164,13 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re MaxKeyValueModifiedKeys: ocr3_1types.MaxMaxKeyValueModifiedKeys, MaxKeyValueModifiedKeysPlusValuesBytes: ocr3_1types.MaxMaxKeyValueModifiedKeysPlusValuesBytes, - MaxBlobPayloadBytes: ocr3_1types.MaxMaxBlobPayloadBytes, + MaxBlobPayloadBytes: MaxBlobPayloadBytes, // Blobs live for blobLifetimeRounds sequence numbers and the pump // broadcasts about one per round, so both budgets are derived from // the configured lifetime plus a margin for asynchronous reaping // (see the libocr docs). MaxPerOracleUnexpiredBlobCount: unexpiredBlobCount, - MaxPerOracleUnexpiredBlobCumulativePayloadBytes: unexpiredBlobCount * ocr3_1types.MaxMaxBlobPayloadBytes, + MaxPerOracleUnexpiredBlobCumulativePayloadBytes: unexpiredBlobCount * MaxBlobPayloadBytes, }, } if err := info.Validate(); err != nil { diff --git a/llo/dev/v31/history_bench_test.go b/llo/dev/v31/history_bench_test.go index d648fc4..e6d0a6b 100644 --- a/llo/dev/v31/history_bench_test.go +++ b/llo/dev/v31/history_bench_test.go @@ -49,32 +49,52 @@ func benchState(tb testing.TB, pairs, depth int) *memKV { key := histKey{streamID: llotypes.StreamID(p + 1), aggregator: benchAggreg} keys = append(keys, key) - // Built in one pass rather than round by round: the window is kept in - // memory and each chunk is persisted as it seals, which is exactly what - // a run of rounds would have written. - w := protocol.NewRingWindow(nil) - _, err := w.SetRequiredCount(uint32(depth)) - require.NoError(tb, err) + // One window instance per record, because a window admits a single + // append per round (protocol.ErrHistoryAlreadyAppended) and each + // instance IS a round. The header and the newest chunk are carried in + // memory across instances instead of through storage, and the chunk is + // persisted as it seals, so what lands in kv is what a run of rounds + // would have written without paying to marshal every intermediate + // round. + var ( + header *protocol.StreamHistoryHeader + tail *protocol.StreamHistoryChunk + ) // Half a chunk past the required depth, so the benchmark measures a // settled window with a half-full newest chunk rather than the boundary // case where the next append happens to start a fresh, nearly empty one. records := depth + protocol.MaxHistoryChunkRecords/2 for i := 1; i <= records; i++ { + w := protocol.NewRingWindow(header) + _, err := w.SetRequiredCount(uint32(depth)) + require.NoError(tb, err) + + // The newest chunk when it still has room; a sealed or absent one + // needs nothing, exactly as historyStore.Append plans it. + for _, sequence := range w.AppendPlan() { + require.NotNil(tb, tail) + require.Equal(tb, sequence, tail.Sequence()) + require.NoError(tb, w.Provide(tail)) + } + appended, err := w.Append(uint64(i)*uint64(1_000_000_000), benchQuote(i)) require.NoError(tb, err) require.True(tb, appended) set := w.WriteSet() - if set.Chunk.Len() == protocol.MaxHistoryChunkRecords || i == records { - _, err := writeHistoryChunk(kv, key.streamID, key.aggregator, set.Chunk) + require.NotNil(tb, set.Chunk) + tail = set.Chunk + if tail.Len() == protocol.MaxHistoryChunkRecords || i == records { + _, err := writeHistoryChunk(kv, key.streamID, key.aggregator, tail) require.NoError(tb, err) } for _, slot := range set.DeletedSlots { require.NoError(tb, deleteHistoryChunk(kv, key.streamID, key.aggregator, slot)) } + header = w.Header() } - _, err = writeHistoryHeader(kv, key.streamID, key.aggregator, w.WriteSet().Header) + _, err := writeHistoryHeader(kv, key.streamID, key.aggregator, header) require.NoError(tb, err) } require.NoError(tb, writeHistoryIndex(kv, keys)) @@ -96,8 +116,7 @@ func BenchmarkHistoryStore_LoadAll(b *testing.B) { kv := benchState(b, benchPairs, benchDepth) lggr := logger.Test(b) - b.ResetTimer() - for range b.N { + for b.Loop() { store, err := newHistoryStore(kv, lggr) if err != nil { b.Fatal(err) @@ -166,9 +185,9 @@ func BenchmarkComputeHistoryRequirements(b *testing.B) { defs[llotypes.ChannelID(c+1)] = llotypes.ChannelDefinition{ ReportFormat: llotypes.ReportFormatEVMABIEncodeUnpackedExpr, Streams: []llotypes.Stream{{StreamID: streamID, Aggregator: benchAggreg}}, - Opts: []byte(fmt.Sprintf( + Opts: fmt.Appendf(nil, `{"abi":[{"type":"int256","expression":"Avg(History(s%d, 300))","expressionStreamID":%d}]}`, - streamID, 900+c)), + streamID, 900+c), } } cache := protocol.NewOptsCache() diff --git a/llo/dev/v31/history_flow_test.go b/llo/dev/v31/history_flow_test.go index ad19ea8..3401153 100644 --- a/llo/dev/v31/history_flow_test.go +++ b/llo/dev/v31/history_flow_test.go @@ -22,7 +22,7 @@ func historyExprChannel(expression string) llotypes.ChannelDefinition { return llotypes.ChannelDefinition{ ReportFormat: llotypes.ReportFormatEVMABIEncodeUnpackedExpr, Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}, - Opts: []byte(fmt.Sprintf(`{"abi":[{"type":"int256","expression":%q,"expressionStreamID":999}]}`, expression)), + Opts: fmt.Appendf(nil, `{"abi":[{"type":"int256","expression":%q,"expressionStreamID":999}]}`, expression), } } diff --git a/llo/dev/v31/history_requirements_test.go b/llo/dev/v31/history_requirements_test.go index 2ea546b..65992ef 100644 --- a/llo/dev/v31/history_requirements_test.go +++ b/llo/dev/v31/history_requirements_test.go @@ -24,7 +24,7 @@ func exprChannel(streams []llotypes.Stream, expressions ...string) llotypes.Chan return llotypes.ChannelDefinition{ ReportFormat: llotypes.ReportFormatEVMABIEncodeUnpackedExpr, Streams: streams, - Opts: []byte(fmt.Sprintf(`{"abi":[%s]}`, abi)), + Opts: fmt.Appendf(nil, `{"abi":[%s]}`, abi), } } diff --git a/llo/dev/v31/kv.go b/llo/dev/v31/kv.go index d4f4fa4..0badbfe 100644 --- a/llo/dev/v31/kv.go +++ b/llo/dev/v31/kv.go @@ -5,6 +5,7 @@ import ( "fmt" "sort" + "github.com/smartcontractkit/chainlink-common/pkg/logger" llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" protocol "github.com/smartcontractkit/chainlink-data-streams/llo/protocol" @@ -282,6 +283,7 @@ func writeHotState( validAfterNanoseconds map[llotypes.ChannelID]uint64, reportable map[llotypes.ChannelID]bool, carryForward map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue, + lggr logger.Logger, ) error { pb := &protocol.LLOHotStateProto{ ObservationTimestampNanoseconds: observationTimestampNs, @@ -329,6 +331,16 @@ func writeHotState( } return pb.StreamAggregates[i].StreamID < pb.StreamAggregates[j].StreamID }) + // Truncation happens here, after the sort, so that every oracle keeps the + // same pairs: (streamID, aggregator) order is total, and the write is the + // only place the whole set is known. See MaxPersistedAggregates. + if dropped := len(pb.StreamAggregates) - protocol.MaxPersistedAggregates; dropped > 0 { + pb.StreamAggregates = pb.StreamAggregates[:protocol.MaxPersistedAggregates] + lggr.Errorw("Too many carry-forward aggregates to persist; dropping the highest (streamID, aggregator) pairs", + "dropped", dropped, + "maxPersistedAggregates", protocol.MaxPersistedAggregates, + ) + } b, err := deterministicMarshal.Marshal(pb) if err != nil { diff --git a/llo/dev/v31/kv_test.go b/llo/dev/v31/kv_test.go index 2c6f960..57fb305 100644 --- a/llo/dev/v31/kv_test.go +++ b/llo/dev/v31/kv_test.go @@ -7,6 +7,7 @@ import ( "github.com/shopspring/decimal" "github.com/stretchr/testify/require" + "github.com/smartcontractkit/chainlink-common/pkg/logger" llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" "github.com/smartcontractkit/chainlink-common/pkg/utils/tests" @@ -58,7 +59,7 @@ func Test_ChannelCache_StaleSeqNrForcesReload(t *testing.T) { kv := newMemKV() defs := llotypes.ChannelDefinitions{1: jsonChannel()} require.NoError(t, writeChannelState(kv, 5, defs)) - require.NoError(t, writeHotState(kv, 0, nil, nil, nil)) + require.NoError(t, writeHotState(kv, 0, nil, nil, nil, logger.Test(t))) cache := protocol.NewChannelCache() s, err := loadKVState(kv, cache) @@ -145,7 +146,7 @@ func Test_KVRecords_DeterministicAndRoundTrip(t *testing.T) { for i := 0; i < 8; i++ { kv := newMemKV() require.NoError(t, writeChannelState(kv, 9, defs)) - require.NoError(t, writeHotState(kv, 1_234, validAfter, reportable, carry)) + require.NoError(t, writeHotState(kv, 1_234, validAfter, reportable, carry, logger.Test(t))) if i == 0 { channelBytes, hotBytes = kv.m[string(keyChannelState)], kv.m[string(keyHotState)] continue @@ -156,7 +157,7 @@ func Test_KVRecords_DeterministicAndRoundTrip(t *testing.T) { kv := newMemKV() require.NoError(t, writeChannelState(kv, 9, defs)) - require.NoError(t, writeHotState(kv, 1_234, validAfter, reportable, carry)) + require.NoError(t, writeHotState(kv, 1_234, validAfter, reportable, carry, logger.Test(t))) require.Equal(t, uint64(9), binary.BigEndian.Uint64(kv.m[string(keyChannelSeqNr)])) s, err := loadKVState(kv, nil) diff --git a/llo/dev/v31/limits_test.go b/llo/dev/v31/limits_test.go new file mode 100644 index 0000000..2cdc6f9 --- /dev/null +++ b/llo/dev/v31/limits_test.go @@ -0,0 +1,180 @@ +package llo + +import ( + "strings" + "testing" + + "github.com/shopspring/decimal" + "github.com/stretchr/testify/require" + + "github.com/smartcontractkit/chainlink-common/pkg/logger" + llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" + + protocol "github.com/smartcontractkit/chainlink-data-streams/llo/protocol" + + "github.com/smartcontractkit/libocr/offchainreporting2plus/ocr3_1types" +) + +// The declared limits in factory.go are only honest if the plugin's own bounds +// keep what it produces underneath them. These tests hold that arithmetic, so +// that raising one of the protocol constants fails the build rather than the +// round: libocr rejects an oversized write or message, and a rejected write +// fails the round for every oracle at once. + +func TestLimits_BlobPayloadGuardMatchesDeclared(t *testing.T) { + // The write-side guard and the declared limit must be the same number, or + // the pump can broadcast a blob every peer's libocr rejects. + require.Equal(t, ocr3_1types.MaxMaxBlobPayloadBytes, MaxBlobPayloadBytes) + + // The anti-bomb bound on the read side is deliberately looser; it bounds a + // different quantity (decompressed bytes) and must not be mistaken for the + // broadcast limit. + require.Greater(t, maxDecompressedBlobPayloadBytes, MaxBlobPayloadBytes) +} + +func TestLimits_BlobPayloadRejectsOversizedFraming(t *testing.T) { + // Incompressible bytes above the declared limit: passes the raw check + // (below maxDecompressedBlobPayloadBytes) but cannot be framed within the + // declared payload limit. + raw := make([]byte, MaxBlobPayloadBytes+1) + // Deterministic pseudo-random bytes, so zstd cannot shrink the payload + // below the limit and the framing check is the one that fires. A periodic + // pattern would compress away and test nothing. + lcg := uint64(1) + for i := range raw { + lcg = lcg*6364136223846793005 + 1442695040888963407 + raw[i] = byte(lcg >> 33) + } + require.Less(t, len(raw), maxDecompressedBlobPayloadBytes) + + _, err := encodeBlobPayload(raw) + require.ErrorContains(t, err, "framed blob payload too large") +} + +func TestLimits_HistoryFitsWriteBudgets(t *testing.T) { + // One pair's window header plus newest chunk is a single value under one + // key, so it must fit the per-key limit. + require.LessOrEqual(t, protocol.MaxHistoryPairRoundBytes, ocr3_1types.MaxMaxKeyValueValueBytes) + + // Every pair rewrites its newest chunk and header each round, and that has + // to leave room in the per-round write set for c/defs and r/agg. + require.Less(t, + protocol.MaxHistoryPairs*protocol.MaxHistoryPairRoundBytes, + ocr3_1types.MaxMaxKeyValueModifiedKeysPlusValuesBytes, + "history alone must not consume the per-round write budget") + + // The admission budget history is held to must itself fit the per-round + // write set. + require.Less(t, protocol.MaxHistoryTotalBytes, ocr3_1types.MaxMaxKeyValueModifiedKeysPlusValuesBytes) + + // Each pair occupies a bounded, statically known number of keys. + require.Less(t, + protocol.MaxHistoryPairs*protocol.MaxHistoryChunkSlots, + ocr3_1types.MaxMaxKeyValueModifiedKeys) +} + +// TestLimits_HotStateWorstCaseFitsPerKeyLimit builds the largest r/agg record +// the persisted-aggregate cap permits and measures it. The cap exists precisely +// to make this record bounded (see protocol.MaxPersistedAggregates); without a +// measurement the cap is only a count, and the byte limit libocr enforces is +// what actually fails the round. +func TestLimits_HotStateWorstCaseFitsPerKeyLimit(t *testing.T) { + // An 18-digit price, the largest value a real feed produces. Note this is + // the honest worst case, not the adversarial one: a decimal's coefficient + // length is not yet bounded, so a byzantine value can still exceed this. + // Bounding the coefficient is a separate, consensus-affecting change. + price := decimal.RequireFromString("123456789012345.678") + + carry := make(map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue, protocol.MaxObservationStreamValuesLength) + aggregators := []llotypes.Aggregator{llotypes.AggregatorMedian, llotypes.AggregatorMode} + for i := range protocol.MaxObservationStreamValuesLength { + sid := llotypes.StreamID(i) + carry[sid] = make(map[llotypes.Aggregator]*protocol.TimestampedStreamValue, len(aggregators)) + for _, agg := range aggregators { + carry[sid][agg] = &protocol.TimestampedStreamValue{ + ObservedAtNanoseconds: 1_700_000_000_000_000_000, + StreamValue: protocol.ToDecimal(price), + } + } + } + // Exactly the cap: every observed stream aggregated two ways. + require.Equal(t, protocol.MaxPersistedAggregates, len(carry)*len(aggregators)) + + validAfter := make(map[llotypes.ChannelID]uint64, protocol.MaxOutcomeChannelDefinitionsLength) + reportable := make(map[llotypes.ChannelID]bool, protocol.MaxOutcomeChannelDefinitionsLength) + for i := range protocol.MaxOutcomeChannelDefinitionsLength { + validAfter[llotypes.ChannelID(i)] = 1_700_000_000_000_000_000 + reportable[llotypes.ChannelID(i)] = true + } + + kv := newMemKV() + require.NoError(t, writeHotState(kv, 1_700_000_000_000_000_000, validAfter, reportable, carry, logger.Test(t))) + + record, err := kv.Read(keyHotState) + require.NoError(t, err) + require.LessOrEqual(t, len(record), ocr3_1types.MaxMaxKeyValueValueBytes, + "worst-case r/agg record (%d bytes) exceeds the per-key limit", len(record)) +} + +// TestLimits_HotStateTruncatesAboveCap covers the cap itself: above it, pairs +// are dropped in (streamID, aggregator) order so every oracle writes the same +// record, rather than every oracle writing an oversized one libocr rejects. +func TestLimits_HotStateTruncatesAboveCap(t *testing.T) { + const over = 3 + tsv := func(ns uint64) *protocol.TimestampedStreamValue { + return &protocol.TimestampedStreamValue{ObservedAtNanoseconds: ns, StreamValue: protocol.ToDecimal(decimal.NewFromInt(1))} + } + + carry := make(map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue, protocol.MaxPersistedAggregates+over) + for i := range protocol.MaxPersistedAggregates + over { + carry[llotypes.StreamID(i)] = map[llotypes.Aggregator]*protocol.TimestampedStreamValue{ + llotypes.AggregatorMedian: tsv(uint64(i) + 1), + } + } + + kv := newMemKV() + require.NoError(t, writeHotState(kv, 1, nil, nil, carry, logger.Test(t))) + + state := &kvState{ + channelDefinitions: llotypes.ChannelDefinitions{}, + validAfterNanoseconds: map[llotypes.ChannelID]uint64{}, + reportedLastRound: map[llotypes.ChannelID]bool{}, + carryForward: map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue{}, + } + require.NoError(t, readHotState(kv, state)) + + require.Len(t, state.carryForward, protocol.MaxPersistedAggregates) + // The kept pairs are the lowest ones: truncation is after the sort, so the + // choice is a function of the pair identities alone and is identical on + // every oracle. + _, keptLowest := state.carryForward[llotypes.StreamID(0)] + require.True(t, keptLowest) + for i := protocol.MaxPersistedAggregates; i < protocol.MaxPersistedAggregates+over; i++ { + _, dropped := state.carryForward[llotypes.StreamID(i)] + require.False(t, dropped, "pair %d should have been dropped", i) + } +} + +// TestLimits_KnownUnboundedInputs documents the bounds that do NOT yet exist, +// so the gap is visible next to the arithmetic that depends on it rather than +// only in a review note. Each of these is a real input to the precursor and the +// c/defs record, and each is bounded only by the checks listed here. +func TestLimits_KnownUnboundedInputs(t *testing.T) { + // Channel opts are raw bytes with no length check on any path. + require.NoError(t, protocol.VerifyChannelDefinitions(nil, llotypes.ChannelDefinitions{ + 1: { + ReportFormat: llotypes.ReportFormatJSON, + Streams: []llotypes.Stream{{StreamID: 1, Aggregator: llotypes.AggregatorMedian}}, + Opts: []byte(strings.Repeat("x", 1<<20)), + }, + }), "opts length is not yet bounded; see the factory limits derivation") + + // Total stream entries across the set are bounded only per channel + // (MaxStreamsPerChannel) and by unique stream IDs + // (MaxObservationStreamValuesLength), so the same stream repeated across + // channels multiplies the entry count without tripping either. + require.Greater(t, + protocol.MaxStreamsPerChannel*protocol.MaxOutcomeChannelDefinitionsLength, + protocol.MaxObservationStreamValuesLength, + "total stream entries are not yet bounded; see the factory limits derivation") +} diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index a521428..20fa579 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -108,7 +108,7 @@ func (p *Plugin) Observation(_ context.Context, seqNr uint64, _ ocrtypes.Attribu return nil, fmt.Errorf("error fetching shouldRetire from cache: %w", err) } - p.voteOnChannels(&obs, state, seqNr) + p.voteOnChannels(&obs, state) streams = observableStreams(state) } @@ -171,7 +171,7 @@ func observableStreams(state *kvState) []llotypes.StreamID { // voteOnChannels populates obs.RemoveChannelIDs / obs.UpdateChannelDefinitions // by comparing the desired channel definitions against current KV state. -func (p *Plugin) voteOnChannels(obs *Observation, state *kvState, seqNr uint64) { +func (p *Plugin) voteOnChannels(obs *Observation, state *kvState) { obs.RemoveChannelIDs = map[llotypes.ChannelID]struct{}{} expectedChannelDefs := p.ChannelDefinitionCache.Definitions(state.channelDefinitions) diff --git a/llo/dev/v31/statetransition.go b/llo/dev/v31/statetransition.go index 728fb88..6a45198 100644 --- a/llo/dev/v31/statetransition.go +++ b/llo/dev/v31/statetransition.go @@ -58,7 +58,7 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A if err := writeChannelState(kvRW, seqNr, nil); err != nil { return nil, err } - if err := writeHotState(kvRW, 0, nil, nil, nil); err != nil { + if err := writeHotState(kvRW, 0, nil, nil, nil, p.Logger); err != nil { return nil, err } return encodePrecursor(precursor{LifeCycleStage: stage}) @@ -69,7 +69,7 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A return nil, fmt.Errorf("failed to load KV state: %w", err) } - timestamps, validPredecessorRetirementReport, shouldRetireVotes, removeChannelVotesByID, updateDefsByHash, updateVotesByHash, streamObservations, err := p.decodeObservations(ctx, aos, seqNr, bf) + timestamps, validPredecessorRetirementReport, shouldRetireVotes, removeChannelVotesByID, updateDefsByHash, updateVotesByHash, streamObservations, err := p.decodeObservations(ctx, aos, bf) if err != nil { return nil, err } @@ -221,7 +221,7 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A return encodePrecursor(out) } -func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.AttributedObservation, seqNr uint64, bf ocr3_1types.BlobFetcher) ( +func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.AttributedObservation, bf ocr3_1types.BlobFetcher) ( timestampsNanoseconds []uint64, validPredecessorRetirementReport *protocol.RetirementReport, shouldRetireVotes int, @@ -538,7 +538,7 @@ func (p *Plugin) flushKV( } } - return writeHotState(kvRW, out.ObservationTimestampNanoseconds, out.ValidAfterNanoseconds, reportable, carryForward) + return writeHotState(kvRW, out.ObservationTimestampNanoseconds, out.ValidAfterNanoseconds, reportable, carryForward, p.Logger) } // channelDefinitionsChanged reports whether the channel set or any individual diff --git a/llo/protocol/limits.go b/llo/protocol/limits.go index 33e40e8..a2638c9 100644 --- a/llo/protocol/limits.go +++ b/llo/protocol/limits.go @@ -182,6 +182,27 @@ const ( // MiB range. 16 MiB leaves generous headroom. MaxDecompressedObservationLength = 16 << 20 + // MaxPersistedAggregates bounds how many (streamID, aggregator) pairs may + // carry a timestamped aggregate forward across rounds in the v3.1 r/agg + // record. Pairs are ordered by (streamID, aggregator) and those beyond the + // cap are not persisted: their streams simply lose carry-forward and are + // re-aggregated from fresh observations each round, which is a degradation + // rather than a halt. + // + // Without it the count is bounded only indirectly, by + // MaxObservationStreamValuesLength unique streams times the number of + // aggregators each may be aggregated by, which reaches tens of thousands of + // pairs -- past libocr's 2 MiB per-key limit for a single record, at which + // point every oracle's write is rejected and the round fails. + // + // 20_000 is twice MaxObservationStreamValuesLength, so it accommodates every + // observed stream being aggregated two ways -- more than any real + // configuration -- while holding the record to the low MiB range. Note this + // caps the pair count, not the bytes: a single record is only bounded once + // the decimal coefficient is bounded too (see MaxHistoryRecordBytes for the + // same caveat). + MaxPersistedAggregates = 2 * MaxObservationStreamValuesLength + // MaxHistoryBackfillObservations bounds the maximum number of // observations to backfill per definition. MaxHistoryBackfillObservations = 64 diff --git a/llo/v30/plugin.go b/llo/v30/plugin.go index 00d7f34..4b2d3af 100644 --- a/llo/v30/plugin.go +++ b/llo/v30/plugin.go @@ -160,39 +160,39 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re protocol.InitMemoryBallast() return &Plugin{ - f.Config, - onchainConfig.PredecessorConfigDigest, - cfg.ConfigDigest, - f.PredecessorRetirementReportCache, - f.ShouldRetireCache, - f.ChannelDefinitionCache, - f.DataSource, - l, - cfg.N, - cfg.F, - obsCodec, - GetOutcomeCodec(offchainConfig), - f.RetirementReportCodec, - f.ReportCodecs, - f.OutcomeTelemetryCh, - f.ReportTelemetryCh, - f.DonID, - protocol.NewOptsCache(), - sync.Mutex{}, - protocol.NewOptsCache(), - cfg.MaxDurationObservation, - offchainConfig.ProtocolVersion, - offchainConfig.DefaultMinReportIntervalNanoseconds, - }, ocr3types.ReportingPluginInfo{ - Name: "LLO", - Limits: ocr3types.ReportingPluginLimits{ - MaxQueryLength: 0, - MaxObservationLength: MaxObservationLength, - MaxOutcomeLength: MaxOutcomeLength, - MaxReportLength: MaxReportLength, - MaxReportCount: protocol.MaxReportCount, - }, - }, nil + f.Config, + onchainConfig.PredecessorConfigDigest, + cfg.ConfigDigest, + f.PredecessorRetirementReportCache, + f.ShouldRetireCache, + f.ChannelDefinitionCache, + f.DataSource, + l, + cfg.N, + cfg.F, + obsCodec, + GetOutcomeCodec(offchainConfig), + f.RetirementReportCodec, + f.ReportCodecs, + f.OutcomeTelemetryCh, + f.ReportTelemetryCh, + f.DonID, + protocol.NewOptsCache(), + sync.Mutex{}, + protocol.NewOptsCache(), + cfg.MaxDurationObservation, + offchainConfig.ProtocolVersion, + offchainConfig.DefaultMinReportIntervalNanoseconds, + }, ocr3types.ReportingPluginInfo{ + Name: "LLO", + Limits: ocr3types.ReportingPluginLimits{ + MaxQueryLength: 0, + MaxObservationLength: MaxObservationLength, + MaxOutcomeLength: MaxOutcomeLength, + MaxReportLength: MaxReportLength, + MaxReportCount: protocol.MaxReportCount, + }, + }, nil } var _ ocr3types.ReportingPlugin[llotypes.ReportInfo] = &Plugin{} From b783d43308ffbb4d0b0116f96450f06a37804553 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Wed, 16 Sep 2026 10:12:53 +0100 Subject: [PATCH 03/40] SPOR-0003 llo/protocol: bound channel opts and total stream entries at admission MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add three consensus-relevant limits: MaxTotalStreamEntries (50_000) over Σ len(cd.Streams) for the whole definition set, MaxChannelOptsBytes (16 KiB) per channel, and MaxTotalOptsBytes (1 MiB) over the set. Sized against a production definitions file (705 live channels, 4_520 stream entries, 304 KiB of opts, a 475 KiB c/defs record) so the caps sit above where real configurations grow. Worst case at every cap marshals to 1.31 MiB, 66% of libocr's 2 MiB per-key limit. --- llo/dev/v31/factory.go | 19 ++-- llo/dev/v31/limits_test.go | 106 +++++++++++++++++++---- llo/protocol/channel_definitions.go | 37 ++++++++ llo/protocol/channel_definitions_test.go | 101 +++++++++++++++++++++ llo/protocol/limits.go | 39 +++++++++ 5 files changed, 277 insertions(+), 25 deletions(-) diff --git a/llo/dev/v31/factory.go b/llo/dev/v31/factory.go index 5bc94a4..df82334 100644 --- a/llo/dev/v31/factory.go +++ b/llo/dev/v31/factory.go @@ -136,16 +136,15 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re // definitions. // MaxReportsPlusPrecursorBytes the precursor embeds every definition plus // every stream aggregate. The aggregate count - // is bounded by protocol.MaxPersistedAggregates; - // the definition set is bounded in channel - // count (MaxOutcomeChannelDefinitionsLength) - // and per-channel stream count - // (MaxStreamsPerChannel) but NOT yet in total - // stream entries or channel opts bytes, and a - // single stream value is not yet bounded in - // decimal coefficient length. Those three - // bounds are what make this number true rather - // than aspirational. + // is bounded by protocol.MaxPersistedAggregates, + // and the definition set by channel count + // (MaxOutcomeChannelDefinitionsLength), total + // stream entries (MaxTotalStreamEntries) and + // opts bytes (MaxTotalOptsBytes). A single + // stream value is still NOT bounded in decimal + // coefficient length, which is the one bound + // left before this number is true rather than + // aspirational. // MaxKeyValueModifiedKeys* the per-round write set is c/defs plus r/agg // plus the history windows, and history is // held to protocol.MaxHistoryTotalBytes so it diff --git a/llo/dev/v31/limits_test.go b/llo/dev/v31/limits_test.go index 2cdc6f9..59a3211 100644 --- a/llo/dev/v31/limits_test.go +++ b/llo/dev/v31/limits_test.go @@ -155,26 +155,102 @@ func TestLimits_HotStateTruncatesAboveCap(t *testing.T) { } } -// TestLimits_KnownUnboundedInputs documents the bounds that do NOT yet exist, -// so the gap is visible next to the arithmetic that depends on it rather than -// only in a review note. Each of these is a real input to the precursor and the -// c/defs record, and each is bounded only by the checks listed here. -func TestLimits_KnownUnboundedInputs(t *testing.T) { - // Channel opts are raw bytes with no length check on any path. - require.NoError(t, protocol.VerifyChannelDefinitions(nil, llotypes.ChannelDefinitions{ +// TestLimits_DefinitionSetBudgetsAreAdmissionOnly pins the shape of the bounds +// the c/defs and precursor sizes depend on: they are enforced when a channel is +// admitted and NOT when it is already committed, because the committed path runs +// on every oracle every round and rejecting there halts the protocol. A +// grandfathered set therefore stays over budget, which is why these numbers +// bound what can be added rather than what can exist. +func TestLimits_DefinitionSetBudgetsAreAdmissionOnly(t *testing.T) { + oversizedOpts := llotypes.ChannelDefinitions{ 1: { ReportFormat: llotypes.ReportFormatJSON, Streams: []llotypes.Stream{{StreamID: 1, Aggregator: llotypes.AggregatorMedian}}, - Opts: []byte(strings.Repeat("x", 1<<20)), + Opts: []byte(strings.Repeat("x", protocol.MaxChannelOptsBytes+1)), }, - }), "opts length is not yet bounded; see the factory limits derivation") + } + + // Committed: accepted, so the round proceeds. + require.NoError(t, protocol.VerifyChannelDefinitions(nil, oversizedOpts)) - // Total stream entries across the set are bounded only per channel - // (MaxStreamsPerChannel) and by unique stream IDs - // (MaxObservationStreamValuesLength), so the same stream repeated across - // channels multiplies the entry count without tripping either. + // Being admitted: refused, so it never becomes committed in the first place. + err := protocol.VerifyChannelDefinitionsForAdmission(nil, oversizedOpts, map[llotypes.ChannelID]struct{}{1: {}}) + require.ErrorContains(t, err, "opts that are too long") + + // The per-channel caps still do not bound total stream entries between + // them; MaxTotalStreamEntries is the bound that does, and it is what the + // definitions-record and precursor sizes are derived from. require.Greater(t, protocol.MaxStreamsPerChannel*protocol.MaxOutcomeChannelDefinitionsLength, - protocol.MaxObservationStreamValuesLength, - "total stream entries are not yet bounded; see the factory limits derivation") + protocol.MaxTotalStreamEntries) +} + +// TestLimits_ChannelStateWorstCaseFitsPerKeyLimit builds the largest definition +// set the admission budgets permit and measures the record it marshals to. The +// budgets were sized from this number, so measuring it is what keeps them +// honest: c/defs is a single value under a single key, and libocr rejects a +// write above the per-key limit. +func TestLimits_ChannelStateWorstCaseFitsPerKeyLimit(t *testing.T) { + const channels = protocol.MaxOutcomeChannelDefinitionsLength + const streamsEach = protocol.MaxTotalStreamEntries / channels + const optsEach = protocol.MaxTotalOptsBytes / channels + + defs := make(llotypes.ChannelDefinitions, channels) + for c := range channels { + streams := make([]llotypes.Stream, 0, streamsEach) + for i := range streamsEach { + streams = append(streams, llotypes.Stream{ + StreamID: llotypes.StreamID(i + 1), + Aggregator: llotypes.AggregatorMedian, + }) + } + defs[llotypes.ChannelID(c+1)] = llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatEVMPremiumLegacy, + Streams: streams, + Opts: []byte(strings.Repeat("x", optsEach)), + } + } + // Every budget at its cap, so the record measured below is the worst case + // admission permits rather than an arbitrary large set. Spreading the opts + // budget evenly loses less than one byte per channel to integer division, + // which is why this is a bound rather than an equality. + require.Equal(t, protocol.MaxTotalStreamEntries, channels*streamsEach) + require.LessOrEqual(t, channels*optsEach, protocol.MaxTotalOptsBytes) + require.Greater(t, channels*optsEach, protocol.MaxTotalOptsBytes-channels) + require.LessOrEqual(t, optsEach, protocol.MaxChannelOptsBytes) + + kv := newMemKV() + require.NoError(t, writeChannelState(kv, 1, defs)) + record, err := kv.Read(keyChannelState) + require.NoError(t, err) + require.LessOrEqual(t, len(record), ocr3_1types.MaxMaxKeyValueValueBytes, + "worst-case c/defs record (%d bytes) exceeds the per-key limit", len(record)) + t.Logf("worst-case c/defs record: %d bytes of the %d per-key limit", + len(record), ocr3_1types.MaxMaxKeyValueValueBytes) +} + +// TestLimits_KnownUnboundedInputs documents the bound that does NOT yet exist, +// so the gap stays visible next to the arithmetic that depends on it. A decimal +// decoded from an untrusted source is bounded in exponent but not in coefficient +// length, so a single stream value is unbounded in bytes on every path that +// carries one: r/agg, the precursor, observations and blobs. Only history +// records are protected, by protocol.MaxHistoryRecordBytes. +func TestLimits_KnownUnboundedInputs(t *testing.T) { + // A 1000-digit coefficient, well inside the permitted exponent range. + huge := decimal.New(1, 0) + for range 1000 { + huge = huge.Mul(decimal.New(10, 0)) + } + require.LessOrEqual(t, huge.Exponent(), int32(protocol.MaxDecimalExponent)) + + encoded, err := protocol.ToDecimal(huge).MarshalBinary() + require.NoError(t, err) + require.Greater(t, len(encoded), protocol.MaxHistoryRecordBytes, + "a single stream value already exceeds the per-history-record bound") + + _, err = protocol.UnmarshalProtoStreamValue(&protocol.LLOStreamValue{ + Type: protocol.LLOStreamValue_Decimal, + Value: encoded, + }) + require.NoError(t, err, "decimal coefficient length is not yet bounded; see the factory limits derivation") } diff --git a/llo/protocol/channel_definitions.go b/llo/protocol/channel_definitions.go index 40b3bdc..534b05d 100644 --- a/llo/protocol/channel_definitions.go +++ b/llo/protocol/channel_definitions.go @@ -53,12 +53,23 @@ func ChangedChannelIDs(current, desired llotypes.ChannelDefinitions) map[llotype // admissionFinding is an admission-only check failure, together with every // channel it implicates, so that it can be filtered by the admitting set. +// +// A whole-set finding (a budget summed over every channel) implicates no +// particular channel: it is a property of the set as a whole, and the channels +// that made it exceed the budget are not distinguishable from the ones that did +// not. Such a finding sets wholeSet and applies whenever anything is being +// admitted, which is what stops a set already over budget from growing while +// leaving a set that is merely already over it alone. type admissionFinding struct { channels []llotypes.ChannelID + wholeSet bool err error } func (f admissionFinding) appliesTo(admitting map[llotypes.ChannelID]struct{}) bool { + if f.wholeSet { + return len(admitting) > 0 + } for _, channelID := range f.channels { if _, ok := admitting[channelID]; ok { return true @@ -88,6 +99,14 @@ func verifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, chan admit := func(err error, channels ...llotypes.ChannelID) { admissionFindings = append(admissionFindings, admissionFinding{channels: channels, err: err}) } + admitSet := func(err error) { + admissionFindings = append(admissionFindings, admissionFinding{wholeSet: true, err: err}) + } + + // Whole-set budgets, accumulated over the channels the loop below visits + // (tombstones excluded: they carry neither streams nor opts that anything + // reads) and checked once at the end. + var totalStreamEntries, totalOptsBytes int uniqueStreamIDs := make(map[llotypes.StreamID]struct{}, len(channelDefs)) // Owners of every stream ID that will hold an aggregate: observed streams @@ -115,6 +134,15 @@ func verifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, chan merr = errors.Join(merr, fmt.Errorf("ChannelDefinition with ID %d has too many streams, got: %d/%d", channelID, len(cd.Streams), MaxStreamsPerChannel)) continue } + totalStreamEntries += len(cd.Streams) + // Opts are opaque bytes, so length is the only thing that can be + // checked here. Admission-only: a committed definition carrying an + // oversized blob is left alone rather than failing verification on every + // oracle, every round. + if len(cd.Opts) > MaxChannelOptsBytes { + admit(fmt.Errorf("ChannelDefinition with ID %d has opts that are too long, got: %d/%d", channelID, len(cd.Opts), MaxChannelOptsBytes), channelID) + } + totalOptsBytes += len(cd.Opts) for _, strm := range cd.Streams { if strm.Aggregator == 0 { merr = errors.Join(merr, fmt.Errorf("ChannelDefinition with ID %d has stream %d with zero aggregator (this may indicate an uninitialized struct)", channelID, strm.StreamID)) @@ -204,6 +232,15 @@ func verifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, chan } } + // Whole-set budgets. Both are what the sizes of the channel-definitions + // record and the precursor actually depend on; see the limits they name. + if totalStreamEntries > MaxTotalStreamEntries { + admitSet(fmt.Errorf("too many stream entries across all channels, got: %d/%d", totalStreamEntries, MaxTotalStreamEntries)) + } + if totalOptsBytes > MaxTotalOptsBytes { + admitSet(fmt.Errorf("too many opts bytes across all channels, got: %d/%d", totalOptsBytes, MaxTotalOptsBytes)) + } + for _, finding := range admissionFindings { if finding.appliesTo(admitting) { merr = errors.Join(merr, finding.err) diff --git a/llo/protocol/channel_definitions_test.go b/llo/protocol/channel_definitions_test.go index edbc7bf..7cbca35 100644 --- a/llo/protocol/channel_definitions_test.go +++ b/llo/protocol/channel_definitions_test.go @@ -2,6 +2,8 @@ package protocol import ( "errors" + "fmt" + "strings" "testing" "github.com/stretchr/testify/require" @@ -389,6 +391,105 @@ func Test_VerifyChannelDefinitions_AdmissionScope(t *testing.T) { }) } +func Test_VerifyChannelDefinitions_SizeBudgets(t *testing.T) { + codecs := map[llotypes.ReportFormat]ReportCodec{0: stubReportCodec{}} + + channel := func(streams int, optsBytes int) llotypes.ChannelDefinition { + cd := llotypes.ChannelDefinition{Streams: make([]llotypes.Stream, 0, streams)} + for i := range streams { + cd.Streams = append(cd.Streams, llotypes.Stream{StreamID: llotypes.StreamID(i + 1), Aggregator: llotypes.AggregatorMedian}) + } + if optsBytes > 0 { + cd.Opts = []byte(strings.Repeat("x", optsBytes)) + } + return cd + } + // Channels sharing one stream ID, so the whole-set entry count grows while + // the unique-stream-ID cap stays untouched. That is the case the per-channel + // and per-set caps miss between them. + sharedStream := func(channels, streamsEach int) llotypes.ChannelDefinitions { + defs := llotypes.ChannelDefinitions{} + for c := range channels { + cd := llotypes.ChannelDefinition{Streams: make([]llotypes.Stream, 0, streamsEach)} + for i := range streamsEach { + cd.Streams = append(cd.Streams, llotypes.Stream{StreamID: llotypes.StreamID(i + 1), Aggregator: llotypes.AggregatorMedian}) + } + defs[llotypes.ChannelID(c+1)] = cd + } + return defs + } + + t.Run("per-channel opts length", func(t *testing.T) { + atLimit := llotypes.ChannelDefinitions{1: channel(1, MaxChannelOptsBytes)} + require.NoError(t, verifyAdmittingAll(codecs, atLimit)) + + over := llotypes.ChannelDefinitions{1: channel(1, MaxChannelOptsBytes+1)} + require.EqualError(t, verifyAdmittingAll(codecs, over), + fmt.Sprintf("ChannelDefinition with ID 1 has opts that are too long, got: %d/%d", MaxChannelOptsBytes+1, MaxChannelOptsBytes)) + }) + + t.Run("total stream entries across the set", func(t *testing.T) { + // Ten channels listing the same 5_000 streams: 50_000 entries against + // 5_000 unique IDs, and every channel well under MaxStreamsPerChannel. + atLimit := sharedStream(10, MaxTotalStreamEntries/10) + require.NoError(t, verifyAdmittingAll(codecs, atLimit)) + + over := sharedStream(11, MaxTotalStreamEntries/10) + require.EqualError(t, verifyAdmittingAll(codecs, over), + fmt.Sprintf("too many stream entries across all channels, got: %d/%d", 11*(MaxTotalStreamEntries/10), MaxTotalStreamEntries)) + }) + + t.Run("total opts bytes across the set", func(t *testing.T) { + // Every channel individually within MaxChannelOptsBytes; only the sum + // exceeds the budget. + set := func(channels int) llotypes.ChannelDefinitions { + defs := llotypes.ChannelDefinitions{} + for c := range channels { + defs[llotypes.ChannelID(c+1)] = channel(1, MaxChannelOptsBytes) + } + return defs + } + atLimit := set(MaxTotalOptsBytes / MaxChannelOptsBytes) + require.NoError(t, verifyAdmittingAll(codecs, atLimit)) + + over := set(MaxTotalOptsBytes/MaxChannelOptsBytes + 1) + require.EqualError(t, verifyAdmittingAll(codecs, over), + fmt.Sprintf("too many opts bytes across all channels, got: %d/%d", MaxTotalOptsBytes+MaxChannelOptsBytes, MaxTotalOptsBytes)) + }) + + t.Run("budgets are admission-only", func(t *testing.T) { + // A committed set over budget must not fail verification: every oracle + // runs this every round, so rejecting it would halt the protocol over + // definitions that are already installed. + over := llotypes.ChannelDefinitions{1: channel(1, MaxChannelOptsBytes+1)} + require.NoError(t, VerifyChannelDefinitions(codecs, over)) + require.NoError(t, VerifyChannelDefinitionsForAdmission(codecs, over, nil)) + require.NoError(t, VerifyChannelDefinitionsForAdmission(codecs, over, map[llotypes.ChannelID]struct{}{})) + }) + + t.Run("a whole-set finding applies to whatever is being admitted", func(t *testing.T) { + // The budget is a property of the set, so the channels that pushed it + // over are not distinguishable. Admitting any channel is refused while + // the set is over budget, which stops it growing; admitting nothing is + // left alone. + over := sharedStream(11, MaxTotalStreamEntries/10) + for _, admitted := range []llotypes.ChannelID{1, 11} { + err := VerifyChannelDefinitionsForAdmission(codecs, over, map[llotypes.ChannelID]struct{}{admitted: {}}) + require.ErrorContains(t, err, "too many stream entries across all channels") + } + }) + + t.Run("tombstones cost nothing", func(t *testing.T) { + // A tombstone carries neither streams nor opts that anything reads, so + // it must not consume either budget. + defs := sharedStream(11, MaxTotalStreamEntries/10) + tombstoned := defs[11] + tombstoned.Tombstone = true + defs[11] = tombstoned + require.NoError(t, verifyAdmittingAll(codecs, defs)) + }) +} + func Test_ChangedChannelIDs(t *testing.T) { streams := []llotypes.Stream{{StreamID: 1, Aggregator: llotypes.AggregatorMedian}} current := llotypes.ChannelDefinitions{ diff --git a/llo/protocol/limits.go b/llo/protocol/limits.go index a2638c9..c4dd0bd 100644 --- a/llo/protocol/limits.go +++ b/llo/protocol/limits.go @@ -39,6 +39,45 @@ const ( // can be supported MaxOutcomeChannelDefinitionsLength = MaxReportCount + // MaxTotalStreamEntries bounds the sum of len(cd.Streams) over the whole + // definition set. + // + // The per-channel and per-set caps do not bound this between them: + // MaxStreamsPerChannel times MaxOutcomeChannelDefinitionsLength permits + // 20 million entries, and MaxObservationStreamValuesLength counts only + // DISTINCT stream IDs, so the same stream listed by many channels costs + // nothing against it. Every entry is carried in the channel-definitions + // record and again in the precursor, so the total is what those sizes + // actually depend on. + // + // 50_000 is five channels at MaxStreamsPerChannel, or five entries for every + // observable stream -- far past any real configuration, while holding the + // definitions record itself to well under a MiB. + MaxTotalStreamEntries = 50_000 + // MaxChannelOptsBytes bounds one channel's opts blob. Opts are opaque JSON + // decoded per report format, so nothing else constrains their length, and + // they travel in the definitions record, the precursor and observations that + // vote to add a channel. + // + // 16 KiB is roughly two orders of magnitude above the largest real opts (an + // ABI plus expressions, a few hundred bytes). + MaxChannelOptsBytes = 16 << 10 + // MaxTotalOptsBytes bounds the sum over the whole set, because + // MaxChannelOptsBytes alone would still permit MaxOutcomeChannelDefinitionsLength + // (2_000) channels times 16 KiB, or 32 MiB. + // + // 1 MiB is ~512 B per channel at the channel-count limit. A production DON + // measured 304 KiB of opts over 705 live channels (431 B each, nearly all + // ABI), so this leaves it room to more than double its channel count before + // the budget binds -- the bound has to be above where real configurations + // grow, or it refuses legitimate admissions rather than abuse. + // + // Worst case it still leaves the definitions record inside libocr's 2 MiB + // per-key limit: 1 MiB of opts plus MaxTotalStreamEntries worth of stream + // entries plus per-channel framing measures 1.31 MiB, 66% of the limit + // (measured by TestLimits_ChannelStateWorstCaseFitsPerKeyLimit). + MaxTotalOptsBytes = 1 << 20 + // Stream history limits. // // A history "pair" is a (streamID, aggregator) tuple: the identity of one From b77c719bb3891130a92c51b6041a0fc041b11ff1 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Wed, 16 Sep 2026 11:07:20 +0100 Subject: [PATCH 04/40] SPOR-0003 llo: bound decimal coefficients on observation decode MaxDecimalExponent bounds where a decimal's point sits, not how many digits precede it, so a value with a legal exponent could carry an arbitrarily long coefficient. A single stream value was therefore unbounded in bytes on every path carrying one. --- llo/dev/v31/blobcompress_fuzz_test.go | 33 +++++ llo/dev/v31/factory.go | 8 +- llo/dev/v31/limits_test.go | 20 ++- llo/dev/v31/observation.go | 2 +- llo/dev/v31/observation_coefficient_test.go | 104 ++++++++++++++++ llo/protocol/limits.go | 26 ++++ llo/protocol/stream_value.go | 77 ++++++++++++ llo/protocol/stream_value_coefficient_test.go | 117 ++++++++++++++++++ llo/protocol/stream_value_fuzz_test.go | 46 +++++++ llo/v30/observation_codec.go | 2 +- llo/v30/observation_codec_coefficient_test.go | 42 +++++++ 11 files changed, 459 insertions(+), 18 deletions(-) create mode 100644 llo/dev/v31/blobcompress_fuzz_test.go create mode 100644 llo/dev/v31/observation_coefficient_test.go create mode 100644 llo/protocol/stream_value_coefficient_test.go create mode 100644 llo/protocol/stream_value_fuzz_test.go create mode 100644 llo/v30/observation_codec_coefficient_test.go diff --git a/llo/dev/v31/blobcompress_fuzz_test.go b/llo/dev/v31/blobcompress_fuzz_test.go new file mode 100644 index 0000000..cc36a5b --- /dev/null +++ b/llo/dev/v31/blobcompress_fuzz_test.go @@ -0,0 +1,33 @@ +package llo + +import ( + "testing" +) + +// FuzzDecodeBlobPayload feeds arbitrary bytes through the blob payload decoder. +// Blob payloads are attacker-controlled, so the contract is that any input +// either returns bytes within the caller's budget or errors -- never a panic, +// and never more bytes than the budget allows (a zstd bomb). +func FuzzDecodeBlobPayload(f *testing.F) { + raw := []byte("stream values would go here") + framed, err := encodeBlobPayload(raw) + if err != nil { + f.Fatal(err) + } + f.Add(framed) + f.Add(append([]byte{blobCodecRaw}, raw...)) + f.Add([]byte{blobCodecZstd}) + f.Add([]byte{}) + + f.Fuzz(func(t *testing.T, payload []byte) { + for _, budget := range []int{0, 1, 1 << 10, maxDecompressedBlobPayloadBytes} { + out, err := decodeBlobPayload(payload, budget) + if err != nil { + continue + } + if len(out) > budget { + t.Fatalf("decoded %d bytes against a budget of %d", len(out), budget) + } + } + }) +} diff --git a/llo/dev/v31/factory.go b/llo/dev/v31/factory.go index df82334..fbb0251 100644 --- a/llo/dev/v31/factory.go +++ b/llo/dev/v31/factory.go @@ -140,11 +140,9 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re // and the definition set by channel count // (MaxOutcomeChannelDefinitionsLength), total // stream entries (MaxTotalStreamEntries) and - // opts bytes (MaxTotalOptsBytes). A single - // stream value is still NOT bounded in decimal - // coefficient length, which is the one bound - // left before this number is true rather than - // aspirational. + // opts bytes (MaxTotalOptsBytes) at admission. + // A stream value's decimal coefficient is bounded + // on observation decode. // MaxKeyValueModifiedKeys* the per-round write set is c/defs plus r/agg // plus the history windows, and history is // held to protocol.MaxHistoryTotalBytes so it diff --git a/llo/dev/v31/limits_test.go b/llo/dev/v31/limits_test.go index 59a3211..b843e3c 100644 --- a/llo/dev/v31/limits_test.go +++ b/llo/dev/v31/limits_test.go @@ -229,12 +229,9 @@ func TestLimits_ChannelStateWorstCaseFitsPerKeyLimit(t *testing.T) { len(record), ocr3_1types.MaxMaxKeyValueValueBytes) } -// TestLimits_KnownUnboundedInputs documents the bound that does NOT yet exist, -// so the gap stays visible next to the arithmetic that depends on it. A decimal -// decoded from an untrusted source is bounded in exponent but not in coefficient -// length, so a single stream value is unbounded in bytes on every path that -// carries one: r/agg, the precursor, observations and blobs. Only history -// records are protected, by protocol.MaxHistoryRecordBytes. +// Observation decode rejects an oversized coefficient (see +// protocol.UnmarshalObservedProtoStreamValue), which bounds everything written +// from an observation going forward. func TestLimits_KnownUnboundedInputs(t *testing.T) { // A 1000-digit coefficient, well inside the permitted exponent range. huge := decimal.New(1, 0) @@ -242,15 +239,16 @@ func TestLimits_KnownUnboundedInputs(t *testing.T) { huge = huge.Mul(decimal.New(10, 0)) } require.LessOrEqual(t, huge.Exponent(), int32(protocol.MaxDecimalExponent)) + require.Greater(t, huge.Coefficient().BitLen(), protocol.MaxDecimalCoefficientBits) encoded, err := protocol.ToDecimal(huge).MarshalBinary() require.NoError(t, err) require.Greater(t, len(encoded), protocol.MaxHistoryRecordBytes, "a single stream value already exceeds the per-history-record bound") - _, err = protocol.UnmarshalProtoStreamValue(&protocol.LLOStreamValue{ - Type: protocol.LLOStreamValue_Decimal, - Value: encoded, - }) - require.NoError(t, err, "decimal coefficient length is not yet bounded; see the factory limits derivation") + pb := &protocol.LLOStreamValue{Type: protocol.LLOStreamValue_Decimal, Value: encoded} + + // Phase 1: an observation carrying it is rejected. + _, err = protocol.UnmarshalObservedProtoStreamValue(pb) + require.ErrorIs(t, err, protocol.ErrDecimalCoefficientOutOfRange) } diff --git a/llo/dev/v31/observation.go b/llo/dev/v31/observation.go index 93f59fc..e642f71 100644 --- a/llo/dev/v31/observation.go +++ b/llo/dev/v31/observation.go @@ -244,5 +244,5 @@ func streamValueFromProtoAllowNil(pb *protocol.LLOStreamValue) (protocol.StreamV if pb == nil { return nil, nil } - return protocol.UnmarshalProtoStreamValue(pb) + return protocol.UnmarshalObservedProtoStreamValue(pb) } diff --git a/llo/dev/v31/observation_coefficient_test.go b/llo/dev/v31/observation_coefficient_test.go new file mode 100644 index 0000000..7e9525b --- /dev/null +++ b/llo/dev/v31/observation_coefficient_test.go @@ -0,0 +1,104 @@ +package llo + +import ( + "math/big" + "testing" + + "github.com/shopspring/decimal" + "github.com/stretchr/testify/require" + + llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" + "github.com/smartcontractkit/chainlink-common/pkg/utils/tests" + + protocol "github.com/smartcontractkit/chainlink-data-streams/llo/protocol" + + ocrtypes "github.com/smartcontractkit/libocr/offchainreporting2plus/types" +) + +// Test_decodeObservation_CoefficientBound covers the blob-carried path, which is +// how v31 observations carry stream values at all: a value over the coefficient +// bound must be rejected there, not merely in the protocol-level decoder. +// +// The rejection is a plain error rather than a blobFetchError, which is what +// makes it deterministic across oracles and therefore safe for +// decodeObservations to drop this observation alone instead of failing the round. +func Test_decodeObservation_CoefficientBound(t *testing.T) { + ctx := tests.Context(t) + + overSized := decimal.NewFromBigInt(new(big.Int).Lsh(big.NewInt(1), protocol.MaxDecimalCoefficientBits), -2) + atLimit := decimal.NewFromBigInt(new(big.Int).Lsh(big.NewInt(1), protocol.MaxDecimalCoefficientBits-1), -2) + + encode := func(t *testing.T, d decimal.Decimal) []byte { + return mustEncodeObs(t, Observation{ + UnixTimestampNanoseconds: 1, + StreamValues: protocol.StreamValues{1: protocol.ToDecimal(d)}, + }) + } + + obs, err := decodeObservation(ctx, encode(t, atLimit), testBlobs) + require.NoError(t, err) + require.Len(t, obs.StreamValues, 1) + + _, err = decodeObservation(ctx, encode(t, overSized), testBlobs) + require.ErrorIs(t, err, protocol.ErrDecimalCoefficientOutOfRange) + + var bfErr *blobFetchError + require.NotErrorAs(t, err, &bfErr, + "must be a deterministic error so the observation is dropped, not the round") +} + +// Test_StateTransition_DropsObservationOverCoefficientBound is the consequence +// that matters: an oracle sending an unbounded value costs its own observation +// and nothing else. The round completes and the remaining observations still +// aggregate, which is why this bound is safe to enforce on the consensus path. +func Test_StateTransition_DropsObservationOverCoefficientBound(t *testing.T) { + ctx := tests.Context(t) + p := testPlugin(t) + kv := newMemKV() + + channel := llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatJSON, + Streams: []llotypes.Stream{{StreamID: 1, Aggregator: llotypes.AggregatorMedian}}, + } + + // Bootstrap, then install the channel so it is effective for the round below. + _, err := p.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, nil), ao(1, nil), ao(2, nil), ao(3, nil)}, kv, testBlobs) + require.NoError(t, err) + _, err = p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, addChannelRound(t, 1_000, 100, channel), kv, testBlobs) + require.NoError(t, err) + _, err = p.StateTransition(ctx, 3, ocrtypes.AttributedQuery{}, addChannelRound(t, 2_000, 100, channel), kv, testBlobs) + require.NoError(t, err) + + good := decimal.NewFromInt(42) + overSized := decimal.NewFromBigInt(new(big.Int).Lsh(big.NewInt(1), protocol.MaxDecimalCoefficientBits), -2) + + obsWith := func(d decimal.Decimal) []byte { + return mustEncodeObs(t, Observation{ + UnixTimestampNanoseconds: 3_000, + StreamValues: protocol.StreamValues{1: protocol.ToDecimal(d)}, + }) + } + // Three honest observations and one unbounded, which is f=1 for this plugin. + aos := []ocrtypes.AttributedObservation{ + ao(0, obsWith(good)), + ao(1, obsWith(good)), + ao(2, obsWith(good)), + ao(3, obsWith(overSized)), + } + + precBytes, err := p.StateTransition(ctx, 4, ocrtypes.AttributedQuery{}, aos, kv, testBlobs) + require.NoError(t, err, "one unbounded observation must not fail the round") + + prec, err := decodePrecursor(precBytes) + require.NoError(t, err) + aggregated, ok := prec.StreamAggregates[1][llotypes.AggregatorMedian] + require.True(t, ok, "the honest observations must still aggregate") + require.Equal(t, good.String(), mustText(t, aggregated)) +} + +func mustText(t *testing.T, sv protocol.StreamValue) string { + t.Helper() + b, err := sv.MarshalText() + require.NoError(t, err) + return string(b) +} diff --git a/llo/protocol/limits.go b/llo/protocol/limits.go index c4dd0bd..351711c 100644 --- a/llo/protocol/limits.go +++ b/llo/protocol/limits.go @@ -35,6 +35,32 @@ const ( // Stream values need only a couple of dozen decimal places, so this // leaves enough headroom. MaxDecimalExponent = 1_000 + // MaxDecimalCoefficientBits bounds the coefficient of a decimal carried by + // an observation, which MaxDecimalExponent does not: the exponent says where + // the point sits, not how many digits precede it, so a value with a legal + // exponent can still carry an arbitrarily long coefficient and be + // arbitrarily large on the wire. + // + // 192 bits is 58 decimal digits, derived from MaxHistoryRecordBytes + // A timestamped quote, the largest shape a stream value takes carrying + // three coefficients at this bound measures 125B as a history record, + // the most that fits the 128B per-record limit. + // One step up (224 bits) measures 137 B, which observation decode would accept + // and history would then refuse, leaving a gap in the series for a value the round agreed on. + // Asserted by TestDecimalCoefficientBoundFitsHistoryRecord. + // + // For scale, the fixtures in the size table below are 57 bits (an 18-digit + // price) and 124 bits (38 digits), so this is ~3x and ~1.5x those. + // Increasing this limit needs to move this constant and MaxHistoryRecordBytes together. + // + // Enforced at observation decode (see UnmarshalObservedProtoStreamValue). + MaxDecimalCoefficientBits = 192 + // MaxStreamValueNesting bounds how deeply a stream value may nest another. + // Only TimestampedStreamValue nests, and only one level is meaningful, so + // this exists to keep the bounds check over untrusted bytes from recursing + // on a value crafted to nest. + MaxStreamValueNesting = 4 + // MaxOutcomeChannelDefinitionsLength is the maximum number of channels that // can be supported MaxOutcomeChannelDefinitionsLength = MaxReportCount diff --git a/llo/protocol/stream_value.go b/llo/protocol/stream_value.go index d35f911..94e7f16 100644 --- a/llo/protocol/stream_value.go +++ b/llo/protocol/stream_value.go @@ -35,8 +35,85 @@ var ( // well-behaved node never encodes such a value; accepting one would let a // single byzantine node force unbounded rescale work on every honest node. ErrDecimalExponentOutOfRange = errors.New("decimal exponent out of range") + // ErrDecimalCoefficientOutOfRange is returned when a decimal carried by an + // observation has a coefficient longer than MaxDecimalCoefficientBits. The + // exponent bound says nothing about coefficient length, so without this a + // single stream value is unbounded in bytes. + ErrDecimalCoefficientOutOfRange = errors.New("decimal coefficient out of range") + // ErrStreamValueNestingTooDeep is returned when a stream value nests deeper + // than MaxStreamValueNesting. + ErrStreamValueNestingTooDeep = errors.New("stream value nesting too deep") ) +// UnmarshalObservedProtoStreamValue decodes a stream value that arrived in a +// peer's observation, and additionally enforces the bounds that only observed +// values are held to today. +// +// Observation decode is the one untrusted entry point where a rejection is +// cheap: it is a pure function of the observation bytes, so every oracle reaches +// the same verdict, and both callers already discard an individual observation +// that fails to decode rather than failing the round (see +// decodeObservations). The same bound applied to values decoded from stored +// state -- a v3.0 outcome, a v3.1 r/agg record -- would instead reject what is +// already persisted and fail decode on every upgraded oracle at once, so those +// paths keep using UnmarshalProtoStreamValue until that change can be +// coordinated across versions. +// +// Bounding observations bounds everything written from them going forward. +func UnmarshalObservedProtoStreamValue(enc *LLOStreamValue) (StreamValue, error) { + sv, err := UnmarshalProtoStreamValue(enc) + if err != nil { + return nil, err + } + if err := checkObservedStreamValue(sv, 0); err != nil { + return nil, err + } + return sv, nil +} + +// checkObservedStreamValue applies the observation-only bounds to every decimal +// a stream value carries, at any nesting depth. +func checkObservedStreamValue(sv StreamValue, depth int) error { + if depth > MaxStreamValueNesting { + return fmt.Errorf("%w: got more than %d levels", ErrStreamValueNestingTooDeep, MaxStreamValueNesting) + } + switch v := sv.(type) { + case nil: + return nil + case *Decimal: + return checkDecimalCoefficient(v.Decimal()) + case *Quote: + if v == nil { + return nil + } + for _, d := range []decimal.Decimal{v.Bid, v.Benchmark, v.Ask} { + if err := checkDecimalCoefficient(d); err != nil { + return err + } + } + return nil + case *TimestampedStreamValue: + if v == nil { + return nil + } + return checkObservedStreamValue(v.StreamValue, depth+1) + default: + // An unknown type carries no decimal this function knows how to reach. + // UnmarshalProtoStreamValue rejects types it does not recognize, so this + // is unreachable rather than a silent pass. + return nil + } +} + +// checkDecimalCoefficient bounds the coefficient length of a decimal carried by +// an observation. See MaxDecimalCoefficientBits. +func checkDecimalCoefficient(d decimal.Decimal) error { + if bits := d.Coefficient().BitLen(); bits > MaxDecimalCoefficientBits { + return fmt.Errorf("%w: got %d bits, expected <= %d", ErrDecimalCoefficientOutOfRange, bits, MaxDecimalCoefficientBits) + } + return nil +} + // checkDecimalExponent bounds the exponent of a decimal decoded from an // untrusted source. See MaxDecimalExponent. func checkDecimalExponent(d decimal.Decimal) error { diff --git a/llo/protocol/stream_value_coefficient_test.go b/llo/protocol/stream_value_coefficient_test.go new file mode 100644 index 0000000..637912d --- /dev/null +++ b/llo/protocol/stream_value_coefficient_test.go @@ -0,0 +1,117 @@ +package protocol + +import ( + "math/big" + "testing" + + "github.com/shopspring/decimal" + "github.com/stretchr/testify/require" +) + +// bigDecimal returns a decimal whose coefficient is exactly bits long, with an +// exponent well inside MaxDecimalExponent so the exponent bound cannot be what +// rejects it. +func bigDecimal(bits int) decimal.Decimal { + coefficient := new(big.Int).Lsh(big.NewInt(1), uint(bits-1)) + return decimal.NewFromBigInt(coefficient, -2) +} + +func protoOf(t *testing.T, sv StreamValue) *LLOStreamValue { + t.Helper() + pb, err := StreamValueToProto(sv) + require.NoError(t, err) + return pb +} + +func Test_UnmarshalObservedProtoStreamValue_CoefficientBound(t *testing.T) { + atLimit := bigDecimal(MaxDecimalCoefficientBits) + over := bigDecimal(MaxDecimalCoefficientBits + 1) + + // The exponent bound is not what does the work here: both values are well + // inside it, which is the whole point of adding a coefficient bound. + require.NoError(t, checkDecimalExponent(atLimit)) + require.NoError(t, checkDecimalExponent(over)) + + t.Run("decimal", func(t *testing.T) { + _, err := UnmarshalObservedProtoStreamValue(protoOf(t, ToDecimal(atLimit))) + require.NoError(t, err) + + _, err = UnmarshalObservedProtoStreamValue(protoOf(t, ToDecimal(over))) + require.ErrorIs(t, err, ErrDecimalCoefficientOutOfRange) + }) + + t.Run("quote rejects any of its three fields", func(t *testing.T) { + small := decimal.NewFromInt(1) + for _, q := range []*Quote{ + {Bid: over, Benchmark: small, Ask: small}, + {Bid: small, Benchmark: over, Ask: small}, + {Bid: small, Benchmark: small, Ask: over}, + } { + _, err := UnmarshalObservedProtoStreamValue(protoOf(t, q)) + require.ErrorIs(t, err, ErrDecimalCoefficientOutOfRange) + } + _, err := UnmarshalObservedProtoStreamValue(protoOf(t, &Quote{Bid: atLimit, Benchmark: atLimit, Ask: atLimit})) + require.NoError(t, err) + }) + + t.Run("timestamped values are checked through the wrapper", func(t *testing.T) { + _, err := UnmarshalObservedProtoStreamValue(protoOf(t, &TimestampedStreamValue{ + ObservedAtNanoseconds: 1, + StreamValue: ToDecimal(over), + })) + require.ErrorIs(t, err, ErrDecimalCoefficientOutOfRange) + + _, err = UnmarshalObservedProtoStreamValue(protoOf(t, &TimestampedStreamValue{ + ObservedAtNanoseconds: 1, + StreamValue: ToDecimal(atLimit), + })) + require.NoError(t, err) + }) + + t.Run("nesting is bounded", func(t *testing.T) { + var sv StreamValue = ToDecimal(decimal.NewFromInt(1)) + for range MaxStreamValueNesting + 1 { + sv = &TimestampedStreamValue{ObservedAtNanoseconds: 1, StreamValue: sv} + } + require.ErrorIs(t, checkObservedStreamValue(sv, 0), ErrStreamValueNestingTooDeep) + }) + + t.Run("stored-state decode is deliberately unchecked", func(t *testing.T) { + // Phase 1 bounds observations only. Rejecting a value already persisted + // would fail decode on every upgraded oracle at once, so this path must + // keep accepting it until the change is coordinated across versions. + sv, err := UnmarshalProtoStreamValue(protoOf(t, ToDecimal(over))) + require.NoError(t, err) + require.NotNil(t, sv) + }) +} + +// TestDecimalCoefficientBoundFitsHistoryRecord keeps MaxDecimalCoefficientBits +// and MaxHistoryRecordBytes consistent. Observation decode admits a value and +// history then has to store it, so a bound the other refuses means a value the +// round agreed on leaves a gap in the series. MaxDecimalCoefficientBits is +// derived from this relationship, so raising either constant without the other +// fails here. +func TestDecimalCoefficientBoundFitsHistoryRecord(t *testing.T) { + // A timestamped quote is the largest shape a stream value takes: three + // coefficients plus the wrapper's timestamp. + worst := func(bits int) StreamValue { + d := bigDecimal(bits) + return &TimestampedStreamValue{ + ObservedAtNanoseconds: 1_700_000_000_000_000_000, + StreamValue: &Quote{Bid: d, Benchmark: d, Ask: d}, + } + } + + size, err := historyRecordSize(1_700_000_000_000_000_000, worst(MaxDecimalCoefficientBits)) + require.NoError(t, err) + require.LessOrEqual(t, size, MaxHistoryRecordBytes, + "a timestamped quote at the coefficient bound (%d B) must fit a history record", size) + + // And the bound is the largest that does fit: the next step up does not. + // Without this the two constants could drift apart silently, with the + // coefficient bound quietly stopping short of what history can hold. + oversize, err := historyRecordSize(1_700_000_000_000_000_000, worst(MaxDecimalCoefficientBits+32)) + require.NoError(t, err) + require.Greater(t, oversize, MaxHistoryRecordBytes) +} diff --git a/llo/protocol/stream_value_fuzz_test.go b/llo/protocol/stream_value_fuzz_test.go new file mode 100644 index 0000000..5f15aa6 --- /dev/null +++ b/llo/protocol/stream_value_fuzz_test.go @@ -0,0 +1,46 @@ +package protocol + +import ( + "testing" + + "google.golang.org/protobuf/proto" +) + +// FuzzUnmarshalObservedProtoStreamValue feeds arbitrary bytes through the +// observation decode path. Observations are attacker-controlled, so the contract +// is that any input either decodes into a value within every bound or returns an +// error -- never a panic, and never an accepted value over a bound. +func FuzzUnmarshalObservedProtoStreamValue(f *testing.F) { + for _, typ := range []LLOStreamValue_Type{ + LLOStreamValue_Decimal, + LLOStreamValue_Quote, + LLOStreamValue_TimestampedStreamValue, + } { + for _, body := range [][]byte{nil, {}, {0x01}, {0xff, 0xff, 0xff, 0xff}} { + b, err := proto.Marshal(&LLOStreamValue{Type: typ, Value: body}) + if err != nil { + f.Fatal(err) + } + f.Add(b) + } + } + + f.Fuzz(func(t *testing.T, data []byte) { + enc := &LLOStreamValue{} + if err := proto.Unmarshal(data, enc); err != nil { + return // not a stream value proto; nothing to check + } + sv, err := UnmarshalObservedProtoStreamValue(enc) + if err != nil { + return + } + if sv == nil { + t.Fatal("decoded a nil stream value without an error") + } + // An accepted value must satisfy the bounds the decoder claims to + // enforce, at every nesting level. + if err := checkObservedStreamValue(sv, 0); err != nil { + t.Fatalf("accepted a value that violates its own bounds: %v", err) + } + }) +} diff --git a/llo/v30/observation_codec.go b/llo/v30/observation_codec.go index 6329924..d528eec 100644 --- a/llo/v30/observation_codec.go +++ b/llo/v30/observation_codec.go @@ -123,7 +123,7 @@ func (c protoObservationCodec) Decode(b types.Observation) (Observation, error) if len(pbuf.StreamValues) > 0 { streamValues = make(protocol.StreamValues, len(pbuf.StreamValues)) for id, enc := range pbuf.StreamValues { - sv, err := protocol.UnmarshalProtoStreamValue(enc) + sv, err := protocol.UnmarshalObservedProtoStreamValue(enc) if err != nil { // Byzantine behavior makes this observation invalid; a // well-behaved node should never encode invalid or nil values diff --git a/llo/v30/observation_codec_coefficient_test.go b/llo/v30/observation_codec_coefficient_test.go new file mode 100644 index 0000000..976e97f --- /dev/null +++ b/llo/v30/observation_codec_coefficient_test.go @@ -0,0 +1,42 @@ +package llo + +import ( + "math/big" + "testing" + + "github.com/shopspring/decimal" + "github.com/stretchr/testify/require" + + "github.com/smartcontractkit/chainlink-common/pkg/logger" + llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" + + protocol "github.com/smartcontractkit/chainlink-data-streams/llo/protocol" +) + +// Test_protoObservationCodec_CoefficientBound covers the v3.0 observation path. +// The bound lives in shared code, so both plugin versions reject the same value, +// which is what keeps a mixed-version DON from disagreeing about more than one +// peer's observation. +func Test_protoObservationCodec_CoefficientBound(t *testing.T) { + codec, err := NewProtoObservationCodec(logger.Nop(), true) + require.NoError(t, err) + + atLimit := decimal.NewFromBigInt(new(big.Int).Lsh(big.NewInt(1), protocol.MaxDecimalCoefficientBits-1), -2) + overSized := decimal.NewFromBigInt(new(big.Int).Lsh(big.NewInt(1), protocol.MaxDecimalCoefficientBits), -2) + + encoded := func(t *testing.T, d decimal.Decimal) []byte { + t.Helper() + b, err := codec.Encode(Observation{ + UnixTimestampNanoseconds: 1, + StreamValues: protocol.StreamValues{llotypes.StreamID(1): protocol.ToDecimal(d)}, + }) + require.NoError(t, err) + return b + } + + _, err = codec.Decode(encoded(t, atLimit)) + require.NoError(t, err) + + _, err = codec.Decode(encoded(t, overSized)) + require.ErrorIs(t, err, protocol.ErrDecimalCoefficientOutOfRange) +} From 3d847aa2890f868d762a48c8514af4c083d0365c Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Wed, 16 Sep 2026 11:42:23 +0100 Subject: [PATCH 05/40] SPOR-0004 llo/dev/v31: isReportable checks through protocol.EffectiveStreams --- llo/dev/v31/backfill.go | 12 +++++++++ llo/dev/v31/plugin_test.go | 53 ++++++++++++++++++++++++++++++++++++++ llo/dev/v31/reports.go | 45 ++++++++++++++++++-------------- 3 files changed, 90 insertions(+), 20 deletions(-) diff --git a/llo/dev/v31/backfill.go b/llo/dev/v31/backfill.go index 56d24a5..2b832e1 100644 --- a/llo/dev/v31/backfill.go +++ b/llo/dev/v31/backfill.go @@ -59,5 +59,17 @@ func selectBackfillCandidate(defs llotypes.ChannelDefinitions, validAfter map[ll if !found { return 0, 0, protocol.HistoryBackfillOpts{}, false } + // The candidate must also be emittable, not merely selectable. Reports + // needs the target's report-timestamp resolution and the row's stream + // values; if either fails there, the report is skipped while the watermark + // has already advanced past the row in the state transition, losing it + // permanently. Both are pure functions of (target definition, row), so + // checking them here keeps selection and emission on one path. + if _, err := protocol.ReportTimestampResolutionNanos(target); err != nil { + return 0, 0, protocol.HistoryBackfillOpts{}, false + } + if _, err := protocol.BuildBackfillStreamValues(target, o.Observations[bestRaw]); err != nil { + return 0, 0, protocol.HistoryBackfillOpts{}, false + } return bestNanos, bestRaw, o, true } diff --git a/llo/dev/v31/plugin_test.go b/llo/dev/v31/plugin_test.go index c2b59ae..1701630 100644 --- a/llo/dev/v31/plugin_test.go +++ b/llo/dev/v31/plugin_test.go @@ -949,3 +949,56 @@ func Test_Observation_RejectsInlineStreamValues(t *testing.T) { var bfErr *blobFetchError require.NotErrorAs(t, err, &bfErr) } + +func Test_IsReportable_EffectiveStreamsFailure(t *testing.T) { + // Reportability and emission share one derivation: isReportable and Reports + // both go through protocol.EffectiveStreams. A channel whose opts cannot be + // decoded has no derivable stream list, so Reports could not assemble + // values for it and reportability must agree, otherwise validAfter advances + // over a round that emitted nothing. + cd := llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatEVMABIEncodeUnpackedExpr, + Opts: []byte(`{"not":`), + Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}, + } + prec := precursor{ + LifeCycleStage: protocol.LifeCycleStageProduction, + ObservationTimestampNanoseconds: 2_000_000_000, + ChannelDefinitions: llotypes.ChannelDefinitions{1: cd}, + ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{1: 1_000_000_000}, + StreamAggregates: protocol.StreamAggregates{ + 100: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(1))}, + }, + } + require.Empty(t, prec.reportableChannels(0, protocol.NewOptsCache(), logger.Test(t))) +} + +func Test_SelectBackfillCandidate_UnemittableRow(t *testing.T) { + const ( + targetCID = llotypes.ChannelID(10) + backfillCID = llotypes.ChannelID(20) + tenSec = uint64(10_000_000_000) + ) + targetCD := llotypes.ChannelDefinition{ReportFormat: llotypes.ReportFormatJSON, Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}} + + // Row carries stream 999, not the target's stream 100, so + // BuildBackfillStreamValues would fail in Reports. The candidate must not + // be selectable: the watermark would otherwise advance past a row that + // never emitted, losing it permanently. + missingStream := llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatHistoryBackfill, + Opts: []byte(`{"targetChannelId":10,"observations":{"5":{"999":"1.5"}}}`), + } + defs := llotypes.ChannelDefinitions{targetCID: targetCD, backfillCID: missingStream} + _, _, _, ok := selectBackfillCandidate(defs, map[llotypes.ChannelID]uint64{backfillCID: 0}, tenSec, backfillCID, nil) + require.False(t, ok, "row missing a target stream must not be selectable") + + // Same row shape, but the value is not parseable as a stream value. + badValue := llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatHistoryBackfill, + Opts: []byte(`{"targetChannelId":10,"observations":{"5":{"100":"not-a-number"}}}`), + } + defs[backfillCID] = badValue + _, _, _, ok = selectBackfillCandidate(defs, map[llotypes.ChannelID]uint64{backfillCID: 0}, tenSec, backfillCID, nil) + require.False(t, ok, "row with an unparseable value must not be selectable") +} diff --git a/llo/dev/v31/reports.go b/llo/dev/v31/reports.go index dc82d1c..b62a2c0 100644 --- a/llo/dev/v31/reports.go +++ b/llo/dev/v31/reports.go @@ -211,28 +211,33 @@ func (o precursor) isReportable(channelID llotypes.ChannelID, minReportInterval } } } - // Calculated streams are derived state, and unlike observed streams a - // missing one cannot be reported around: the codec has nothing to encode, so - // Reports skips the report. Without this check the channel would still be - // counted as reported and validAfter would advance over a round that emitted - // nothing — a silent coverage gap. This is independent of - // DisableNilStreamValues, which is about observed values. + // Reportability and emission must agree on what the report carries. Reports + // derives the report's values from protocol.EffectiveStreams, so the same + // derivation has to succeed here: if it fails there, the report is skipped + // while the channel is still counted as reported and validAfter advances + // over a round that emitted nothing — a silent coverage gap. // - // The check is against the streams the channel's opts declare, which is also - // what protocol.EffectiveStreams derives the report's trailing values from. - // A channel whose expressions failed (bad input, undecodable opts, eval - // error, or history still warming up) has no aggregate for them. - if protocol.HasCalculatedStreams(cd) { - calculatedStreamIDs, err := protocol.CalculatedStreamIDs(optsCache, cd, channelID) - if err != nil { - lggr.Warnw("IsReportable=false; cannot resolve calculated stream IDs", "channelID", channelID, "err", err) - return false + // EffectiveStreams is a pure function of (definition, opts), so it is safe + // in the state transition. Codec lookup and Encode are not: p.ReportCodecs + // is node-local, and reading it here would make the state transition + // node-dependent. Those failure modes remain outside this predicate. + streams, err := protocol.EffectiveStreams(optsCache, cd, channelID) + if err != nil { + lggr.Warnw("IsReportable=false; cannot derive effective streams", "channelID", channelID, "err", err) + return false + } + // Calculated streams are derived state, and unlike observed streams a + // missing one cannot be reported around: the codec has nothing to encode. + // A channel whose expressions failed (bad input, eval error, or history + // still warming up) has no aggregate for them. Observed streams stay + // nil-permitted here; that is what DisableNilStreamValues above governs. + for _, strm := range streams { + if strm.Aggregator != llotypes.AggregatorCalculated { + continue } - for _, sid := range calculatedStreamIDs { - if o.StreamAggregates[sid][llotypes.AggregatorCalculated] == nil { - lggr.Warnw("IsReportable=false; nil calculated stream value", "channelID", channelID, "streamID", sid) - return false - } + if o.StreamAggregates[strm.StreamID][strm.Aggregator] == nil { + lggr.Warnw("IsReportable=false; nil calculated stream value", "channelID", channelID, "streamID", strm.StreamID) + return false } } validAfter, ok := o.ValidAfterNanoseconds[channelID] From 3619afd3e81f0c23b3d80d62a1657404b00232e0 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Wed, 16 Sep 2026 13:05:47 +0100 Subject: [PATCH 06/40] SPOR-0004 llo/v31: gate reportability on report codec coverage consensus fact Reportability is a consensus decision; encoding is not. isReportable is a pure function of the precursor and its result is persisted as reportedLastRound, which the next round reads to advance validAfter. But p.ReportCodecs is node-local plugin state that the state transition cannot read without forking, so a channel whose format no codec covers stays reportable, Reports drops it with a log line, and validAfter advances over a round that emitted nothing. Carry codec coverage as a replicated fact instead. Each observation advertises the report formats its oracle can encode; StateTransition tallies the advertisements and snapshots the tally on the precursor, so reportableChannels and the persisted reportedLastRound read the identical number. isReportable then requires 2f+1 advertised supporters for the format a channel's report is encoded with. --- llo/dev/v31/observation.go | 54 +++++++ llo/dev/v31/plugin.go | 27 ++++ llo/dev/v31/plugin_test.go | 239 +++++++++++++++++++++++++++++-- llo/dev/v31/precursor.go | 33 +++++ llo/dev/v31/reports.go | 36 ++++- llo/dev/v31/statetransition.go | 12 +- llo/protocol/limits.go | 5 + llo/protocol/plugin_codecs.pb.go | 127 +++++++++++++--- llo/protocol/plugin_codecs.proto | 15 ++ 9 files changed, 503 insertions(+), 45 deletions(-) diff --git a/llo/dev/v31/observation.go b/llo/dev/v31/observation.go index e642f71..1aa2b45 100644 --- a/llo/dev/v31/observation.go +++ b/llo/dev/v31/observation.go @@ -4,6 +4,7 @@ import ( "context" "encoding/binary" "fmt" + "sort" llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" @@ -24,6 +25,11 @@ type Observation struct { RemoveChannelIDs map[llotypes.ChannelID]struct{} UpdateChannelDefinitions llotypes.ChannelDefinitions StreamValues protocol.StreamValues + // SupportedReportFormats are the report formats this oracle has a report + // codec for. Encoding is node-local state that the state transition cannot + // read without forking; advertising it here turns it into a replicated fact + // that reportability can gate on (see isReportable). + SupportedReportFormats []llotypes.ReportFormat } // observationWireVersion is the leading byte of the v31 observation framing. @@ -64,6 +70,10 @@ func encodeObservation(obs Observation, handles [][]byte) (ocrtypes.Observation, } } + // Sorted and deduped: the wire bytes need not be deterministic, but a + // canonical list keeps goldens stable and matches what decode enforces. + main.SupportedReportFormats = sortedUniqueFormats(obs.SupportedReportFormats) + mainBytes, err := proto.Marshal(main) if err != nil { return nil, fmt.Errorf("marshal observation: %w", err) @@ -216,9 +226,53 @@ func observationFromProto(main *protocol.LLOObservationProto) (Observation, erro if len(main.StreamValues) > 0 { return Observation{}, fmt.Errorf("observation carries %d inline stream values: v31 requires blob-carried values", len(main.StreamValues)) } + if len(main.SupportedReportFormats) > protocol.MaxObservationSupportedReportFormatsLength { + return Observation{}, fmt.Errorf("observation advertises too many report formats: %d (max %d)", len(main.SupportedReportFormats), protocol.MaxObservationSupportedReportFormatsLength) + } + + obs.SupportedReportFormats = sortedUniqueFormatsFromWire(main.SupportedReportFormats) return obs, nil } +// sortedUniqueFormats returns the formats sorted ascending with duplicates +// removed. nil in, nil out, so an oracle advertising nothing stays absent from +// the wire rather than carrying an empty list. +func sortedUniqueFormats(in []llotypes.ReportFormat) []uint32 { + if len(in) == 0 { + return nil + } + seen := make(map[llotypes.ReportFormat]struct{}, len(in)) + out := make([]uint32, 0, len(in)) + for _, f := range in { + if _, dup := seen[f]; dup { + continue + } + seen[f] = struct{}{} + out = append(out, uint32(f)) + } + sort.Slice(out, func(i, j int) bool { return out[i] < out[j] }) + return out +} + +// sortedUniqueFormatsFromWire is sortedUniqueFormats for the wire +// representation: sorted ascending, duplicates removed. +func sortedUniqueFormatsFromWire(in []uint32) []llotypes.ReportFormat { + if len(in) == 0 { + return nil + } + seen := make(map[uint32]struct{}, len(in)) + out := make([]llotypes.ReportFormat, 0, len(in)) + for _, f := range in { + if _, dup := seen[f]; dup { + continue + } + seen[f] = struct{}{} + out = append(out, llotypes.ReportFormat(f)) + } + sort.Slice(out, func(i, j int) bool { return out[i] < out[j] }) + return out +} + func streamValuesToProto(in protocol.StreamValues) (map[uint32]*protocol.LLOStreamValue, error) { if len(in) == 0 { return nil, nil diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index 20fa579..2510ada 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -4,6 +4,7 @@ import ( "context" "errors" "fmt" + "sort" "time" "golang.org/x/exp/maps" @@ -139,9 +140,32 @@ func (p *Plugin) Observation(_ context.Context, seqNr uint64, _ ocrtypes.Attribu } obs.UnixTimestampNanoseconds = uint64(obsTSNanos) + // Advertised every round, including when retired: a statement about this + // binary, not about the round or about what this node wants admitted. + // voteOnChannels above is deliberately unaware of p.ReportCodecs -- whether + // a channel can be encoded DON-wide is decided from these advertisements in + // the state transition, not locally per voter. + obs.SupportedReportFormats = supportedReportFormats(p.ReportCodecs) + return encodeObservation(obs, handles) } +// supportedReportFormats lists the report formats this node can encode, taken +// from the codecs it was constructed with. Truncated to the advertisable bound +// so the observation stays within its size budget; a real codec map is far +// smaller than the bound, so this never fires in practice. +func supportedReportFormats(codecs map[llotypes.ReportFormat]protocol.ReportCodec) []llotypes.ReportFormat { + out := make([]llotypes.ReportFormat, 0, len(codecs)) + for format := range codecs { + out = append(out, format) + } + sort.Slice(out, func(i, j int) bool { return out[i] < out[j] }) + if len(out) > protocol.MaxObservationSupportedReportFormatsLength { + out = out[:protocol.MaxObservationSupportedReportFormatsLength] + } + return out +} + // observableStreams lists the streams a round should observe: every stream of // every live channel, minus calculated streams (which are derived in // StateTransition rather than observed). @@ -232,6 +256,9 @@ func (p *Plugin) ValidateObservation(ctx context.Context, seqNr uint64, _ ocrtyp if len(observation.RemoveChannelIDs) > protocol.MaxObservationRemoveChannelIDsLength { return fmt.Errorf("RemoveChannelIDs is too long: %v vs %v", len(observation.RemoveChannelIDs), protocol.MaxObservationRemoveChannelIDsLength) } + if len(observation.SupportedReportFormats) > protocol.MaxObservationSupportedReportFormatsLength { + return fmt.Errorf("SupportedReportFormats is too long: %v vs %v", len(observation.SupportedReportFormats), protocol.MaxObservationSupportedReportFormatsLength) + } // Only the baseline checks run here. A definition is installed on more than // f votes for its exact hash, so at least one honest oracle must have voted diff --git a/llo/dev/v31/plugin_test.go b/llo/dev/v31/plugin_test.go index 1701630..ca18c2c 100644 --- a/llo/dev/v31/plugin_test.go +++ b/llo/dev/v31/plugin_test.go @@ -131,6 +131,16 @@ func ao(observer int, obsBytes []byte) ocrtypes.AttributedObservation { return ocrtypes.AttributedObservation{Observer: commontypes.OracleID(observer), Observation: obsBytes} } +// testSupportedReportFormats is the codec coverage a fixture oracle advertises +// by default: every format the tests in this package build channels for. +var testSupportedReportFormats = []llotypes.ReportFormat{ + llotypes.ReportFormatJSON, + llotypes.ReportFormatEVMPremiumLegacy, + llotypes.ReportFormatEVMABIEncodeUnpacked, + llotypes.ReportFormatEVMABIEncodeUnpackedExpr, + llotypes.ReportFormatHistoryBackfill, +} + // testBlobs is the shared in-memory blob store used by StateTransition // fixtures. It is content-addressed and mutex-guarded, so tests can share it. var testBlobs = llotest.NewBlobBroadcastFetcher() @@ -140,6 +150,13 @@ var testBlobs = llotest.NewBlobBroadcastFetcher() // referenced by handle. Pass testBlobs as the fetcher to StateTransition. func mustEncodeObs(t *testing.T, obs Observation) []byte { t.Helper() + // A real oracle always advertises the formats it can encode, and channel + // reportability requires 2f+1 of them to do so. Fixtures that do not care + // about the support gate get the full set, so they behave like a healthy + // DON; tests that exercise the gate set the field explicitly. + if obs.SupportedReportFormats == nil { + obs.SupportedReportFormats = testSupportedReportFormats + } if len(obs.StreamValues) == 0 { b, err := encodeObservation(obs, nil) require.NoError(t, err) @@ -429,7 +446,7 @@ func Test_SecondsResolutionOverlap(t *testing.T) { for _, tc := range tests { t.Run(tc.name, func(t *testing.T) { p := mkPrec(tc.format, tc.opts, tc.validAfter, tc.obsTs) - got := p.reportableChannels(0, protocol.NewOptsCache(), logger.Test(t)) + got := p.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), logger.Test(t)) if tc.reportable { require.Equal(t, []llotypes.ChannelID{1}, got) } else { @@ -457,14 +474,14 @@ func Test_DisableNilStreamValues(t *testing.T) { // Missing stream 200 -> not reportable. missing := base(protocol.StreamAggregates{100: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(1))}}) - require.Empty(t, missing.reportableChannels(0, protocol.NewOptsCache(), logger.Test(t))) + require.Empty(t, missing.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), logger.Test(t))) // Both streams present -> reportable. full := base(protocol.StreamAggregates{ 100: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(1))}, 200: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(2))}, }) - require.Equal(t, []llotypes.ChannelID{1}, full.reportableChannels(0, protocol.NewOptsCache(), logger.Test(t))) + require.Equal(t, []llotypes.ChannelID{1}, full.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), logger.Test(t))) } func Test_DisableNilStreamValues_CalculatedStreams(t *testing.T) { @@ -521,17 +538,17 @@ func Test_DisableNilStreamValues_CalculatedStreams(t *testing.T) { // ProcessCalculatedStreams bailed before writing the calculated // aggregate; the definition alone looks complete. o := mkPrec(true, validOpts, baseStreams, baseAggregates()) - require.Empty(t, o.reportableChannels(0, populatedCache(o), logger.Test(t))) + require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), logger.Test(t))) }) t.Run("inline calculated stream but nil aggregate -> not reportable", func(t *testing.T) { o := mkPrec(true, validOpts, withCalculated, baseAggregates()) - require.Empty(t, o.reportableChannels(0, populatedCache(o), logger.Test(t))) + require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), logger.Test(t))) }) t.Run("fully evaluated -> reportable", func(t *testing.T) { o := mkPrec(true, validOpts, withCalculated, evaluatedAggregates()) - require.Equal(t, []llotypes.ChannelID{1}, o.reportableChannels(0, populatedCache(o), logger.Test(t))) + require.Equal(t, []llotypes.ChannelID{1}, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), logger.Test(t))) }) t.Run("DisableNilStreamValues=false, evaluation failed -> not reportable", func(t *testing.T) { @@ -541,32 +558,32 @@ func Test_DisableNilStreamValues_CalculatedStreams(t *testing.T) { // report. Treating the channel as reportable would advance validAfter // over a round that emitted nothing. o := mkPrec(false, validOpts, baseStreams, baseAggregates()) - require.Empty(t, o.reportableChannels(0, populatedCache(o), logger.Test(t))) + require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), logger.Test(t))) }) t.Run("DisableNilStreamValues=false, fully evaluated -> reportable", func(t *testing.T) { o := mkPrec(false, validOpts, withCalculated, evaluatedAggregates()) - require.Equal(t, []llotypes.ChannelID{1}, o.reportableChannels(0, populatedCache(o), logger.Test(t))) + require.Equal(t, []llotypes.ChannelID{1}, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), logger.Test(t))) }) t.Run("malformed opts -> not reportable", func(t *testing.T) { o := mkPrec(true, []byte(`{"abi":`), withCalculated, evaluatedAggregates()) - require.Empty(t, o.reportableChannels(0, populatedCache(o), logger.Test(t))) + require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), logger.Test(t))) }) t.Run("opts declare no expressions -> not reportable", func(t *testing.T) { o := mkPrec(true, []byte(`{"abi":[]}`), withCalculated, evaluatedAggregates()) - require.Empty(t, o.reportableChannels(0, populatedCache(o), logger.Test(t))) + require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), logger.Test(t))) }) t.Run("cache miss falls back to channel opts -> reportable", func(t *testing.T) { o := mkPrec(true, validOpts, withCalculated, evaluatedAggregates()) - require.Equal(t, []llotypes.ChannelID{1}, o.reportableChannels(0, protocol.NewOptsCache(), logger.Test(t))) + require.Equal(t, []llotypes.ChannelID{1}, o.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), logger.Test(t))) }) t.Run("cache miss falls back to channel opts -> not reportable when unevaluated", func(t *testing.T) { o := mkPrec(true, validOpts, baseStreams, baseAggregates()) - require.Empty(t, o.reportableChannels(0, protocol.NewOptsCache(), logger.Test(t))) + require.Empty(t, o.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), logger.Test(t))) }) } @@ -753,7 +770,7 @@ func Test_HistoryBackfill(t *testing.T) { ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{targetCID: tenSec /* target not reportable */, backfillCID: 0}, StreamAggregates: protocol.StreamAggregates{}, } - b, err := encodePrecursor(prec) + b, err := encodePrecursor(prec.withSupport(3)) require.NoError(t, err) reports, err := p.Reports(ctx, 2, b) require.NoError(t, err) @@ -914,7 +931,7 @@ func Test_CalculatedStreams_ReportValues(t *testing.T) { cache.ResetTo(definitions) calculated.ProcessCalculatedStreams(p.Logger, prec.ChannelDefinitions, prec.StreamAggregates, prec.ObservationTimestampNanoseconds, cache, nil) - precBytes, err := encodePrecursor(prec) + precBytes, err := encodePrecursor(prec.withSupport(3)) require.NoError(t, err) reports, err := p.Reports(ctx, 2, precBytes) require.NoError(t, err) @@ -970,7 +987,7 @@ func Test_IsReportable_EffectiveStreamsFailure(t *testing.T) { 100: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(1))}, }, } - require.Empty(t, prec.reportableChannels(0, protocol.NewOptsCache(), logger.Test(t))) + require.Empty(t, prec.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), logger.Test(t))) } func Test_SelectBackfillCandidate_UnemittableRow(t *testing.T) { @@ -1002,3 +1019,195 @@ func Test_SelectBackfillCandidate_UnemittableRow(t *testing.T) { _, _, _, ok = selectBackfillCandidate(defs, map[llotypes.ChannelID]uint64{backfillCID: 0}, tenSec, backfillCID, nil) require.False(t, ok, "row with an unparseable value must not be selectable") } + +// withSupport returns a copy of the precursor with n oracles advertising a +// report codec for every format its channels need, so tests that are not about +// the support gate are unaffected by it. Backfill channels are resolved to +// their target's format, which is the one their report is encoded with. +func (o precursor) withSupport(n int) precursor { + o.SupportByFormat = make(map[llotypes.ReportFormat]int, len(o.ChannelDefinitions)) + for _, cd := range o.ChannelDefinitions { + o.SupportByFormat[cd.ReportFormat] = n + } + return o +} + +func Test_ReportFormatSupportGate(t *testing.T) { + const cid = llotypes.ChannelID(1) + cd := llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatJSON, + Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}, + } + base := func(support int) precursor { + return precursor{ + LifeCycleStage: protocol.LifeCycleStageProduction, + ObservationTimestampNanoseconds: 2_000_000_000, + ChannelDefinitions: llotypes.ChannelDefinitions{cid: cd}, + ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{cid: 1_000_000_000}, + StreamAggregates: protocol.StreamAggregates{ + 100: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(1))}, + }, + SupportByFormat: map[llotypes.ReportFormat]int{llotypes.ReportFormatJSON: support}, + } + } + + // f=1 requires 2f+1 = 3 advertised supporters: 2f would only guarantee f+1 + // real encoders if none of the advertisements were lies. + require.Equal(t, []llotypes.ChannelID{cid}, base(3).reportableChannels(0, 1, protocol.NewOptsCache(), logger.Test(t))) + require.Empty(t, base(2).reportableChannels(0, 1, protocol.NewOptsCache(), logger.Test(t))) + require.Empty(t, base(0).reportableChannels(0, 1, protocol.NewOptsCache(), logger.Test(t))) + + // A format no oracle advertises is never reportable, however healthy the + // channel otherwise is. + noEntry := base(3) + noEntry.SupportByFormat = map[llotypes.ReportFormat]int{} + require.Empty(t, noEntry.reportableChannels(0, 1, protocol.NewOptsCache(), logger.Test(t))) + + // Support is keyed by format, not channel: an unrelated format's coverage + // does not carry the channel. + wrongFormat := base(0) + wrongFormat.SupportByFormat = map[llotypes.ReportFormat]int{llotypes.ReportFormatEVMPremiumLegacy: 4} + require.Empty(t, wrongFormat.reportableChannels(0, 1, protocol.NewOptsCache(), logger.Test(t))) +} + +func Test_ReportFormatSupportGate_Backfill(t *testing.T) { + const ( + targetCID = llotypes.ChannelID(10) + backfillCID = llotypes.ChannelID(20) + tenSec = uint64(10_000_000_000) + ) + defs := llotypes.ChannelDefinitions{ + targetCID: {ReportFormat: llotypes.ReportFormatJSON, Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}}, + backfillCID: { + ReportFormat: llotypes.ReportFormatHistoryBackfill, + Opts: []byte(`{"targetChannelId":10,"observations":{"5":{"100":"1.5"}}}`), + }, + } + base := func(support map[llotypes.ReportFormat]int) precursor { + return precursor{ + LifeCycleStage: protocol.LifeCycleStageProduction, + ObservationTimestampNanoseconds: tenSec, + ChannelDefinitions: defs, + ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{targetCID: tenSec /* target not reportable */, backfillCID: 0}, + StreamAggregates: protocol.StreamAggregates{}, + SupportByFormat: support, + } + } + + // The backfill report is encoded with the TARGET's codec, so the target's + // format is what must be covered. Coverage of history_backfill itself is + // irrelevant: no codec encodes it. + targetCovered := base(map[llotypes.ReportFormat]int{llotypes.ReportFormatJSON: 3}) + require.Equal(t, []llotypes.ChannelID{backfillCID}, targetCovered.reportableChannels(0, 1, protocol.NewOptsCache(), logger.Test(t))) + + backfillCoveredOnly := base(map[llotypes.ReportFormat]int{llotypes.ReportFormatHistoryBackfill: 4}) + require.Empty(t, backfillCoveredOnly.reportableChannels(0, 1, protocol.NewOptsCache(), logger.Test(t))) +} + +func Test_ReportFormatSupportGate_StopsValidAfterAdvance(t *testing.T) { + // The gate's point: an uncovered channel must not have validAfter advance + // over a round that emitted nothing. reportedLastRound is what the next + // round reads to decide the advance, so assert on that. + const cid = llotypes.ChannelID(1) + cd := llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatJSON, + Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}, + } + prec := precursor{ + LifeCycleStage: protocol.LifeCycleStageProduction, + ObservationTimestampNanoseconds: 2_000_000_000, + ChannelDefinitions: llotypes.ChannelDefinitions{cid: cd}, + ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{cid: 1_000_000_000}, + StreamAggregates: protocol.StreamAggregates{ + 100: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(1))}, + }, + } + + covered := prec + covered.SupportByFormat = map[llotypes.ReportFormat]int{llotypes.ReportFormatJSON: 3} + require.True(t, covered.isReportable(cid, 0, 1, protocol.NewOptsCache(), logger.Test(t))) + + uncovered := prec + uncovered.SupportByFormat = map[llotypes.ReportFormat]int{llotypes.ReportFormatJSON: 2} + require.False(t, uncovered.isReportable(cid, 0, 1, protocol.NewOptsCache(), logger.Test(t))) +} + +func Test_Observation_SupportedReportFormats_RoundTrip(t *testing.T) { + ctx := tests.Context(t) + + // Duplicates are collapsed and the list is sorted, so one oracle can only + // ever contribute one vote per format to the tally. + obs := Observation{ + UnixTimestampNanoseconds: 1, + SupportedReportFormats: []llotypes.ReportFormat{ + llotypes.ReportFormatJSON, + llotypes.ReportFormatEVMPremiumLegacy, + llotypes.ReportFormatJSON, + }, + } + b, err := encodeObservation(obs, nil) + require.NoError(t, err) + got, err := decodeObservation(ctx, b, nil) + require.NoError(t, err) + require.Equal(t, []llotypes.ReportFormat{llotypes.ReportFormatEVMPremiumLegacy, llotypes.ReportFormatJSON}, got.SupportedReportFormats) + + // An oracle advertising nothing decodes as advertising nothing, rather than + // as an empty-but-present list. + b, err = encodeObservation(Observation{UnixTimestampNanoseconds: 1}, nil) + require.NoError(t, err) + got, err = decodeObservation(ctx, b, nil) + require.NoError(t, err) + require.Nil(t, got.SupportedReportFormats) + + // Over-length lists are rejected at decode rather than reaching the tally. + tooMany := make([]uint32, protocol.MaxObservationSupportedReportFormatsLength+1) + for i := range tooMany { + tooMany[i] = uint32(i) + } + raw, err := proto.Marshal(&protocol.LLOObservationProto{UnixTimestampNanoseconds: 1, SupportedReportFormats: tooMany}) + require.NoError(t, err) + _, err = decodeObservation(ctx, frameObservation(nil, raw), nil) + require.ErrorContains(t, err, "advertises too many report formats") +} + +func Test_Precursor_SupportByFormat_RoundTrip(t *testing.T) { + prec := precursor{ + LifeCycleStage: protocol.LifeCycleStageProduction, + SupportByFormat: map[llotypes.ReportFormat]int{ + llotypes.ReportFormatJSON: 4, + llotypes.ReportFormatEVMPremiumLegacy: 3, + }, + } + // Encoding must be deterministic despite map iteration order: libocr + // digests the precursor bytes, so two oracles disagreeing on byte order + // would fail attestation. + b1, err := encodePrecursor(prec) + require.NoError(t, err) + b2, err := encodePrecursor(prec) + require.NoError(t, err) + require.Equal(t, b1, b2) + + got, err := decodePrecursor(b1) + require.NoError(t, err) + require.Equal(t, prec.SupportByFormat, got.SupportByFormat) +} + +func Test_StateTransition_TalliesReportFormatSupport(t *testing.T) { + ctx := tests.Context(t) + p := testPlugin(t) + kv := newMemKV() + + _, err := p.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, nil), ao(1, nil), ao(2, nil)}, kv, testBlobs) + require.NoError(t, err) + + // Two oracles advertise JSON, one advertises nothing: the tally is a count + // of advertisements, and an oracle that advertises nothing is not counted. + obsJSON := mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 1, SupportedReportFormats: []llotypes.ReportFormat{llotypes.ReportFormatJSON}}) + obsNone := mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 1, SupportedReportFormats: []llotypes.ReportFormat{}}) + precBytes, err := p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, obsJSON), ao(1, obsJSON), ao(2, obsNone)}, kv, testBlobs) + require.NoError(t, err) + + prec, err := decodePrecursor(precBytes) + require.NoError(t, err) + require.Equal(t, 2, prec.SupportByFormat[llotypes.ReportFormatJSON]) +} diff --git a/llo/dev/v31/precursor.go b/llo/dev/v31/precursor.go index ee94336..d632750 100644 --- a/llo/dev/v31/precursor.go +++ b/llo/dev/v31/precursor.go @@ -26,6 +26,14 @@ type precursor struct { // ChannelDefinitions came from. It lets Reports tell whether the decoded-opts // cache already matches these definitions without walking every channel. ChannelStateSeqNr uint64 + // SupportByFormat is the number of oracles that advertised a report codec + // for each report format in the round that produced this precursor. + // + // Snapshotted rather than recomputed so that reportability and the + // validAfter advance for a round read the identical number: Reports runs + // against this precursor, and StateTransition persists reportedLastRound + // from the same object (see isReportable). + SupportByFormat map[llotypes.ReportFormat]int } func encodePrecursor(p precursor) (ocr3_1types.ReportsPlusPrecursor, error) { @@ -83,6 +91,22 @@ func encodePrecursor(p precursor) (ocr3_1types.ReportsPlusPrecursor, error) { }) } + if len(p.SupportByFormat) > 0 { + pb.SupportByFormat = make([]*protocol.LLOReportFormatSupportProto, 0, len(p.SupportByFormat)) + for format, count := range p.SupportByFormat { + if count < 0 { + return nil, fmt.Errorf("negative support count for report format %v: %d", format, count) + } + pb.SupportByFormat = append(pb.SupportByFormat, &protocol.LLOReportFormatSupportProto{ + ReportFormat: uint32(format), + OracleCount: uint32(count), + }) + } + sort.Slice(pb.SupportByFormat, func(i, j int) bool { + return pb.SupportByFormat[i].ReportFormat < pb.SupportByFormat[j].ReportFormat + }) + } + b, err := deterministicMarshal.Marshal(pb) if err != nil { return nil, fmt.Errorf("marshal precursor: %w", err) @@ -112,6 +136,15 @@ func decodePrecursor(b ocr3_1types.ReportsPlusPrecursor) (precursor, error) { for _, va := range pb.ValidAfterNanoseconds { p.ValidAfterNanoseconds[va.ChannelID] = va.ValidAfterNanoseconds } + if len(pb.SupportByFormat) > protocol.MaxObservationSupportedReportFormatsLength { + return precursor{}, fmt.Errorf("precursor carries too many report format support entries: %d (max %d)", len(pb.SupportByFormat), protocol.MaxObservationSupportedReportFormatsLength) + } + if len(pb.SupportByFormat) > 0 { + p.SupportByFormat = make(map[llotypes.ReportFormat]int, len(pb.SupportByFormat)) + for _, sup := range pb.SupportByFormat { + p.SupportByFormat[llotypes.ReportFormat(sup.ReportFormat)] = int(sup.OracleCount) + } + } for _, sa := range pb.StreamAggregates { sv, err := protocol.UnmarshalProtoStreamValue(sa.StreamValue) if err != nil { diff --git a/llo/dev/v31/reports.go b/llo/dev/v31/reports.go index b62a2c0..c8f6e4b 100644 --- a/llo/dev/v31/reports.go +++ b/llo/dev/v31/reports.go @@ -62,7 +62,7 @@ func (p *Plugin) Reports(ctx context.Context, seqNr uint64, rawPrecursor ocr3_1t }) } - for _, cid := range out.reportableChannels(p.DefaultMinReportIntervalNanoseconds, channelOpts, p.Logger) { + for _, cid := range out.reportableChannels(p.DefaultMinReportIntervalNanoseconds, p.F, channelOpts, p.Logger) { cd := out.ChannelDefinitions[cid] if cd.ReportFormat == llotypes.ReportFormatHistoryBackfill { @@ -177,12 +177,27 @@ func (p *Plugin) Reports(ctx context.Context, seqNr uint64, rawPrecursor ocr3_1t return rwis, nil } +// formatIsEncodable reports whether enough oracles advertised a report codec +// for format that a report encoded with it would actually be certified. +// +// The tally is read from the precursor, not from p.ReportCodecs, so this stays +// a pure function of replicated state. An oracle that cannot encode still +// computes reportable=true when enough peers can. +func (o precursor) formatIsEncodable(format llotypes.ReportFormat, f int, channelID llotypes.ChannelID, lggr logger.Logger) bool { + if supporters := o.SupportByFormat[format]; supporters < 2*f+1 { + lggr.Warnw("IsReportable=false; too few oracles advertise a report codec for this format", + "channelID", channelID, "reportFormat", format, "supporters", supporters, "required", 2*f+1) + return false + } + return true +} + // reportableChannels returns the sorted set of channels reportable in this // (current) round (see isReportable). -func (o precursor) reportableChannels(minReportInterval uint64, optsCache *protocol.OptsCache, lggr logger.Logger) []llotypes.ChannelID { +func (o precursor) reportableChannels(minReportInterval uint64, f int, optsCache *protocol.OptsCache, lggr logger.Logger) []llotypes.ChannelID { reportable := make([]llotypes.ChannelID, 0, len(o.ChannelDefinitions)) for channelID := range o.ChannelDefinitions { - if o.isReportable(channelID, minReportInterval, optsCache, lggr) { + if o.isReportable(channelID, minReportInterval, f, optsCache, lggr) { reportable = append(reportable, channelID) } } @@ -190,7 +205,7 @@ func (o precursor) reportableChannels(minReportInterval uint64, optsCache *proto return reportable } -func (o precursor) isReportable(channelID llotypes.ChannelID, minReportInterval uint64, optsCache *protocol.OptsCache, lggr logger.Logger) bool { +func (o precursor) isReportable(channelID llotypes.ChannelID, minReportInterval uint64, f int, optsCache *protocol.OptsCache, lggr logger.Logger) bool { if o.LifeCycleStage == protocol.LifeCycleStageRetired { return false } @@ -199,8 +214,14 @@ func (o precursor) isReportable(channelID llotypes.ChannelID, minReportInterval return false } if cd.ReportFormat == llotypes.ReportFormatHistoryBackfill { - _, _, _, ok := selectBackfillCandidate(o.ChannelDefinitions, o.ValidAfterNanoseconds, o.ObservationTimestampNanoseconds, channelID, optsCache) - return ok + _, _, opts, ok := selectBackfillCandidate(o.ChannelDefinitions, o.ValidAfterNanoseconds, o.ObservationTimestampNanoseconds, channelID, optsCache) + if !ok { + return false + } + // Backfill reports are encoded with the target channel's codec, so the + // target's format is the one that must be encodable DON-wide. Selection + // above already established the target exists. + return o.formatIsEncodable(o.ChannelDefinitions[opts.TargetChannelID].ReportFormat, f, channelID, lggr) } // When DisableNilStreamValues is set, every stream must have a (non-nil) // aggregate value for the channel to be reportable. @@ -240,6 +261,9 @@ func (o precursor) isReportable(channelID llotypes.ChannelID, minReportInterval return false } } + if !o.formatIsEncodable(cd.ReportFormat, f, channelID, lggr) { + return false + } validAfter, ok := o.ValidAfterNanoseconds[channelID] if !ok { return false diff --git a/llo/dev/v31/statetransition.go b/llo/dev/v31/statetransition.go index 6a45198..d2f78a1 100644 --- a/llo/dev/v31/statetransition.go +++ b/llo/dev/v31/statetransition.go @@ -69,7 +69,7 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A return nil, fmt.Errorf("failed to load KV state: %w", err) } - timestamps, validPredecessorRetirementReport, shouldRetireVotes, removeChannelVotesByID, updateDefsByHash, updateVotesByHash, streamObservations, err := p.decodeObservations(ctx, aos, bf) + timestamps, validPredecessorRetirementReport, shouldRetireVotes, removeChannelVotesByID, updateDefsByHash, updateVotesByHash, supportByFormat, streamObservations, err := p.decodeObservations(ctx, aos, bf) if err != nil { return nil, err } @@ -88,6 +88,7 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A ChannelStateSeqNr: prev.channelStateSeqNr, ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{}, StreamAggregates: protocol.StreamAggregates{}, + SupportByFormat: supportByFormat, } // Lifecycle stage & promotion. @@ -228,10 +229,12 @@ func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.Attribut removeChannelVotesByID map[llotypes.ChannelID]int, updateChannelDefinitionsByHash map[[32]byte]protocol.ChannelDefinitionWithID, updateChannelVotesByHash map[[32]byte]int, + supportVotesByFormat map[llotypes.ReportFormat]int, streamObservations map[llotypes.StreamID][]protocol.StreamValue, err error, ) { removeChannelVotesByID = make(map[llotypes.ChannelID]int) + supportVotesByFormat = make(map[llotypes.ReportFormat]int) updateChannelDefinitionsByHash = make(map[[32]byte]protocol.ChannelDefinitionWithID) updateChannelVotesByHash = make(map[[32]byte]int) streamObservations = make(map[llotypes.StreamID][]protocol.StreamValue) @@ -270,6 +273,11 @@ func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.Attribut } timestampsNanoseconds = append(timestampsNanoseconds, observation.UnixTimestampNanoseconds) + // Deduped by decodeObservation, so one oracle contributes at most one + // vote per format. + for _, format := range observation.SupportedReportFormats { + supportVotesByFormat[format]++ + } for channelID := range observation.RemoveChannelIDs { removeChannelVotesByID[channelID]++ } @@ -527,7 +535,7 @@ func (p *Plugin) flushKV( // round can advance validAfter faithfully (see prevReportable). reportable := make(map[llotypes.ChannelID]bool, len(out.ChannelDefinitions)) for id := range out.ChannelDefinitions { - reportable[id] = out.isReportable(id, p.DefaultMinReportIntervalNanoseconds, prev.opts, p.Logger) + reportable[id] = out.isReportable(id, p.DefaultMinReportIntervalNanoseconds, p.F, prev.opts, p.Logger) } // Stream history: write modified windows, delete pairs no live channel diff --git a/llo/protocol/limits.go b/llo/protocol/limits.go index 351711c..f6ce5f2 100644 --- a/llo/protocol/limits.go +++ b/llo/protocol/limits.go @@ -27,6 +27,11 @@ const ( MaxObservationUpdateChannelDefinitionsLength = 5 // Maximum number of streams that can be observed per round MaxObservationStreamValuesLength = 10_000 + // MaxObservationSupportedReportFormatsLength bounds the report formats an + // observation may advertise support for. A real codec map holds a handful + // of entries; the headroom keeps the bound from needing revision while + // still stopping a peer from padding its observation. + MaxObservationSupportedReportFormatsLength = 32 // Maximum allowed number of streams per channel MaxStreamsPerChannel = 10_000 // MaxDecimalExponent bounds the absolute value of the base-10 exponent of diff --git a/llo/protocol/plugin_codecs.pb.go b/llo/protocol/plugin_codecs.pb.go index 159a4b2..1b85958 100644 --- a/llo/protocol/plugin_codecs.pb.go +++ b/llo/protocol/plugin_codecs.pb.go @@ -1,7 +1,7 @@ // Code generated by protoc-gen-go. DO NOT EDIT. // versions: // protoc-gen-go v1.36.12 -// protoc v7.35.1 +// protoc v7.36.1 // source: plugin_codecs.proto package protocol @@ -89,8 +89,14 @@ type LLOObservationProto struct { // uniqueness. UpdateChannelDefinitions map[uint32]*LLOChannelDefinitionProto `protobuf:"bytes,5,rep,name=updateChannelDefinitions,proto3" json:"updateChannelDefinitions,omitempty" protobuf_key:"varint,1,opt,name=key" protobuf_val:"bytes,2,opt,name=value"` StreamValues map[uint32]*LLOStreamValue `protobuf:"bytes,6,rep,name=streamValues,proto3" json:"streamValues,omitempty" protobuf_key:"varint,1,opt,name=key" protobuf_val:"bytes,2,opt,name=value"` - unknownFields protoimpl.UnknownFields - sizeCache protoimpl.SizeCache + // Report formats this oracle has a report codec for. Advertised so that + // reportability can require enough oracles to be able to encode a channel + // before validAfter advances over it. Encoding is node-local state and + // cannot be read in the state transition; this carries it as a replicated + // fact instead. + SupportedReportFormats []uint32 `protobuf:"varint,8,rep,packed,name=supportedReportFormats,proto3" json:"supportedReportFormats,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache } func (x *LLOObservationProto) Reset() { @@ -172,6 +178,13 @@ func (x *LLOObservationProto) GetStreamValues() map[uint32]*LLOStreamValue { return nil } +func (x *LLOObservationProto) GetSupportedReportFormats() []uint32 { + if x != nil { + return x.SupportedReportFormats + } + return nil +} + type LLOStreamValue struct { state protoimpl.MessageState `protogen:"open.v1"` Type LLOStreamValue_Type `protobuf:"varint,1,opt,name=type,proto3,enum=v1.LLOStreamValue_Type" json:"type,omitempty"` @@ -1255,8 +1268,10 @@ type LLOPrecursorProto struct { ValidAfterNanoseconds []*LLOChannelIDAndValidAfterNanosecondsProto `protobuf:"bytes,4,rep,name=validAfterNanoseconds,proto3" json:"validAfterNanoseconds,omitempty"` StreamAggregates []*LLOStreamAggregate `protobuf:"bytes,5,rep,name=streamAggregates,proto3" json:"streamAggregates,omitempty"` ChannelStateSeqNr uint64 `protobuf:"varint,6,opt,name=channelStateSeqNr,proto3" json:"channelStateSeqNr,omitempty"` - unknownFields protoimpl.UnknownFields - sizeCache protoimpl.SizeCache + // supportByFormat MUST be sorted ascending by reportFormat. + SupportByFormat []*LLOReportFormatSupportProto `protobuf:"bytes,7,rep,name=supportByFormat,proto3" json:"supportByFormat,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache } func (x *LLOPrecursorProto) Reset() { @@ -1331,11 +1346,72 @@ func (x *LLOPrecursorProto) GetChannelStateSeqNr() uint64 { return 0 } +func (x *LLOPrecursorProto) GetSupportByFormat() []*LLOReportFormatSupportProto { + if x != nil { + return x.SupportByFormat + } + return nil +} + +// LLOReportFormatSupportProto is the number of oracles that advertised a report +// codec for one report format in the round that produced the precursor. +type LLOReportFormatSupportProto struct { + state protoimpl.MessageState `protogen:"open.v1"` + ReportFormat uint32 `protobuf:"varint,1,opt,name=reportFormat,proto3" json:"reportFormat,omitempty"` + OracleCount uint32 `protobuf:"varint,2,opt,name=oracleCount,proto3" json:"oracleCount,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache +} + +func (x *LLOReportFormatSupportProto) Reset() { + *x = LLOReportFormatSupportProto{} + mi := &file_plugin_codecs_proto_msgTypes[19] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) +} + +func (x *LLOReportFormatSupportProto) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*LLOReportFormatSupportProto) ProtoMessage() {} + +func (x *LLOReportFormatSupportProto) ProtoReflect() protoreflect.Message { + mi := &file_plugin_codecs_proto_msgTypes[19] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use LLOReportFormatSupportProto.ProtoReflect.Descriptor instead. +func (*LLOReportFormatSupportProto) Descriptor() ([]byte, []int) { + return file_plugin_codecs_proto_rawDescGZIP(), []int{19} +} + +func (x *LLOReportFormatSupportProto) GetReportFormat() uint32 { + if x != nil { + return x.ReportFormat + } + return 0 +} + +func (x *LLOReportFormatSupportProto) GetOracleCount() uint32 { + if x != nil { + return x.OracleCount + } + return 0 +} + var File_plugin_codecs_proto protoreflect.FileDescriptor const file_plugin_codecs_proto_rawDesc = "" + "\n" + - "\x13plugin_codecs.proto\x12\x02v1\"\xb2\x05\n" + + "\x13plugin_codecs.proto\x12\x02v1\"\xea\x05\n" + "\x13LLOObservationProto\x12D\n" + "\x1dattestedPredecessorRetirement\x18\x01 \x01(\fR\x1dattestedPredecessorRetirement\x12\"\n" + "\fshouldRetire\x18\x02 \x01(\bR\fshouldRetire\x12F\n" + @@ -1343,7 +1419,8 @@ const file_plugin_codecs_proto_rawDesc = "" + "\x18unixTimestampNanoseconds\x18\a \x01(\x04R\x18unixTimestampNanoseconds\x12*\n" + "\x10removeChannelIDs\x18\x04 \x03(\rR\x10removeChannelIDs\x12q\n" + "\x18updateChannelDefinitions\x18\x05 \x03(\v25.v1.LLOObservationProto.UpdateChannelDefinitionsEntryR\x18updateChannelDefinitions\x12M\n" + - "\fstreamValues\x18\x06 \x03(\v2).v1.LLOObservationProto.StreamValuesEntryR\fstreamValues\x1aj\n" + + "\fstreamValues\x18\x06 \x03(\v2).v1.LLOObservationProto.StreamValuesEntryR\fstreamValues\x126\n" + + "\x16supportedReportFormats\x18\b \x03(\rR\x16supportedReportFormats\x1aj\n" + "\x1dUpdateChannelDefinitionsEntry\x12\x10\n" + "\x03key\x18\x01 \x01(\rR\x03key\x123\n" + "\x05value\x18\x02 \x01(\v2\x1d.v1.LLOChannelDefinitionProtoR\x05value:\x028\x01\x1aS\n" + @@ -1424,14 +1501,18 @@ const file_plugin_codecs_proto_rawDesc = "" + "\x1fobservationTimestampNanoseconds\x18\x01 \x01(\x04R\x1fobservationTimestampNanoseconds\x12c\n" + "\x15validAfterNanoseconds\x18\x02 \x03(\v2-.v1.LLOChannelIDAndValidAfterNanosecondsProtoR\x15validAfterNanoseconds\x122\n" + "\x14reportableChannelIDs\x18\x03 \x03(\rR\x14reportableChannelIDs\x12B\n" + - "\x10streamAggregates\x18\x04 \x03(\v2\x16.v1.LLOStreamAggregateR\x10streamAggregates\"\xb0\x03\n" + + "\x10streamAggregates\x18\x04 \x03(\v2\x16.v1.LLOStreamAggregateR\x10streamAggregates\"\xfb\x03\n" + "\x11LLOPrecursorProto\x12&\n" + "\x0elifeCycleStage\x18\x01 \x01(\tR\x0elifeCycleStage\x12H\n" + "\x1fobservationTimestampNanoseconds\x18\x02 \x01(\x04R\x1fobservationTimestampNanoseconds\x12R\n" + "\x12channelDefinitions\x18\x03 \x03(\v2\".v1.LLOChannelIDAndDefinitionProtoR\x12channelDefinitions\x12c\n" + "\x15validAfterNanoseconds\x18\x04 \x03(\v2-.v1.LLOChannelIDAndValidAfterNanosecondsProtoR\x15validAfterNanoseconds\x12B\n" + "\x10streamAggregates\x18\x05 \x03(\v2\x16.v1.LLOStreamAggregateR\x10streamAggregates\x12,\n" + - "\x11channelStateSeqNr\x18\x06 \x01(\x04R\x11channelStateSeqNrB\fZ\n" + + "\x11channelStateSeqNr\x18\x06 \x01(\x04R\x11channelStateSeqNr\x12I\n" + + "\x0fsupportByFormat\x18\a \x03(\v2\x1f.v1.LLOReportFormatSupportProtoR\x0fsupportByFormat\"c\n" + + "\x1bLLOReportFormatSupportProto\x12\"\n" + + "\freportFormat\x18\x01 \x01(\rR\freportFormat\x12 \n" + + "\voracleCount\x18\x02 \x01(\rR\voracleCountB\fZ\n" + ".;protocolb\x06proto3" var ( @@ -1447,7 +1528,7 @@ func file_plugin_codecs_proto_rawDescGZIP() []byte { } var file_plugin_codecs_proto_enumTypes = make([]protoimpl.EnumInfo, 1) -var file_plugin_codecs_proto_msgTypes = make([]protoimpl.MessageInfo, 21) +var file_plugin_codecs_proto_msgTypes = make([]protoimpl.MessageInfo, 22) var file_plugin_codecs_proto_goTypes = []any{ (LLOStreamValue_Type)(0), // 0: v1.LLOStreamValue.Type (*LLOObservationProto)(nil), // 1: v1.LLOObservationProto @@ -1469,12 +1550,13 @@ var file_plugin_codecs_proto_goTypes = []any{ (*LLOChannelStateProto)(nil), // 17: v1.LLOChannelStateProto (*LLOHotStateProto)(nil), // 18: v1.LLOHotStateProto (*LLOPrecursorProto)(nil), // 19: v1.LLOPrecursorProto - nil, // 20: v1.LLOObservationProto.UpdateChannelDefinitionsEntry - nil, // 21: v1.LLOObservationProto.StreamValuesEntry + (*LLOReportFormatSupportProto)(nil), // 20: v1.LLOReportFormatSupportProto + nil, // 21: v1.LLOObservationProto.UpdateChannelDefinitionsEntry + nil, // 22: v1.LLOObservationProto.StreamValuesEntry } var file_plugin_codecs_proto_depIdxs = []int32{ - 20, // 0: v1.LLOObservationProto.updateChannelDefinitions:type_name -> v1.LLOObservationProto.UpdateChannelDefinitionsEntry - 21, // 1: v1.LLOObservationProto.streamValues:type_name -> v1.LLOObservationProto.StreamValuesEntry + 21, // 0: v1.LLOObservationProto.updateChannelDefinitions:type_name -> v1.LLOObservationProto.UpdateChannelDefinitionsEntry + 22, // 1: v1.LLOObservationProto.streamValues:type_name -> v1.LLOObservationProto.StreamValuesEntry 0, // 2: v1.LLOStreamValue.type:type_name -> v1.LLOStreamValue.Type 2, // 3: v1.LLOTimestampedStreamValue.streamValue:type_name -> v1.LLOStreamValue 2, // 4: v1.LLOStreamHistoryRecord.value:type_name -> v1.LLOStreamValue @@ -1494,13 +1576,14 @@ var file_plugin_codecs_proto_depIdxs = []int32{ 13, // 18: v1.LLOPrecursorProto.channelDefinitions:type_name -> v1.LLOChannelIDAndDefinitionProto 15, // 19: v1.LLOPrecursorProto.validAfterNanoseconds:type_name -> v1.LLOChannelIDAndValidAfterNanosecondsProto 16, // 20: v1.LLOPrecursorProto.streamAggregates:type_name -> v1.LLOStreamAggregate - 8, // 21: v1.LLOObservationProto.UpdateChannelDefinitionsEntry.value:type_name -> v1.LLOChannelDefinitionProto - 2, // 22: v1.LLOObservationProto.StreamValuesEntry.value:type_name -> v1.LLOStreamValue - 23, // [23:23] is the sub-list for method output_type - 23, // [23:23] is the sub-list for method input_type - 23, // [23:23] is the sub-list for extension type_name - 23, // [23:23] is the sub-list for extension extendee - 0, // [0:23] is the sub-list for field type_name + 20, // 21: v1.LLOPrecursorProto.supportByFormat:type_name -> v1.LLOReportFormatSupportProto + 8, // 22: v1.LLOObservationProto.UpdateChannelDefinitionsEntry.value:type_name -> v1.LLOChannelDefinitionProto + 2, // 23: v1.LLOObservationProto.StreamValuesEntry.value:type_name -> v1.LLOStreamValue + 24, // [24:24] is the sub-list for method output_type + 24, // [24:24] is the sub-list for method input_type + 24, // [24:24] is the sub-list for extension type_name + 24, // [24:24] is the sub-list for extension extendee + 0, // [0:24] is the sub-list for field type_name } func init() { file_plugin_codecs_proto_init() } @@ -1514,7 +1597,7 @@ func file_plugin_codecs_proto_init() { GoPackagePath: reflect.TypeOf(x{}).PkgPath(), RawDescriptor: unsafe.Slice(unsafe.StringData(file_plugin_codecs_proto_rawDesc), len(file_plugin_codecs_proto_rawDesc)), NumEnums: 1, - NumMessages: 21, + NumMessages: 22, NumExtensions: 0, NumServices: 0, }, diff --git a/llo/protocol/plugin_codecs.proto b/llo/protocol/plugin_codecs.proto index 05aa6c4..c0e4a40 100644 --- a/llo/protocol/plugin_codecs.proto +++ b/llo/protocol/plugin_codecs.proto @@ -27,6 +27,12 @@ message LLOObservationProto { // uniqueness. map updateChannelDefinitions = 5; map streamValues = 6; + // Report formats this oracle has a report codec for. Advertised so that + // reportability can require enough oracles to be able to encode a channel + // before validAfter advances over it. Encoding is node-local state and + // cannot be read in the state transition; this carries it as a replicated + // fact instead. + repeated uint32 supportedReportFormats = 8; } message LLOStreamValue { @@ -211,4 +217,13 @@ message LLOPrecursorProto { repeated LLOChannelIDAndValidAfterNanosecondsProto validAfterNanoseconds = 4; repeated LLOStreamAggregate streamAggregates = 5; uint64 channelStateSeqNr = 6; + // supportByFormat MUST be sorted ascending by reportFormat. + repeated LLOReportFormatSupportProto supportByFormat = 7; +} + +// LLOReportFormatSupportProto is the number of oracles that advertised a report +// codec for one report format in the round that produced the precursor. +message LLOReportFormatSupportProto { + uint32 reportFormat = 1; + uint32 oracleCount = 2; } From 9e25c053b3456455e8d97481a2a8568c347c1b3b Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Wed, 16 Sep 2026 15:36:15 +0100 Subject: [PATCH 07/40] SPOR-0006 llo: do not halt Observation on baseline verification of already committed definitions Log the finding and continue in v30 and v31 instead of returning an error, so channel votes are still cast and the offending channel can be removed. The channels stay observed. --- llo/dev/v31/plugin.go | 29 +++- llo/dev/v31/plugin_test.go | 165 +++++++++++++++++++++++ llo/protocol/channel_definitions.go | 139 +++++++++++++++---- llo/protocol/channel_definitions_test.go | 94 +++++++++++++ llo/v30/plugin_observation.go | 19 ++- llo/v30/plugin_observation_test.go | 89 +++++++++++- 6 files changed, 500 insertions(+), 35 deletions(-) diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index 2510ada..2efb2a8 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -93,8 +93,16 @@ func (p *Plugin) Observation(_ context.Context, seqNr uint64, _ ocrtypes.Attribu if state.lifeCycleStage == protocol.LifeCycleStageRetired { p.Logger.Debugw("Node is retired, will generate empty observation", "stage", "Observation", "seqNr", seqNr) } else { - if err = protocol.VerifyChannelDefinitions(p.ReportCodecs, state.channelDefinitions); err != nil { - return nil, fmt.Errorf("state.channelDefinitions is invalid: %w", err) + // Committed state is replicated, so failing verification here on every + // node would halt the DON with no way out, as the nodes would never vote + // to remove offending channels. + // For a DON where all participants share the same version this is unreachable, + // as ValidateObservation runs the same baseline checks over the merged set, + // but a version skew can make it reachable. + // Report the finding and carry on, which is the same treatment the + // admission-only findings get in voteOnChannels. + if badChannels, verifyErr := protocol.UnverifiableChannelIDs(p.ReportCodecs, state.channelDefinitions); len(badChannels) > 0 || verifyErr != nil { + p.Logger.Errorw("Committed channel definitions fail baseline verification on this build", "stage", "Observation", "seqNr", seqNr, "channelIDs", sortedChannelIDSet(badChannels), "err", verifyErr) } if p.PredecessorConfigDigest != nil && state.lifeCycleStage == protocol.LifeCycleStageStaging { @@ -193,6 +201,16 @@ func observableStreams(state *kvState) []llotypes.StreamID { return streams } +// sortedChannelIDSet renders a channel ID set in ascending order, for logs. +func sortedChannelIDSet(set map[llotypes.ChannelID]struct{}) []llotypes.ChannelID { + ids := make([]llotypes.ChannelID, 0, len(set)) + for channelID := range set { + ids = append(ids, channelID) + } + sortChannelIDs(ids) + return ids +} + // voteOnChannels populates obs.RemoveChannelIDs / obs.UpdateChannelDefinitions // by comparing the desired channel definitions against current KV state. func (p *Plugin) voteOnChannels(obs *Observation, state *kvState) { @@ -281,6 +299,13 @@ func (p *Plugin) ValidateObservation(ctx context.Context, seqNr uint64, _ ocrtyp } defsForVerify = merged } + // Unlike the committed-state check in Observation, this one stays fatal: + // rejecting a peer observation is not a halt, and these checks are the only + // thing standing between a proposer and a malformed committed definition. + // Under version skew a stricter build rejects observations that carry + // updates from a staler one, which costs update-voting liveness only -- an + // observation that votes no update has nothing to verify here, so rounds + // themselves are unaffected. if err := protocol.VerifyChannelDefinitions(p.ReportCodecs, defsForVerify); err != nil { return fmt.Errorf("UpdateChannelDefinitions is invalid: %w", err) } diff --git a/llo/dev/v31/plugin_test.go b/llo/dev/v31/plugin_test.go index ca18c2c..1ee0f9f 100644 --- a/llo/dev/v31/plugin_test.go +++ b/llo/dev/v31/plugin_test.go @@ -4,6 +4,8 @@ import ( "context" "encoding/binary" "errors" + "strconv" + "sync" "testing" "time" @@ -1211,3 +1213,166 @@ func Test_StateTransition_TalliesReportFormatSupport(t *testing.T) { require.NoError(t, err) require.Equal(t, 2, prec.SupportByFormat[llotypes.ReportFormatJSON]) } + +// strictJSONCodec is reportcodec.JSONReportCodec with an extra Verify rule this +// build has and the build that admitted the definition did not: the version +// skew that makes a baseline failure on committed state reachable. +type strictJSONCodec struct { + reportcodec.JSONReportCodec + rejectStream llotypes.StreamID +} + +func (c strictJSONCodec) Verify(cd llotypes.ChannelDefinition) error { + for _, strm := range cd.Streams { + if strm.StreamID == c.rejectStream { + return errors.New("this build rejects stream " + strconv.Itoa(int(strm.StreamID))) + } + } + return c.JSONReportCodec.Verify(cd) +} + +// recordingDataSource records the stream IDs it was asked to observe. +type recordingDataSource struct { + mu sync.Mutex + seen map[llotypes.StreamID]struct{} +} + +func (d *recordingDataSource) Observe(_ context.Context, sv protocol.StreamValues, _ DSOpts) error { + d.mu.Lock() + defer d.mu.Unlock() + if d.seen == nil { + d.seen = map[llotypes.StreamID]struct{}{} + } + for streamID := range sv { + d.seen[streamID] = struct{}{} + sv[streamID] = protocol.ToDecimal(decimal.NewFromInt(1)) + } + return nil +} + +func (d *recordingDataSource) streams() []llotypes.StreamID { + d.mu.Lock() + defer d.mu.Unlock() + out := make([]llotypes.StreamID, 0, len(d.seen)) + for streamID := range d.seen { + out = append(out, streamID) + } + return out +} + +// Test_Observation_UnverifiableCommittedChannelIsNotFatal covers the +// version-skew case: a channel committed under an older build fails this +// build's codec.Verify. The node must not halt -- it keeps observing and, +// crucially, still votes the offending channel out, which is the only way the +// DON recovers. +func Test_Observation_UnverifiableCommittedChannelIsNotFatal(t *testing.T) { + ctx := tests.Context(t) + p := testPlugin(t) + kv := newMemKV() + + healthy := llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatJSON, + Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}, + } + rejected := llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatJSON, + Streams: []llotypes.Stream{{StreamID: 999, Aggregator: llotypes.AggregatorMedian}}, + } + + // Rounds 1-2: both channels are admitted by a build that accepts them. + _, err := p.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, nil), ao(1, nil), ao(2, nil)}, kv, testBlobs) + require.NoError(t, err) + addObs := Observation{ + UnixTimestampNanoseconds: 1_000, + UpdateChannelDefinitions: llotypes.ChannelDefinitions{1: healthy, 2: rejected}, + } + addAOs := []ocrtypes.AttributedObservation{} + for i := 0; i < 4; i++ { + addAOs = append(addAOs, ao(i, mustEncodeObs(t, addObs))) + } + _, err = p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, addAOs, kv, testBlobs) + require.NoError(t, err) + require.Contains(t, kvChannelDefs(t, kv), llotypes.ChannelID(2)) + + // Now this node upgrades to a build whose codec rejects channel 2, and the + // definitions file drops it so the node has something to vote for. + p.ReportCodecs = map[llotypes.ReportFormat]protocol.ReportCodec{ + llotypes.ReportFormatJSON: strictJSONCodec{rejectStream: 999}, + } + p.ChannelCache = protocol.NewChannelCache() + p.ChannelDefinitionCache = &mockChannelDefinitionCache{defs: llotypes.ChannelDefinitions{1: healthy}} + p.ShouldRetireCache = &mockShouldRetireCache{} + ds := &recordingDataSource{} + attachPump(t, p, ds, newFakeBroadcaster()) + + obsBytes, err := p.Observation(ctx, 3, ocrtypes.AttributedQuery{}, kv, nil) + require.NoError(t, err, "a committed definition this build rejects must not halt the node") + obs, err := decodeObservation(ctx, obsBytes, testBlobs) + require.NoError(t, err) + + // The removal vote is the recovery path, and it is only cast because the + // verification failure above was not fatal. + require.Contains(t, obs.RemoveChannelIDs, llotypes.ChannelID(2)) + require.NotContains(t, obs.UpdateChannelDefinitions, llotypes.ChannelID(2)) + + // The rejected channel's streams are still observed: withholding them would + // only starve the nodes still on the old build, which considers the channel + // valid and reportable. + require.Eventually(t, func() bool { return p.pump.Cycles() >= 1 }, tests.WaitTimeout(t), 10*time.Millisecond) + require.ElementsMatch(t, []llotypes.StreamID{100, 999}, ds.streams()) +} + +// Test_FullRound_RecoverFromUnverifiableChannel is the end-to-end recovery: +// every node upgrades to a build that rejects a committed channel, and the DON +// keeps making rounds and votes the channel out. +func Test_FullRound_RecoverFromUnverifiableChannel(t *testing.T) { + ctx := tests.Context(t) + p := testPlugin(t) + kv := newMemKV() + + healthy := llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatJSON, + Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}, + } + rejected := llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatJSON, + Streams: []llotypes.Stream{{StreamID: 999, Aggregator: llotypes.AggregatorMedian}}, + } + + _, err := p.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, nil), ao(1, nil), ao(2, nil)}, kv, testBlobs) + require.NoError(t, err) + addObs := Observation{ + UnixTimestampNanoseconds: 1_000, + UpdateChannelDefinitions: llotypes.ChannelDefinitions{1: healthy, 2: rejected}, + } + addAOs := []ocrtypes.AttributedObservation{} + for i := 0; i < 4; i++ { + addAOs = append(addAOs, ao(i, mustEncodeObs(t, addObs))) + } + _, err = p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, addAOs, kv, testBlobs) + require.NoError(t, err) + + p.ReportCodecs = map[llotypes.ReportFormat]protocol.ReportCodec{ + llotypes.ReportFormatJSON: strictJSONCodec{rejectStream: 999}, + } + p.ChannelCache = protocol.NewChannelCache() + + // Every oracle votes the rejected channel out, and the round is accepted: + // the same observation must also pass ValidateObservation on the new build. + removeObs := Observation{ + UnixTimestampNanoseconds: 2_000, + RemoveChannelIDs: map[llotypes.ChannelID]struct{}{2: {}}, + StreamValues: protocol.StreamValues{100: protocol.ToDecimal(decimal.NewFromInt(42))}, + } + removeAOs := []ocrtypes.AttributedObservation{} + for i := 0; i < 4; i++ { + removeAOs = append(removeAOs, ao(i, mustEncodeObs(t, removeObs))) + } + for _, aObs := range removeAOs { + require.NoError(t, p.ValidateObservation(ctx, 3, ocrtypes.AttributedQuery{}, aObs, kv, testBlobs)) + } + _, err = p.StateTransition(ctx, 3, ocrtypes.AttributedQuery{}, removeAOs, kv, testBlobs) + require.NoError(t, err) + require.NotContains(t, kvChannelDefs(t, kv), llotypes.ChannelID(2)) + require.Contains(t, kvChannelDefs(t, kv), llotypes.ChannelID(1)) +} diff --git a/llo/protocol/channel_definitions.go b/llo/protocol/channel_definitions.go index 534b05d..5ef72c4 100644 --- a/llo/protocol/channel_definitions.go +++ b/llo/protocol/channel_definitions.go @@ -78,9 +78,106 @@ func (f admissionFinding) appliesTo(admitting map[llotypes.ChannelID]struct{}) b return false } -func verifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, channelDefs llotypes.ChannelDefinitions, admitting map[llotypes.ChannelID]struct{}) (merr error) { +// UnverifiableChannelIDs returns the channels of an already-committed +// definition set that fail a baseline check, so that a node can skip them +// instead of halting. +// +// Baseline checks are the ones every definition set must satisfy, admitted or +// committed, and VerifyChannelDefinitions reports them as a single error. That +// is the right answer on the admission path, where the set can simply be +// rejected. +// Committed state is replicated, so failing verification on every +// node would halt the DON with no way out, as the nodes would never vote +// to remove offending channels. +// For a DON where all participants share the same version this is unreachable, +// as ValidateObservation runs the same baseline checks over the merged set, +// but a version skew can make it reachable. +// Report the finding and carry on, which is the same treatment the +// admission-only findings get in voteOnChannels. +// +// Attributing the findings instead lets the caller drop just those channels and +// keep going, which is the same treatment the admission-only findings already +// get. The returned error carries the whole-set baseline findings, which +// implicate no particular channel and so cannot be skipped selectively. +func UnverifiableChannelIDs(codecs map[llotypes.ReportFormat]ReportCodec, channelDefs llotypes.ChannelDefinitions) (map[llotypes.ChannelID]struct{}, error) { + res := analyzeChannelDefinitions(codecs, channelDefs) + ids := make(map[llotypes.ChannelID]struct{}, len(res.channelErrs)) + for channelID := range res.channelErrs { + ids[channelID] = struct{}{} + } + return ids, res.setErr() +} + +// verifyResult is the outcome of analyzing a definition set: baseline findings +// attributed to the channel that produced them, baseline findings that belong +// to the set as a whole, and the admission-only findings, which are filtered by +// the admitting set only when an error is materialized. +type verifyResult struct { + channelErrs map[llotypes.ChannelID]error + wholeSetErr error + uniqueStreamIDs int + admissionFindings []admissionFinding +} + +// setErr reports the baseline findings that implicate no particular channel. +// The unique-stream-ID budget is one of them, but it is only meaningful once +// the per-channel findings are clear (a definition that failed verification may +// have contributed stream IDs that a corrected one would not), so it is +// reported only when nothing else failed. +func (r verifyResult) setErr() error { + if r.wholeSetErr != nil { + return r.wholeSetErr + } + if len(r.channelErrs) == 0 && r.uniqueStreamIDs > MaxObservationStreamValuesLength { + return fmt.Errorf("too many unique stream IDs, got: %d/%d", r.uniqueStreamIDs, MaxObservationStreamValuesLength) + } + return nil +} + +// err joins every finding that applies into the single error the verification +// entry points return. Channel findings are joined in ascending channel ID +// order so that a rejected definitions file produces the same error on every +// oracle. +func (r verifyResult) err(admitting map[llotypes.ChannelID]struct{}) error { + if r.wholeSetErr != nil { + return r.wholeSetErr + } + + var merr error + channelIDs := make([]llotypes.ChannelID, 0, len(r.channelErrs)) + for channelID := range r.channelErrs { + channelIDs = append(channelIDs, channelID) + } + sort.Slice(channelIDs, func(i, j int) bool { return channelIDs[i] < channelIDs[j] }) + for _, channelID := range channelIDs { + merr = errors.Join(merr, r.channelErrs[channelID]) + } + + for _, finding := range r.admissionFindings { + if finding.appliesTo(admitting) { + merr = errors.Join(merr, finding.err) + } + } + + if merr != nil { + return merr + } + if r.uniqueStreamIDs > MaxObservationStreamValuesLength { + return fmt.Errorf("too many unique stream IDs, got: %d/%d", r.uniqueStreamIDs, MaxObservationStreamValuesLength) + } + return nil +} + +func verifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, channelDefs llotypes.ChannelDefinitions, admitting map[llotypes.ChannelID]struct{}) error { + return analyzeChannelDefinitions(codecs, channelDefs).err(admitting) +} + +func analyzeChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, channelDefs llotypes.ChannelDefinitions) (res verifyResult) { + res.channelErrs = make(map[llotypes.ChannelID]error) + if len(channelDefs) > MaxOutcomeChannelDefinitionsLength { - return fmt.Errorf("too many channels, got: %d/%d", len(channelDefs), MaxOutcomeChannelDefinitionsLength) + res.wholeSetErr = fmt.Errorf("too many channels, got: %d/%d", len(channelDefs), MaxOutcomeChannelDefinitionsLength) + return res } // Verify in ascending channel ID order so that the errors a rejected @@ -91,16 +188,23 @@ func verifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, chan } sort.Slice(channelIDs, func(i, j int) bool { return channelIDs[i] < channelIDs[j] }) + // Baseline findings are attributed to the channel that produced them; a + // cross-definition baseline finding is attributed to every channel it + // implicates, so that skipping any one of them clears it. + base := func(err error, channels ...llotypes.ChannelID) { + for _, channelID := range channels { + res.channelErrs[channelID] = errors.Join(res.channelErrs[channelID], err) + } + } // Admission-only findings are collected as they are discovered and filtered // against admitting once at the end. The bookkeeping the cross-definition // checks rely on is built for the whole set either way, so that a finding // does not depend on which channels are being admitted. - var admissionFindings []admissionFinding admit := func(err error, channels ...llotypes.ChannelID) { - admissionFindings = append(admissionFindings, admissionFinding{channels: channels, err: err}) + res.admissionFindings = append(res.admissionFindings, admissionFinding{channels: channels, err: err}) } admitSet := func(err error) { - admissionFindings = append(admissionFindings, admissionFinding{wholeSet: true, err: err}) + res.admissionFindings = append(res.admissionFindings, admissionFinding{wholeSet: true, err: err}) } // Whole-set budgets, accumulated over the channels the loop below visits @@ -127,11 +231,11 @@ func verifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, chan } if len(cd.Streams) == 0 { - merr = errors.Join(merr, fmt.Errorf("ChannelDefinition with ID %d has no streams", channelID)) + base(fmt.Errorf("ChannelDefinition with ID %d has no streams", channelID), channelID) continue } if len(cd.Streams) > MaxStreamsPerChannel { - merr = errors.Join(merr, fmt.Errorf("ChannelDefinition with ID %d has too many streams, got: %d/%d", channelID, len(cd.Streams), MaxStreamsPerChannel)) + base(fmt.Errorf("ChannelDefinition with ID %d has too many streams, got: %d/%d", channelID, len(cd.Streams), MaxStreamsPerChannel), channelID) continue } totalStreamEntries += len(cd.Streams) @@ -145,7 +249,7 @@ func verifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, chan totalOptsBytes += len(cd.Opts) for _, strm := range cd.Streams { if strm.Aggregator == 0 { - merr = errors.Join(merr, fmt.Errorf("ChannelDefinition with ID %d has stream %d with zero aggregator (this may indicate an uninitialized struct)", channelID, strm.StreamID)) + base(fmt.Errorf("ChannelDefinition with ID %d has stream %d with zero aggregator (this may indicate an uninitialized struct)", channelID, strm.StreamID), channelID) continue } // An aggregator this binary does not know has no aggregator @@ -185,7 +289,7 @@ func verifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, chan if codec, ok := codecs[cd.ReportFormat]; ok { verifyErr = codec.Verify(cd) if verifyErr != nil { - merr = errors.Join(merr, fmt.Errorf("invalid ChannelDefinition with ID %d: %w", channelID, verifyErr)) + base(fmt.Errorf("invalid ChannelDefinition with ID %d: %w", channelID, verifyErr), channelID) } if av, ok := codec.(AdmissionVerifier); ok && verifyErr == nil { if err := av.VerifyForAdmission(cd); err != nil { @@ -209,7 +313,7 @@ func verifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, chan } if cd.ReportFormat == llotypes.ReportFormatHistoryBackfill { if err := ValidateHistoryBackfillAgainstDefinitions(cd, channelDefs, 0); err != nil { - merr = errors.Join(merr, fmt.Errorf("invalid history backfill channel %d: %w", channelID, err)) + base(fmt.Errorf("invalid history backfill channel %d: %w", channelID, err), channelID) } if err := ValidateHistoryBackfillTarget(cd, channelDefs); err != nil { admit(fmt.Errorf("invalid history backfill channel %d: %w", channelID, err), channelID) @@ -241,19 +345,8 @@ func verifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, chan admitSet(fmt.Errorf("too many opts bytes across all channels, got: %d/%d", totalOptsBytes, MaxTotalOptsBytes)) } - for _, finding := range admissionFindings { - if finding.appliesTo(admitting) { - merr = errors.Join(merr, finding.err) - } - } - - if merr != nil { - return merr - } - if len(uniqueStreamIDs) > MaxObservationStreamValuesLength { - return fmt.Errorf("too many unique stream IDs, got: %d/%d", len(uniqueStreamIDs), MaxObservationStreamValuesLength) - } - return nil + res.uniqueStreamIDs = len(uniqueStreamIDs) + return res } func sortedStreamIDs(m map[llotypes.StreamID]llotypes.ChannelID) []llotypes.StreamID { diff --git a/llo/protocol/channel_definitions_test.go b/llo/protocol/channel_definitions_test.go index 7cbca35..c4441f7 100644 --- a/llo/protocol/channel_definitions_test.go +++ b/llo/protocol/channel_definitions_test.go @@ -507,3 +507,97 @@ func Test_ChangedChannelIDs(t *testing.T) { require.Empty(t, ChangedChannelIDs(current, current)) require.Empty(t, ChangedChannelIDs(current, nil)) } + +// rejectingCodec fails Verify for the channels whose IDs are in reject, keyed +// by the first stream ID of the definition (channel IDs are not passed to +// Verify). +type rejectingCodec struct { + rejectStream llotypes.StreamID +} + +func (rejectingCodec) Encode(Report, llotypes.ChannelDefinition, *OptsCache) ([]byte, error) { + return nil, nil +} + +func (c rejectingCodec) Verify(cd llotypes.ChannelDefinition) error { + if len(cd.Streams) > 0 && cd.Streams[0].StreamID == c.rejectStream { + return errors.New("codec says no") + } + return nil +} + +func Test_UnverifiableChannelIDs(t *testing.T) { + channel := func(streamID llotypes.StreamID) llotypes.ChannelDefinition { + return llotypes.ChannelDefinition{Streams: []llotypes.Stream{{StreamID: streamID, Aggregator: llotypes.AggregatorMedian}}} + } + + t.Run("attributes a baseline failure to exactly its channel", func(t *testing.T) { + codecs := map[llotypes.ReportFormat]ReportCodec{0: rejectingCodec{rejectStream: 2}} + defs := llotypes.ChannelDefinitions{1: channel(1), 2: channel(2), 3: channel(3)} + + ids, err := UnverifiableChannelIDs(codecs, defs) + require.NoError(t, err) + require.Equal(t, map[llotypes.ChannelID]struct{}{2: {}}, ids) + + // The same set is still a hard error on the admission path. + require.EqualError(t, VerifyChannelDefinitions(codecs, defs), "invalid ChannelDefinition with ID 2: codec says no") + }) + + t.Run("attributes every kind of baseline finding", func(t *testing.T) { + codecs := map[llotypes.ReportFormat]ReportCodec{0: stubReportCodec{}} + tooManyStreams := llotypes.ChannelDefinition{Streams: make([]llotypes.Stream, MaxStreamsPerChannel+1)} + for i := range tooManyStreams.Streams { + tooManyStreams.Streams[i] = llotypes.Stream{StreamID: llotypes.StreamID(i + 100), Aggregator: llotypes.AggregatorMedian} + } + defs := llotypes.ChannelDefinitions{ + 1: channel(1), + 2: {}, // no streams + 3: {Streams: []llotypes.Stream{{StreamID: 3}}}, // zero aggregator + 4: tooManyStreams, + } + + ids, err := UnverifiableChannelIDs(codecs, defs) + require.NoError(t, err) + require.Equal(t, map[llotypes.ChannelID]struct{}{2: {}, 3: {}, 4: {}}, ids) + }) + + t.Run("admission-only findings are not attributed", func(t *testing.T) { + // Two channels sharing a feed ID is admission-only: a committed set + // carrying it keeps reporting, so neither channel is skipped. + codecs := map[llotypes.ReportFormat]ReportCodec{0: mockFeedIDCodec{feedID: [32]byte{0xab}, ok: true}} + defs := llotypes.ChannelDefinitions{1: channel(1), 2: channel(2)} + + ids, err := UnverifiableChannelIDs(codecs, defs) + require.NoError(t, err) + require.Empty(t, ids) + require.Error(t, verifyAdmittingAll(codecs, defs)) + }) + + t.Run("a tombstone is never unverifiable", func(t *testing.T) { + codecs := map[llotypes.ReportFormat]ReportCodec{0: rejectingCodec{rejectStream: 2}} + defs := llotypes.ChannelDefinitions{1: channel(1), 2: {Tombstone: true}} + + ids, err := UnverifiableChannelIDs(codecs, defs) + require.NoError(t, err) + require.Empty(t, ids) + }) + + t.Run("a whole-set finding is returned as an error with no channels to skip", func(t *testing.T) { + codecs := map[llotypes.ReportFormat]ReportCodec{0: stubReportCodec{}} + defs := make(llotypes.ChannelDefinitions, MaxOutcomeChannelDefinitionsLength+1) + for i := range MaxOutcomeChannelDefinitionsLength + 1 { + defs[llotypes.ChannelID(i+1)] = channel(llotypes.StreamID(i + 1)) + } + + ids, err := UnverifiableChannelIDs(codecs, defs) + require.Error(t, err) + require.Contains(t, err.Error(), "too many channels") + require.Empty(t, ids) + }) + + t.Run("an empty definition set is clean", func(t *testing.T) { + ids, err := UnverifiableChannelIDs(map[llotypes.ReportFormat]ReportCodec{}, llotypes.ChannelDefinitions{}) + require.NoError(t, err) + require.Empty(t, ids) + }) +} diff --git a/llo/v30/plugin_observation.go b/llo/v30/plugin_observation.go index fd38117..d92c24f 100644 --- a/llo/v30/plugin_observation.go +++ b/llo/v30/plugin_observation.go @@ -40,13 +40,8 @@ func (p *Plugin) observation(ctx context.Context, outctx ocr3types.OutcomeContex if previousOutcome.LifeCycleStage == protocol.LifeCycleStageRetired { p.Logger.Debugw("Node is retired, will generate empty observation", "stage", "Observation", "seqNr", outctx.SeqNr) } else { - if err = protocol.VerifyChannelDefinitions(p.ReportCodecs, previousOutcome.ChannelDefinitions); err != nil { - // This is not expected, unless the majority of nodes are using a - // different verification method than this one. - // - // If it does happen, it's an invariant violation and we cannot - // generate an observation. - return nil, fmt.Errorf("previousOutcome.Definitions is invalid: %w", err) + if badChannels, verifyErr := protocol.UnverifiableChannelIDs(p.ReportCodecs, previousOutcome.ChannelDefinitions); len(badChannels) > 0 || verifyErr != nil { + p.Logger.Errorw("Agreed channel definitions fail baseline verification on this build", "stage", "Observation", "seqNr", outctx.SeqNr, "channelIDs", sortedChannelIDSet(badChannels), "err", verifyErr) } // Only try to fetch this from the cache if this instance if configured @@ -181,6 +176,16 @@ func (p *Plugin) observation(ctx context.Context, outctx ocr3types.OutcomeContex return serialized, nil } +// sortedChannelIDSet renders a channel ID set in ascending order, for logs. +func sortedChannelIDSet(set map[llotypes.ChannelID]struct{}) []llotypes.ChannelID { + ids := make([]llotypes.ChannelID, 0, len(set)) + for channelID := range set { + ids = append(ids, channelID) + } + sortChannelIDs(ids) + return ids +} + type Observation struct { // Attested (i.e. signed by f+1 oracles) retirement report from predecessor // protocol instance diff --git a/llo/v30/plugin_observation_test.go b/llo/v30/plugin_observation_test.go index 07ca30b..2ad9296 100644 --- a/llo/v30/plugin_observation_test.go +++ b/llo/v30/plugin_observation_test.go @@ -296,7 +296,7 @@ func testObservation(t *testing.T, outcomeCodec OutcomeCodec) { assert.Equal(t, ds.s, decoded.StreamValues) }) - t.Run("in case previous outcome channel definitions is invalid, returns error", func(t *testing.T) { + t.Run("in case previous outcome channel definitions fails a whole-set baseline check, does not halt", func(t *testing.T) { dfns := make(llotypes.ChannelDefinitions) for i := uint32(0); i < 2*protocol.MaxOutcomeChannelDefinitionsLength; i++ { dfns[i] = llotypes.ChannelDefinition{ @@ -311,9 +311,62 @@ func testObservation(t *testing.T, outcomeCodec OutcomeCodec) { encodedPreviousOutcome, err := p.OutcomeCodec.Encode(previousOutcome) require.NoError(t, err) + // Agreed state that this build considers invalid is logged and worked + // around, never fatal: halting here would stop every node on this + // build at once, and the removal votes are cast in this same call. outctx := ocr3types.OutcomeContext{SeqNr: 3, PreviousOutcome: encodedPreviousOutcome} - _, err = p.Observation(context.Background(), outctx, query) - require.EqualError(t, err, "previousOutcome.Definitions is invalid: too many channels, got: 4000/2000") + obs, err := p.Observation(context.Background(), outctx, query) + require.NoError(t, err) + require.NotEmpty(t, obs) + }) + + t.Run("a committed channel this build rejects is not fatal and is still voted out", func(t *testing.T) { + rejected := llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatJSON, + Streams: []llotypes.Stream{{StreamID: 999, Aggregator: llotypes.AggregatorMedian}}, + } + committed := llotypes.ChannelDefinitions{ + 1: smallDefinitions[1], + 9: rejected, + } + previousOutcome := Outcome{ + LifeCycleStage: llotypes.LifeCycleStage("test"), + ObservationTimestampNanoseconds: testStartTSNanos, + ChannelDefinitions: committed, + } + encodedPreviousOutcome, err := p.OutcomeCodec.Encode(previousOutcome) + require.NoError(t, err) + + // This build's codec rejects channel 9; the build that admitted it + // did not. The definitions file has already dropped it. + strict := &Plugin{ + Config: Config{true}, + OutcomeCodec: outcomeCodec, + ShouldRetireCache: &mockShouldRetireCache{}, + Logger: logger.Test(t), + ObservationCodec: obsCodec, + ReportCodecs: map[llotypes.ReportFormat]protocol.ReportCodec{ + llotypes.ReportFormatJSON: strictJSONVerifyCodec{rejectStream: 999}, + }, + ChannelDefinitionCache: &mockChannelDefinitionCache{definitions: llotypes.ChannelDefinitions{1: smallDefinitions[1]}}, + // Fills only the streams it was asked for, so the assertions + // below reflect what the observation actually requested. + DataSource: &echoDataSource{}, + } + + outctx := ocr3types.OutcomeContext{SeqNr: 3, PreviousOutcome: encodedPreviousOutcome} + obsBytes, err := strict.Observation(context.Background(), outctx, query) + require.NoError(t, err, "a committed definition this build rejects must not halt the node") + decoded, err := strict.ObservationCodec.Decode(obsBytes) + require.NoError(t, err) + + // The removal vote is the recovery path. + require.Contains(t, decoded.RemoveChannelIDs, llotypes.ChannelID(9)) + // Channel 9's stream is still observed: withholding it would only + // starve the nodes still on the old build, which considers the + // channel valid and reportable. + require.Contains(t, decoded.StreamValues, llotypes.StreamID(999)) + require.Contains(t, decoded.StreamValues, llotypes.StreamID(1)) }) t.Run("in case ChannelDefinitionsCache returns invalid definitions, does not vote to change anything", func(t *testing.T) { @@ -559,3 +612,33 @@ func testObservation(t *testing.T, outcomeCodec OutcomeCodec) { assert.Equal(t, ds.s, decoded.StreamValues) }) } + +// echoDataSource returns a value for exactly the streams it is asked to observe. +type echoDataSource struct{} + +func (echoDataSource) Observe(_ context.Context, streamValues protocol.StreamValues, _ DSOpts) error { + for streamID := range streamValues { + streamValues[streamID] = protocol.ToDecimal(decimal.NewFromInt(1)) + } + return nil +} + +// strictJSONVerifyCodec rejects any definition carrying rejectStream, standing +// in for a build whose Verify is stricter than the one that admitted the +// definition. +type strictJSONVerifyCodec struct { + rejectStream llotypes.StreamID +} + +func (strictJSONVerifyCodec) Encode(protocol.Report, llotypes.ChannelDefinition, *protocol.OptsCache) ([]byte, error) { + return nil, nil +} + +func (c strictJSONVerifyCodec) Verify(cd llotypes.ChannelDefinition) error { + for _, strm := range cd.Streams { + if strm.StreamID == c.rejectStream { + return errors.New("this build rejects this definition") + } + } + return nil +} From 04c3a42568043cf6017ce57e139655ef135e1cb7 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Thu, 17 Sep 2026 10:43:43 +0100 Subject: [PATCH 08/40] SPOR-0007 llo: do not fail Observation on retirement cache errors --- llo/dev/v31/flow_test.go | 15 ++++++++++++--- llo/dev/v31/plugin.go | 11 +++++++++-- llo/dev/v31/plugin_test.go | 29 +++++++++++++++++++++++++++++ llo/v30/plugin_observation.go | 10 ++++++++-- llo/v30/plugin_observation_test.go | 30 +++++++++++++++++++++++++++--- 5 files changed, 85 insertions(+), 10 deletions(-) diff --git a/llo/dev/v31/flow_test.go b/llo/dev/v31/flow_test.go index f72ca03..4fba1ef 100644 --- a/llo/dev/v31/flow_test.go +++ b/llo/dev/v31/flow_test.go @@ -107,10 +107,13 @@ func (m *blockingDataSource) concurrent() int { return m.maxSeen } -type mockShouldRetireCache struct{ retire bool } +type mockShouldRetireCache struct { + retire bool + err error +} func (m *mockShouldRetireCache) ShouldRetire(ocrtypes.ConfigDigest) (bool, error) { - return m.retire, nil + return m.retire, m.err } type mockOnchainConfigCodec struct{} @@ -120,9 +123,15 @@ func (mockOnchainConfigCodec) Decode([]byte) (protocol.OnchainConfig, error) { } func (mockOnchainConfigCodec) Encode(protocol.OnchainConfig) ([]byte, error) { return nil, nil } -type mockPredecessorRetirementReportCache struct{ report protocol.RetirementReport } +type mockPredecessorRetirementReportCache struct { + report protocol.RetirementReport + err error +} func (m *mockPredecessorRetirementReportCache) AttestedRetirementReport(ocrtypes.ConfigDigest) ([]byte, error) { + if m.err != nil { + return nil, m.err + } return []byte("attested"), nil } func (m *mockPredecessorRetirementReportCache) CheckAttestedRetirementReport(ocrtypes.ConfigDigest, []byte) (protocol.RetirementReport, error) { diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index 2efb2a8..afa2f02 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -108,13 +108,20 @@ func (p *Plugin) Observation(_ context.Context, seqNr uint64, _ ocrtypes.Attribu if p.PredecessorConfigDigest != nil && state.lifeCycleStage == protocol.LifeCycleStageStaging { obs.AttestedPredecessorRetirement, err = p.PredecessorRetirementReportCache.AttestedRetirementReport(*p.PredecessorConfigDigest) if err != nil { - return nil, fmt.Errorf("error fetching attested retirement report from cache: %w", err) + // Best-effort: the state transition only needs one node to + // supply a valid retirement report, so omit it rather than + // failing the round. + obs.AttestedPredecessorRetirement = nil + p.Logger.Errorw("Failed to fetch attested retirement report from cache, omitting it from this observation", "stage", "Observation", "seqNr", seqNr, "err", err) } } obs.ShouldRetire, err = p.ShouldRetireCache.ShouldRetire(p.ConfigDigest) if err != nil { - return nil, fmt.Errorf("error fetching shouldRetire from cache: %w", err) + // Best-effort: retirement is decided by a quorum of votes, so + // abstain rather than failing the round. + obs.ShouldRetire = false + p.Logger.Errorw("Failed to fetch shouldRetire from cache, not voting to retire this round", "stage", "Observation", "seqNr", seqNr, "err", err) } p.voteOnChannels(&obs, state) diff --git a/llo/dev/v31/plugin_test.go b/llo/dev/v31/plugin_test.go index 1ee0f9f..225ddce 100644 --- a/llo/dev/v31/plugin_test.go +++ b/llo/dev/v31/plugin_test.go @@ -1376,3 +1376,32 @@ func Test_FullRound_RecoverFromUnverifiableChannel(t *testing.T) { require.NotContains(t, kvChannelDefs(t, kv), llotypes.ChannelID(2)) require.Contains(t, kvChannelDefs(t, kv), llotypes.ChannelID(1)) } + +// Test_Observation_RetirementCacheErrorsAreNotFatal guards the liveness fix: +// both retirement caches read from node-local, asynchronously populated state, +// so a transient failure must not fail the round. The vote is best-effort — a +// single node supplying a valid retirement report is enough, and retirement +// needs a quorum of votes — so the node abstains and carries on. +func Test_Observation_RetirementCacheErrorsAreNotFatal(t *testing.T) { + ctx := tests.Context(t) + p := testPlugin(t) + predecessor := ocrtypes.ConfigDigest{0xAB} + p.PredecessorConfigDigest = &predecessor + p.PredecessorRetirementReportCache = &mockPredecessorRetirementReportCache{err: errors.New("rpc failure")} + p.ShouldRetireCache = &mockShouldRetireCache{retire: true, err: errors.New("rpc failure")} + p.ChannelDefinitionCache = &mockChannelDefinitionCache{defs: llotypes.ChannelDefinitions{}} + kv := newMemKV() + + // Bootstrap -> staging, which is the only stage that reads the predecessor + // retirement report cache. + _, err := p.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, nil), ao(1, nil), ao(2, nil)}, kv, testBlobs) + require.NoError(t, err) + + obsBytes, err := p.Observation(ctx, 2, ocrtypes.AttributedQuery{}, kv, nil) + require.NoError(t, err, "a failing retirement cache must not halt the node") + obs, err := decodeObservation(ctx, obsBytes, testBlobs) + require.NoError(t, err) + + require.Empty(t, obs.AttestedPredecessorRetirement) + require.False(t, obs.ShouldRetire) +} diff --git a/llo/v30/plugin_observation.go b/llo/v30/plugin_observation.go index d92c24f..104d810 100644 --- a/llo/v30/plugin_observation.go +++ b/llo/v30/plugin_observation.go @@ -50,13 +50,19 @@ func (p *Plugin) observation(ctx context.Context, outctx ocr3types.OutcomeContex var err2 error obs.AttestedPredecessorRetirement, err2 = p.PredecessorRetirementReportCache.AttestedRetirementReport(*p.PredecessorConfigDigest) if err2 != nil { - return nil, fmt.Errorf("error fetching attested retirement report from cache: %w", err2) + // Best-effort: Outcome only needs one node to supply a valid + // retirement report, so omit it rather than failing the round. + obs.AttestedPredecessorRetirement = nil + p.Logger.Errorw("Failed to fetch attested retirement report from cache, omitting it from this observation", "stage", "Observation", "seqNr", outctx.SeqNr, "err", err2) } } obs.ShouldRetire, err = p.ShouldRetireCache.ShouldRetire(p.ConfigDigest) if err != nil { - return nil, fmt.Errorf("error fetching shouldRetire from cache: %w", err) + // Best-effort: retirement is decided by a quorum of votes, so + // abstain rather than failing the round. + obs.ShouldRetire = false + p.Logger.Errorw("Failed to fetch shouldRetire from cache, not voting to retire this round", "stage", "Observation", "seqNr", outctx.SeqNr, "err", err) } if obs.ShouldRetire && p.Config.VerboseLogging { p.Logger.Debugw("Voting to retire", "seqNr", outctx.SeqNr, "stage", "Observation") diff --git a/llo/v30/plugin_observation_test.go b/llo/v30/plugin_observation_test.go index 2ad9296..e2cd4d9 100644 --- a/llo/v30/plugin_observation_test.go +++ b/llo/v30/plugin_observation_test.go @@ -516,7 +516,7 @@ func testObservation(t *testing.T, outcomeCodec OutcomeCodec) { assert.Equal(t, []byte("foo"), decoded.AttestedPredecessorRetirement) }) - t.Run("if predecessor retirement report cache returns error, returns error", func(t *testing.T) { + t.Run("if predecessor retirement report cache returns error, omits it from the observation", func(t *testing.T) { prrc := &mockPredecessorRetirementReportCache{ err: errors.New("retirement report not found error"), } @@ -532,8 +532,32 @@ func testObservation(t *testing.T, outcomeCodec OutcomeCodec) { require.NoError(t, err) outctx := ocr3types.OutcomeContext{SeqNr: 2, PreviousOutcome: encodedPreviousOutcome} - _, err = p.Observation(context.Background(), outctx, query) - require.EqualError(t, err, "error fetching attested retirement report from cache: retirement report not found error") + obs, err := p.Observation(context.Background(), outctx, query) + require.NoError(t, err) + decoded, err := p.ObservationCodec.Decode(obs) + require.NoError(t, err) + + assert.Empty(t, decoded.AttestedPredecessorRetirement) + }) + t.Run("if shouldRetire cache returns error, does not vote to retire", func(t *testing.T) { + p.PredecessorRetirementReportCache = &mockPredecessorRetirementReportCache{} + p.ShouldRetireCache = &mockShouldRetireCache{shouldRetire: true, err: errors.New("should retire check failed")} + defer func() { p.ShouldRetireCache = &mockShouldRetireCache{} }() + previousOutcome := Outcome{ + LifeCycleStage: protocol.LifeCycleStageStaging, + ObservationTimestampNanoseconds: testStartTSNanos, + ChannelDefinitions: cdc.definitions, + } + encodedPreviousOutcome, err := p.OutcomeCodec.Encode(previousOutcome) + require.NoError(t, err) + + outctx := ocr3types.OutcomeContext{SeqNr: 2, PreviousOutcome: encodedPreviousOutcome} + obs, err := p.Observation(context.Background(), outctx, query) + require.NoError(t, err) + decoded, err := p.ObservationCodec.Decode(obs) + require.NoError(t, err) + + assert.False(t, decoded.ShouldRetire) }) t.Run("in production lifecycle stage, does not add attestedRetirementReport to observation", func(t *testing.T) { prrc := &mockPredecessorRetirementReportCache{ From 2ca024eb60df000e750822bb96c35b9781db9c04 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Thu, 17 Sep 2026 14:27:04 +0100 Subject: [PATCH 09/40] SPOR-0009, SPOR-0013 llo/dev/v31: decouple snapshot freshness from blob lifetime Split MaxSnapshotRounds, the local gate on how stale a node's own stream values may be when it references them, from BlobLifetimeRounds, the expiration hint that decides how long peers can fetch the blob. Require BlobFetchMarginRounds between them so a referenceable handle is always still fetchable. Derive the wall-clock snapshot age bound from the round period measured across consecutive Take calls rather than from MaxDurationObservation, which is unrelated to the round cadence and can reject every snapshot. Leave the bound inert until a period is measured, and escalate a sustained miss streak to an error log. Use the snapshot timestamp as the observation timestamp. --- llo/dev/v31/blobcompress.go | 5 +- llo/dev/v31/blobcompress_fuzz_test.go | 2 +- llo/dev/v31/blobpump.go | 212 +++++++++++++++++++------- llo/dev/v31/blobpump_test.go | 56 +++++-- llo/dev/v31/doc.go | 12 +- llo/dev/v31/factory.go | 53 ++++--- llo/dev/v31/flow_test.go | 1 + llo/dev/v31/plugin.go | 21 ++- llo/dev/v31/plugin_test.go | 13 +- llo/dev/v31/statetransition.go | 4 +- 10 files changed, 279 insertions(+), 100 deletions(-) diff --git a/llo/dev/v31/blobcompress.go b/llo/dev/v31/blobcompress.go index 49b1cf3..10f3724 100644 --- a/llo/dev/v31/blobcompress.go +++ b/llo/dev/v31/blobcompress.go @@ -91,9 +91,8 @@ func encodeBlobPayload(raw []byte) ([]byte, error) { out = append(out, blobCodecRaw) out = append(out, raw...) } - // The framed payload is what is broadcast, so it -- not the raw bytes -- is - // what has to fit the declared limit. A payload that compresses poorly can - // pass the check above and still land over it. + // The framed payload is what is broadcast, has to fit the declared limit. + // A payload that compresses poorly can pass the check above and still land over it. if len(out) > MaxBlobPayloadBytes { return nil, fmt.Errorf("framed blob payload too large: %d > %d bytes", len(out), MaxBlobPayloadBytes) } diff --git a/llo/dev/v31/blobcompress_fuzz_test.go b/llo/dev/v31/blobcompress_fuzz_test.go index cc36a5b..42f3c9b 100644 --- a/llo/dev/v31/blobcompress_fuzz_test.go +++ b/llo/dev/v31/blobcompress_fuzz_test.go @@ -6,7 +6,7 @@ import ( // FuzzDecodeBlobPayload feeds arbitrary bytes through the blob payload decoder. // Blob payloads are attacker-controlled, so the contract is that any input -// either returns bytes within the caller's budget or errors -- never a panic, +// either returns bytes within the caller's budget or errors. Never a panic, // and never more bytes than the budget allows (a zstd bomb). func FuzzDecodeBlobPayload(f *testing.F) { raw := []byte("stream values would go here") diff --git a/llo/dev/v31/blobpump.go b/llo/dev/v31/blobpump.go index 1b20a15..a85b8ac 100644 --- a/llo/dev/v31/blobpump.go +++ b/llo/dev/v31/blobpump.go @@ -21,24 +21,48 @@ import ( // Defaults for the blob pump. See PluginFactoryParams for the overrides. const ( - // DefaultBlobLifetimeRounds is how many sequence numbers past the one a - // pump cycle started for a broadcast blob is hinted to live. It must exceed - // 1: a snapshot gathered for seqNr N is normally consumed at N+1, and the - // blob is fetched by the other oracles during that later round. - DefaultBlobLifetimeRounds = 3 + // DefaultMaxSnapshotRounds bounds freshness and is enforced LOCALLY, by this + // node's Take: it decides how stale the parked stream values may be when this + // oracle references them in an observation. It has no effect on the blob + // transport. Shortening it makes this node discard its own snapshot sooner, + // it does not make the blob unfetchable for anyone. A snapshot gathered for + // seqNr N is usable through N+MaxSnapshotRounds-1; the default of 2 means + // consume at N+1, tolerating one skipped round. Tune against the report + // format's staleness budget. + DefaultMaxSnapshotRounds = 2 + // DefaultBlobLifetimeRounds bounds fetchability and is enforced REMOTELY, by + // libocr's blob transport: it is the expiration hint passed to BroadcastBlob, + // after which peers can no longer fetch the blob and the handle in an + // observation resolves to nothing. It does not bound staleness: a node's own + // freshness gate is DefaultMaxSnapshotRounds. Tune against network fetch + // latency and reaping lag, not against data freshness. + DefaultBlobLifetimeRounds = 4 + // BlobFetchMarginRounds couples the two: the number of rounds that must remain + // between the last seqNr at which this node may reference a snapshot (local, + // MaxSnapshotRounds) and the seqNr at which its blob expires for peers + // (remote, BlobLifetimeRounds). Guarantees a handle that is still locally + // usable is still remotely fetchable, with slack for slow or lagging peers. + BlobFetchMarginRounds = 3 // DefaultBlobObservationDurationMultiplier scales MaxDurationObservation // into the pump's per-cycle budget. The pump runs off the OCR critical // path, so it can afford to wait longer than a synchronous observation. DefaultBlobObservationDurationMultiplier = 2 - // DefaultBlobSnapshotAgeMultiplier scales MaxDurationObservation into the - // wall-clock age at which a parked snapshot is discarded. Generous on - // purpose: ordinary jitter should reuse the previous snapshot rather than - // discard it, since blob expiry already bounds staleness in rounds. - DefaultBlobSnapshotAgeMultiplier = 5 + // SnapshotAgeSlack scales the measured round period into the wall-clock age + // at which a parked snapshot is discarded. The bound is derived from the + // observed round period rather than from MaxDurationObservation, which is + // unrelated to the round cadence: a bound shorter than one round period + // would reject every snapshot and silently stop the node contributing + // stream values. Slack makes this a jitter guard, not a second freshness + // gate, since staleness in rounds is already bounded by MaxSnapshotRounds. + SnapshotAgeSlack = 2 // MaxBlobLifetimeRounds bounds BlobLifetimeRounds. The pump broadcasts // roughly one blob per round, so the unexpired-blob budget declared to // libocr grows with the lifetime; this keeps that budget sane. MaxBlobLifetimeRounds = 64 + // MissStreakLogThreshold is how many consecutive rounds may find no usable + // snapshot before the pump escalates from debug to error logging. A node + // that never contributes stream values is a silent failure otherwise. + MissStreakLogThreshold = 5 // BlobReapingMarginRounds is added to blobLifetimeRounds when deriving the // per-oracle unexpired-blob budget, covering blobs that are expired but not // yet reaped (reaping is asynchronous, on the order of tens of seconds). @@ -69,8 +93,12 @@ type blobSnapshot struct { observedAt time.Time // forSeqNr is the sequence number known when the cycle started. forSeqNr uint64 - // expiresAt is the blob expiration hint; the snapshot must not be used at - // or beyond this sequence number. + // usableBefore is the local freshness gate: this node must not reference the + // snapshot at or beyond this sequence number. Purely local, not on the wire. + usableBefore uint64 + // expiresAt is the expiration hint given to the blob transport: peers cannot + // fetch the blob at or beyond this sequence number. Always greater than + // usableBefore by at least BlobFetchMarginRounds. expiresAt uint64 // streamCount is the number of streams the cycle observed (for logging). streamCount int @@ -87,52 +115,54 @@ type blobSnapshot struct { // bounded by sequence number, so refreshing while no rounds are running would // produce snapshots that are already too old to use. type blobPump struct { - bbf ocr3_1types.BlobBroadcastFetcher - ds DataSource - lggr logger.Logger - configDigest ocrtypes.ConfigDigest - verboseLogging bool - observationTimeout time.Duration - maxSnapshotAge time.Duration - blobLifetimeRounds uint64 + blobPumpParams + lggr logger.Logger trigger chan struct{} ctx context.Context cancel context.CancelFunc wg sync.WaitGroup - inFlight atomic.Bool - misses atomic.Uint64 - cycles atomic.Uint64 + inFlight atomic.Bool + misses atomic.Uint64 + cycles atomic.Uint64 + missStreak atomic.Uint64 mu sync.Mutex input pumpInput ready *blobSnapshot + // lastTakeAt and roundPeriod estimate the round cadence from the interval + // between consecutive Take calls (one per round), which is the only signal + // the plugin has: deltaRound is not part of the reporting plugin config. + lastTakeAt time.Time + roundPeriod time.Duration +} + +// blobPumpParams is the pump's configuration, resolved by the factory. +type blobPumpParams struct { + bbf ocr3_1types.BlobBroadcastFetcher + ds DataSource + configDigest ocrtypes.ConfigDigest + verboseLogging bool + observationTimeout time.Duration + // maxSnapshotAge is the wall-clock jitter guard. Positive pins it to an + // explicit duration, zero derives it from the measured round period, and + // negative disables it, leaving maxSnapshotRounds as the only bound. + maxSnapshotAge time.Duration + // maxSnapshotRounds is the local freshness gate. See DefaultMaxSnapshotRounds. + maxSnapshotRounds uint64 + // blobLifetimeRounds is the remote fetchability bound. See DefaultBlobLifetimeRounds. + blobLifetimeRounds uint64 } -func newBlobPump( - bbf ocr3_1types.BlobBroadcastFetcher, - ds DataSource, - lggr logger.Logger, - configDigest ocrtypes.ConfigDigest, - verboseLogging bool, - observationTimeout time.Duration, - maxSnapshotAge time.Duration, - blobLifetimeRounds uint64, -) *blobPump { +func newBlobPump(lggr logger.Logger, params blobPumpParams) *blobPump { ctx, cancel := context.WithCancel(context.Background()) return &blobPump{ - bbf: bbf, - ds: ds, - lggr: logger.Sugared(lggr).Named("BlobPump"), - configDigest: configDigest, - verboseLogging: verboseLogging, - observationTimeout: observationTimeout, - maxSnapshotAge: maxSnapshotAge, - blobLifetimeRounds: blobLifetimeRounds, - trigger: make(chan struct{}, 1), - ctx: ctx, - cancel: cancel, + blobPumpParams: params, + lggr: logger.Sugared(lggr).Named("BlobPump"), + trigger: make(chan struct{}, 1), + ctx: ctx, + cancel: cancel, } } @@ -173,35 +203,99 @@ func (p *blobPump) SetInput(in pumpInput) { // value is the reason a snapshot was not returned, for logging. func (p *blobPump) Take(seqNr uint64) (*blobSnapshot, string) { if !p.enabled() { - p.misses.Add(1) + p.miss() return nil, "blob pump disabled" } + now := time.Now() + p.mu.Lock() snap := p.ready p.ready = nil + // Measure before resolving the limit, both under the same lock. On the very + // first Take there is no previous call to measure against, so roundPeriod is + // still zero and the derived age check is inert. That Take is also the one + // that kicks the first cycle, though, so nothing is parked yet and the round + // misses on "no snapshot parked" regardless. By the second Take, which is + // the first that can see a snapshot, the gap has been measured and the check + // is live. There is no round in which a snapshot exists and roundPeriod is + // still zero. + p.recordRoundLocked(now) + ageLimit := p.snapshotAgeLimitLocked() p.mu.Unlock() defer p.kick() switch { case snap == nil: - p.misses.Add(1) + p.miss() if p.inFlight.Load() { return nil, "cycle in flight" } return nil, "no snapshot parked" - case seqNr >= snap.expiresAt: - p.misses.Add(1) - return nil, fmt.Sprintf("blob expired (forSeqNr=%d expiresAt=%d)", snap.forSeqNr, snap.expiresAt) - case p.maxSnapshotAge > 0 && time.Since(snap.observedAt) > p.maxSnapshotAge: - p.misses.Add(1) - return nil, fmt.Sprintf("snapshot too old (age=%s max=%s)", time.Since(snap.observedAt), p.maxSnapshotAge) + case seqNr >= snap.usableBefore: + p.miss() + return nil, fmt.Sprintf("snapshot too stale (forSeqNr=%d usableBefore=%d expiresAt=%d)", snap.forSeqNr, snap.usableBefore, snap.expiresAt) + case ageLimit > 0 && now.Sub(snap.observedAt) > ageLimit: + p.miss() + return nil, fmt.Sprintf("snapshot too old (age=%s max=%s)", now.Sub(snap.observedAt), ageLimit) default: + p.missStreak.Store(0) return snap, "" } } +// miss records a round that found no usable snapshot. Sustained misses means +// this node is not contributing at all. Record misses and log when above MissStreakLogThreshold. +func (p *blobPump) miss() { + p.misses.Add(1) + if streak := p.missStreak.Add(1); streak >= MissStreakLogThreshold && streak%MissStreakLogThreshold == 0 { + p.lggr.Errorw("Blob pump has found no usable snapshot for consecutive rounds; this node is contributing no stream values", + "missStreak", streak, "misses", p.misses.Load(), "cycles", p.cycles.Load(), "maxSnapshotAge", p.maxSnapshotAge, "maxSnapshotRounds", p.maxSnapshotRounds) + } +} + +// recordRoundLocked folds the interval since the previous Take into the round +// period estimate. Take is called once per round, so consecutive calls measure +// the round cadence. +// +// Rounds that observe no streams skip Take entirely, so the next gap spans several +// round periods and overestimates; a stall inflates one gap badly, and estimation +// takes about four rounds to decay it. +// Both leave the age bound too generous for a while rather than too tight, +// which is the direction that cannot silently stop this node contributing. +func (p *blobPump) recordRoundLocked(now time.Time) { + if !p.lastTakeAt.IsZero() { + gap := now.Sub(p.lastTakeAt) + if p.roundPeriod == 0 { + p.roundPeriod = gap + } else { + p.roundPeriod = (3*p.roundPeriod + gap) / 4 + } + } + p.lastTakeAt = now +} + +// snapshotAgeLimitLocked resolves the wall-clock bound for this round. Zero +// means no bound at all, and is the fallback whenever the cadence is unknown: +// an unmeasured round period falls back to maxSnapshotRounds as the only +// staleness bound rather than to an invented duration. Rejecting every snapshot +// is the worse failure, because it stops the node contributing stream values +// while every round still succeeds, so it shows up as a counter and nothing +// else. +func (p *blobPump) snapshotAgeLimitLocked() time.Duration { + switch { + case p.maxSnapshotAge < 0: + return 0 + case p.maxSnapshotAge > 0: + return p.maxSnapshotAge + case p.roundPeriod > 0: + return time.Duration(p.maxSnapshotRounds) * p.roundPeriod * SnapshotAgeSlack + default: + return 0 + } +} + // Misses reports how many rounds found no usable snapshot. func (p *blobPump) Misses() uint64 { return p.misses.Load() } @@ -266,7 +360,7 @@ func (p *blobPump) cycle() { p.mu.Unlock() if p.verboseLogging { - p.lggr.Debugw("Blob pump parked snapshot", "seqNr", in.seqNr, "expiresAt", snap.expiresAt, "streams", snap.streamCount, "handleBytes", len(snap.handleBytes)) + p.lggr.Debugw("Blob pump parked snapshot", "seqNr", in.seqNr, "usableBefore", snap.usableBefore, "expiresAt", snap.expiresAt, "streams", snap.streamCount, "handleBytes", len(snap.handleBytes)) } } @@ -293,6 +387,7 @@ func (p *blobPump) observe(in pumpInput) (*blobSnapshot, error) { return nil, fmt.Errorf("no stream values observed for %d streams", len(in.streams)) } + usableBefore := in.seqNr + p.maxSnapshotRounds expiresAt := in.seqNr + p.blobLifetimeRounds handle, err := p.bbf.BroadcastBlob(ctx, payload, ocr3_1types.BlobExpirationHintSequenceNumber{SeqNr: expiresAt}) if err != nil { @@ -304,11 +399,12 @@ func (p *blobPump) observe(in pumpInput) (*blobSnapshot, error) { } return &blobSnapshot{ - handleBytes: handleBytes, - observedAt: observedAt, - forSeqNr: in.seqNr, - expiresAt: expiresAt, - streamCount: len(sv), + handleBytes: handleBytes, + observedAt: observedAt, + forSeqNr: in.seqNr, + usableBefore: usableBefore, + expiresAt: expiresAt, + streamCount: len(sv), }, nil } diff --git a/llo/dev/v31/blobpump_test.go b/llo/dev/v31/blobpump_test.go index e0830b4..4003b95 100644 --- a/llo/dev/v31/blobpump_test.go +++ b/llo/dev/v31/blobpump_test.go @@ -23,7 +23,16 @@ import ( func testPump(t *testing.T, ds DataSource, bbf ocr3_1types.BlobBroadcastFetcher, maxAge time.Duration) *blobPump { t.Helper() - p := newBlobPump(bbf, ds, logger.Test(t), ocrtypes.ConfigDigest{1}, true, tests.WaitTimeout(t), maxAge, DefaultBlobLifetimeRounds) + p := newBlobPump(logger.Test(t), blobPumpParams{ + bbf: bbf, + ds: ds, + configDigest: ocrtypes.ConfigDigest{1}, + verboseLogging: true, + observationTimeout: tests.WaitTimeout(t), + maxSnapshotAge: maxAge, + maxSnapshotRounds: DefaultMaxSnapshotRounds, + blobLifetimeRounds: DefaultBlobLifetimeRounds, + }) p.Start() t.Cleanup(p.Close) return p @@ -56,6 +65,7 @@ func Test_blobPump_TakeKicksNextCycle(t *testing.T) { snap, reason = p.Take(3) require.NotNil(t, snap, "reason: %s", reason) require.Equal(t, uint64(2), snap.forSeqNr) + require.Equal(t, uint64(2+DefaultMaxSnapshotRounds), snap.usableBefore) require.Equal(t, uint64(2+DefaultBlobLifetimeRounds), snap.expiresAt) require.NotEmpty(t, snap.handleBytes) require.Equal(t, uint64(1), p.Misses()) @@ -105,22 +115,25 @@ func Test_blobPump_TakeIsSingleUse(t *testing.T) { } func Test_blobPump_RejectsStaleSnapshots(t *testing.T) { - t.Run("expired by sequence number", func(t *testing.T) { + // The local freshness gate is usableBefore, which falls well short of the + // blob's own expiry: the snapshot stops being referenceable while the blob + // is still fetchable by peers. + t.Run("too stale by sequence number", func(t *testing.T) { p := testPump(t, mockDS(), newFakeBroadcaster(), time.Minute) p.mu.Lock() - p.ready = &blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now(), forSeqNr: 2, expiresAt: 5} + p.ready = &blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now(), forSeqNr: 2, usableBefore: 4, expiresAt: 6} p.mu.Unlock() - snap, reason := p.Take(5) + snap, reason := p.Take(4) require.Nil(t, snap) - require.Contains(t, reason, "blob expired") + require.Contains(t, reason, "too stale") require.Equal(t, uint64(1), p.Misses()) }) t.Run("expired by wall clock", func(t *testing.T) { p := testPump(t, mockDS(), newFakeBroadcaster(), time.Nanosecond) p.mu.Lock() - p.ready = &blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, expiresAt: 100} + p.ready = &blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100} p.mu.Unlock() snap, reason := p.Take(3) @@ -129,13 +142,38 @@ func Test_blobPump_RejectsStaleSnapshots(t *testing.T) { }) t.Run("age check disabled", func(t *testing.T) { - p := testPump(t, mockDS(), newFakeBroadcaster(), 0) + p := testPump(t, mockDS(), newFakeBroadcaster(), -1) p.mu.Lock() - p.ready = &blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, expiresAt: 100} + p.ready = &blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100} p.mu.Unlock() snap, _ := p.Take(3) - require.NotNil(t, snap, "with the age check disabled only blob expiry bounds staleness") + require.NotNil(t, snap, "with the age check disabled only maxSnapshotRounds bounds staleness") + }) + + // With no explicit age the bound is derived from the measured round period, + // so a pump that has seen no rounds yet must not reject on age: guessing a + // cadence would silently stop the node contributing stream values. + t.Run("derived age check is inert until a round period is measured", func(t *testing.T) { + p := testPump(t, mockDS(), newFakeBroadcaster(), 0) + p.mu.Lock() + p.ready = &blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100} + p.mu.Unlock() + + snap, reason := p.Take(3) + require.NotNil(t, snap, "reason: %s", reason) + }) + + t.Run("derived age check rejects once the round period is known", func(t *testing.T) { + p := testPump(t, mockDS(), newFakeBroadcaster(), 0) + p.mu.Lock() + p.roundPeriod = time.Millisecond + p.ready = &blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100} + p.mu.Unlock() + + snap, reason := p.Take(3) + require.Nil(t, snap) + require.Contains(t, reason, "too old") }) } diff --git a/llo/dev/v31/doc.go b/llo/dev/v31/doc.go index 19105a3..4a36464 100644 --- a/llo/dev/v31/doc.go +++ b/llo/dev/v31/doc.go @@ -94,9 +94,15 @@ // discards a snapshot, so the pump rate tracks the round rate without knowing // deltaRound, and cycles are serial so only one Observe is ever in flight. // -// A snapshot is therefore gathered one round before it is used, and its -// usability is bounded by the blob expiration hint (forSeqNr + -// BlobLifetimeRounds) plus a generous wall-clock age check. A round that finds +// A snapshot is therefore gathered one round before it is used. Two separate +// bounds apply to it. MaxSnapshotRounds is local: it decides how stale the +// values may be when this node references them (forSeqNr + MaxSnapshotRounds), +// and is what a report format's staleness budget should be tuned against. +// BlobLifetimeRounds is remote: it is the expiration hint given to the blob +// transport (forSeqNr + BlobLifetimeRounds), deciding how long peers can still +// fetch the blob, and sits BlobFetchMarginRounds beyond the last seqNr at which +// the handle can be referenced. A wall-clock age check derived from the +// measured round period guards against jitter on top. A round that finds // nothing usable — cold start, a failed cycle, or a stale snapshot — emits an // observation with no stream values. That is not a halt: quorum counts // observations, not values. The cost lands in aggregation, which needs >F values diff --git a/llo/dev/v31/factory.go b/llo/dev/v31/factory.go index fbb0251..bfd219d 100644 --- a/llo/dev/v31/factory.go +++ b/llo/dev/v31/factory.go @@ -35,18 +35,26 @@ type PluginFactoryParams struct { ReportTelemetryCh chan<- *protocol.LLOReportTelemetry // DonID is optional and used only for telemetry and logging. DonID uint32 - // BlobLifetimeRounds overrides DefaultBlobLifetimeRounds if non-zero. Must - // be >1, since a snapshot gathered for one sequence number is consumed by a - // later one. + // MaxSnapshotRounds overrides DefaultMaxSnapshotRounds if non-zero. Bounds + // how stale this node's own stream values may be when it references them, + // and nothing else; it does not affect what peers can fetch. Must be >0, + // since a snapshot gathered for one sequence number is consumed by a later + // one, and must leave BlobFetchMarginRounds below BlobLifetimeRounds. + MaxSnapshotRounds uint64 + // BlobLifetimeRounds overrides DefaultBlobLifetimeRounds if non-zero. Bounds + // how long peers can still fetch a broadcast blob, and nothing else; it does + // not bound staleness, MaxSnapshotRounds does. BlobLifetimeRounds uint64 // MaxDurationBlobObservation overrides the pump's per-cycle observation // budget (default: DefaultBlobObservationDurationMultiplier * // cfg.MaxDurationObservation). MaxDurationBlobObservation time.Duration - // MaxBlobSnapshotAge overrides the wall-clock age at which a parked snapshot - // is discarded (default: DefaultBlobSnapshotAgeMultiplier * - // cfg.MaxDurationObservation). A negative value disables the age check, - // leaving blob expiry as the only staleness bound. + // MaxBlobSnapshotAge pins the wall-clock age at which a parked snapshot is + // discarded. Left at zero the pump derives it from the round period it + // measures, which is the only safe default: any bound derived from + // MaxDurationObservation is unrelated to the round cadence and can reject + // every snapshot. A negative value disables the check, leaving + // MaxSnapshotRounds as the only staleness bound. MaxBlobSnapshotAge time.Duration } @@ -74,24 +82,26 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re // Initialize the memory ballast protocol.InitMemoryBallast() + maxSnapshotRounds := f.MaxSnapshotRounds + if maxSnapshotRounds == 0 { + maxSnapshotRounds = DefaultMaxSnapshotRounds + } blobLifetimeRounds := f.BlobLifetimeRounds - if blobLifetimeRounds <= 1 { + if blobLifetimeRounds == 0 { blobLifetimeRounds = DefaultBlobLifetimeRounds } if blobLifetimeRounds > MaxBlobLifetimeRounds { return nil, nil, fmt.Errorf("BlobLifetimeRounds (%d) exceeds MaxBlobLifetimeRounds (%d)", blobLifetimeRounds, MaxBlobLifetimeRounds) } + // A snapshot is last referenced at forSeqNr+maxSnapshotRounds-1 and its blob + // expires at forSeqNr+blobLifetimeRounds, so this is the fetch margin. + if blobLifetimeRounds+1 < maxSnapshotRounds+BlobFetchMarginRounds { + return nil, nil, fmt.Errorf("BlobLifetimeRounds (%d) leaves less than %d rounds of fetch margin past MaxSnapshotRounds (%d)", blobLifetimeRounds, BlobFetchMarginRounds, maxSnapshotRounds) + } blobObservationTimeout := f.MaxDurationBlobObservation if blobObservationTimeout <= 0 { blobObservationTimeout = DefaultBlobObservationDurationMultiplier * cfg.MaxDurationObservation } - maxSnapshotAge := f.MaxBlobSnapshotAge - switch { - case maxSnapshotAge == 0: - maxSnapshotAge = DefaultBlobSnapshotAgeMultiplier * cfg.MaxDurationObservation - case maxSnapshotAge < 0: - maxSnapshotAge = 0 // disabled; blob expiry still bounds staleness - } p := &Plugin{ Config: f.Config, @@ -118,12 +128,21 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re p.ChannelCache = protocol.NewChannelCache() // Setup the blobpump - p.pump = newBlobPump(bbf, f.DataSource, l, cfg.ConfigDigest, f.Config.VerboseLogging, blobObservationTimeout, maxSnapshotAge, blobLifetimeRounds) + p.pump = newBlobPump(l, blobPumpParams{ + bbf: bbf, + ds: f.DataSource, + configDigest: cfg.ConfigDigest, + verboseLogging: f.Config.VerboseLogging, + observationTimeout: blobObservationTimeout, + maxSnapshotAge: f.MaxBlobSnapshotAge, + maxSnapshotRounds: maxSnapshotRounds, + blobLifetimeRounds: blobLifetimeRounds, + }) p.pump.Start() unexpiredBlobCount := perOracleUnexpiredBlobCount(blobLifetimeRounds) // Declared limits. Each is the libocr maximum, which is only honest if the - // plugin's own admission rules keep what it produces underneath it -- libocr + // plugin's own admission rules keep what it produces underneath it: libocr // rejects an oversized message or write set, which fails the round for every // oracle. The derivations, and which of them are currently enforced, are: // diff --git a/llo/dev/v31/flow_test.go b/llo/dev/v31/flow_test.go index 4fba1ef..f32bdb8 100644 --- a/llo/dev/v31/flow_test.go +++ b/llo/dev/v31/flow_test.go @@ -173,6 +173,7 @@ func Test_Factory_NewReportingPlugin(t *testing.T) { require.Equal(t, 1, pl.F) require.NotNil(t, pl.ChannelCache) require.NotNil(t, pl.pump) + require.Equal(t, uint64(DefaultMaxSnapshotRounds), pl.pump.maxSnapshotRounds) require.Equal(t, uint64(DefaultBlobLifetimeRounds), pl.pump.blobLifetimeRounds) require.NoError(t, pl.Close()) } diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index afa2f02..8fb0d6a 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -136,20 +136,31 @@ func (p *Plugin) Observation(_ context.Context, seqNr uint64, _ ocrtypes.Attribu // affected streams that round's aggregate (which needs >F values), not the // round itself. var handles [][]byte + var snap *blobSnapshot // A nil pump means stream values were never wired up (or the plugin was built // without the factory); rounds that observe no streams have nothing for the // pump to gather, so neither publishes input nor consumes a snapshot. if p.pump != nil && len(streams) > 0 { p.pump.SetInput(pumpInput{streams: streams, seqNr: seqNr, lifeCycleStage: state.lifeCycleStage}) - if snap, reason := p.pump.Take(seqNr); snap != nil { + var reason string + if snap, reason = p.pump.Take(seqNr); snap != nil { handles = append(handles, snap.handleBytes) } else { p.Logger.Debugw("No usable stream-value snapshot for this round", "stage", "Observation", "seqNr", seqNr, "reason", reason, "misses", p.pump.Misses(), "cycles", p.pump.Cycles()) } } - obsTSNanos := time.Now().UnixNano() + // Timestamp the data, not the round. The pump gathers stream values off the + // critical path, so they were read before this round started; stamping + // time.Now() would have the report claim the values are newer than they are + // for every aggregate that does not carry its own timestamp. A round with no + // snapshot carries only votes, for which the round time is the right stamp. + obsTime := time.Now() + if snap != nil { + obsTime = snap.observedAt + } + obsTSNanos := obsTime.UnixNano() if obsTSNanos < 0 { return nil, fmt.Errorf("negative observation timestamps are not supported, got: %d", obsTSNanos) } @@ -157,8 +168,8 @@ func (p *Plugin) Observation(_ context.Context, seqNr uint64, _ ocrtypes.Attribu // Advertised every round, including when retired: a statement about this // binary, not about the round or about what this node wants admitted. - // voteOnChannels above is deliberately unaware of p.ReportCodecs -- whether - // a channel can be encoded DON-wide is decided from these advertisements in + // voteOnChannels above is deliberately unaware of p.ReportCodecs: whether a + // channel can be encoded DON-wide is decided from these advertisements in // the state transition, not locally per voter. obs.SupportedReportFormats = supportedReportFormats(p.ReportCodecs) @@ -310,7 +321,7 @@ func (p *Plugin) ValidateObservation(ctx context.Context, seqNr uint64, _ ocrtyp // rejecting a peer observation is not a halt, and these checks are the only // thing standing between a proposer and a malformed committed definition. // Under version skew a stricter build rejects observations that carry - // updates from a staler one, which costs update-voting liveness only -- an + // updates from a staler one, which costs update-voting liveness only. An // observation that votes no update has nothing to verify here, so rounds // themselves are unaffected. if err := protocol.VerifyChannelDefinitions(p.ReportCodecs, defsForVerify); err != nil { diff --git a/llo/dev/v31/plugin_test.go b/llo/dev/v31/plugin_test.go index 225ddce..383c07c 100644 --- a/llo/dev/v31/plugin_test.go +++ b/llo/dev/v31/plugin_test.go @@ -123,7 +123,16 @@ func testPlugin(t *testing.T) *Plugin { func attachPump(t *testing.T, p *Plugin, ds DataSource, bbf ocr3_1types.BlobBroadcastFetcher) *blobPump { t.Helper() p.DataSource = ds - p.pump = newBlobPump(bbf, ds, logger.Test(t), p.ConfigDigest, true, tests.WaitTimeout(t), time.Minute, DefaultBlobLifetimeRounds) + p.pump = newBlobPump(logger.Test(t), blobPumpParams{ + bbf: bbf, + ds: ds, + configDigest: p.ConfigDigest, + verboseLogging: true, + observationTimeout: tests.WaitTimeout(t), + maxSnapshotAge: time.Minute, + maxSnapshotRounds: DefaultMaxSnapshotRounds, + blobLifetimeRounds: DefaultBlobLifetimeRounds, + }) p.pump.Start() t.Cleanup(func() { require.NoError(t, p.Close()) }) return p.pump @@ -1262,7 +1271,7 @@ func (d *recordingDataSource) streams() []llotypes.StreamID { // Test_Observation_UnverifiableCommittedChannelIsNotFatal covers the // version-skew case: a channel committed under an older build fails this -// build's codec.Verify. The node must not halt -- it keeps observing and, +// build's codec.Verify. The node must not halt. It keeps observing and, // crucially, still votes the offending channel out, which is the only way the // DON recovers. func Test_Observation_UnverifiableCommittedChannelIsNotFatal(t *testing.T) { diff --git a/llo/dev/v31/statetransition.go b/llo/dev/v31/statetransition.go index d2f78a1..204a770 100644 --- a/llo/dev/v31/statetransition.go +++ b/llo/dev/v31/statetransition.go @@ -364,8 +364,8 @@ func applyChannelVotes( // reclaimed. // // The agreed value of each pair is also recorded into stream history for the -// pairs that require it. History records what the round actually agreed on -- -// the same value written into StreamAggregates -- so a window is always a series +// pairs that require it. History records what the round actually agreed on, +// the same value written into StreamAggregates, so a window is always a series // of values that reached consensus. A pair with no aggregate this round // (aggregation failed, stream absent) contributes nothing: a gap in the series // is honest, whereas repeating the previous value would silently weight it From 4635b16a95824d461e613477ed1ea30e145c11e5 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Thu, 17 Sep 2026 19:16:31 +0100 Subject: [PATCH 10/40] SPOR-0002 llo/v31: configure the per-stream contribution floor explicitly --- llo/dev/v31/factory.go | 15 ++++- llo/dev/v31/flow_test.go | 38 ++++++++++- llo/dev/v31/plugin.go | 13 ++++ llo/dev/v31/plugin_test.go | 60 ++++++++++++++++++ llo/dev/v31/statetransition.go | 2 +- llo/dev/v31/telemetry.go | 5 +- llo/protocol/aggregators.go | 42 ++++++------- llo/protocol/aggregators_test.go | 84 +++++++++++++------------ llo/protocol/llo_offchain_config.pb.go | 29 +++++++-- llo/protocol/llo_offchain_config.proto | 7 +++ llo/protocol/llo_plugin_telemetry.pb.go | 20 ++++-- llo/protocol/llo_plugin_telemetry.proto | 4 ++ llo/protocol/offchain_config.go | 19 ++++++ llo/protocol/offchain_config_test.go | 44 +++++++++++++ llo/v30/plugin_outcome.go | 12 +++- 15 files changed, 315 insertions(+), 79 deletions(-) diff --git a/llo/dev/v31/factory.go b/llo/dev/v31/factory.go index bfd219d..d8c394c 100644 --- a/llo/dev/v31/factory.go +++ b/llo/dev/v31/factory.go @@ -2,6 +2,7 @@ package llo import ( "context" + "errors" "fmt" "time" @@ -77,7 +78,7 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re } l := logger.Sugared(f.Logger).With("lloProtocolVersion", offchainConfig.ProtocolVersion, "configDigest", cfg.ConfigDigest, "lloOCRVersion", "3.1") - l.Infow("llo/dev/v31.NewReportingPlugin", "onchainConfig", onchainConfig, "offchainConfig", offchainConfig) + l.Infow("llo/dev/v31.NewReportingPlugin", "onchainConfig", onchainConfig, "offchainConfig", offchainConfig, "f", cfg.F, "n", cfg.N) // Initialize the memory ballast protocol.InitMemoryBallast() @@ -98,6 +99,17 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re if blobLifetimeRounds+1 < maxSnapshotRounds+BlobFetchMarginRounds { return nil, nil, fmt.Errorf("BlobLifetimeRounds (%d) leaves less than %d rounds of fetch margin past MaxSnapshotRounds (%d)", blobLifetimeRounds, BlobFetchMarginRounds, maxSnapshotRounds) } + // The contribution floor is a replicated state transition parameter, it + // must come from the offchainConfig and set explicitly. + if offchainConfig.AggregationFaultTolerance == nil { + return nil, nil, errors.New("NewReportingPlugin: offchain config must set aggregationFaultTolerance explicitly") + } + aggregationFaultTolerance := int(*offchainConfig.AggregationFaultTolerance) + if aggregationFaultTolerance > cfg.F { + return nil, nil, fmt.Errorf("aggregationFaultTolerance (%d) must not exceed consensus F (%d): a floor of %d contributions can never be met from %d observations", + aggregationFaultTolerance, cfg.F, 2*aggregationFaultTolerance+1, 2*cfg.F+1) + } + blobObservationTimeout := f.MaxDurationBlobObservation if blobObservationTimeout <= 0 { blobObservationTimeout = DefaultBlobObservationDurationMultiplier * cfg.MaxDurationObservation @@ -121,6 +133,7 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re ReportTelemetryCh: f.ReportTelemetryCh, ProtocolVersion: offchainConfig.ProtocolVersion, DefaultMinReportIntervalNanoseconds: offchainConfig.DefaultMinReportIntervalNanoseconds, + AggregationFaultTolerance: aggregationFaultTolerance, } // Definitions and the opts decoded from them are cached together, as one diff --git a/llo/dev/v31/flow_test.go b/llo/dev/v31/flow_test.go index f32bdb8..27ac860 100644 --- a/llo/dev/v31/flow_test.go +++ b/llo/dev/v31/flow_test.go @@ -154,13 +154,47 @@ func addChannelRound(t *testing.T, ts uint64, cid llotypes.ChannelID, cd llotype // --- tests --- +// mustEncodeOffchainConfig encodes a v0 offchain config with an explicit +// aggregation fault tolerance, which LLO v31 requires. +func mustEncodeOffchainConfig(t *testing.T, aggregationFaultTolerance uint32) []byte { + t.Helper() + b, err := protocol.OffchainConfig{AggregationFaultTolerance: &aggregationFaultTolerance}.Encode() + require.NoError(t, err) + return b +} + +func Test_Factory_NewReportingPlugin_aggregationFaultTolerance(t *testing.T) { + ctx := tests.Context(t) + f := NewPluginFactory(PluginFactoryParams{ + OnchainConfigCodec: mockOnchainConfigCodec{}, + Logger: logger.Test(t), + }) + + t.Run("refuses to start when unset", func(t *testing.T) { + _, _, err := f.NewReportingPlugin(ctx, ocr3types.ReportingPluginConfig{N: 4, F: 1, ConfigDigest: ocrtypes.ConfigDigest{9}}, nil) + require.EqualError(t, err, "NewReportingPlugin: offchain config must set aggregationFaultTolerance explicitly") + }) + t.Run("refuses to start when it exceeds consensus F", func(t *testing.T) { + _, _, err := f.NewReportingPlugin(ctx, ocr3types.ReportingPluginConfig{N: 4, F: 1, ConfigDigest: ocrtypes.ConfigDigest{9}, OffchainConfig: mustEncodeOffchainConfig(t, 2)}, nil) + require.EqualError(t, err, "aggregationFaultTolerance (2) must not exceed consensus F (1): a floor of 5 contributions can never be met from 3 observations") + }) + t.Run("accepts zero", func(t *testing.T) { + p, _, err := f.NewReportingPlugin(ctx, ocr3types.ReportingPluginConfig{N: 4, F: 1, ConfigDigest: ocrtypes.ConfigDigest{9}, OffchainConfig: mustEncodeOffchainConfig(t, 0)}, nil) + require.NoError(t, err) + pl, ok := p.(*Plugin) + require.True(t, ok) + require.Equal(t, 1, pl.minContributions()) + require.NoError(t, pl.Close()) + }) +} + func Test_Factory_NewReportingPlugin(t *testing.T) { ctx := tests.Context(t) f := NewPluginFactory(PluginFactoryParams{ OnchainConfigCodec: mockOnchainConfigCodec{}, Logger: logger.Test(t), }) - p, info, err := f.NewReportingPlugin(ctx, ocr3types.ReportingPluginConfig{N: 4, F: 1, ConfigDigest: ocrtypes.ConfigDigest{9}}, nil) + p, info, err := f.NewReportingPlugin(ctx, ocr3types.ReportingPluginConfig{N: 4, F: 1, ConfigDigest: ocrtypes.ConfigDigest{9}, OffchainConfig: mustEncodeOffchainConfig(t, 1)}, nil) require.NoError(t, err) info1, ok := info.(interface{ Validate() error }) @@ -171,6 +205,8 @@ func Test_Factory_NewReportingPlugin(t *testing.T) { require.True(t, ok) require.Equal(t, 4, pl.N) require.Equal(t, 1, pl.F) + require.Equal(t, 1, pl.AggregationFaultTolerance) + require.Equal(t, 3, pl.minContributions()) require.NotNil(t, pl.ChannelCache) require.NotNil(t, pl.pump) require.Equal(t, uint64(DefaultMaxSnapshotRounds), pl.pump.maxSnapshotRounds) diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index 8fb0d6a..9a274ec 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -62,8 +62,21 @@ type Plugin struct { // From offchain config ProtocolVersion uint32 DefaultMinReportIntervalNanoseconds uint64 + // AggregationFaultTolerance is how many Byzantine contributors per-stream + // aggregation tolerates. It sets the contribution floor, see + // minContributions. + AggregationFaultTolerance int } +// minContributions is the contribution floor: the fewest contributions a stream +// aggregate may be built from. 2*AggregationFaultTolerance+1 keeps the result +// inside the honest value range when up to AggregationFaultTolerance +// contributors are Byzantine. +// +// Distinct from the consensus quorum, which counts attributed observations, not +// the per-stream contributions inside them. +func (p *Plugin) minContributions() int { return 2*p.AggregationFaultTolerance + 1 } + // Query is empty: LLO oracles do not coordinate on what to observe. func (p *Plugin) Query(ctx context.Context, seqNr uint64, _ ocr3_1types.KeyValueStateReader, _ ocr3_1types.BlobBroadcastFetcher) (ocrtypes.Query, error) { return nil, nil diff --git a/llo/dev/v31/plugin_test.go b/llo/dev/v31/plugin_test.go index 383c07c..d3744fb 100644 --- a/llo/dev/v31/plugin_test.go +++ b/llo/dev/v31/plugin_test.go @@ -115,6 +115,8 @@ func testPlugin(t *testing.T) *Plugin { ChannelCache: protocol.NewChannelCache(), ProtocolVersion: 0, DefaultMinReportIntervalNanoseconds: 0, + // Floor of 3 contributions, which is what N=4, F=1 can sustain. + AggregationFaultTolerance: 1, } } @@ -344,6 +346,64 @@ func Test_FullRound_AddChannelThenReport(t *testing.T) { assert.Equal(t, llotypes.ReportFormatJSON, reports4[0].ReportWithInfo.Info.ReportFormat) } +func Test_StateTransition_ContributionFloor(t *testing.T) { + ctx := tests.Context(t) + p := testPlugin(t) // F=1, AggregationFaultTolerance=1, floor 3 + require.Equal(t, 3, p.minContributions()) + kv := newMemKV() + + channelDef := llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatJSON, + Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}, + } + _, err := p.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, nil), ao(1, nil), ao(2, nil)}, kv, testBlobs) + require.NoError(t, err) + addAOs := []ocrtypes.AttributedObservation{} + for i := 0; i < 4; i++ { + addAOs = append(addAOs, ao(i, mustEncodeObs(t, Observation{ + UnixTimestampNanoseconds: 1_000, + UpdateChannelDefinitions: llotypes.ChannelDefinitions{1: channelDef}, + }))) + } + _, err = p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, addAOs, kv, testBlobs) + require.NoError(t, err) + + // contributors oracles carry stream 100; the rest take part in the round + // without contributing a value for it, which is exactly the population + // collapse the floor guards against. + valObs := func(ts uint64, contributors int) []ocrtypes.AttributedObservation { + aos := []ocrtypes.AttributedObservation{} + for i := 0; i < 4; i++ { + obs := Observation{UnixTimestampNanoseconds: ts} + if i < contributors { + obs.StreamValues = protocol.StreamValues{100: protocol.ToDecimal(decimal.NewFromInt(42))} + } + aos = append(aos, ao(i, mustEncodeObs(t, obs))) + } + return aos + } + + // Three contributions meet the floor, so the stream aggregates. + prec, err := p.StateTransition(ctx, 3, ocrtypes.AttributedQuery{}, valObs(2_000, 3), kv, testBlobs) + require.NoError(t, err) + out, err := decodePrecursor(prec) + require.NoError(t, err) + require.Contains(t, out.StreamAggregates, llotypes.StreamID(100)) + + // Two do not: no aggregate, so no report. The channel itself stays + // reportable, since nil observed stream values are permitted unless the + // definition sets DisableNilStreamValues; the report is then dropped at + // encode time, exactly as for any other missing observed value. + prec, err = p.StateTransition(ctx, 4, ocrtypes.AttributedQuery{}, valObs(3_000, 2), kv, testBlobs) + require.NoError(t, err) + out, err = decodePrecursor(prec) + require.NoError(t, err) + require.NotContains(t, out.StreamAggregates, llotypes.StreamID(100)) + reports, err := p.Reports(ctx, 4, prec) + require.NoError(t, err) + require.Empty(t, reports) +} + func Test_Precursor_RoundTrip_And_Determinism(t *testing.T) { p := precursor{ LifeCycleStage: protocol.LifeCycleStageProduction, diff --git a/llo/dev/v31/statetransition.go b/llo/dev/v31/statetransition.go index 204a770..aeac9ff 100644 --- a/llo/dev/v31/statetransition.go +++ b/llo/dev/v31/statetransition.go @@ -422,7 +422,7 @@ func (p *Plugin) aggregate( } continue } - result, aerr := aggF(streamObservations[sid], p.F) + result, aerr := aggF(streamObservations[sid], p.minContributions()) switch v := result.(type) { case *protocol.TimestampedStreamValue: diff --git a/llo/dev/v31/telemetry.go b/llo/dev/v31/telemetry.go index 02b2466..18fb976 100644 --- a/llo/dev/v31/telemetry.go +++ b/llo/dev/v31/telemetry.go @@ -18,7 +18,7 @@ func (p *Plugin) captureOutcomeTelemetry(out precursor, seqNr uint64) { if p.OutcomeTelemetryCh == nil { return } - ot, err := makeOutcomeTelemetry(out, p.ConfigDigest, seqNr, p.DonID) + ot, err := makeOutcomeTelemetry(out, p.ConfigDigest, seqNr, p.DonID, p.minContributions()) if err != nil { p.Logger.Warnw("Error making outcome telemetry", "err", err) return @@ -30,7 +30,7 @@ func (p *Plugin) captureOutcomeTelemetry(out precursor, seqNr uint64) { } } -func makeOutcomeTelemetry(out precursor, configDigest ocrtypes.ConfigDigest, seqNr uint64, donID uint32) (*protocol.LLOOutcomeTelemetry, error) { +func makeOutcomeTelemetry(out precursor, configDigest ocrtypes.ConfigDigest, seqNr uint64, donID uint32, minContributions int) (*protocol.LLOOutcomeTelemetry, error) { ot := &protocol.LLOOutcomeTelemetry{ LifeCycleStage: string(out.LifeCycleStage), ObservationTimestampNanoseconds: out.ObservationTimestampNanoseconds, @@ -40,6 +40,7 @@ func makeOutcomeTelemetry(out precursor, configDigest ocrtypes.ConfigDigest, seq SeqNr: seqNr, ConfigDigest: configDigest[:], DonId: donID, + MinContributions: uint32(minContributions), } for id, cd := range out.ChannelDefinitions { ot.ChannelDefinitions[id] = protocol.ChannelDefinitionToProto(cd) diff --git a/llo/protocol/aggregators.go b/llo/protocol/aggregators.go index a94c3a2..a6a7060 100644 --- a/llo/protocol/aggregators.go +++ b/llo/protocol/aggregators.go @@ -11,7 +11,9 @@ import ( llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" ) -type AggregatorFunc func(values []StreamValue, f int) (StreamValue, error) +// AggregatorFunc aggregates contributions, requiring at least +// minContributions usable values of the selected type. +type AggregatorFunc func(contributions []StreamValue, minContributions int) (StreamValue, error) func GetAggregatorFunc(a llotypes.Aggregator) AggregatorFunc { switch a { @@ -25,7 +27,8 @@ func GetAggregatorFunc(a llotypes.Aggregator) AggregatorFunc { return nil } } -func MedianAggregator(values []StreamValue, f int) (StreamValue, error) { + +func MedianAggregator(values []StreamValue, minContributions int) (StreamValue, error) { typ, typValues := mostCommonType(values) switch typ { @@ -49,7 +52,7 @@ func MedianAggregator(values []StreamValue, f int) (StreamValue, error) { timestamps[i] = v.ObservedAtNanoseconds } - medianValue, err := MedianAggregator(svalues, f) + medianValue, err := MedianAggregator(svalues, minContributions) if err != nil { return nil, err } @@ -71,12 +74,10 @@ func MedianAggregator(values []StreamValue, f int) (StreamValue, error) { continue } } - if len(observations) <= f { - // In the worst case, we have 2f+1 observations, of which up to f - // are allowed to be invalid/missing. If we have less than f+1 - // usable observations, we cannot securely generate a median at - // all. - return nil, fmt.Errorf("not enough observations to calculate median, expected at least f+1, got %d", len(observations)) + if len(observations) < minContributions { + // Below the contribution floor the median is not guaranteed to sit + // inside the honest value range, so refuse to produce one. + return nil, fmt.Errorf("not enough contributions to calculate median: got %d, need %d", len(observations), minContributions) } sort.Slice(observations, func(i, j int) bool { return observations[i].Cmp(observations[j]) < 0 }) // We use a "rank-k" median here, instead one could average in case of @@ -91,9 +92,10 @@ func MedianAggregator(values []StreamValue, f int) (StreamValue, error) { // ModeAggregator works on arbitrary StreamValue types // It picks the most common value -// There must be at least f+1 observations in agreement in order to produce a value -// nil observations are ignored -func ModeAggregator(values []StreamValue, f int) (StreamValue, error) { +// There must be at least minContributions contributions in agreement in order +// to produce a value +// nil contributions are ignored +func ModeAggregator(values []StreamValue, minContributions int) (StreamValue, error) { largestBucketType, largestBucket := mostCommonType(values) // find the most common value in the bucket @@ -119,8 +121,8 @@ func ModeAggregator(values []StreamValue, f int) (StreamValue, error) { } } - if modeCount < f+1 { - return nil, fmt.Errorf("not enough observations in agreement to calculate mode, expected at least f+1, most common value had %d", modeCount) + if modeCount < minContributions { + return nil, fmt.Errorf("not enough contributions in agreement to calculate mode: most common value had %d, need %d", modeCount, minContributions) } if len(modeSerialized) == 0 { return nil, nil @@ -153,7 +155,7 @@ func mostCommonType(values []StreamValue) (LLOStreamValue_Type, []StreamValue) { return mostCommonType, largestBucket } -func QuoteAggregator(values []StreamValue, f int) (StreamValue, error) { +func QuoteAggregator(values []StreamValue, minContributions int) (StreamValue, error) { var observations []*Quote for _, value := range values { if v, ok := value.(*Quote); !ok { @@ -164,12 +166,10 @@ func QuoteAggregator(values []StreamValue, f int) (StreamValue, error) { } // Exclude Quotes that violate bid<=mid<=ask } - if len(observations) <= f { - // In the worst case, we have 2f+1 observations, of which up to f - // are allowed to be invalid/missing. If we have less than f+1 - // usable observations, we cannot securely generate a median at - // all. - return nil, fmt.Errorf("not enough valid observations to aggregate quote, expected at least f+1, got %d", len(observations)) + if len(observations) < minContributions { + // Below the contribution floor the rank medians below are not + // guaranteed to sit inside the honest value range. + return nil, fmt.Errorf("not enough valid contributions to aggregate quote: got %d, need %d", len(observations), minContributions) } // Calculate "rank-k" median for benchmark, bid and ask separately. // This is guaranteed not to return values that violate bid<=mid<=ask due diff --git a/llo/protocol/aggregators_test.go b/llo/protocol/aggregators_test.go index 270212d..623d3fd 100644 --- a/llo/protocol/aggregators_test.go +++ b/llo/protocol/aggregators_test.go @@ -18,17 +18,19 @@ func Test_MedianAggregator(t *testing.T) { ToDecimal(decimal.NewFromFloat(5.5)), } - f := 1 + // Contribution floor, expressed directly: v3.0 passes F+1, v3.1 passes + // 2*AggregationFaultTolerance+1. + minContributions := 2 t.Run("returns median with even number of values", func(t *testing.T) { - sv, err := MedianAggregator(values, f) + sv, err := MedianAggregator(values, minContributions) require.NoError(t, err) assert.IsType(t, &Decimal{}, sv) assert.Equal(t, "4.4", sv.(*Decimal).String()) }) t.Run("returns higher value with odd number of values", func(t *testing.T) { - sv, err := MedianAggregator(values[:5], f) + sv, err := MedianAggregator(values[:5], minContributions) require.NoError(t, err) assert.IsType(t, &Decimal{}, sv) assert.Equal(t, "3.3", sv.(*Decimal).String()) @@ -44,7 +46,7 @@ func Test_MedianAggregator(t *testing.T) { ToDecimal(decimal.NewFromFloat(5.5)), } - sv, err := MedianAggregator(mixedValues, f) + sv, err := MedianAggregator(mixedValues, minContributions) require.NoError(t, err) assert.IsType(t, &Decimal{}, sv) assert.Equal(t, "4.4", sv.(*Decimal).String()) @@ -60,31 +62,31 @@ func Test_MedianAggregator(t *testing.T) { &TimestampedStreamValue{ObservedAtNanoseconds: 106, StreamValue: ToDecimal(decimal.NewFromFloat(5.6))}, } - sv, err := MedianAggregator(mixedValues, f) + sv, err := MedianAggregator(mixedValues, minContributions) require.NoError(t, err) assert.IsType(t, &TimestampedStreamValue{}, sv) assert.Equal(t, "5.6", sv.(*TimestampedStreamValue).StreamValue.(*Decimal).String()) assert.Equal(t, uint64(102), sv.(*TimestampedStreamValue).ObservedAtNanoseconds) }) - t.Run("fails with fewer than f+1 values", func(t *testing.T) { - _, err := MedianAggregator(values[:2], 3) - require.EqualError(t, err, "not enough observations to calculate median, expected at least f+1, got 2") + t.Run("fails below the contribution floor", func(t *testing.T) { + _, err := MedianAggregator(values[:2], 4) + require.EqualError(t, err, "not enough contributions to calculate median: got 2, need 4") }) t.Run("fails with unsupported StreamValue type", func(t *testing.T) { - _, err := MedianAggregator([]StreamValue{nil, nil, nil}, 1) - require.EqualError(t, err, "not enough observations to calculate median, expected at least f+1, got 0") + _, err := MedianAggregator([]StreamValue{nil, nil, nil}, 2) + require.EqualError(t, err, "not enough contributions to calculate median: got 0, need 2") }) } func Test_ModeAggregator(t *testing.T) { tcs := []struct { - name string - values []StreamValue - f int - output StreamValue - errStr string + name string + values []StreamValue + minContributions int + output StreamValue + errStr string }{ { name: "returns mode value with 3f+1 values in agreement", @@ -94,8 +96,8 @@ func Test_ModeAggregator(t *testing.T) { ToDecimal(decimal.NewFromFloat(1.1)), ToDecimal(decimal.NewFromFloat(1.1)), }, - f: 1, - output: ToDecimal(decimal.NewFromFloat(1.1)), + minContributions: 2, + output: ToDecimal(decimal.NewFromFloat(1.1)), }, { name: "returns mode value with 3f values in agreement", @@ -105,8 +107,8 @@ func Test_ModeAggregator(t *testing.T) { ToDecimal(decimal.NewFromFloat(1.1)), ToDecimal(decimal.NewFromFloat(2.2)), }, - f: 1, - output: ToDecimal(decimal.NewFromFloat(1.1)), + minContributions: 2, + output: ToDecimal(decimal.NewFromFloat(1.1)), }, { name: "returns mode value using tie-breaker with split agreement", @@ -116,25 +118,25 @@ func Test_ModeAggregator(t *testing.T) { ToDecimal(decimal.NewFromFloat(2.2)), ToDecimal(decimal.NewFromFloat(2.2)), }, - f: 1, - output: ToDecimal(decimal.NewFromFloat(1.1)), + minContributions: 2, + output: ToDecimal(decimal.NewFromFloat(1.1)), }, { - name: "returns error if not enough observations", - values: []StreamValue{}, - f: 1, - errStr: "not enough observations in agreement to calculate mode, expected at least f+1, most common value had 0", + name: "returns error if not enough observations", + values: []StreamValue{}, + minContributions: 2, + errStr: "not enough contributions in agreement to calculate mode: most common value had 0, need 2", }, { - name: "returns error if less than f in agreement", + name: "returns error if fewer than the floor agree", values: []StreamValue{ ToDecimal(decimal.NewFromFloat(1.1)), ToDecimal(decimal.NewFromFloat(1.2)), ToDecimal(decimal.NewFromFloat(2.2)), ToDecimal(decimal.NewFromFloat(3.2)), }, - f: 1, - errStr: "not enough observations in agreement to calculate mode, expected at least f+1, most common value had 1", + minContributions: 2, + errStr: "not enough contributions in agreement to calculate mode: most common value had 1, need 2", }, { name: "handles mixed types, tie-breaking on first type", @@ -144,8 +146,8 @@ func Test_ModeAggregator(t *testing.T) { &Quote{Benchmark: decimal.NewFromFloat(1.2)}, &Quote{Benchmark: decimal.NewFromFloat(1.2)}, }, - f: 1, - output: ToDecimal(decimal.NewFromFloat(1.1)), + minContributions: 2, + output: ToDecimal(decimal.NewFromFloat(1.1)), }, { name: "handles mixed types where Quote is most common", @@ -155,8 +157,8 @@ func Test_ModeAggregator(t *testing.T) { &Quote{Bid: decimal.NewFromFloat(1.2), Benchmark: decimal.NewFromFloat(2.2), Ask: decimal.NewFromFloat(3.2)}, &Quote{Bid: decimal.NewFromFloat(1.2), Benchmark: decimal.NewFromFloat(2.2), Ask: decimal.NewFromFloat(3.2)}, }, - f: 1, - output: &Quote{Bid: decimal.NewFromFloat(1.2), Benchmark: decimal.NewFromFloat(2.2), Ask: decimal.NewFromFloat(3.2)}, + minContributions: 2, + output: &Quote{Bid: decimal.NewFromFloat(1.2), Benchmark: decimal.NewFromFloat(2.2), Ask: decimal.NewFromFloat(3.2)}, }, { name: "nils are not counted", @@ -169,13 +171,13 @@ func Test_ModeAggregator(t *testing.T) { nil, nil, }, - f: 2, - output: ToDecimal(decimal.NewFromFloat(1.1)), + minContributions: 3, + output: ToDecimal(decimal.NewFromFloat(1.1)), }, } for _, tc := range tcs { t.Run(tc.name, func(t *testing.T) { - sv, err := ModeAggregator(tc.values, tc.f) + sv, err := ModeAggregator(tc.values, tc.minContributions) if tc.errStr == "" { require.NoError(t, err) assert.Equal(t, tc.output, sv) @@ -195,7 +197,7 @@ func Test_QuoteAggregator(t *testing.T) { &Quote{Bid: (decimal.NewFromFloat(10.01)), Benchmark: (decimal.NewFromFloat(10.03)), Ask: (decimal.NewFromFloat(10.10))}, } - sv, err := QuoteAggregator(values, 1) + sv, err := QuoteAggregator(values, 2) require.NoError(t, err) assert.IsType(t, &Quote{}, sv) q := sv.(*Quote) @@ -211,7 +213,7 @@ func Test_QuoteAggregator(t *testing.T) { &Quote{Bid: (decimal.NewFromFloat(7.7)), Benchmark: (decimal.NewFromFloat(8.8)), Ask: (decimal.NewFromFloat(8.7))}, // invalid &Quote{Bid: (decimal.NewFromFloat(12.12)), Benchmark: (decimal.NewFromFloat(11.11)), Ask: (decimal.NewFromFloat(12.12))}, // invalid } - sv, err := QuoteAggregator(values, 1) + sv, err := QuoteAggregator(values, 2) require.NoError(t, err) assert.IsType(t, &Quote{}, sv) q := sv.(*Quote) @@ -220,9 +222,9 @@ func Test_QuoteAggregator(t *testing.T) { assert.Equal(t, "6.6", q.Ask.String()) }) - t.Run("fails with fewer than f+1 values", func(t *testing.T) { - _, err := QuoteAggregator([]StreamValue{&Quote{}, &Quote{}}, 2) - require.EqualError(t, err, "not enough valid observations to aggregate quote, expected at least f+1, got 2") + t.Run("fails below the contribution floor", func(t *testing.T) { + _, err := QuoteAggregator([]StreamValue{&Quote{}, &Quote{}}, 3) + require.EqualError(t, err, "not enough valid contributions to aggregate quote: got 2, need 3") }) t.Run("ignores non-Quote type", func(t *testing.T) { @@ -232,7 +234,7 @@ func Test_QuoteAggregator(t *testing.T) { ToDecimal(decimal.NewFromFloat(7.7)), ToDecimal(decimal.NewFromFloat(8.8)), } - sv, err := QuoteAggregator(values, 1) + sv, err := QuoteAggregator(values, 2) require.NoError(t, err) assert.IsType(t, &Quote{}, sv) q := sv.(*Quote) diff --git a/llo/protocol/llo_offchain_config.pb.go b/llo/protocol/llo_offchain_config.pb.go index 64a2b07..6ae8061 100644 --- a/llo/protocol/llo_offchain_config.pb.go +++ b/llo/protocol/llo_offchain_config.pb.go @@ -1,7 +1,7 @@ // Code generated by protoc-gen-go. DO NOT EDIT. // versions: -// protoc-gen-go v1.36.11 -// protoc v7.35.1 +// protoc-gen-go v1.36.12 +// protoc v7.36.1 // source: llo_offchain_config.proto package protocol @@ -26,8 +26,15 @@ type LLOOffchainConfigProto struct { ProtocolVersion uint32 `protobuf:"varint,1,opt,name=protocolVersion,proto3" json:"protocolVersion,omitempty"` DefaultMinReportIntervalNanoseconds uint64 `protobuf:"varint,2,opt,name=defaultMinReportIntervalNanoseconds,proto3" json:"defaultMinReportIntervalNanoseconds,omitempty"` EnableObservationCompression bool `protobuf:"varint,3,opt,name=enableObservationCompression,proto3" json:"enableObservationCompression,omitempty"` - unknownFields protoimpl.UnknownFields - sizeCache protoimpl.SizeCache + // How many Byzantine contributors per-stream aggregation tolerates in + // LLO v31: an aggregate is built only from at least + // 2*aggregationFaultTolerance+1 contributions. Distinct from the OCR + // consensus F, which governs observations, not contributions. Explicitly + // presence-tracked: unset and 0 are different configs, and LLO v31 refuses + // to start when unset. + AggregationFaultTolerance *uint32 `protobuf:"varint,4,opt,name=aggregationFaultTolerance,proto3,oneof" json:"aggregationFaultTolerance,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache } func (x *LLOOffchainConfigProto) Reset() { @@ -81,15 +88,24 @@ func (x *LLOOffchainConfigProto) GetEnableObservationCompression() bool { return false } +func (x *LLOOffchainConfigProto) GetAggregationFaultTolerance() uint32 { + if x != nil && x.AggregationFaultTolerance != nil { + return *x.AggregationFaultTolerance + } + return 0 +} + var File_llo_offchain_config_proto protoreflect.FileDescriptor const file_llo_offchain_config_proto_rawDesc = "" + "\n" + - "\x19llo_offchain_config.proto\x12\x02v1\"\xd8\x01\n" + + "\x19llo_offchain_config.proto\x12\x02v1\"\xb9\x02\n" + "\x16LLOOffchainConfigProto\x12(\n" + "\x0fprotocolVersion\x18\x01 \x01(\rR\x0fprotocolVersion\x12P\n" + "#defaultMinReportIntervalNanoseconds\x18\x02 \x01(\x04R#defaultMinReportIntervalNanoseconds\x12B\n" + - "\x1cenableObservationCompression\x18\x03 \x01(\bR\x1cenableObservationCompressionB\fZ\n" + + "\x1cenableObservationCompression\x18\x03 \x01(\bR\x1cenableObservationCompression\x12A\n" + + "\x19aggregationFaultTolerance\x18\x04 \x01(\rH\x00R\x19aggregationFaultTolerance\x88\x01\x01B\x1c\n" + + "\x1a_aggregationFaultToleranceB\fZ\n" + ".;protocolb\x06proto3" var ( @@ -121,6 +137,7 @@ func file_llo_offchain_config_proto_init() { if File_llo_offchain_config_proto != nil { return } + file_llo_offchain_config_proto_msgTypes[0].OneofWrappers = []any{} type x struct{} out := protoimpl.TypeBuilder{ File: protoimpl.DescBuilder{ diff --git a/llo/protocol/llo_offchain_config.proto b/llo/protocol/llo_offchain_config.proto index 73b86a2..b98ddd9 100644 --- a/llo/protocol/llo_offchain_config.proto +++ b/llo/protocol/llo_offchain_config.proto @@ -7,4 +7,11 @@ message LLOOffchainConfigProto { uint32 protocolVersion = 1; uint64 defaultMinReportIntervalNanoseconds = 2; bool enableObservationCompression = 3; + // How many Byzantine contributors per-stream aggregation tolerates in + // LLO v31: an aggregate is built only from at least + // 2*aggregationFaultTolerance+1 contributions. Distinct from the OCR + // consensus F, which governs observations, not contributions. Explicitly + // presence-tracked: unset and 0 are different configs, and LLO v31 refuses + // to start when unset. + optional uint32 aggregationFaultTolerance = 4; } diff --git a/llo/protocol/llo_plugin_telemetry.pb.go b/llo/protocol/llo_plugin_telemetry.pb.go index b1f5ec3..d1acc11 100644 --- a/llo/protocol/llo_plugin_telemetry.pb.go +++ b/llo/protocol/llo_plugin_telemetry.pb.go @@ -1,7 +1,7 @@ // Code generated by protoc-gen-go. DO NOT EDIT. // versions: -// protoc-gen-go v1.36.11 -// protoc v7.35.1 +// protoc-gen-go v1.36.12 +// protoc v7.36.1 // source: llo_plugin_telemetry.proto package protocol @@ -36,6 +36,10 @@ type LLOOutcomeTelemetry struct { SeqNr uint64 `protobuf:"varint,9,opt,name=seq_nr,json=seqNr,proto3" json:"seq_nr,omitempty"` ConfigDigest []byte `protobuf:"bytes,10,opt,name=config_digest,json=configDigest,proto3" json:"config_digest,omitempty"` DonId uint32 `protobuf:"varint,11,opt,name=don_id,json=donId,proto3" json:"don_id,omitempty"` + // Contribution floor the round's stream aggregates were produced under: + // 2*aggregationFaultTolerance+1. Zero in LLO v30, which derives the floor + // from the consensus F instead. + MinContributions uint32 `protobuf:"varint,12,opt,name=min_contributions,json=minContributions,proto3" json:"min_contributions,omitempty"` unknownFields protoimpl.UnknownFields sizeCache protoimpl.SizeCache } @@ -126,6 +130,13 @@ func (x *LLOOutcomeTelemetry) GetDonId() uint32 { return 0 } +func (x *LLOOutcomeTelemetry) GetMinContributions() uint32 { + if x != nil { + return x.MinContributions + } + return 0 +} + type LLOAggregatorStreamValue struct { state protoimpl.MessageState `protogen:"open.v1"` AggregatorValues map[uint32]*LLOStreamValue `protobuf:"bytes,1,rep,name=aggregator_values,json=aggregatorValues,proto3" json:"aggregator_values,omitempty" protobuf_key:"varint,1,opt,name=key" protobuf_val:"bytes,2,opt,name=value"` @@ -299,7 +310,7 @@ var File_llo_plugin_telemetry_proto protoreflect.FileDescriptor const file_llo_plugin_telemetry_proto_rawDesc = "" + "\n" + - "\x1allo_plugin_telemetry.proto\x12\x02v1\x1a\x13plugin_codecs.proto\"\x9b\x06\n" + + "\x1allo_plugin_telemetry.proto\x12\x02v1\x1a\x13plugin_codecs.proto\"\xc8\x06\n" + "\x13LLOOutcomeTelemetry\x12(\n" + "\x10life_cycle_stage\x18\x01 \x01(\tR\x0elifeCycleStage\x12J\n" + "!observation_timestamp_nanoseconds\x18\x02 \x01(\x04R\x1fobservationTimestampNanoseconds\x12`\n" + @@ -309,7 +320,8 @@ const file_llo_plugin_telemetry_proto_rawDesc = "" + "\x06seq_nr\x18\t \x01(\x04R\x05seqNr\x12#\n" + "\rconfig_digest\x18\n" + " \x01(\fR\fconfigDigest\x12\x15\n" + - "\x06don_id\x18\v \x01(\rR\x05donId\x1ad\n" + + "\x06don_id\x18\v \x01(\rR\x05donId\x12+\n" + + "\x11min_contributions\x18\f \x01(\rR\x10minContributions\x1ad\n" + "\x17ChannelDefinitionsEntry\x12\x10\n" + "\x03key\x18\x01 \x01(\rR\x03key\x123\n" + "\x05value\x18\x02 \x01(\v2\x1d.v1.LLOChannelDefinitionProtoR\x05value:\x028\x01\x1aH\n" + diff --git a/llo/protocol/llo_plugin_telemetry.proto b/llo/protocol/llo_plugin_telemetry.proto index e8c87d6..0a321fd 100644 --- a/llo/protocol/llo_plugin_telemetry.proto +++ b/llo/protocol/llo_plugin_telemetry.proto @@ -20,6 +20,10 @@ message LLOOutcomeTelemetry { uint64 seq_nr = 9; bytes config_digest = 10; uint32 don_id = 11; + // Contribution floor the round's stream aggregates were produced under: + // 2*aggregationFaultTolerance+1. Zero in LLO v30, which derives the floor + // from the consensus F instead. + uint32 min_contributions = 12; } message LLOAggregatorStreamValue { diff --git a/llo/protocol/offchain_config.go b/llo/protocol/offchain_config.go index 164412e..7dad0d8 100644 --- a/llo/protocol/offchain_config.go +++ b/llo/protocol/offchain_config.go @@ -3,6 +3,7 @@ package protocol import ( "errors" "fmt" + "math" "google.golang.org/protobuf/proto" ) @@ -20,6 +21,15 @@ type OffchainConfig struct { DefaultMinReportIntervalNanoseconds uint64 // EnableObservationCompression enables observation compression. EnableObservationCompression bool + // AggregationFaultTolerance is how many Byzantine contributors per-stream + // aggregation tolerates in LLO v31: it sets the contribution floor to + // 2*AggregationFaultTolerance+1. nil means unset: LLO v30 ignores it, LLO + // v3.1 refuses to construct a plugin. + // + // There is no default. The right value depends on the DON's observed + // per-stream omission rate, so a default would either silently weaken + // safety or silently stall reporting. + AggregationFaultTolerance *uint32 } func DecodeOffchainConfig(b []byte) (o OffchainConfig, err error) { @@ -40,6 +50,7 @@ func DecodeOffchainConfig(b []byte) (o OffchainConfig, err error) { o.ProtocolVersion = pbuf.ProtocolVersion o.DefaultMinReportIntervalNanoseconds = pbuf.DefaultMinReportIntervalNanoseconds o.EnableObservationCompression = pbuf.EnableObservationCompression + o.AggregationFaultTolerance = pbuf.AggregationFaultTolerance // NOTE: Validate must run on the decoded values. A node that cannot honour the // configured version must refuse to run rather than diverge from the nodes // that can. @@ -54,6 +65,7 @@ func (c OffchainConfig) Encode() ([]byte, error) { ProtocolVersion: c.ProtocolVersion, DefaultMinReportIntervalNanoseconds: c.DefaultMinReportIntervalNanoseconds, EnableObservationCompression: c.EnableObservationCompression, + AggregationFaultTolerance: c.AggregationFaultTolerance, } return proto.Marshal(pbuf) } @@ -74,5 +86,12 @@ func (c OffchainConfig) Validate() error { default: return fmt.Errorf("unknown protocol version: %d", c.ProtocolVersion) } + // AggregationFaultTolerance is validated where it is used: LLO v30 ignores + // it, LLO v31 requires it (see llo/dev/v31.NewReportingPlugin). Nothing here + // can tell which plugin will consume this config. The bound below only + // keeps the int conversion and the 2x+1 arithmetic safe. + if c.AggregationFaultTolerance != nil && *c.AggregationFaultTolerance > math.MaxInt8 { + return fmt.Errorf("aggregationFaultTolerance out of range: %d", *c.AggregationFaultTolerance) + } return nil } diff --git a/llo/protocol/offchain_config_test.go b/llo/protocol/offchain_config_test.go index 3e428a1..e7f11cf 100644 --- a/llo/protocol/offchain_config_test.go +++ b/llo/protocol/offchain_config_test.go @@ -8,6 +8,50 @@ import ( "github.com/stretchr/testify/require" ) +func Test_OffchainConfig_AggregationFaultTolerance(t *testing.T) { + t.Run("unset and zero are distinct configs", func(t *testing.T) { + unset, err := OffchainConfig{ProtocolVersion: 1, DefaultMinReportIntervalNanoseconds: 1}.Encode() + require.NoError(t, err) + zero := uint32(0) + explicitZero, err := OffchainConfig{ProtocolVersion: 1, DefaultMinReportIntervalNanoseconds: 1, AggregationFaultTolerance: &zero}.Encode() + require.NoError(t, err) + assert.NotEqual(t, unset, explicitZero) + + decodedUnset, err := DecodeOffchainConfig(unset) + require.NoError(t, err) + assert.Nil(t, decodedUnset.AggregationFaultTolerance) + + decodedZero, err := DecodeOffchainConfig(explicitZero) + require.NoError(t, err) + require.NotNil(t, decodedZero.AggregationFaultTolerance) + assert.Equal(t, uint32(0), *decodedZero.AggregationFaultTolerance) + }) + t.Run("round-trips a set value", func(t *testing.T) { + aft := uint32(5) + b, err := OffchainConfig{ProtocolVersion: 2, DefaultMinReportIntervalNanoseconds: 1, AggregationFaultTolerance: &aft}.Encode() + require.NoError(t, err) + decoded, err := DecodeOffchainConfig(b) + require.NoError(t, err) + require.NotNil(t, decoded.AggregationFaultTolerance) + assert.Equal(t, uint32(5), *decoded.AggregationFaultTolerance) + }) + t.Run("invalid onchain bytes decode to unset", func(t *testing.T) { + b, err := hex.DecodeString("7b2265787069726174696f6e57696e646f77223a38363430302c2262617365555344466565223a22302e3332227d") + require.NoError(t, err) + decoded, err := DecodeOffchainConfig(b) + require.NoError(t, err) + assert.Nil(t, decoded.AggregationFaultTolerance) + }) + t.Run("out of range is rejected", func(t *testing.T) { + aft := uint32(1 << 20) + err := OffchainConfig{ProtocolVersion: 1, DefaultMinReportIntervalNanoseconds: 1, AggregationFaultTolerance: &aft}.Validate() + require.EqualError(t, err, "aggregationFaultTolerance out of range: 1048576") + }) + t.Run("is not required by Validate, since v3.0 ignores it", func(t *testing.T) { + require.NoError(t, OffchainConfig{ProtocolVersion: 1, DefaultMinReportIntervalNanoseconds: 1}.Validate()) + }) +} + func Test_OffchainConfig(t *testing.T) { t.Run("decoding invalid bytes", func(t *testing.T) { b, err := hex.DecodeString("7b2265787069726174696f6e57696e646f77223a38363430302c2262617365555344466565223a22302e3332227d") diff --git a/llo/v30/plugin_outcome.go b/llo/v30/plugin_outcome.go index 9ea3efe..8f81a47 100644 --- a/llo/v30/plugin_outcome.go +++ b/llo/v30/plugin_outcome.go @@ -280,7 +280,15 @@ func (p *Plugin) outcome(outctx ocr3types.OutcomeContext, query types.Query, aos } } - // Perform the aggregation + // Perform the aggregation. + // + // The contribution floor is F+1 values per (streamID, aggregator) + // pair, which is the intended behavior for v30. The rationale is + // liveness: stream values are sparse, since a data source is + // contractually allowed to leave a stream unset, and + // ObservationQuorum lets the round proceed on with 2F+1 + // attributed observations, so a stream aggregate rests on a + // subset of oracles than the round itself requires. aggF := protocol.GetAggregatorFunc(agg) if aggF == nil { // Unknown aggregator, e.g. one added by a newer version. Admission @@ -288,7 +296,7 @@ func (p *Plugin) outcome(outctx ocr3types.OutcomeContext, query types.Query, aos // protocol: skip the pair, keeping any carried-forward value. continue } - result, err := aggF(streamObservations[sid], p.F) + result, err := aggF(streamObservations[sid], p.F+1) // Handle aggregation results switch v := result.(type) { From 6018b7fa2e21c5b1ac0138fb413c45d30b536175 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 07:43:42 +0100 Subject: [PATCH 11/40] SPOR-0011 llo/v31: verify the definition set an observation advocates ValidateObservation merged the observation's updates onto committed state but kept the channels it voted to remove. That set is one no vote can produce, so a swap of one channel for another while the set sits on a whole-set budget failed verification on a ceiling its own vote clears. Take the removals out of the merged set. Whole-set budgets cannot be enforced per observation anyway as votes from different oracles combine, and what gets committed is decided by the per-hash threshold. --- llo/dev/v31/flow_test.go | 45 ++++++++++++++++++++++++++++++++++++++++ llo/dev/v31/plugin.go | 9 ++++++++ 2 files changed, 54 insertions(+) diff --git a/llo/dev/v31/flow_test.go b/llo/dev/v31/flow_test.go index 27ac860..2d487db 100644 --- a/llo/dev/v31/flow_test.go +++ b/llo/dev/v31/flow_test.go @@ -453,3 +453,48 @@ func Test_StateTransition_Retirement(t *testing.T) { require.Equal(t, llotypes.ReportFormatRetirement, reports[0].ReportWithInfo.Info.ReportFormat) require.Equal(t, protocol.LifeCycleStageRetired, reports[0].ReportWithInfo.Info.LifeCycleStage) } + +// Test_ValidateObservation_RemoveAddSwapAtBudget pins that the set verified is +// the one the observation advocates: a swap that keeps the definition set +// inside a whole-set budget must validate, even though the committed set plus +// the update alone exceeds it. Ignoring the removals would reject every honest +// observation voting the swap, and with it the round's observation quorum. +func Test_ValidateObservation_RemoveAddSwapAtBudget(t *testing.T) { + ctx := tests.Context(t) + p := testPlugin(t) + kv := newMemKV() + + // streamRangeChannel builds a channel holding n distinct stream IDs starting + // at first, so the unique-stream-ID budget can be sat exactly on. + streamRangeChannel := func(first llotypes.StreamID, n int) llotypes.ChannelDefinition { + streams := make([]llotypes.Stream, 0, n) + for i := 0; i < n; i++ { + streams = append(streams, llotypes.Stream{StreamID: first + llotypes.StreamID(i), Aggregator: llotypes.AggregatorMedian}) + } + return llotypes.ChannelDefinition{ReportFormat: llotypes.ReportFormatJSON, Streams: streams} + } + + half := protocol.MaxObservationStreamValuesLength / 2 + committed := llotypes.ChannelDefinitions{ + 1: streamRangeChannel(1, half), + 2: streamRangeChannel(llotypes.StreamID(half)+1, half), + } + require.NoError(t, protocol.VerifyChannelDefinitions(p.ReportCodecs, committed), "committed set must sit exactly on the budget") + require.NoError(t, writeChannelState(kv, 1, committed)) + + // Swap channel 2 out for channel 3, which holds as many streams as the one + // it replaces, so the resulting set is back on the budget, not over it. + replacement := llotypes.ChannelDefinitions{3: streamRangeChannel(llotypes.StreamID(2*half)+1, half)} + swap := Observation{ + UnixTimestampNanoseconds: 1_000, + RemoveChannelIDs: map[llotypes.ChannelID]struct{}{2: {}}, + UpdateChannelDefinitions: replacement, + } + require.NoError(t, p.ValidateObservation(ctx, 2, ocrtypes.AttributedQuery{}, ao(0, mustEncodeObs(t, swap)), kv, testBlobs)) + + // The same update without the removal vote does exceed the budget, which is + // what makes the assertion above about the removals and not about slack in + // the limit. + addOnly := Observation{UnixTimestampNanoseconds: 1_000, UpdateChannelDefinitions: replacement} + require.Error(t, p.ValidateObservation(ctx, 2, ocrtypes.AttributedQuery{}, ao(0, mustEncodeObs(t, addOnly)), kv, testBlobs)) +} diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index 9a274ec..a036f99 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -315,6 +315,12 @@ func (p *Plugin) ValidateObservation(ctx context.Context, seqNr uint64, _ ocrtyp // verification accepted (see voteOnChannels). Repeating the admission-only // checks here would add nothing, and would make oracles running different // versions of those checks disagree on whether an observation is valid. + // The set verified is the one this observation advocates: committed state + // with its updates applied and its removals taken out. + // + // Whole-set budgets cannot be enforced per observation anyway, votes from + // different oracles combine, and what gets committed is decided by the + // per-hash threshold, not by any single observation. defsForVerify := observation.UpdateChannelDefinitions if len(observation.UpdateChannelDefinitions) > 0 { state, serr := loadColdKVState(kvReader, p.ChannelCache) @@ -323,6 +329,9 @@ func (p *Plugin) ValidateObservation(ctx context.Context, seqNr uint64, _ ocrtyp } merged := make(llotypes.ChannelDefinitions, len(state.channelDefinitions)+len(observation.UpdateChannelDefinitions)) for id, def := range state.channelDefinitions { + if _, removed := observation.RemoveChannelIDs[id]; removed { + continue + } merged[id] = def } for id, def := range observation.UpdateChannelDefinitions { From efeeabe696ec205bd395a7ab1dbe5a47a20b6cc4 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 07:59:13 +0100 Subject: [PATCH 12/40] SPOR-0015 llo/protocol: state why the channel cache evicts by insertion order --- llo/protocol/channel_cache.go | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/llo/protocol/channel_cache.go b/llo/protocol/channel_cache.go index da63f2e..2361d83 100644 --- a/llo/protocol/channel_cache.go +++ b/llo/protocol/channel_cache.go @@ -129,6 +129,12 @@ func (c *ChannelCache) store(gen *ChannelGeneration) *ChannelGeneration { c.gens = make(map[uint64]*ChannelGeneration, channelGenerationsRetained) } c.gens[gen.seqNr] = gen + // Evict by insertion order, not by sequence number: what a round asks for + // next is the record it is replaying, which for a node restoring from a + // snapshot is an older one than the cache already holds (see the type doc). + // Retaining the highest sequence numbers instead would evict exactly the + // generations such a node keeps asking for. In steady state the two agree, + // insertion being monotonic there. c.order = append(c.order, gen.seqNr) for len(c.order) > channelGenerationsRetained { delete(c.gens, c.order[0]) From 77965689b4620ab085ac7193b82889cad1c9b3cc Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 08:29:48 +0100 Subject: [PATCH 13/40] SPOR-0014 llo/dev/v31: bound how long the blob pump waits at Close --- llo/dev/v31/blobpump.go | 39 ++++++++++++++++++++----- llo/dev/v31/blobpump_test.go | 56 +++++++++++++++++++++++++++++++++++- llo/dev/v31/plugin.go | 4 +-- 3 files changed, 89 insertions(+), 10 deletions(-) diff --git a/llo/dev/v31/blobpump.go b/llo/dev/v31/blobpump.go index a85b8ac..6f95a02 100644 --- a/llo/dev/v31/blobpump.go +++ b/llo/dev/v31/blobpump.go @@ -69,6 +69,11 @@ const ( BlobReapingMarginRounds = 16 // MinPerOracleUnexpiredBlobCount is the floor for the derived budget. MinPerOracleUnexpiredBlobCount = 32 + // closeTimeoutSlackMultiplier scales the pump's observation timeout into how + // long Close waits for an in-flight cycle. + closeTimeoutSlackMultiplier = 5 + // minCloseTimeout fixes the minimum time Close waits waits for an in-flight cycle. + minCloseTimeout = 1 * time.Second ) // perOracleUnexpiredBlobCount derives the per-oracle unexpired-blob budget from @@ -118,10 +123,11 @@ type blobPump struct { blobPumpParams lggr logger.Logger - trigger chan struct{} - ctx context.Context - cancel context.CancelFunc - wg sync.WaitGroup + trigger chan struct{} + ctx context.Context + cancel context.CancelFunc + wg sync.WaitGroup + closeTimeout time.Duration inFlight atomic.Bool misses atomic.Uint64 @@ -163,6 +169,7 @@ func newBlobPump(lggr logger.Logger, params blobPumpParams) *blobPump { trigger: make(chan struct{}, 1), ctx: ctx, cancel: cancel, + closeTimeout: max(params.observationTimeout*closeTimeoutSlackMultiplier, minCloseTimeout), } } @@ -183,10 +190,28 @@ func (p *blobPump) Start() { go p.run() } -// Close stops the pump and waits for any in-flight cycle to unwind. -func (p *blobPump) Close() { +// Close stops the pump and waits for any in-flight cycle to unwind, bounded by +// closeTimeout. Reports false when the cycle did not unwind, meaning its +// goroutine is still blocked in a DataSource or broadcaster call that ignored +// the cancelled context: abandon it rather than let it hang the caller's +// shutdown path. +func (p *blobPump) Close() bool { p.cancel() - p.wg.Wait() + + done := make(chan struct{}) + go func() { + p.wg.Wait() + close(done) + }() + + select { + case <-done: + return true + case <-time.After(p.closeTimeout): + p.lggr.Errorw("Blob pump cycle did not unwind after context cancellation; abandoning goroutine", + "closeTimeout", p.closeTimeout, "observationTimeout", p.observationTimeout) + return false + } } // SetInput publishes the round context for subsequent cycles. Cheap; called diff --git a/llo/dev/v31/blobpump_test.go b/llo/dev/v31/blobpump_test.go index 4003b95..abc5129 100644 --- a/llo/dev/v31/blobpump_test.go +++ b/llo/dev/v31/blobpump_test.go @@ -3,11 +3,13 @@ package llo import ( "context" "errors" + "sync" "sync/atomic" "testing" "time" "github.com/shopspring/decimal" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" "github.com/smartcontractkit/chainlink-common/pkg/logger" @@ -34,7 +36,7 @@ func testPump(t *testing.T, ds DataSource, bbf ocr3_1types.BlobBroadcastFetcher, blobLifetimeRounds: DefaultBlobLifetimeRounds, }) p.Start() - t.Cleanup(p.Close) + t.Cleanup(func() { assert.True(t, p.Close(), "pump did not stop cleanly") }) return p } @@ -317,3 +319,55 @@ func Test_blobPump_SurvivesDataSourcePanic(t *testing.T) { require.Eventually(t, func() bool { return ds.calls.Load() >= 2 }, tests.WaitTimeout(t), 10*time.Millisecond) require.False(t, p.inFlight.Load()) } + +// stuckDataSource ignores its context and blocks until released, modelling a +// host DataSource that does not honor cancellation. +type stuckDataSource struct { + entered chan struct{} + release chan struct{} + once sync.Once +} + +func (s *stuckDataSource) Observe(ctx context.Context, sv protocol.StreamValues, opts DSOpts) error { + s.once.Do(func() { close(s.entered) }) + <-s.release + return errors.New("released") +} + +// Test_blobPump_CloseDoesNotHangOnStuckDataSource asserts Close gives up after +// closeTimeout rather than waiting forever on a DataSource that ignores its +// context. The DataSource is shared between the blue and green instances, so a +// Close that never returns would also keep the other instance and the source +// itself from closing. +func Test_blobPump_CloseDoesNotHangOnStuckDataSource(t *testing.T) { + ds := &stuckDataSource{entered: make(chan struct{}), release: make(chan struct{})} + defer close(ds.release) + + p := newBlobPump(logger.Test(t), blobPumpParams{ + bbf: newFakeBroadcaster(), + ds: ds, + observationTimeout: time.Minute, + maxSnapshotRounds: DefaultMaxSnapshotRounds, + blobLifetimeRounds: DefaultBlobLifetimeRounds, + }) + require.Equal(t, closeTimeoutSlackMultiplier*time.Minute, p.closeTimeout, "closeTimeout must derive from the observation timeout") + p.closeTimeout = 100 * time.Millisecond + p.Start() + + p.SetInput(pumpInputFor(2)) + p.Take(2) + select { + case <-ds.entered: + case <-time.After(tests.WaitTimeout(t)): + t.Fatal("DataSource.Observe was never called") + } + + done := make(chan bool, 1) + go func() { done <- p.Close() }() + select { + case ok := <-done: + require.False(t, ok, "Close must report the cycle did not unwind") + case <-time.After(tests.WaitTimeout(t)): + t.Fatal("Close hung on a DataSource that ignores context") + } +} diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index a036f99..fe200b8 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -385,8 +385,8 @@ func (p *Plugin) ShouldTransmitAcceptedReport(context.Context, uint64, ocr3types } func (p *Plugin) Close() error { - if p.pump != nil { - p.pump.Close() + if p.pump != nil && !p.pump.Close() { + return fmt.Errorf("blob pump did not stop within %s", p.pump.closeTimeout) } return nil } From 1659335a9d3f0c06ca060c20172cea5797056984 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 08:33:48 +0100 Subject: [PATCH 14/40] SPOR-0016 llo/dev/v31: cover the layout reset that warms nothing --- llo/dev/v31/history.go | 6 ++- llo/dev/v31/history_test.go | 75 +++++++++++++++++++++++++++++++++++++ 2 files changed, 80 insertions(+), 1 deletion(-) diff --git a/llo/dev/v31/history.go b/llo/dev/v31/history.go index 1c63bdb..dc14739 100644 --- a/llo/dev/v31/history.go +++ b/llo/dev/v31/history.go @@ -366,7 +366,11 @@ func (s *historyStore) Flush(w ocr3_1types.KeyValueStateReadWriter) error { // it. A DON that has never used history writes no history keys at all, and // stamping a version onto an otherwise untouched state would be the only // exception to that. - if s.layoutReset && (indexChanged || len(s.index) > 0) { + // + // The index covers that condition on its own: a reset that abandoned any + // stored pair set indexChanged up front, and the only other way this round + // touches history is a pair it warmed, which adds it to the index. + if s.layoutReset && indexChanged { if err := writeHistoryLayoutVersion(w); err != nil { return fmt.Errorf("write history layout version: %w", err) } diff --git a/llo/dev/v31/history_test.go b/llo/dev/v31/history_test.go index e6da482..2a78e68 100644 --- a/llo/dev/v31/history_test.go +++ b/llo/dev/v31/history_test.go @@ -1076,3 +1076,78 @@ func TestHistoryStore_DeterministicAcrossALap(t *testing.T) { assert.Equal(t, run(), run(), "two oracles running identical rounds must end with byte-identical state") } + +// TestHistoryStore_LayoutChangeWithEmptyIndex checks a stale version alone does +// not make the round write: a DON that has never stored history has no index to +// rewrite, so there is nothing for a version to describe yet. +func TestHistoryStore_LayoutChangeWithEmptyIndex(t *testing.T) { + t.Parallel() + + kv := newCountingKV() + require.NoError(t, kv.Write(keyHistoryVersion, []byte{historyLayoutVersion + 1})) + kv.writes = map[string]int{} + + s := newTestHistoryStore(t, kv) + require.NoError(t, s.Flush(kv)) + assert.Empty(t, kv.writes, "a reset with nothing stored and nothing warmed must write no history keys") + + version, err := readHistoryLayoutVersion(kv) + require.NoError(t, err) + assert.Equal(t, byte(historyLayoutVersion+1), version, "the stale version stands until this layout stores something") + + // The next round that stores something stamps the version and the index. + s = newTestHistoryStore(t, kv) + require.NoError(t, s.SetRequired(1, testAggMedian, 3)) + _, err = s.Append(1, testAggMedian, 1_000, testDecimal(4)) + require.NoError(t, err) + require.NoError(t, s.Flush(kv)) + + version, err = readHistoryLayoutVersion(kv) + require.NoError(t, err) + assert.Equal(t, byte(historyLayoutVersion), version) +} + +// TestHistoryStore_LayoutChangeWarmingNothing checks a reset that warms nothing +// still settles the stale index it inherited: every abandoned pair is reclaimed, +// the index is rewritten empty, and the version is stamped so the reset does not +// repeat. +func TestHistoryStore_LayoutChangeWarmingNothing(t *testing.T) { + t.Parallel() + + kv := newCountingKV() + key := histKey{streamID: 1, aggregator: testAggMedian} + + seed := newTestHistoryStore(t, kv) + require.NoError(t, seed.SetRequired(key.streamID, key.aggregator, 3)) + _, err := seed.Append(key.streamID, key.aggregator, 1_000, testDecimal(7)) + require.NoError(t, err) + require.NoError(t, seed.Flush(kv)) + require.Len(t, readHistoryRecords(t, kv, key.streamID, key.aggregator), 1) + + // A node comes up on a different layout and no channel requires the pair. + require.NoError(t, kv.Write(keyHistoryVersion, []byte{historyLayoutVersion + 1})) + + s := newTestHistoryStore(t, kv) + require.NoError(t, s.Flush(kv)) + + assert.Nil(t, readHistory(t, kv, key.streamID, key.aggregator), "the abandoned header must be deleted") + for _, k := range historyKeys(key.streamID, key.aggregator) { + b, err := kv.Read(k) + require.NoError(t, err) + assert.Empty(t, b, "key %x must be deleted", k) + } + + index, err := readHistoryIndex(kv) + require.NoError(t, err) + assert.Empty(t, index, "the stale index must be rewritten from what this round warmed, which is nothing") + + version, err := readHistoryLayoutVersion(kv) + require.NoError(t, err) + assert.Equal(t, byte(historyLayoutVersion), version) + + // The reset settled, so a later idle round writes nothing at all. + writes := len(kv.writes) + s = newTestHistoryStore(t, kv) + require.NoError(t, s.Flush(kv)) + assert.Len(t, kv.writes, writes, "the settled reset must not write again") +} From 96be4c454bffef90a5002639a5152ddecce6d24c Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 08:59:46 +0100 Subject: [PATCH 15/40] SPOR-0018 llo: memoize the per-definition channel verification checks Verification decodes channels opts up to four times per analysis (the codec Verify and VerifyForAdmission, FeedID, and the calculated stream IDs), and a round analyses largely the same set three times over: the committed definitions and the desired ones in Observation, and the set each update-carrying observation advocates in ValidateObservation. Share one cache across all three call sites, so the parallel ValidateObservation calls hit what Observation already decoded, and prunes it against the committed set, which is the authority on which channels exist. Over 2000 channels carrying 250 bytes of opts, The Observation two analyses go from 12.1ms to 1.08ms and each ValidateObservation from 5.9ms to 0.51ms, with per-round allocations dropping from 40185 to 199. --- llo/dev/v31/factory.go | 1 + llo/dev/v31/plugin.go | 18 +- llo/protocol/channel_analysis_cache.go | 172 +++++++++++ llo/protocol/channel_analysis_cache_test.go | 272 ++++++++++++++++++ llo/protocol/channel_cache.go | 12 +- llo/protocol/channel_definitions.go | 85 +++--- .../channel_definitions_bench_test.go | 183 ++++++++++++ 7 files changed, 701 insertions(+), 42 deletions(-) create mode 100644 llo/protocol/channel_analysis_cache.go create mode 100644 llo/protocol/channel_analysis_cache_test.go create mode 100644 llo/protocol/channel_definitions_bench_test.go diff --git a/llo/dev/v31/factory.go b/llo/dev/v31/factory.go index d8c394c..0c43ed2 100644 --- a/llo/dev/v31/factory.go +++ b/llo/dev/v31/factory.go @@ -139,6 +139,7 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re // Definitions and the opts decoded from them are cached together, as one // immutable generation per c/seqnr, so a round can never mix the two. p.ChannelCache = protocol.NewChannelCache() + p.ChannelAnalysisCache = protocol.NewChannelAnalysisCache() // Setup the blobpump p.pump = newBlobPump(l, blobPumpParams{ diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index fe200b8..97d5b2a 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -51,6 +51,15 @@ type Plugin struct { // could swap decoded opts out from under it. ChannelCache *protocol.ChannelCache + // ChannelAnalysisCache memoizes the per-definition verification checks, whose + // cost is dominated by decoding channel opts. Verification runs three times + // per round over largely identical sets: the committed definitions and the + // desired ones in Observation, and the set each update-carrying observation + // advocates in ValidateObservation. One cache is shared by all of them, so + // the parallel ValidateObservation calls hit what Observation already + // decoded. May be nil, in which case nothing is memoized. + ChannelAnalysisCache *protocol.ChannelAnalysisCache + // Optional telemetry sinks; best-effort, non-blocking. OutcomeTelemetryCh chan<- *protocol.LLOOutcomeTelemetry ReportTelemetryCh chan<- *protocol.LLOReportTelemetry @@ -114,7 +123,10 @@ func (p *Plugin) Observation(_ context.Context, seqNr uint64, _ ocrtypes.Attribu // but a version skew can make it reachable. // Report the finding and carry on, which is the same treatment the // admission-only findings get in voteOnChannels. - if badChannels, verifyErr := protocol.UnverifiableChannelIDs(p.ReportCodecs, state.channelDefinitions); len(badChannels) > 0 || verifyErr != nil { + // The committed set is the authority on which channels exist, so this is + // where entries for channels that are gone are dropped. + p.ChannelAnalysisCache.Prune(state.channelDefinitions) + if badChannels, verifyErr := protocol.UnverifiableChannelIDsWithCache(p.ReportCodecs, state.channelDefinitions, p.ChannelAnalysisCache); len(badChannels) > 0 || verifyErr != nil { p.Logger.Errorw("Committed channel definitions fail baseline verification on this build", "stage", "Observation", "seqNr", seqNr, "channelIDs", sortedChannelIDSet(badChannels), "err", verifyErr) } @@ -252,7 +264,7 @@ func (p *Plugin) voteOnChannels(obs *Observation, state *kvState) { // admission-only checks; the ones already committed are not, or a // grandfathered channel would freeze channel voting entirely. admitting := protocol.ChangedChannelIDs(state.channelDefinitions, expectedChannelDefs) - if err := protocol.VerifyChannelDefinitionsForAdmission(p.ReportCodecs, expectedChannelDefs, admitting); err != nil { + if err := protocol.VerifyChannelDefinitionsForAdmissionWithCache(p.ReportCodecs, expectedChannelDefs, admitting, p.ChannelAnalysisCache); err != nil { // Don't halt on an invalid channel-definitions file; just don't vote. p.Logger.Errorw("ChannelDefinitionCache.Definitions is invalid", "err", err) return @@ -346,7 +358,7 @@ func (p *Plugin) ValidateObservation(ctx context.Context, seqNr uint64, _ ocrtyp // updates from a staler one, which costs update-voting liveness only. An // observation that votes no update has nothing to verify here, so rounds // themselves are unaffected. - if err := protocol.VerifyChannelDefinitions(p.ReportCodecs, defsForVerify); err != nil { + if err := protocol.VerifyChannelDefinitionsWithCache(p.ReportCodecs, defsForVerify, p.ChannelAnalysisCache); err != nil { return fmt.Errorf("UpdateChannelDefinitions is invalid: %w", err) } diff --git a/llo/protocol/channel_analysis_cache.go b/llo/protocol/channel_analysis_cache.go new file mode 100644 index 0000000..3e51a0a --- /dev/null +++ b/llo/protocol/channel_analysis_cache.go @@ -0,0 +1,172 @@ +package protocol + +import ( + "sync" + + llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" +) + +// channelFacts is everything analyzeChannelDefinitions derives from a single +// channel definition by decoding its opts: the codec's verdicts, the feed ID the +// channel publishes under, and the calculated stream IDs its expressions +// declare. +// +// Every one of these is contractually a pure function of the definition alone +// (see ReportCodec.Verify, AdmissionVerifier and FeedIDer), which is what makes +// them cacheable. Findings that involve more than one definition are not here: +// they are recomputed from these facts on every analysis, so that a finding +// never depends on which set was analyzed first. +type channelFacts struct { + verifyErr error + admissionErr error + feedID [32]byte + hasFeedID bool + feedIDErr error + calculatedIDs []llotypes.StreamID + calculatedErr error +} + +// ChannelAnalysisCache memoizes channelFacts so that an unchanged channel +// definition has its opts decoded once rather than on every analysis. +// +// Verification runs several times per round over overlapping sets: the +// committed definitions and the desired ones in Observation, and the set each +// update-carrying observation advocates in ValidateObservation, which runs in +// its own goroutine. Decoding the opts dominates all of them, and between them +// the definitions are almost entirely the same and almost never change. +// +// Entries are keyed by channel ID and validated against the definition they +// were derived from, so a hit is an exact identity check rather than a digest +// comparison. A digest would be cheaper by a hair and would trade that +// exactness for the possibility of serving a verdict for a definition that was +// never verified, on a set whose contents are decided by vote. +// +// Definitions being changed alternate between the committed and the desired +// value within a round, so the cache holds a small number of entries per +// channel rather than one. The overlapping sets agree on everything else. +// +// A nil *ChannelAnalysisCache is usable and simply memoizes nothing. +type ChannelAnalysisCache struct { + mu sync.Mutex + facts map[llotypes.ChannelID][]cachedChannelFacts +} + +// cachedChannelFacts pairs the facts with the definition they were derived +// from. cd is the identity of the entry, not payload: a lookup is a hit only +// when the definition presented equals this one. +type cachedChannelFacts struct { + cd llotypes.ChannelDefinition + facts channelFacts +} + +// channelFactsRetained bounds how many definitions of one channel are +// remembered. Two is what a round asks for while a channel is being changed +// (the committed value and the proposed one); the third leaves room for a +// second proposal in flight. +const channelFactsRetained = 3 + +func NewChannelAnalysisCache() *ChannelAnalysisCache { + return &ChannelAnalysisCache{ + facts: make(map[llotypes.ChannelID][]cachedChannelFacts), + } +} + +// Prune drops the entries of every channel channelDefs does not hold. Verifying +// a set does not by itself mean the channels it omits are gone, so this is for +// callers that know the committed set: on the v31 hot path, the round that +// loads it. +func (c *ChannelAnalysisCache) Prune(channelDefs llotypes.ChannelDefinitions) { + if c == nil { + return + } + c.mu.Lock() + defer c.mu.Unlock() + for channelID := range c.facts { + if _, ok := channelDefs[channelID]; !ok { + delete(c.facts, channelID) + } + } +} + +// get returns the facts derived from exactly cd, if they are cached. +func (c *ChannelAnalysisCache) get(channelID llotypes.ChannelID, cd llotypes.ChannelDefinition) (channelFacts, bool) { + if c == nil { + return channelFacts{}, false + } + c.mu.Lock() + defer c.mu.Unlock() + for _, entry := range c.facts[channelID] { + if entry.cd.Equals(cd) { + return entry.facts, true + } + } + return channelFacts{}, false +} + +// put records facts as derived from cd, evicting the channel's oldest entry +// once the bound is reached. cd is cloned: the caller's definition may share +// memory with a set that is mutated later, and the entry's identity has to stay +// what it was derived from. +func (c *ChannelAnalysisCache) put(channelID llotypes.ChannelID, cd llotypes.ChannelDefinition, facts channelFacts) { + if c == nil { + return + } + c.mu.Lock() + defer c.mu.Unlock() + + entries := c.facts[channelID] + for _, entry := range entries { + if entry.cd.Equals(cd) { + // A concurrent analysis got there first. The facts are a pure + // function of the definition, so the two agree. + return + } + } + if c.facts == nil { + c.facts = make(map[llotypes.ChannelID][]cachedChannelFacts) + } + entries = append(entries, cachedChannelFacts{cd: cloneChannelDefinition(cd), facts: facts}) + if len(entries) > channelFactsRetained { + entries = entries[len(entries)-channelFactsRetained:] + } + c.facts[channelID] = entries +} + +// channelFactsFor returns the facts for cd, decoding its opts only when they are +// not already cached. +func channelFactsFor(cache *ChannelAnalysisCache, codecs map[llotypes.ReportFormat]ReportCodec, channelID llotypes.ChannelID, cd llotypes.ChannelDefinition) channelFacts { + if facts, ok := cache.get(channelID, cd); ok { + return facts + } + facts := deriveChannelFacts(codecs, channelID, cd) + cache.put(channelID, cd, facts) + return facts +} + +// deriveChannelFacts decodes the channel's opts and applies every check that +// needs nothing but this definition. The errors are returned unwrapped: the +// caller wraps them with the channel ID, so that the same facts read the same +// way whichever set they are being reported against. +func deriveChannelFacts(codecs map[llotypes.ReportFormat]ReportCodec, channelID llotypes.ChannelID, cd llotypes.ChannelDefinition) (facts channelFacts) { + if HasCalculatedStreams(cd) { + facts.calculatedIDs, facts.calculatedErr = CalculatedStreamIDs(nil, cd, channelID) + } + + codec, ok := codecs[cd.ReportFormat] + if !ok { + return facts + } + + facts.verifyErr = codec.Verify(cd) + if facts.verifyErr != nil { + // Verify and FeedID are only consulted for definitions Verify accepted. + return facts + } + if av, ok := codec.(AdmissionVerifier); ok { + facts.admissionErr = av.VerifyForAdmission(cd) + } + if feedIDer, ok := codec.(FeedIDer); ok { + facts.feedID, facts.hasFeedID, facts.feedIDErr = feedIDer.FeedID(cd) + } + return facts +} diff --git a/llo/protocol/channel_analysis_cache_test.go b/llo/protocol/channel_analysis_cache_test.go new file mode 100644 index 0000000..014dc8c --- /dev/null +++ b/llo/protocol/channel_analysis_cache_test.go @@ -0,0 +1,272 @@ +package protocol + +import ( + "errors" + "fmt" + "sync" + "sync/atomic" + "testing" + + "github.com/stretchr/testify/require" + + llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" +) + +// countingReportCodec counts how often the per-definition checks are consulted, +// which is what the cache is there to avoid. +type countingReportCodec struct { + verifies atomic.Int64 + admissions atomic.Int64 + feedIDs atomic.Int64 + + verifyErr error + admissionErr error + feedIDErr error +} + +func (c *countingReportCodec) Encode(Report, llotypes.ChannelDefinition, *OptsCache) ([]byte, error) { + return nil, nil +} + +func (c *countingReportCodec) Verify(llotypes.ChannelDefinition) error { + c.verifies.Add(1) + return c.verifyErr +} + +func (c *countingReportCodec) VerifyForAdmission(llotypes.ChannelDefinition) error { + c.admissions.Add(1) + return c.admissionErr +} + +func (c *countingReportCodec) FeedID(cd llotypes.ChannelDefinition) ([32]byte, bool, error) { + c.feedIDs.Add(1) + if c.feedIDErr != nil { + return [32]byte{}, false, c.feedIDErr + } + var feedID [32]byte + copy(feedID[:], cd.Opts) + return feedID, true, nil +} + +func (c *countingReportCodec) calls() (int64, int64, int64) { + return c.verifies.Load(), c.admissions.Load(), c.feedIDs.Load() +} + +func countingCodecs(codec *countingReportCodec) map[llotypes.ReportFormat]ReportCodec { + return map[llotypes.ReportFormat]ReportCodec{llotypes.ReportFormat(0): codec} +} + +func cacheTestDef(opts string) llotypes.ChannelDefinition { + return llotypes.ChannelDefinition{ + Streams: []llotypes.Stream{{StreamID: 1, Aggregator: llotypes.AggregatorMedian}}, + Opts: []byte(opts), + } +} + +func Test_ChannelAnalysisCache(t *testing.T) { + t.Run("an unchanged definition is checked once across analyses", func(t *testing.T) { + codec := &countingReportCodec{} + defs := llotypes.ChannelDefinitions{1: cacheTestDef(`{"a":1}`)} + cache := NewChannelAnalysisCache() + + for i := 0; i < 5; i++ { + require.NoError(t, VerifyChannelDefinitionsWithCache(countingCodecs(codec), defs, cache)) + require.NoError(t, VerifyChannelDefinitionsForAdmissionWithCache(countingCodecs(codec), defs, map[llotypes.ChannelID]struct{}{1: {}}, cache)) + } + + verifies, admissions, feedIDs := codec.calls() + require.Equal(t, int64(1), verifies) + require.Equal(t, int64(1), admissions) + require.Equal(t, int64(1), feedIDs) + }) + + t.Run("a changed definition is checked again", func(t *testing.T) { + codec := &countingReportCodec{} + cache := NewChannelAnalysisCache() + + require.NoError(t, VerifyChannelDefinitionsWithCache(countingCodecs(codec), llotypes.ChannelDefinitions{1: cacheTestDef(`{"a":1}`)}, cache)) + require.NoError(t, VerifyChannelDefinitionsWithCache(countingCodecs(codec), llotypes.ChannelDefinitions{1: cacheTestDef(`{"a":2}`)}, cache)) + + verifies, _, _ := codec.calls() + require.Equal(t, int64(2), verifies) + }) + + t.Run("a definition differing only in streams is checked again", func(t *testing.T) { + codec := &countingReportCodec{} + cache := NewChannelAnalysisCache() + + cd := cacheTestDef(`{"a":1}`) + require.NoError(t, VerifyChannelDefinitionsWithCache(countingCodecs(codec), llotypes.ChannelDefinitions{1: cd}, cache)) + cd.Streams = append(cd.Streams, llotypes.Stream{StreamID: 2, Aggregator: llotypes.AggregatorMedian}) + require.NoError(t, VerifyChannelDefinitionsWithCache(countingCodecs(codec), llotypes.ChannelDefinitions{1: cd}, cache)) + + verifies, _, _ := codec.calls() + require.Equal(t, int64(2), verifies) + }) + + t.Run("the committed and the proposed value of a changing channel are both retained", func(t *testing.T) { + codec := &countingReportCodec{} + cache := NewChannelAnalysisCache() + committed := llotypes.ChannelDefinitions{1: cacheTestDef(`{"a":1}`)} + proposed := llotypes.ChannelDefinitions{1: cacheTestDef(`{"a":2}`)} + + // What a round does while a channel is being changed: verify the + // committed set, then the one an observation advocates, repeatedly. + for i := 0; i < 5; i++ { + require.NoError(t, VerifyChannelDefinitionsWithCache(countingCodecs(codec), committed, cache)) + require.NoError(t, VerifyChannelDefinitionsWithCache(countingCodecs(codec), proposed, cache)) + } + + verifies, _, _ := codec.calls() + require.Equal(t, int64(2), verifies) + }) + + t.Run("caching the entry does not alias the caller's definition", func(t *testing.T) { + codec := &countingReportCodec{} + cache := NewChannelAnalysisCache() + + cd := cacheTestDef(`{"a":1}`) + defs := llotypes.ChannelDefinitions{1: cd} + require.NoError(t, VerifyChannelDefinitionsWithCache(countingCodecs(codec), defs, cache)) + + // Mutating the definition the cache was handed must not rewrite the + // identity of the entry, or a different definition would read as a hit. + cd.Opts[3] = '9' + cd.Streams[0].StreamID = 7 + require.NoError(t, VerifyChannelDefinitionsWithCache(countingCodecs(codec), llotypes.ChannelDefinitions{1: cd}, cache)) + + verifies, _, _ := codec.calls() + require.Equal(t, int64(2), verifies) + }) + + t.Run("Prune drops channels the committed set no longer holds", func(t *testing.T) { + codec := &countingReportCodec{} + cache := NewChannelAnalysisCache() + defs := llotypes.ChannelDefinitions{1: cacheTestDef(`{"a":1}`)} + + require.NoError(t, VerifyChannelDefinitionsWithCache(countingCodecs(codec), defs, cache)) + cache.Prune(llotypes.ChannelDefinitions{}) + require.NoError(t, VerifyChannelDefinitionsWithCache(countingCodecs(codec), defs, cache)) + + verifies, _, _ := codec.calls() + require.Equal(t, int64(2), verifies) + }) + + t.Run("a nil cache memoizes nothing", func(t *testing.T) { + codec := &countingReportCodec{} + defs := llotypes.ChannelDefinitions{1: cacheTestDef(`{"a":1}`)} + + require.NoError(t, VerifyChannelDefinitionsWithCache(countingCodecs(codec), defs, nil)) + require.NoError(t, VerifyChannelDefinitionsWithCache(countingCodecs(codec), defs, nil)) + + verifies, _, _ := codec.calls() + require.Equal(t, int64(2), verifies) + + var nilCache *ChannelAnalysisCache + nilCache.Prune(defs) + require.NoError(t, VerifyChannelDefinitionsWithCache(countingCodecs(codec), defs, nilCache)) + }) +} + +// Test_ChannelAnalysisCache_SameFindings pins the cache to being invisible: the +// error a set produces must not depend on whether it, or an overlapping set, +// was analyzed before. +func Test_ChannelAnalysisCache_SameFindings(t *testing.T) { + sharedFeedID := &countingReportCodec{} + + for _, tc := range []struct { + name string + codec *countingReportCodec + defs llotypes.ChannelDefinitions + }{ + { + name: "valid set", + codec: &countingReportCodec{}, + defs: llotypes.ChannelDefinitions{1: cacheTestDef(`{"a":1}`), 2: cacheTestDef(`{"a":2}`)}, + }, + { + name: "baseline failure", + codec: &countingReportCodec{verifyErr: errors.New("bad opts")}, + defs: llotypes.ChannelDefinitions{1: cacheTestDef(`{"a":1}`), 2: cacheTestDef(`{"a":2}`)}, + }, + { + name: "admission failure", + codec: &countingReportCodec{admissionErr: errors.New("not admissible")}, + defs: llotypes.ChannelDefinitions{1: cacheTestDef(`{"a":1}`), 2: cacheTestDef(`{"a":2}`)}, + }, + { + name: "feed ID failure", + codec: &countingReportCodec{feedIDErr: errors.New("no feed ID")}, + defs: llotypes.ChannelDefinitions{1: cacheTestDef(`{"a":1}`)}, + }, + { + name: "duplicate feed ID across channels", + codec: sharedFeedID, + defs: llotypes.ChannelDefinitions{1: cacheTestDef(`{"a":1}`), 2: cacheTestDef(`{"a":1}`)}, + }, + } { + t.Run(tc.name, func(t *testing.T) { + codecs := countingCodecs(tc.codec) + admitting := map[llotypes.ChannelID]struct{}{1: {}, 2: {}} + + want := VerifyChannelDefinitions(codecs, tc.defs) + wantAdmission := VerifyChannelDefinitionsForAdmission(codecs, tc.defs, admitting) + wantBad, wantSetErr := UnverifiableChannelIDs(codecs, tc.defs) + + cache := NewChannelAnalysisCache() + // Warm the cache the way a round does, through a different entry + // point than the one being compared. + _, _ = UnverifiableChannelIDsWithCache(codecs, tc.defs, cache) + + got := VerifyChannelDefinitionsWithCache(codecs, tc.defs, cache) + gotAdmission := VerifyChannelDefinitionsForAdmissionWithCache(codecs, tc.defs, admitting, cache) + gotBad, gotSetErr := UnverifiableChannelIDsWithCache(codecs, tc.defs, cache) + + requireSameError(t, want, got) + requireSameError(t, wantAdmission, gotAdmission) + requireSameError(t, wantSetErr, gotSetErr) + require.Equal(t, wantBad, gotBad) + }) + } +} + +func requireSameError(t *testing.T, want, got error) { + t.Helper() + if want == nil { + require.NoError(t, got) + return + } + require.EqualError(t, got, want.Error()) +} + +// Test_ChannelAnalysisCache_Concurrent exercises the access pattern of a round: +// Observation warming the cache while several ValidateObservation goroutines +// read it and each advocate a different update. +func Test_ChannelAnalysisCache_Concurrent(t *testing.T) { + codec := &countingReportCodec{} + codecs := countingCodecs(codec) + cache := NewChannelAnalysisCache() + + committed := make(llotypes.ChannelDefinitions, 50) + for i := llotypes.ChannelID(1); i <= 50; i++ { + committed[i] = cacheTestDef(fmt.Sprintf(`{"a":%d}`, i)) + } + want := VerifyChannelDefinitions(codecs, committed) + + var wg sync.WaitGroup + for i := 0; i < 16; i++ { + wg.Add(1) + go func(i int) { + defer wg.Done() + for j := 0; j < 20; j++ { + requireSameError(t, want, VerifyChannelDefinitionsWithCache(codecs, committed, cache)) + + updated := CloneChannelDefinitions(committed) + channelID := llotypes.ChannelID(i%50 + 1) + updated[channelID] = cacheTestDef(fmt.Sprintf(`{"a":%d,"b":%d}`, channelID, j)) + requireSameError(t, VerifyChannelDefinitions(codecs, updated), VerifyChannelDefinitionsWithCache(codecs, updated, cache)) + } + }(i) + } + wg.Wait() +} diff --git a/llo/protocol/channel_cache.go b/llo/protocol/channel_cache.go index 2361d83..aee0aae 100644 --- a/llo/protocol/channel_cache.go +++ b/llo/protocol/channel_cache.go @@ -149,9 +149,15 @@ func (c *ChannelCache) store(gen *ChannelGeneration) *ChannelGeneration { func CloneChannelDefinitions(in llotypes.ChannelDefinitions) llotypes.ChannelDefinitions { out := make(llotypes.ChannelDefinitions, len(in)) for id, cd := range in { - cd.Streams = slices.Clone(cd.Streams) - cd.Opts = slices.Clone(cd.Opts) - out[id] = cd + out[id] = cloneChannelDefinition(cd) } return out } + +// cloneChannelDefinition deep-copies one definition: the Streams slice and the +// raw opts bytes are copied, so the copy shares no memory with the original. +func cloneChannelDefinition(cd llotypes.ChannelDefinition) llotypes.ChannelDefinition { + cd.Streams = slices.Clone(cd.Streams) + cd.Opts = slices.Clone(cd.Opts) + return cd +} diff --git a/llo/protocol/channel_definitions.go b/llo/protocol/channel_definitions.go index 5ef72c4..ceee013 100644 --- a/llo/protocol/channel_definitions.go +++ b/llo/protocol/channel_definitions.go @@ -11,7 +11,13 @@ import ( // VerifyChannelDefinitions applies the checks that any definition set must // satisfy, whether it is being admitted or has already been committed. func VerifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, channelDefs llotypes.ChannelDefinitions) error { - return verifyChannelDefinitions(codecs, channelDefs, nil) + return verifyChannelDefinitions(codecs, channelDefs, nil, nil) +} + +// VerifyChannelDefinitionsWithCache is VerifyChannelDefinitions, memoizing the +// per-definition checks in cache. A nil cache memoizes nothing. +func VerifyChannelDefinitionsWithCache(codecs map[llotypes.ReportFormat]ReportCodec, channelDefs llotypes.ChannelDefinitions, cache *ChannelAnalysisCache) error { + return verifyChannelDefinitions(codecs, channelDefs, nil, cache) } // VerifyChannelDefinitionsForAdmission additionally applies the admission-only @@ -34,7 +40,14 @@ func VerifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, chan // -- not a consensus-critical one, so oracles running different versions of the // admission-only checks disagree only about what they vote for. func VerifyChannelDefinitionsForAdmission(codecs map[llotypes.ReportFormat]ReportCodec, channelDefs llotypes.ChannelDefinitions, admitting map[llotypes.ChannelID]struct{}) error { - return verifyChannelDefinitions(codecs, channelDefs, admitting) + return verifyChannelDefinitions(codecs, channelDefs, admitting, nil) +} + +// VerifyChannelDefinitionsForAdmissionWithCache is +// VerifyChannelDefinitionsForAdmission, memoizing the per-definition checks in +// cache. A nil cache memoizes nothing. +func VerifyChannelDefinitionsForAdmissionWithCache(codecs map[llotypes.ReportFormat]ReportCodec, channelDefs llotypes.ChannelDefinitions, admitting map[llotypes.ChannelID]struct{}, cache *ChannelAnalysisCache) error { + return verifyChannelDefinitions(codecs, channelDefs, admitting, cache) } // ChangedChannelIDs returns the IDs of the channels desired holds that current @@ -100,7 +113,13 @@ func (f admissionFinding) appliesTo(admitting map[llotypes.ChannelID]struct{}) b // get. The returned error carries the whole-set baseline findings, which // implicate no particular channel and so cannot be skipped selectively. func UnverifiableChannelIDs(codecs map[llotypes.ReportFormat]ReportCodec, channelDefs llotypes.ChannelDefinitions) (map[llotypes.ChannelID]struct{}, error) { - res := analyzeChannelDefinitions(codecs, channelDefs) + return UnverifiableChannelIDsWithCache(codecs, channelDefs, nil) +} + +// UnverifiableChannelIDsWithCache is UnverifiableChannelIDs, memoizing the +// per-definition checks in cache. A nil cache memoizes nothing. +func UnverifiableChannelIDsWithCache(codecs map[llotypes.ReportFormat]ReportCodec, channelDefs llotypes.ChannelDefinitions, cache *ChannelAnalysisCache) (map[llotypes.ChannelID]struct{}, error) { + res := analyzeChannelDefinitions(codecs, channelDefs, cache) ids := make(map[llotypes.ChannelID]struct{}, len(res.channelErrs)) for channelID := range res.channelErrs { ids[channelID] = struct{}{} @@ -168,11 +187,16 @@ func (r verifyResult) err(admitting map[llotypes.ChannelID]struct{}) error { return nil } -func verifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, channelDefs llotypes.ChannelDefinitions, admitting map[llotypes.ChannelID]struct{}) error { - return analyzeChannelDefinitions(codecs, channelDefs).err(admitting) +func verifyChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, channelDefs llotypes.ChannelDefinitions, admitting map[llotypes.ChannelID]struct{}, cache *ChannelAnalysisCache) error { + return analyzeChannelDefinitions(codecs, channelDefs, cache).err(admitting) } -func analyzeChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, channelDefs llotypes.ChannelDefinitions) (res verifyResult) { +// analyzeChannelDefinitions applies every check to the set. The checks that +// need nothing but a single definition are taken from cache when it already +// holds them for that exact definition (see ChannelAnalysisCache); everything +// that involves more than one definition is computed here on every call, so +// that a finding never depends on what was analyzed before. +func analyzeChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, channelDefs llotypes.ChannelDefinitions, cache *ChannelAnalysisCache) (res verifyResult) { res.channelErrs = make(map[llotypes.ChannelID]error) if len(channelDefs) > MaxOutcomeChannelDefinitionsLength { @@ -272,12 +296,13 @@ func analyzeChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, cha } } } + facts := channelFactsFor(cache, codecs, channelID, cd) + if HasCalculatedStreams(cd) { - ids, err := CalculatedStreamIDs(nil, cd, channelID) - if err != nil { - admit(fmt.Errorf("invalid ChannelDefinition with ID %d: %w", channelID, err), channelID) + if facts.calculatedErr != nil { + admit(fmt.Errorf("invalid ChannelDefinition with ID %d: %w", channelID, facts.calculatedErr), channelID) } - for _, streamID := range ids { + for _, streamID := range facts.calculatedIDs { if owner, ok := calculatedBy[streamID]; ok { admit(fmt.Errorf("ChannelDefinition with ID %d declares calculated stream %d already declared by channel %d", channelID, streamID, owner), channelID, owner) continue @@ -285,30 +310,21 @@ func analyzeChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, cha calculatedBy[streamID] = channelID } } - var verifyErr error - if codec, ok := codecs[cd.ReportFormat]; ok { - verifyErr = codec.Verify(cd) - if verifyErr != nil { - base(fmt.Errorf("invalid ChannelDefinition with ID %d: %w", channelID, verifyErr), channelID) - } - if av, ok := codec.(AdmissionVerifier); ok && verifyErr == nil { - if err := av.VerifyForAdmission(cd); err != nil { - admit(fmt.Errorf("invalid ChannelDefinition with ID %d: %w", channelID, err), channelID) - } - } + if facts.verifyErr != nil { + base(fmt.Errorf("invalid ChannelDefinition with ID %d: %w", channelID, facts.verifyErr), channelID) } - if feedIDer, ok := codecs[cd.ReportFormat].(FeedIDer); ok && verifyErr == nil { - feedID, hasFeedID, err := feedIDer.FeedID(cd) - switch { - case err != nil: - admit(fmt.Errorf("invalid ChannelDefinition with ID %d: failed to resolve feed ID: %w", channelID, err), channelID) - case !hasFeedID: - default: - if owner, ok := feedIDBy[feedID]; ok { - admit(fmt.Errorf("ChannelDefinition with ID %d has feed ID 0x%x already used by channel %d", channelID, feedID, owner), channelID, owner) - } else { - feedIDBy[feedID] = channelID - } + if facts.admissionErr != nil { + admit(fmt.Errorf("invalid ChannelDefinition with ID %d: %w", channelID, facts.admissionErr), channelID) + } + switch { + case facts.feedIDErr != nil: + admit(fmt.Errorf("invalid ChannelDefinition with ID %d: failed to resolve feed ID: %w", channelID, facts.feedIDErr), channelID) + case !facts.hasFeedID: + default: + if owner, ok := feedIDBy[facts.feedID]; ok { + admit(fmt.Errorf("ChannelDefinition with ID %d has feed ID 0x%x already used by channel %d", channelID, facts.feedID, owner), channelID, owner) + } else { + feedIDBy[facts.feedID] = channelID } } if cd.ReportFormat == llotypes.ReportFormatHistoryBackfill { @@ -319,9 +335,6 @@ func analyzeChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, cha admit(fmt.Errorf("invalid history backfill channel %d: %w", channelID, err), channelID) } } - if verifyErr != nil { - continue - } } // A calculated stream that shares its ID with an observed stream would write diff --git a/llo/protocol/channel_definitions_bench_test.go b/llo/protocol/channel_definitions_bench_test.go new file mode 100644 index 0000000..0de9e9e --- /dev/null +++ b/llo/protocol/channel_definitions_bench_test.go @@ -0,0 +1,183 @@ +package protocol + +import ( + "encoding/json" + "fmt" + "testing" + + llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" +) + +// benchOpts mirrors the shape of a real channel's opts: a feed ID plus one ABI +// entry per stream. Decoding it is what verification actually spends its time +// on. +type benchOpts struct { + FeedID *[32]byte `json:"feedID"` + ABI []benchABI `json:"abi"` + BaseUSDFee string `json:"baseUSDFee"` + ExpirationWindow uint32 `json:"expirationWindow"` +} + +type benchABI struct { + StreamID uint32 `json:"streamID"` + Multiplier string `json:"multiplier"` + Type string `json:"type"` +} + +// benchReportCodec decodes the opts on every call, the way every real codec +// does (see reportcodec/evm). +type benchReportCodec struct{} + +func (benchReportCodec) Encode(Report, llotypes.ChannelDefinition, *OptsCache) ([]byte, error) { + return nil, nil +} + +func (benchReportCodec) Verify(cd llotypes.ChannelDefinition) error { + var o benchOpts + if err := json.Unmarshal(cd.Opts, &o); err != nil { + return fmt.Errorf("failed to decode opts: %w", err) + } + if len(o.ABI) != len(cd.Streams) { + return fmt.Errorf("ABI length mismatch; expected: %d, got: %d", len(cd.Streams), len(o.ABI)) + } + return nil +} + +func (benchReportCodec) FeedID(cd llotypes.ChannelDefinition) ([32]byte, bool, error) { + var o benchOpts + if err := json.Unmarshal(cd.Opts, &o); err != nil { + return [32]byte{}, false, fmt.Errorf("failed to decode opts: %w", err) + } + if o.FeedID == nil { + return [32]byte{}, false, nil + } + return *o.FeedID, true, nil +} + +// benchChannelDefs builds n channels of three streams each, with distinct feed +// IDs and roughly 250 bytes of opts apiece. +func benchChannelDefs(tb testing.TB, n int) llotypes.ChannelDefinitions { + tb.Helper() + defs := make(llotypes.ChannelDefinitions, n) + for i := 0; i < n; i++ { + channelID := llotypes.ChannelID(i + 1) + var feedID [32]byte + feedID[0], feedID[1], feedID[2], feedID[3] = byte(i), byte(i>>8), byte(i>>16), byte(i>>24) + + streams := make([]llotypes.Stream, 0, 3) + abi := make([]benchABI, 0, 3) + for j := 0; j < 3; j++ { + streamID := llotypes.StreamID(channelID*10 + llotypes.StreamID(j)) + streams = append(streams, llotypes.Stream{StreamID: streamID, Aggregator: llotypes.AggregatorMedian}) + abi = append(abi, benchABI{StreamID: streamID, Multiplier: "100000000000000000000000000", Type: "int192"}) + } + opts, err := json.Marshal(benchOpts{ + FeedID: &feedID, + ABI: abi, + BaseUSDFee: "0.1", + ExpirationWindow: 86400, + }) + if err != nil { + tb.Fatal(err) + } + defs[channelID] = llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatEVMStreamlined, + Streams: streams, + Opts: opts, + } + } + return defs +} + +func benchCodecs() map[llotypes.ReportFormat]ReportCodec { + return map[llotypes.ReportFormat]ReportCodec{ + llotypes.ReportFormatEVMStreamlined: benchReportCodec{}, + } +} + +// BenchmarkVerifyChannelDefinitions measures one analysis of an unchanged set, +// which is what every round pays several times over. +func BenchmarkVerifyChannelDefinitions(b *testing.B) { + codecs := benchCodecs() + defs := benchChannelDefs(b, 2000) + + b.Run("uncached", func(b *testing.B) { + for b.Loop() { + if err := VerifyChannelDefinitions(codecs, defs); err != nil { + b.Fatal(err) + } + } + }) + + b.Run("cached", func(b *testing.B) { + cache := NewChannelAnalysisCache() + if err := VerifyChannelDefinitionsWithCache(codecs, defs, cache); err != nil { + b.Fatal(err) + } + for b.Loop() { + if err := VerifyChannelDefinitionsWithCache(codecs, defs, cache); err != nil { + b.Fatal(err) + } + } + }) +} + +// BenchmarkObservationChannelVerification measures what Observation does per +// round: attribute the baseline findings of the committed set, then verify the +// desired set for admission. +func BenchmarkObservationChannelVerification(b *testing.B) { + codecs := benchCodecs() + committed := benchChannelDefs(b, 2000) + desired := CloneChannelDefinitions(committed) + + run := func(b *testing.B, cache *ChannelAnalysisCache) { + for b.Loop() { + if _, err := UnverifiableChannelIDsWithCache(codecs, committed, cache); err != nil { + b.Fatal(err) + } + admitting := ChangedChannelIDs(committed, desired) + if err := VerifyChannelDefinitionsForAdmissionWithCache(codecs, desired, admitting, cache); err != nil { + b.Fatal(err) + } + } + } + + b.Run("uncached", func(b *testing.B) { run(b, nil) }) + b.Run("cached", func(b *testing.B) { + cache := NewChannelAnalysisCache() + run(b, cache) + }) +} + +// BenchmarkValidateObservationChannelVerification measures the verification an +// update-carrying observation triggers, with the cache warmed by the committed +// set the way Observation warms it. +func BenchmarkValidateObservationChannelVerification(b *testing.B) { + codecs := benchCodecs() + committed := benchChannelDefs(b, 2000) + + // One channel's opts are changed, which is what an observation that votes an + // update advocates: the committed set with that update applied. + updated := CloneChannelDefinitions(committed) + cd := updated[1] + cd.Opts = append([]byte(nil), committed[2].Opts...) + updated[1] = cd + + b.Run("uncached", func(b *testing.B) { + for b.Loop() { + // Errors are expected here (the update duplicates a feed ID); the + // cost is what is being measured. + _ = VerifyChannelDefinitions(codecs, updated) + } + }) + + b.Run("cached", func(b *testing.B) { + cache := NewChannelAnalysisCache() + if _, err := UnverifiableChannelIDsWithCache(codecs, committed, cache); err != nil { + b.Fatal(err) + } + for b.Loop() { + _ = VerifyChannelDefinitionsWithCache(codecs, updated, cache) + } + }) +} From c116f939a25d87a12492005bf672d61ea19491b1 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 10:25:59 +0100 Subject: [PATCH 16/40] SPOR-0019 llo/dev/v31: cut repeated and serialized blob work Memoize blob payloads per round, scoped to the sequence number, so a handle referenced by an observation is fetched, decompressed and unmarshaled once for ValidateObservation and StateTransition together instead of once per phase. Retry a refused blob broadcast inside the pump cycle, bounded by BlobBroadcastAttempts and by the cycle observation timeout, so stream values that were gathered fine are not thrown away by one transport failure. --- llo/dev/v31/blobmemo.go | 92 ++++++++++++ llo/dev/v31/blobmemo_test.go | 147 ++++++++++++++++++++ llo/dev/v31/blobpump.go | 56 +++++++- llo/dev/v31/blobpump_test.go | 90 ++++++++++++ llo/dev/v31/doc.go | 6 +- llo/dev/v31/factory.go | 1 + llo/dev/v31/flow_test.go | 4 +- llo/dev/v31/observation.go | 83 +++++++---- llo/dev/v31/observation_coefficient_test.go | 4 +- llo/dev/v31/plugin.go | 8 +- llo/dev/v31/plugin_test.go | 30 ++-- llo/dev/v31/statetransition.go | 25 +++- 12 files changed, 491 insertions(+), 55 deletions(-) create mode 100644 llo/dev/v31/blobmemo.go create mode 100644 llo/dev/v31/blobmemo_test.go diff --git a/llo/dev/v31/blobmemo.go b/llo/dev/v31/blobmemo.go new file mode 100644 index 0000000..9f0cea9 --- /dev/null +++ b/llo/dev/v31/blobmemo.go @@ -0,0 +1,92 @@ +package llo + +import ( + "sync" + + protocol "github.com/smartcontractkit/chainlink-data-streams/llo/protocol" +) + +// blobPayloadCache memoizes the stream values decoded from blob payloads within +// one sequence number. ValidateObservation and StateTransition decode the same +// observations in the same round, and every FetchBlob re-verifies the blob's +// certificate and re-reads its payload before the plugin decompresses and +// unmarshals it again, so the second decode is entirely repeated work. +// +// Entries are keyed by the marshaled blob handle and scoped to a single +// sequence number. That scope is what keeps the memo consistent with the blob +// transport, which refuses a handle that has expired as of the round's sequence +// number: a hit can never resurrect a blob the round itself would have +// rejected. What one round can hold is bounded by what its observations may +// reference, at most maxObservationBlobHandles handles per observation, each +// contributing at most maxObservationDecompressedBytes, and the whole map is +// dropped when the sequence number advances. +// +// Decoded stream values are treated as immutable: a hit copies the map entries +// into the observation rather than handing out the memoized map. +type blobPayloadCache struct { + mu sync.Mutex + seqNr uint64 + entries map[string]blobPayloadEntry +} + +// blobPayloadEntry is one memoized blob payload. size is the decompressed byte +// count, memoized alongside the values because it is charged against the +// observation's decompression budget: a hit must consume exactly what the +// original decode consumed, or the same observation would decode differently on +// a hit than on a miss. +type blobPayloadEntry struct { + values protocol.StreamValues + size int +} + +func newBlobPayloadCache() *blobPayloadCache { + return &blobPayloadCache{} +} + +// round returns the memo scoped to seqNr, discarding whatever was held for an +// earlier one. Returns nil for a nil cache, which disables memoization. +func (c *blobPayloadCache) round(seqNr uint64) *roundBlobPayloads { + if c == nil { + return nil + } + c.mu.Lock() + defer c.mu.Unlock() + if c.seqNr != seqNr || c.entries == nil { + c.seqNr = seqNr + c.entries = make(map[string]blobPayloadEntry) + } + return &roundBlobPayloads{cache: c, seqNr: seqNr} +} + +// roundBlobPayloads is a handle on the memo for one sequence number. Reads and +// writes through a handle whose round has been superseded are dropped, so a +// round can never see another round's payloads. +type roundBlobPayloads struct { + cache *blobPayloadCache + seqNr uint64 +} + +func (r *roundBlobPayloads) get(handle []byte) (blobPayloadEntry, bool) { + if r == nil { + return blobPayloadEntry{}, false + } + r.cache.mu.Lock() + defer r.cache.mu.Unlock() + if r.cache.seqNr != r.seqNr { + return blobPayloadEntry{}, false + } + entry, ok := r.cache.entries[string(handle)] + return entry, ok +} + +func (r *roundBlobPayloads) put(handle []byte, entry blobPayloadEntry) { + if r == nil { + return + } + r.cache.mu.Lock() + defer r.cache.mu.Unlock() + if r.cache.seqNr != r.seqNr { + return + } + r.cache.entries[string(handle)] = entry +} diff --git a/llo/dev/v31/blobmemo_test.go b/llo/dev/v31/blobmemo_test.go new file mode 100644 index 0000000..b306df5 --- /dev/null +++ b/llo/dev/v31/blobmemo_test.go @@ -0,0 +1,147 @@ +package llo + +import ( + "testing" + + "github.com/shopspring/decimal" + "github.com/stretchr/testify/require" + + llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" + "github.com/smartcontractkit/chainlink-common/pkg/utils/tests" + + "github.com/smartcontractkit/chainlink-data-streams/llo/dev/v31/llotest" + protocol "github.com/smartcontractkit/chainlink-data-streams/llo/protocol" + + "github.com/smartcontractkit/libocr/offchainreporting2plus/ocr3_1types" + ocrtypes "github.com/smartcontractkit/libocr/offchainreporting2plus/types" +) + +// blobObservation broadcasts sv as a blob through bc and returns an observation +// referencing the resulting handle. +func blobObservation(t *testing.T, bc *llotest.BlobBroadcastFetcher, sv protocol.StreamValues) ocrtypes.Observation { + t.Helper() + payload, err := marshalStreamValues(sv) + require.NoError(t, err) + handle, err := bc.BroadcastBlob(tests.Context(t), payload, ocr3_1types.BlobExpirationHintSequenceNumber{SeqNr: 100}) + require.NoError(t, err) + handleBytes, err := handle.MarshalBinary() + require.NoError(t, err) + enc, err := encodeObservation(Observation{UnixTimestampNanoseconds: 1}, [][]byte{handleBytes}) + require.NoError(t, err) + return enc +} + +func testStreamValues(n int) protocol.StreamValues { + sv := protocol.StreamValues{} + for i := 0; i < n; i++ { + sv[llotypes.StreamID(i)] = protocol.ToDecimal(decimal.NewFromInt(int64(i))) + } + return sv +} + +// Test_BlobPayloadCache_SkipsSecondFetch covers the reason the memo exists: the +// same handle decoded twice in one round fetches once. +func Test_BlobPayloadCache_SkipsSecondFetch(t *testing.T) { + ctx := tests.Context(t) + bc := newFakeBroadcaster() + sv := testStreamValues(50) + enc := blobObservation(t, bc, sv) + + cache := newBlobPayloadCache() + + first, err := decodeObservation(ctx, enc, bc, cache.round(7)) + require.NoError(t, err) + require.Equal(t, 1, bc.Fetches()) + + second, err := decodeObservation(ctx, enc, bc, cache.round(7)) + require.NoError(t, err) + require.Equal(t, 1, bc.Fetches(), "the second decode of the round must be served from the memo") + + // A hit and a miss must produce the same observation, or the round's outcome + // would depend on whether the memo was populated. + require.Len(t, second.StreamValues, len(first.StreamValues)) + for id, want := range first.StreamValues { + require.True(t, equalStreamValue(want, second.StreamValues[id]), "stream %d", id) + } + + // Without a memo the fetch happens again, which is what the memo replaces. + _, err = decodeObservation(ctx, enc, bc, nil) + require.NoError(t, err) + require.Equal(t, 2, bc.Fetches()) +} + +// Test_BlobPayloadCache_ScopedToSeqNr asserts entries do not survive the round +// they were decoded for: the blob transport gates fetches on the round's +// sequence number, and the memo must not reach around that. +func Test_BlobPayloadCache_ScopedToSeqNr(t *testing.T) { + ctx := tests.Context(t) + bc := newFakeBroadcaster() + enc := blobObservation(t, bc, testStreamValues(10)) + + cache := newBlobPayloadCache() + _, err := decodeObservation(ctx, enc, bc, cache.round(7)) + require.NoError(t, err) + require.Equal(t, 1, bc.Fetches()) + + _, err = decodeObservation(ctx, enc, bc, cache.round(8)) + require.NoError(t, err) + require.Equal(t, 2, bc.Fetches(), "a later round must not be served the previous round's payload") + + // The superseded handle is inert rather than a window back into round 7. + stale := cache.round(7) + require.NotNil(t, cache.round(9)) + _, ok := stale.get([]byte("anything")) + require.False(t, ok) + stale.put([]byte("anything"), blobPayloadEntry{}) + _, ok = cache.round(9).get([]byte("anything")) + require.False(t, ok) +} + +// Test_BlobPayloadCache_FailuresNotMemoized covers the two decode failures: a +// fetch failure is node-local and must be retried, and a malformed payload is +// deterministic so nothing is gained by remembering it. +func Test_BlobPayloadCache_FailuresNotMemoized(t *testing.T) { + ctx := tests.Context(t) + bc := newFakeBroadcaster() + enc := blobObservation(t, bc, testStreamValues(10)) + cache := newBlobPayloadCache() + + // No fetcher: a reference that could not be resolved must not be cached as + // an absence, so a later decode in the same round still fetches. + _, err := decodeObservation(ctx, enc, nil, cache.round(7)) + var bfErr *blobFetchError + require.ErrorAs(t, err, &bfErr) + + _, err = decodeObservation(ctx, enc, bc, cache.round(7)) + require.NoError(t, err) + require.Equal(t, 1, bc.Fetches()) +} + +// Test_BlobPayloadCache_BudgetChargedOnHit asserts a hit consumes the same +// decompression budget the miss did, so an observation naming one handle +// repeatedly is bounded identically either way. +func Test_BlobPayloadCache_BudgetChargedOnHit(t *testing.T) { + ctx := tests.Context(t) + bc := newFakeBroadcaster() + sv := testStreamValues(20) + payload, err := marshalStreamValues(sv) + require.NoError(t, err) + handle, err := bc.BroadcastBlob(ctx, payload, ocr3_1types.BlobExpirationHintSequenceNumber{SeqNr: 100}) + require.NoError(t, err) + handleBytes, err := handle.MarshalBinary() + require.NoError(t, err) + + // The same handle four times, so the budget is charged four times. + enc, err := encodeObservation(Observation{UnixTimestampNanoseconds: 1}, [][]byte{handleBytes, handleBytes, handleBytes, handleBytes}) + require.NoError(t, err) + + withMemo, memoErr := decodeObservation(ctx, enc, bc, newBlobPayloadCache().round(7)) + require.Equal(t, 1, bc.Fetches(), "repeats within one observation are memo hits") + withoutMemo, plainErr := decodeObservation(ctx, enc, bc, nil) + + // Whatever the budget verdict is, it must be the same with and without the + // memo; the payloads here are small, so both are expected to succeed. + require.Equal(t, plainErr == nil, memoErr == nil) + require.NoError(t, memoErr) + require.Len(t, withMemo.StreamValues, len(withoutMemo.StreamValues)) +} diff --git a/llo/dev/v31/blobpump.go b/llo/dev/v31/blobpump.go index 6f95a02..ee2df25 100644 --- a/llo/dev/v31/blobpump.go +++ b/llo/dev/v31/blobpump.go @@ -59,6 +59,13 @@ const ( // roughly one blob per round, so the unexpired-blob budget declared to // libocr grows with the lifetime; this keeps that budget sane. MaxBlobLifetimeRounds = 64 + // BlobBroadcastAttempts is how many times one cycle tries to broadcast the + // payload it gathered. + BlobBroadcastAttempts = 3 + // BlobBroadcastRetryBackoff is the wait before the second broadcast attempt, + // doubled for each attempt after it. All attempts share the cycle's + // observation timeout, so this cannot extend how long a cycle runs. + BlobBroadcastRetryBackoff = 50 * time.Millisecond // MissStreakLogThreshold is how many consecutive rounds may find no usable // snapshot before the pump escalates from debug to error logging. A node // that never contributes stream values is a silent failure otherwise. @@ -413,10 +420,9 @@ func (p *blobPump) observe(in pumpInput) (*blobSnapshot, error) { } usableBefore := in.seqNr + p.maxSnapshotRounds - expiresAt := in.seqNr + p.blobLifetimeRounds - handle, err := p.bbf.BroadcastBlob(ctx, payload, ocr3_1types.BlobExpirationHintSequenceNumber{SeqNr: expiresAt}) + handle, expiresAt, err := p.broadcast(ctx, payload, in.seqNr) if err != nil { - return nil, fmt.Errorf("BroadcastBlob error: %w", err) + return nil, err } handleBytes, err := handle.MarshalBinary() if err != nil { @@ -433,6 +439,50 @@ func (p *blobPump) observe(in pumpInput) (*blobSnapshot, error) { }, nil } +// broadcast hands the payload to the blob transport, retrying a failed +// broadcast within the cycle's own timeout, and returns the handle together +// with the expiration hint it was broadcast under. +// +// forSeqNr is the round the values were gathered for. The expiration hint is +// recomputed per attempt from the latest round published by Observation, so a +// retry that lands rounds later does not hand peers a blob that expires as of +// a sequence number already behind them. The hint only moves forward, and it +// bounds fetchability, not staleness: how stale the values themselves may be is +// still decided locally by usableBefore. +func (p *blobPump) broadcast(ctx context.Context, payload []byte, forSeqNr uint64) (ocr3_1types.BlobHandle, uint64, error) { + var lastErr error + for attempt := 1; attempt <= BlobBroadcastAttempts; attempt++ { + expiresAt := max(forSeqNr, p.latestSeqNr()) + p.blobLifetimeRounds + handle, err := p.bbf.BroadcastBlob(ctx, payload, ocr3_1types.BlobExpirationHintSequenceNumber{SeqNr: expiresAt}) + if err == nil { + if attempt > 1 { + p.lggr.Infow("Blob broadcast succeeded after retry", "attempt", attempt, "forSeqNr", forSeqNr, "expiresAt", expiresAt) + } + return handle, expiresAt, nil + } + lastErr = err + if attempt == BlobBroadcastAttempts { + break + } + backoff := BlobBroadcastRetryBackoff << (attempt - 1) + p.lggr.Warnw("Blob broadcast failed; retrying within this cycle", "attempt", attempt, "attempts", BlobBroadcastAttempts, "backoff", backoff, "forSeqNr", forSeqNr, "err", err) + select { + case <-ctx.Done(): + return ocr3_1types.BlobHandle{}, 0, fmt.Errorf("BroadcastBlob error: %w (last attempt: %w)", ctx.Err(), lastErr) + case <-time.After(backoff): + } + } + return ocr3_1types.BlobHandle{}, 0, fmt.Errorf("BroadcastBlob error after %d attempts: %w", BlobBroadcastAttempts, lastErr) +} + +// latestSeqNr is the most recent round Observation published, which may be +// ahead of the round a cycle started for. +func (p *blobPump) latestSeqNr() uint64 { + p.mu.Lock() + defer p.mu.Unlock() + return p.input.seqNr +} + // marshalStreamValues serializes stream values into the stream-values-only // proto that is carried by a blob, framed by encodeBlobPayload (which // compresses it when that shrinks the payload). Returns nil when nothing was diff --git a/llo/dev/v31/blobpump_test.go b/llo/dev/v31/blobpump_test.go index abc5129..bbfb78b 100644 --- a/llo/dev/v31/blobpump_test.go +++ b/llo/dev/v31/blobpump_test.go @@ -371,3 +371,93 @@ func Test_blobPump_CloseDoesNotHangOnStuckDataSource(t *testing.T) { t.Fatal("Close hung on a DataSource that ignores context") } } + +// flakyBroadcaster fails the first failures broadcasts and then delegates. It +// also runs onAttempt before each call, so a test can move the round forward +// between attempts. +type flakyBroadcaster struct { + *llotest.BlobBroadcastFetcher + failures atomic.Int64 + onAttempt func(attempt int) + attempts atomic.Int64 +} + +func (f *flakyBroadcaster) BroadcastBlob(ctx context.Context, payload []byte, hint ocr3_1types.BlobExpirationHint) (ocr3_1types.BlobHandle, error) { + attempt := int(f.attempts.Add(1)) + if f.onAttempt != nil { + f.onAttempt(attempt) + } + if f.failures.Add(-1) >= 0 { + return ocr3_1types.BlobHandle{}, errors.New("broadcast unavailable") + } + return f.BlobBroadcastFetcher.BroadcastBlob(ctx, payload, hint) +} + +// Test_blobPump_RetriesFailedBroadcast covers the recovery this retry exists +// for: values gathered fine are not thrown away because the transport refused +// the first broadcast. +func Test_blobPump_RetriesFailedBroadcast(t *testing.T) { + ds := mockDS() + bc := &flakyBroadcaster{BlobBroadcastFetcher: newFakeBroadcaster()} + bc.failures.Store(1) + + p := testPump(t, ds, bc, time.Minute) + p.SetInput(pumpInputFor(2)) + _, _ = p.Take(2) + + require.Eventually(t, func() bool { return p.Cycles() >= 1 }, tests.WaitTimeout(t), 10*time.Millisecond) + require.EqualValues(t, 2, bc.attempts.Load(), "the failed attempt must be retried, once") + + snap, reason := p.Take(3) + require.NotNil(t, snap, reason) + require.EqualValues(t, 2, snap.forSeqNr, "the retry does not change the round the values were gathered for") +} + +// Test_blobPump_BroadcastRetryRefreshesExpiry asserts a retry that lands after +// the round moved on hands peers a hint derived from the current round, not +// from the round the values were gathered for. +func Test_blobPump_BroadcastRetryRefreshesExpiry(t *testing.T) { + ds := mockDS() + inner := newFakeBroadcaster() + bc := &flakyBroadcaster{BlobBroadcastFetcher: inner} + bc.failures.Store(1) + + var p *blobPump + bc.onAttempt = func(attempt int) { + if attempt == 1 { + // Rounds advanced while the first attempt was failing. + p.SetInput(pumpInputFor(5)) + } + } + p = testPump(t, ds, bc, time.Minute) + p.SetInput(pumpInputFor(2)) + _, _ = p.Take(2) + + require.Eventually(t, func() bool { return p.Cycles() >= 1 }, tests.WaitTimeout(t), 10*time.Millisecond) + hints := inner.Hints() + require.Len(t, hints, 1) + require.Equal(t, ocr3_1types.BlobExpirationHintSequenceNumber{SeqNr: 5 + DefaultBlobLifetimeRounds}, hints[0]) + + // Fetchability moved forward; local freshness did not. + snap, reason := p.Take(3) + require.NotNil(t, snap, reason) + require.EqualValues(t, 2+DefaultMaxSnapshotRounds, snap.usableBefore) + require.EqualValues(t, 5+DefaultBlobLifetimeRounds, snap.expiresAt) +} + +// Test_blobPump_BroadcastRetriesAreBounded asserts a transport that stays down +// costs a bounded number of attempts and parks nothing. +func Test_blobPump_BroadcastRetriesAreBounded(t *testing.T) { + ds := mockDS() + bc := &flakyBroadcaster{BlobBroadcastFetcher: newFakeBroadcaster()} + bc.failures.Store(1 << 30) + + p := testPump(t, ds, bc, time.Minute) + p.SetInput(pumpInputFor(2)) + _, _ = p.Take(2) + + require.Eventually(t, func() bool { return bc.attempts.Load() >= BlobBroadcastAttempts }, tests.WaitTimeout(t), 10*time.Millisecond) + require.Zero(t, p.Cycles()) + snap, _ := p.Take(3) + require.Nil(t, snap) +} diff --git a/llo/dev/v31/doc.go b/llo/dev/v31/doc.go index 4a36464..c7a1cc0 100644 --- a/llo/dev/v31/doc.go +++ b/llo/dev/v31/doc.go @@ -101,7 +101,11 @@ // BlobLifetimeRounds is remote: it is the expiration hint given to the blob // transport (forSeqNr + BlobLifetimeRounds), deciding how long peers can still // fetch the blob, and sits BlobFetchMarginRounds beyond the last seqNr at which -// the handle can be referenced. A wall-clock age check derived from the +// the handle can be referenced. A broadcast that the transport refuses is +// retried inside the cycle (BlobBroadcastAttempts), which is what keeps the +// round trip off the OCR critical path; a retry recomputes the hint from the +// round current at that attempt, so the values stay bounded by MaxSnapshotRounds +// while the blob stays fetchable for the round that will reference it. A wall-clock age check derived from the // measured round period guards against jitter on top. A round that finds // nothing usable — cold start, a failed cycle, or a stale snapshot — emits an // observation with no stream values. That is not a halt: quorum counts diff --git a/llo/dev/v31/factory.go b/llo/dev/v31/factory.go index 0c43ed2..9de991d 100644 --- a/llo/dev/v31/factory.go +++ b/llo/dev/v31/factory.go @@ -140,6 +140,7 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re // immutable generation per c/seqnr, so a round can never mix the two. p.ChannelCache = protocol.NewChannelCache() p.ChannelAnalysisCache = protocol.NewChannelAnalysisCache() + p.BlobPayloads = newBlobPayloadCache() // Setup the blobpump p.pump = newBlobPump(l, blobPumpParams{ diff --git a/llo/dev/v31/flow_test.go b/llo/dev/v31/flow_test.go index 2d487db..96c4566 100644 --- a/llo/dev/v31/flow_test.go +++ b/llo/dev/v31/flow_test.go @@ -249,7 +249,7 @@ func Test_Observation_And_Validate_Flow(t *testing.T) { first, err := p.Observation(ctx, 3, ocrtypes.AttributedQuery{}, kv, nil) require.NoError(t, err) require.NotEmpty(t, first) - decodedFirst, err := decodeObservation(ctx, first, bc) + decodedFirst, err := decodeObservation(ctx, first, bc, nil) require.NoError(t, err) require.Empty(t, decodedFirst.StreamValues) @@ -260,7 +260,7 @@ func Test_Observation_And_Validate_Flow(t *testing.T) { obsBytes, err := p.Observation(ctx, 4, ocrtypes.AttributedQuery{}, kv, nil) require.NoError(t, err) require.NotEmpty(t, obsBytes) - decoded, err := decodeObservation(ctx, obsBytes, bc) + decoded, err := decodeObservation(ctx, obsBytes, bc, nil) require.NoError(t, err) require.Contains(t, decoded.StreamValues, llotypes.StreamID(100)) diff --git a/llo/dev/v31/observation.go b/llo/dev/v31/observation.go index 1aa2b45..81cb16a 100644 --- a/llo/dev/v31/observation.go +++ b/llo/dev/v31/observation.go @@ -116,7 +116,11 @@ func (e *blobFetchError) Unwrap() error { return e.err } // decodeObservation reverses encodeObservation, fetching any referenced blobs. // Observations carrying inline stream values are rejected: v31 disseminates // values exclusively via blobs. -func decodeObservation(ctx context.Context, raw ocrtypes.Observation, bf ocr3_1types.BlobFetcher) (Observation, error) { +// +// memo, when non-nil, memoizes decoded blob payloads for the round so the same +// handle is fetched and decompressed once instead of once per plugin phase. It +// changes cost only: an observation decodes identically on a hit and on a miss. +func decodeObservation(ctx context.Context, raw ocrtypes.Observation, bf ocr3_1types.BlobFetcher, memo *roundBlobPayloads) (Observation, error) { if len(raw) == 0 { return Observation{}, nil } @@ -133,7 +137,13 @@ func decodeObservation(ctx context.Context, raw ocrtypes.Observation, bf ocr3_1t } rest = rest[k:] - handles := make([]ocr3_1types.BlobHandle, 0, nHandles) + // The marshaled bytes are kept alongside the decoded handle: they are the + // memo key, and re-marshaling to obtain it would be wasted work. + type blobRef struct { + handle ocr3_1types.BlobHandle + key []byte + } + handles := make([]blobRef, 0, nHandles) for i := uint64(0); i < nHandles; i++ { l, k2 := binary.Uvarint(rest) if k2 <= 0 || uint64(len(rest[k2:])) < l { @@ -144,7 +154,7 @@ func decodeObservation(ctx context.Context, raw ocrtypes.Observation, bf ocr3_1t if err := h.UnmarshalBinary(rest[:l]); err != nil { return Observation{}, fmt.Errorf("unmarshal blob handle: %w", err) } - handles = append(handles, h) + handles = append(handles, blobRef{handle: h, key: rest[:l]}) rest = rest[l:] } @@ -163,33 +173,52 @@ func decodeObservation(ctx context.Context, raw ocrtypes.Observation, bf ocr3_1t // naming several blobs. budget := maxObservationDecompressedBytes for _, h := range handles { - if bf == nil { - return Observation{}, &blobFetchError{fmt.Errorf("observation references a blob but no fetcher was provided")} - } - payload, ferr := bf.FetchBlob(ctx, h) - if ferr != nil { - return Observation{}, &blobFetchError{fmt.Errorf("fetch blob: %w", ferr)} - } - // Framing/codec faults are deterministic across oracles (every one sees - // the same bytes), so they stay plain errors and drop this observation - // alone, unlike the fetch failure above. - raw, err := decodeBlobPayload(payload, budget) - if err != nil { - return Observation{}, err - } - budget -= len(raw) - chunk := &protocol.LLOObservationProto{} - if err := proto.Unmarshal(raw, chunk); err != nil { - return Observation{}, fmt.Errorf("unmarshal blob payload: %w", err) - } - if obs.StreamValues == nil { - obs.StreamValues = make(protocol.StreamValues, len(chunk.StreamValues)) - } - for id, pbSv := range chunk.StreamValues { - sv, err := streamValueFromProtoAllowNil(pbSv) + entry, ok := memo.get(h.key) + if ok { + // A hit is charged the same decompressed size the miss was, so a + // handle named twice exhausts the budget exactly as before. + if entry.size > budget { + return Observation{}, fmt.Errorf("decompressed blob payload too large: %d > %d bytes", entry.size, budget) + } + } else { + if bf == nil { + return Observation{}, &blobFetchError{fmt.Errorf("observation references a blob but no fetcher was provided")} + } + payload, ferr := bf.FetchBlob(ctx, h.handle) + if ferr != nil { + return Observation{}, &blobFetchError{fmt.Errorf("fetch blob: %w", ferr)} + } + // Framing/codec faults are deterministic across oracles (every one sees + // the same bytes), so they stay plain errors and drop this observation + // alone, unlike the fetch failure above. + raw, err := decodeBlobPayload(payload, budget) if err != nil { return Observation{}, err } + chunk := &protocol.LLOObservationProto{} + if err := proto.Unmarshal(raw, chunk); err != nil { + return Observation{}, fmt.Errorf("unmarshal blob payload: %w", err) + } + values := make(protocol.StreamValues, len(chunk.StreamValues)) + for id, pbSv := range chunk.StreamValues { + sv, err := streamValueFromProtoAllowNil(pbSv) + if err != nil { + return Observation{}, err + } + values[id] = sv + } + // Only a payload that decoded cleanly is memoized. A fetch failure is + // node-local and transient, and a decode failure is deterministic and + // recomputed for free, so neither is worth remembering. + entry = blobPayloadEntry{values: values, size: len(raw)} + memo.put(h.key, entry) + } + + budget -= entry.size + if obs.StreamValues == nil { + obs.StreamValues = make(protocol.StreamValues, len(entry.values)) + } + for id, sv := range entry.values { obs.StreamValues[id] = sv } } diff --git a/llo/dev/v31/observation_coefficient_test.go b/llo/dev/v31/observation_coefficient_test.go index 7e9525b..7544418 100644 --- a/llo/dev/v31/observation_coefficient_test.go +++ b/llo/dev/v31/observation_coefficient_test.go @@ -35,11 +35,11 @@ func Test_decodeObservation_CoefficientBound(t *testing.T) { }) } - obs, err := decodeObservation(ctx, encode(t, atLimit), testBlobs) + obs, err := decodeObservation(ctx, encode(t, atLimit), testBlobs, nil) require.NoError(t, err) require.Len(t, obs.StreamValues, 1) - _, err = decodeObservation(ctx, encode(t, overSized), testBlobs) + _, err = decodeObservation(ctx, encode(t, overSized), testBlobs, nil) require.ErrorIs(t, err, protocol.ErrDecimalCoefficientOutOfRange) var bfErr *blobFetchError diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index 97d5b2a..dc9f565 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -60,6 +60,12 @@ type Plugin struct { // decoded. May be nil, in which case nothing is memoized. ChannelAnalysisCache *protocol.ChannelAnalysisCache + // BlobPayloads memoizes blob payloads decoded within one round, so a handle + // referenced by an observation is fetched and decompressed once for + // ValidateObservation and StateTransition together. May be nil, in which + // case nothing is memoized. + BlobPayloads *blobPayloadCache + // Optional telemetry sinks; best-effort, non-blocking. OutcomeTelemetryCh chan<- *protocol.LLOOutcomeTelemetry ReportTelemetryCh chan<- *protocol.LLOReportTelemetry @@ -303,7 +309,7 @@ func (p *Plugin) ValidateObservation(ctx context.Context, seqNr uint64, _ ocrtyp return nil } - observation, err := decodeObservation(ctx, ao.Observation, bf) + observation, err := decodeObservation(ctx, ao.Observation, bf, p.BlobPayloads.round(seqNr)) if err != nil { return fmt.Errorf("observation decode error: %w", err) } diff --git a/llo/dev/v31/plugin_test.go b/llo/dev/v31/plugin_test.go index d3744fb..73ab5bc 100644 --- a/llo/dev/v31/plugin_test.go +++ b/llo/dev/v31/plugin_test.go @@ -201,7 +201,7 @@ func Test_Observation_WireRoundTrip(t *testing.T) { enc, err := encodeObservation(obs, nil) require.NoError(t, err) - got, err := decodeObservation(ctx, enc, nil) + got, err := decodeObservation(ctx, enc, nil, nil) require.NoError(t, err) assert.Equal(t, obs.ShouldRetire, got.ShouldRetire) @@ -223,7 +223,7 @@ func Test_Observation_StreamValuesNeverInline(t *testing.T) { enc, err := encodeObservation(obs, nil) require.NoError(t, err) - got, err := decodeObservation(ctx, enc, nil) + got, err := decodeObservation(ctx, enc, nil, nil) require.NoError(t, err) require.Empty(t, got.StreamValues, "stream values must travel in a blob, never inline") } @@ -249,14 +249,14 @@ func Test_Observation_BlobRoundTrip(t *testing.T) { enc, err := encodeObservation(Observation{UnixTimestampNanoseconds: 1}, [][]byte{handleBytes}) require.NoError(t, err) - got, err := decodeObservation(ctx, enc, bc) + got, err := decodeObservation(ctx, enc, bc, nil) require.NoError(t, err) require.Len(t, got.StreamValues, len(sv)) require.True(t, equalStreamValue(sv[100], got.StreamValues[100])) // Without a fetcher the reference is unusable, and that must be classified // as a node-local blob-fetch failure rather than a malformed observation. - _, err = decodeObservation(ctx, enc, nil) + _, err = decodeObservation(ctx, enc, nil, nil) var bfErr *blobFetchError require.ErrorAs(t, err, &bfErr) } @@ -478,7 +478,7 @@ func Test_decodeObservation_RejectsHugeHandleCount(t *testing.T) { var tmp [binary.MaxVarintLen64]byte n := binary.PutUvarint(tmp[:], ^uint64(0)) // max uint64 buf = append(buf, tmp[:n]...) - _, err := decodeObservation(ctx, buf, nil) + _, err := decodeObservation(ctx, buf, nil, nil) require.Error(t, err, "must reject an oversized handle count instead of allocating") require.Contains(t, err.Error(), "too many blobs") } @@ -878,12 +878,12 @@ func Test_decodeObservation_BlobFetchErrorIsClassified(t *testing.T) { var bfErr *blobFetchError // A failing fetcher -> node-local error, must be a *blobFetchError. - _, err = decodeObservation(ctx, frame, &errBroadcaster{}) + _, err = decodeObservation(ctx, frame, &errBroadcaster{}, nil) require.Error(t, err) require.True(t, errors.As(err, &bfErr), "fetch failure must be a blobFetchError so StateTransition propagates it") // A nil fetcher (blob referenced but unfetchable) -> also a *blobFetchError. - _, err = decodeObservation(ctx, frame, nil) + _, err = decodeObservation(ctx, frame, nil, nil) require.Error(t, err) require.True(t, errors.As(err, &bfErr)) } @@ -897,13 +897,13 @@ func Test_decodeObservation_MalformedIsNotBlobFetchError(t *testing.T) { var bfErr *blobFetchError // Unknown wire version. - _, err := decodeObservation(ctx, []byte{0x02, 0x00}, &errBroadcaster{}) + _, err := decodeObservation(ctx, []byte{0x02, 0x00}, &errBroadcaster{}, nil) require.Error(t, err) require.False(t, errors.As(err, &bfErr), "malformed framing must stay droppable, not a blobFetchError") // Well-framed but garbage handle bytes: UnmarshalBinary fails deterministically. frame := frameObservation([][]byte{{0xFF}}, nil) - _, err = decodeObservation(ctx, frame, &errBroadcaster{}) + _, err = decodeObservation(ctx, frame, &errBroadcaster{}, nil) require.Error(t, err) require.False(t, errors.As(err, &bfErr)) } @@ -1030,7 +1030,7 @@ func Test_Observation_RejectsInlineStreamValues(t *testing.T) { }) require.NoError(t, err) - _, err = decodeObservation(ctx, frameObservation(nil, mainBytes), nil) + _, err = decodeObservation(ctx, frameObservation(nil, mainBytes), nil, nil) require.ErrorContains(t, err, "inline stream values") // Deterministic across oracles, so it must not be a blob-fetch failure. @@ -1218,7 +1218,7 @@ func Test_Observation_SupportedReportFormats_RoundTrip(t *testing.T) { } b, err := encodeObservation(obs, nil) require.NoError(t, err) - got, err := decodeObservation(ctx, b, nil) + got, err := decodeObservation(ctx, b, nil, nil) require.NoError(t, err) require.Equal(t, []llotypes.ReportFormat{llotypes.ReportFormatEVMPremiumLegacy, llotypes.ReportFormatJSON}, got.SupportedReportFormats) @@ -1226,7 +1226,7 @@ func Test_Observation_SupportedReportFormats_RoundTrip(t *testing.T) { // as an empty-but-present list. b, err = encodeObservation(Observation{UnixTimestampNanoseconds: 1}, nil) require.NoError(t, err) - got, err = decodeObservation(ctx, b, nil) + got, err = decodeObservation(ctx, b, nil, nil) require.NoError(t, err) require.Nil(t, got.SupportedReportFormats) @@ -1237,7 +1237,7 @@ func Test_Observation_SupportedReportFormats_RoundTrip(t *testing.T) { } raw, err := proto.Marshal(&protocol.LLOObservationProto{UnixTimestampNanoseconds: 1, SupportedReportFormats: tooMany}) require.NoError(t, err) - _, err = decodeObservation(ctx, frameObservation(nil, raw), nil) + _, err = decodeObservation(ctx, frameObservation(nil, raw), nil, nil) require.ErrorContains(t, err, "advertises too many report formats") } @@ -1376,7 +1376,7 @@ func Test_Observation_UnverifiableCommittedChannelIsNotFatal(t *testing.T) { obsBytes, err := p.Observation(ctx, 3, ocrtypes.AttributedQuery{}, kv, nil) require.NoError(t, err, "a committed definition this build rejects must not halt the node") - obs, err := decodeObservation(ctx, obsBytes, testBlobs) + obs, err := decodeObservation(ctx, obsBytes, testBlobs, nil) require.NoError(t, err) // The removal vote is the recovery path, and it is only cast because the @@ -1468,7 +1468,7 @@ func Test_Observation_RetirementCacheErrorsAreNotFatal(t *testing.T) { obsBytes, err := p.Observation(ctx, 2, ocrtypes.AttributedQuery{}, kv, nil) require.NoError(t, err, "a failing retirement cache must not halt the node") - obs, err := decodeObservation(ctx, obsBytes, testBlobs) + obs, err := decodeObservation(ctx, obsBytes, testBlobs, nil) require.NoError(t, err) require.Empty(t, obs.AttestedPredecessorRetirement) diff --git a/llo/dev/v31/statetransition.go b/llo/dev/v31/statetransition.go index aeac9ff..39b2e84 100644 --- a/llo/dev/v31/statetransition.go +++ b/llo/dev/v31/statetransition.go @@ -6,6 +6,7 @@ import ( "errors" "fmt" "sort" + "sync" "github.com/smartcontractkit/chainlink-common/pkg/logger" llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" @@ -69,7 +70,7 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A return nil, fmt.Errorf("failed to load KV state: %w", err) } - timestamps, validPredecessorRetirementReport, shouldRetireVotes, removeChannelVotesByID, updateDefsByHash, updateVotesByHash, supportByFormat, streamObservations, err := p.decodeObservations(ctx, aos, bf) + timestamps, validPredecessorRetirementReport, shouldRetireVotes, removeChannelVotesByID, updateDefsByHash, updateVotesByHash, supportByFormat, streamObservations, err := p.decodeObservations(ctx, aos, bf, p.BlobPayloads.round(seqNr)) if err != nil { return nil, err } @@ -222,7 +223,7 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A return encodePrecursor(out) } -func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.AttributedObservation, bf ocr3_1types.BlobFetcher) ( +func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.AttributedObservation, bf ocr3_1types.BlobFetcher, memo *roundBlobPayloads) ( timestampsNanoseconds []uint64, validPredecessorRetirementReport *protocol.RetirementReport, shouldRetireVotes int, @@ -239,8 +240,24 @@ func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.Attribut updateChannelVotesByHash = make(map[[32]byte]int) streamObservations = make(map[llotypes.StreamID][]protocol.StreamValue) - for _, ao := range aos { - observation, derr := decodeObservation(ctx, ao.Observation, bf) + // Decode concurrently: each observation may reference blobs that are not yet + // assembled locally, and waiting for one serially delays every other. The + // tally below still runs in aos order, so the outcome does not depend on + // which decode finished first. + decoded := make([]Observation, len(aos)) + decodeErrs := make([]error, len(aos)) + var wg sync.WaitGroup + wg.Add(len(aos)) + for i, ao := range aos { + go func() { + defer wg.Done() + decoded[i], decodeErrs[i] = decodeObservation(ctx, ao.Observation, bf, memo) + }() + } + wg.Wait() + + for i, ao := range aos { + observation, derr := decoded[i], decodeErrs[i] if derr != nil { var bfErr *blobFetchError if errors.As(derr, &bfErr) { From e1de001d0859fd35e12f77aaa1e705735fc7b975 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 10:54:30 +0100 Subject: [PATCH 17/40] SPOR-REC: enable race detector in test-ci --- Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Makefile b/Makefile index 808a049..153d2bf 100644 --- a/Makefile +++ b/Makefile @@ -13,7 +13,7 @@ test: .PHONY: test-ci test-ci: testdb - go test ./... -covermode=atomic -coverpkg=./... -coverprofile=./coverage.txt -json | tee output.txt + go test ./... -covermode=atomic -race -coverpkg=./... -coverprofile=./coverage.txt -json | tee output.txt .PHONY: lint lint: From 02f3431eec8eed4352c1957220d784ee10603572 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 11:49:31 +0100 Subject: [PATCH 18/40] SPOR-REC llo/dev/v31: add golden tests for the precursor and KV records To regenerate run GOLDEN_UPDATE=1 go test ./llo/dev/v31/ -run Golden --- llo/dev/v31/golden_test.go | 250 ++++++++++++++++++ .../v31/testdata/golden/kv_channel_seqnr.bin | Bin 0 -> 8 bytes .../v31/testdata/golden/kv_channel_state.bin | 3 + .../v31/testdata/golden/kv_history_chunk.bin | Bin 0 -> 45 bytes .../v31/testdata/golden/kv_history_header.bin | 1 + .../v31/testdata/golden/kv_history_index.bin | Bin 0 -> 16 bytes .../testdata/golden/kv_history_version.bin | 1 + llo/dev/v31/testdata/golden/kv_hot_state.bin | Bin 0 -> 98 bytes llo/dev/v31/testdata/golden/kv_lifecycle.bin | 1 + .../v31/testdata/golden/precursor_empty.bin | 0 .../v31/testdata/golden/precursor_full.bin | Bin 0 -> 193 bytes 11 files changed, 256 insertions(+) create mode 100644 llo/dev/v31/golden_test.go create mode 100644 llo/dev/v31/testdata/golden/kv_channel_seqnr.bin create mode 100644 llo/dev/v31/testdata/golden/kv_channel_state.bin create mode 100644 llo/dev/v31/testdata/golden/kv_history_chunk.bin create mode 100644 llo/dev/v31/testdata/golden/kv_history_header.bin create mode 100644 llo/dev/v31/testdata/golden/kv_history_index.bin create mode 100644 llo/dev/v31/testdata/golden/kv_history_version.bin create mode 100644 llo/dev/v31/testdata/golden/kv_hot_state.bin create mode 100644 llo/dev/v31/testdata/golden/kv_lifecycle.bin create mode 100644 llo/dev/v31/testdata/golden/precursor_empty.bin create mode 100644 llo/dev/v31/testdata/golden/precursor_full.bin diff --git a/llo/dev/v31/golden_test.go b/llo/dev/v31/golden_test.go new file mode 100644 index 0000000..566f302 --- /dev/null +++ b/llo/dev/v31/golden_test.go @@ -0,0 +1,250 @@ +package llo + +import ( + "encoding/hex" + "os" + "path/filepath" + "testing" + + "github.com/shopspring/decimal" + "github.com/stretchr/testify/require" + + "github.com/smartcontractkit/chainlink-common/pkg/logger" + llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" + + protocol "github.com/smartcontractkit/chainlink-data-streams/llo/protocol" +) + +// Golden tests freeze the wire format of the two records the plugin cannot +// change unilaterally: the precursor StateTransition hands to Reports, and the +// KeyValueState records every oracle replicates. A format change that is not +// accompanied by a deliberate golden update diverges oracles rather than +// failing loudly, so the bytes are asserted directly. +// +// Regenerate with: GOLDEN_UPDATE=1 go test ./llo/dev/v31/ -run Golden + +const goldenDir = "testdata/golden" + +// goldenBytes compares got against the committed golden file, or rewrites the +// file when GOLDEN_UPDATE is set. +func goldenBytes(t *testing.T, name string, got []byte) { + t.Helper() + path := filepath.Join(goldenDir, name) + if os.Getenv("GOLDEN_UPDATE") != "" { + require.NoError(t, os.WriteFile(path, got, 0o600)) + return + } + want, err := os.ReadFile(path) + require.NoError(t, err, "golden file not found; run with GOLDEN_UPDATE=1 to generate") + require.Equal(t, hex.EncodeToString(want), hex.EncodeToString(got), "encoding of %s changed; every oracle must agree on these bytes", name) +} + +// goldenPrecursor is the fully populated precursor the golden cases encode. It +// exercises every field, both stream value types, and out-of-order map keys so +// that the sorting the encoder does is part of what is frozen. +func goldenPrecursor() precursor { + return precursor{ + LifeCycleStage: llotypes.LifeCycleStage("production"), + ObservationTimestampNanoseconds: 1_700_000_000_000_000_000, + ChannelStateSeqNr: 42, + ChannelDefinitions: llotypes.ChannelDefinitions{ + 3: { + ReportFormat: llotypes.ReportFormatEVMPremiumLegacy, + Streams: []llotypes.Stream{{StreamID: 300, Aggregator: llotypes.AggregatorQuote}}, + Opts: []byte(`{"baseUSDFee":"0.1"}`), + }, + 1: { + ReportFormat: llotypes.ReportFormatJSON, + Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}, {StreamID: 200, Aggregator: llotypes.AggregatorMode}}, + Opts: []byte(`{"foo":"bar"}`), + }, + }, + ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{3: 30, 1: 10}, + StreamAggregates: protocol.StreamAggregates{ + 300: {llotypes.AggregatorQuote: &protocol.Quote{ + Bid: decimal.NewFromInt(1010), + Benchmark: decimal.NewFromInt(1011), + Ask: decimal.NewFromInt(1012), + }}, + 100: { + llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(123)), + llotypes.AggregatorMode: protocol.ToDecimal(decimal.NewFromInt(124)), + }, + }, + SupportByFormat: map[llotypes.ReportFormat]int{ + llotypes.ReportFormatEVMPremiumLegacy: 3, + llotypes.ReportFormatJSON: 4, + }, + } +} + +func Test_Golden_Precursor(t *testing.T) { + for _, tc := range []struct { + name string + p precursor + }{ + {"precursor_empty.bin", precursor{}}, + {"precursor_full.bin", goldenPrecursor()}, + } { + t.Run(tc.name, func(t *testing.T) { + b, err := encodePrecursor(tc.p) + require.NoError(t, err) + goldenBytes(t, tc.name, b) + + // The golden bytes must also decode back to the same projection, so + // an accidental encode/decode asymmetry cannot hide behind a + // regenerated file. + want, err := os.ReadFile(filepath.Join(goldenDir, tc.name)) + require.NoError(t, err) + got, err := decodePrecursor(want) + require.NoError(t, err) + require.Equal(t, normalizePrecursor(tc.p), got) + }) + } +} + +// normalizePrecursor fills in the empty maps decodePrecursor always returns, so +// that a nil-map input compares equal to its decoded form. +func normalizePrecursor(p precursor) precursor { + if p.ChannelDefinitions == nil { + p.ChannelDefinitions = llotypes.ChannelDefinitions{} + } + if p.ValidAfterNanoseconds == nil { + p.ValidAfterNanoseconds = map[llotypes.ChannelID]uint64{} + } + if p.StreamAggregates == nil { + p.StreamAggregates = protocol.StreamAggregates{} + } + return p +} + +func Test_Golden_KVRecords(t *testing.T) { + p := goldenPrecursor() + kv := newMemKV() + + require.NoError(t, writeLifecycle(kv, p.LifeCycleStage)) + require.NoError(t, writeChannelState(kv, p.ChannelStateSeqNr, p.ChannelDefinitions)) + require.NoError(t, writeHotState(kv, + p.ObservationTimestampNanoseconds, + p.ValidAfterNanoseconds, + map[llotypes.ChannelID]bool{3: true, 1: false, 2: true}, + map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue{ + 300: {llotypes.AggregatorQuote: {ObservedAtNanoseconds: 3, StreamValue: &protocol.Quote{ + Bid: decimal.NewFromInt(1010), + Benchmark: decimal.NewFromInt(1011), + Ask: decimal.NewFromInt(1012), + }}}, + 100: {llotypes.AggregatorMedian: {ObservedAtNanoseconds: 1, StreamValue: protocol.ToDecimal(decimal.NewFromInt(123))}}, + }, + logger.Test(t), + )) + require.NoError(t, writeHistoryLayoutVersion(kv)) + require.NoError(t, writeHistoryIndex(kv, []histKey{ + {streamID: 100, aggregator: llotypes.AggregatorMedian}, + {streamID: 300, aggregator: llotypes.AggregatorQuote}, + })) + + for _, tc := range []struct { + name string + key []byte + }{ + {"kv_lifecycle.bin", keyLifecycle}, + {"kv_channel_state.bin", keyChannelState}, + {"kv_channel_seqnr.bin", keyChannelSeqNr}, + {"kv_hot_state.bin", keyHotState}, + {"kv_history_version.bin", keyHistoryVersion}, + {"kv_history_index.bin", keyHistoryIndex}, + } { + t.Run(tc.name, func(t *testing.T) { + b, ok := kv.m[string(tc.key)] + require.True(t, ok, "record %s was not written", tc.key) + goldenBytes(t, tc.name, b) + }) + } + + // The records must load back into the same projection the writers were + // handed. + s, err := loadKVState(kv, nil) + require.NoError(t, err) + require.Equal(t, p.LifeCycleStage, s.lifeCycleStage) + require.Equal(t, p.ChannelStateSeqNr, s.channelStateSeqNr) + require.Equal(t, p.ChannelDefinitions, s.channelDefinitions) + require.Equal(t, p.ObservationTimestampNanoseconds, s.observationTimestampNs) + require.Equal(t, p.ValidAfterNanoseconds, s.validAfterNanoseconds) + require.Equal(t, map[llotypes.ChannelID]bool{3: true, 2: true}, s.reportedLastRound) + require.Len(t, s.carryForward, 2) + + version, err := readHistoryLayoutVersion(kv) + require.NoError(t, err) + require.Equal(t, historyLayoutVersion, version) + index, err := readHistoryIndex(kv) + require.NoError(t, err) + require.Equal(t, []histKey{ + {streamID: 100, aggregator: llotypes.AggregatorMedian}, + {streamID: 300, aggregator: llotypes.AggregatorQuote}, + }, index) +} + +func Test_Golden_KVHistoryRecords(t *testing.T) { + const ( + sid = llotypes.StreamID(100) + agg = llotypes.AggregatorMedian + ) + + kv := newMemKV() + + // One append per round, each round re-reading the stored window: that is the + // only way to grow a chunk past a single record, and it exercises the stored + // form rather than an in-memory shortcut. + var set protocol.RingWriteSet + for i := 1; i <= 3; i++ { + w := readHistory(t, kv, sid, agg) + if w == nil { + w = protocol.NewRingWindow(nil) + } + _, err := w.SetRequiredCount(4) + require.NoError(t, err) + appended, err := w.Append(uint64(i)*1_000, protocol.ToDecimal(decimal.NewFromInt(int64(i)))) + require.NoError(t, err) + require.True(t, appended) + + set = w.WriteSet() + require.NotNil(t, set.Header) + require.NotNil(t, set.Chunk) + _, err = writeHistoryHeader(kv, sid, agg, set.Header) + require.NoError(t, err) + _, err = writeHistoryChunk(kv, sid, agg, set.Chunk) + require.NoError(t, err) + } + + goldenBytes(t, "kv_history_header.bin", kv.m[string(historyHeaderKey(sid, agg))]) + goldenBytes(t, "kv_history_chunk.bin", kv.m[string(historyChunkKey(sid, agg, set.Chunk.Slot()))]) + + header, err := readHistoryHeader(kv, sid, agg) + require.NoError(t, err) + require.Equal(t, set.Header.Sequences(), header.Sequences()) + require.Equal(t, set.Header.Counts(), header.Counts()) + chunk, err := readHistoryChunk(kv, sid, agg, set.Chunk.Slot()) + require.NoError(t, err) + require.Len(t, chunk.Records(), 3) +} + +// Test_Golden_KVKeys freezes the key layout. Keys are part of the replicated +// schema too: renaming one silently orphans the stored value on every node. +func Test_Golden_KVKeys(t *testing.T) { + for _, tc := range []struct { + want string + key string + }{ + {"c/lifecycle", string(keyLifecycle)}, + {"c/defs", string(keyChannelState)}, + {"c/seqnr", string(keyChannelSeqNr)}, + {"r/agg", string(keyHotState)}, + {"hidx", string(keyHistoryIndex)}, + {"hv", string(keyHistoryVersion)}, + } { + require.Equal(t, tc.want, tc.key) + } + require.Equal(t, "68682f0000006400000001", hex.EncodeToString(historyHeaderKey(100, llotypes.AggregatorMedian))) + require.Equal(t, "68632f000000640000000100000007", hex.EncodeToString(historyChunkKey(100, llotypes.AggregatorMedian, 7))) +} diff --git a/llo/dev/v31/testdata/golden/kv_channel_seqnr.bin b/llo/dev/v31/testdata/golden/kv_channel_seqnr.bin new file mode 100644 index 0000000000000000000000000000000000000000..541378f226c8ff8ef166e1effb2144e92d541f30 GIT binary patch literal 8 LcmZQz00S)m05Sk8 literal 0 HcmV?d00001 diff --git a/llo/dev/v31/testdata/golden/kv_channel_state.bin b/llo/dev/v31/testdata/golden/kv_channel_state.bin new file mode 100644 index 0000000..56c0f22 --- /dev/null +++ b/llo/dev/v31/testdata/golden/kv_channel_state.bin @@ -0,0 +1,3 @@ + +"dÈ {"foo":"bar"} +#¬{"baseUSDFee":"0.1"} \ No newline at end of file diff --git a/llo/dev/v31/testdata/golden/kv_history_chunk.bin b/llo/dev/v31/testdata/golden/kv_history_chunk.bin new file mode 100644 index 0000000000000000000000000000000000000000..311ee5e241355dd1bd79a8324b773f82f2b57e54 GIT binary patch literal 45 jcmWgQ<#@p^#397S00c~oLcAOo_~Be8Aa{p2oXZRVWMTug literal 0 HcmV?d00001 diff --git a/llo/dev/v31/testdata/golden/kv_history_header.bin b/llo/dev/v31/testdata/golden/kv_history_header.bin new file mode 100644 index 0000000..cd1164a --- /dev/null +++ b/llo/dev/v31/testdata/golden/kv_history_header.bin @@ -0,0 +1 @@ +"è(¸ \ No newline at end of file diff --git a/llo/dev/v31/testdata/golden/kv_history_index.bin b/llo/dev/v31/testdata/golden/kv_history_index.bin new file mode 100644 index 0000000000000000000000000000000000000000..2fc23f39e5a6d91747ebcc8376c7fba4eb113692 GIT binary patch literal 16 UcmZQzU`SzLU|<9y9U#pN00t5OmH+?% literal 0 HcmV?d00001 diff --git a/llo/dev/v31/testdata/golden/kv_history_version.bin b/llo/dev/v31/testdata/golden/kv_history_version.bin new file mode 100644 index 0000000..6b2aaa7 --- /dev/null +++ b/llo/dev/v31/testdata/golden/kv_history_version.bin @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/llo/dev/v31/testdata/golden/kv_hot_state.bin b/llo/dev/v31/testdata/golden/kv_hot_state.bin new file mode 100644 index 0000000000000000000000000000000000000000..f4f189acb75d9bd9ad4b640853323c139c013e56 GIT binary patch literal 98 zcmd;RXjrlF@%-nf#f4Zn7zMb1B(s2=6cZD(k{CydkN^jh5Dy2V5Qh*O0}wD(OE4}L5~&7?7pH~>ySSyM0u>qP87kE(v2ZX7Z~;kX z0XZ!`juasdAvOjeV5*j2MB&#+Flng)4Fa1e&BYE@#QaGJ%J?h=WqgrfHqf#HnaT_# HnFLq>#2_Iv literal 0 HcmV?d00001 From 0890f8232791f5025b7ffde6c76e0c3edd143ca2 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 11:58:59 +0100 Subject: [PATCH 19/40] SPOR-REC llo/dev/v31: fuzz the observation, precursor and KV decoders --- llo/dev/v31/decode_fuzz_test.go | 235 ++++++++++++++++++++++++++++++++ 1 file changed, 235 insertions(+) create mode 100644 llo/dev/v31/decode_fuzz_test.go diff --git a/llo/dev/v31/decode_fuzz_test.go b/llo/dev/v31/decode_fuzz_test.go new file mode 100644 index 0000000..186e3ce --- /dev/null +++ b/llo/dev/v31/decode_fuzz_test.go @@ -0,0 +1,235 @@ +package llo + +import ( + "context" + "testing" + + "github.com/shopspring/decimal" + + llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" + + protocol "github.com/smartcontractkit/chainlink-data-streams/llo/protocol" + + "github.com/smartcontractkit/libocr/offchainreporting2plus/ocr3_1types" + ocrtypes "github.com/smartcontractkit/libocr/offchainreporting2plus/types" +) + +// The decoders fuzzed here all consume bytes the local node did not produce: +// observations come from peers, and the precursor and the KeyValueState records +// come from replicated state a restored snapshot or an earlier version may have +// written. The contract in every case is the same: return a value or an error, +// never panic, and never let a decoded value exceed the bounds the protocol +// enforces. + +// payloadFetcher returns the same payload for every handle, so a fuzzer can +// drive the blob path of decodeObservation with arbitrary bytes. +type payloadFetcher struct{ payload []byte } + +func (f payloadFetcher) FetchBlob(context.Context, ocr3_1types.BlobHandle) ([]byte, error) { + return f.payload, nil +} + +var _ ocr3_1types.BlobFetcher = payloadFetcher{} + +// FuzzDecodeObservation feeds arbitrary bytes through the observation framing, +// proto decode, and blob merge. Observations are attacker-controlled: a +// byzantine peer picks these bytes, and every other oracle decodes them. +func FuzzDecodeObservation(f *testing.F) { + obs := Observation{ + AttestedPredecessorRetirement: []byte("retirement"), + ShouldRetire: true, + UnixTimestampNanoseconds: 1_700_000_000_000_000_000, + RemoveChannelIDs: map[llotypes.ChannelID]struct{}{7: {}}, + UpdateChannelDefinitions: llotypes.ChannelDefinitions{1: jsonChannel()}, + SupportedReportFormats: []llotypes.ReportFormat{llotypes.ReportFormatJSON, llotypes.ReportFormatEVMPremiumLegacy}, + } + encoded, err := encodeObservation(obs, nil) + if err != nil { + f.Fatal(err) + } + + handle := make([]byte, 32) + withHandle, err := encodeObservation(obs, [][]byte{handle}) + if err != nil { + f.Fatal(err) + } + + payload, err := marshalStreamValues(protocol.StreamValues{100: protocol.ToDecimal(decimal.NewFromInt(123))}) + if err != nil { + f.Fatal(err) + } + + f.Add([]byte(encoded), []byte(nil)) + f.Add([]byte(withHandle), payload) + f.Add([]byte{observationWireVersion}, []byte(nil)) + f.Add([]byte{observationWireVersion, 0xff}, []byte(nil)) + f.Add([]byte{}, []byte(nil)) + f.Add([]byte("not an observation"), []byte("not a payload")) + + f.Fuzz(func(t *testing.T, raw, payload []byte) { + decoded, err := decodeObservation(context.Background(), ocrtypes.Observation(raw), payloadFetcher{payload: payload}, nil) + if err != nil { + return + } + // v31 never accepts inline values, and the formats it accepts are + // bounded and canonicalized, because both feed the state transition. + if len(decoded.SupportedReportFormats) > protocol.MaxObservationSupportedReportFormatsLength { + t.Fatalf("decoded %d report formats, max %d", len(decoded.SupportedReportFormats), protocol.MaxObservationSupportedReportFormatsLength) + } + for i := 1; i < len(decoded.SupportedReportFormats); i++ { + if decoded.SupportedReportFormats[i-1] >= decoded.SupportedReportFormats[i] { + t.Fatalf("report formats are not strictly ascending: %v", decoded.SupportedReportFormats) + } + } + for id, cd := range decoded.UpdateChannelDefinitions { + if len(cd.Streams) > protocol.MaxStreamsPerChannel { + t.Fatalf("channel %d decoded with %d streams, max %d", id, len(cd.Streams), protocol.MaxStreamsPerChannel) + } + } + }) +} + +// FuzzDecodePrecursor feeds arbitrary bytes through the precursor decoder. The +// precursor crosses from StateTransition to Reports as opaque bytes, so a +// decoded one must be re-encodable and stable: Reports is the only consumer and +// it has no other source for this state. +func FuzzDecodePrecursor(f *testing.F) { + full, err := encodePrecursor(goldenPrecursor()) + if err != nil { + f.Fatal(err) + } + empty, err := encodePrecursor(precursor{}) + if err != nil { + f.Fatal(err) + } + f.Add([]byte(full)) + f.Add([]byte(empty)) + f.Add([]byte{0x0a, 0x00}) + f.Add([]byte("not a precursor")) + f.Add([]byte{}) + + f.Fuzz(func(t *testing.T, b []byte) { + p, err := decodePrecursor(b) + if err != nil { + return + } + if len(p.SupportByFormat) > protocol.MaxObservationSupportedReportFormatsLength { + t.Fatalf("decoded %d report format support entries, max %d", len(p.SupportByFormat), protocol.MaxObservationSupportedReportFormatsLength) + } + // Re-encoding a decoded precursor must be a fixed point: encode is + // deterministic, so a value that survives decode has exactly one + // encoding, and every oracle must agree on it. + reencoded, err := encodePrecursor(p) + if err != nil { + t.Fatalf("re-encode a decoded precursor: %v", err) + } + again, err := decodePrecursor(reencoded) + if err != nil { + t.Fatalf("re-decode a re-encoded precursor: %v", err) + } + twice, err := encodePrecursor(again) + if err != nil { + t.Fatalf("re-encode twice: %v", err) + } + if string(reencoded) != string(twice) { + t.Fatal("re-encoding a decoded precursor is not stable") + } + }) +} + +// FuzzLoadKVState feeds arbitrary bytes into every KeyValueState record the +// plugin reads at the start of a round. The store is replicated and may have +// been written by a different version or restored from a snapshot, so a +// corrupt record must fail the load rather than crash the node. +func FuzzLoadKVState(f *testing.F) { + seeded := newMemKV() + defs := llotypes.ChannelDefinitions{1: jsonChannel()} + if err := writeChannelState(seeded, 9, defs); err != nil { + f.Fatal(err) + } + f.Add([]byte("production"), seeded.m[string(keyChannelState)], beU64(9), []byte(nil)) + f.Add([]byte{}, []byte{}, []byte{}, []byte{}) + f.Add([]byte("staging"), []byte("not a proto"), []byte{0x01}, []byte{0x01, 0x02}) + f.Add([]byte(nil), []byte{0x0a, 0x02, 0x08, 0x01}, beU64(1), append(beU32(100), beU32(1)...)) + + f.Fuzz(func(t *testing.T, lifecycle, channelState, seqNr, hotState []byte) { + kv := newMemKV() + if err := kv.Write(keyLifecycle, lifecycle); err != nil { + t.Fatal(err) + } + if err := kv.Write(keyChannelState, channelState); err != nil { + t.Fatal(err) + } + if err := kv.Write(keyChannelSeqNr, seqNr); err != nil { + t.Fatal(err) + } + if err := kv.Write(keyHotState, hotState); err != nil { + t.Fatal(err) + } + + // The cold load is what Observation and ValidateObservation run, and the + // full load is what StateTransition runs; both must survive the same + // bytes. + if _, err := loadColdKVState(kv, protocol.NewChannelCache()); err != nil { + return + } + s, err := loadKVState(kv, nil) + if err != nil { + return + } + for id, cd := range s.channelDefinitions { + if len(cd.Streams) > protocol.MaxStreamsPerChannel { + t.Fatalf("channel %d decoded with %d streams, max %d", id, len(cd.Streams), protocol.MaxStreamsPerChannel) + } + } + }) +} + +// FuzzDecodeHistoryRecords feeds arbitrary bytes into the three history records: +// the index, a window header, and a ring chunk. Corrupt history is discarded +// and re-warmed rather than trusted, so the decoders must reject it without +// sizing an allocation from it. +func FuzzDecodeHistoryRecords(f *testing.F) { + f.Add(append(beU32(100), beU32(1)...), []byte(nil), []byte(nil)) + f.Add([]byte{0x01}, []byte("not a proto"), []byte("not a proto")) + f.Add([]byte{}, []byte{}, []byte{}) + + f.Fuzz(func(t *testing.T, index, header, chunk []byte) { + const ( + sid = llotypes.StreamID(100) + agg = llotypes.AggregatorMedian + ) + + kv := newMemKV() + if err := kv.Write(keyHistoryIndex, index); err != nil { + t.Fatal(err) + } + if err := kv.Write(historyHeaderKey(sid, agg), header); err != nil { + t.Fatal(err) + } + if err := kv.Write(historyChunkKey(sid, agg, 0), chunk); err != nil { + t.Fatal(err) + } + + if keys, err := readHistoryIndex(kv); err == nil && len(keys) > protocol.MaxHistoryPairs { + t.Fatalf("decoded %d history pairs, max %d", len(keys), protocol.MaxHistoryPairs) + } + if _, err := readHistoryLayoutVersion(kv); err != nil { + t.Fatalf("reading the layout version must not fail: %v", err) + } + + decodedHeader, err := readHistoryHeader(kv, sid, agg) + if err == nil && decodedHeader != nil { + if len(decodedHeader.Sequences()) != len(decodedHeader.Counts()) { + t.Fatalf("header decoded with %d sequences and %d counts", len(decodedHeader.Sequences()), len(decodedHeader.Counts())) + } + // A decoded header must be usable as a window: that is the only + // thing the store does with it. + protocol.NewRingWindow(decodedHeader).AppendPlan() + } + decodedChunk, err := readHistoryChunk(kv, sid, agg, 0) + if err == nil && decodedChunk != nil && decodedChunk.Len() > protocol.MaxHistoryChunkRecords { + t.Fatalf("chunk decoded with %d records, max %d", decodedChunk.Len(), protocol.MaxHistoryChunkRecords) + } + }) +} From c506bedabb98e40e3d90393e3b0e960a184c57df Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 12:03:01 +0100 Subject: [PATCH 20/40] SPOR-REC llo/dev/v31: add tests for warm and cold restarts --- llo/dev/v31/restart_test.go | 110 ++++++++++++++++++++++++++++++++++++ 1 file changed, 110 insertions(+) create mode 100644 llo/dev/v31/restart_test.go diff --git a/llo/dev/v31/restart_test.go b/llo/dev/v31/restart_test.go new file mode 100644 index 0000000..814c8a8 --- /dev/null +++ b/llo/dev/v31/restart_test.go @@ -0,0 +1,110 @@ +package llo + +import ( + "fmt" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" + "github.com/smartcontractkit/chainlink-common/pkg/utils/tests" + + ocrtypes "github.com/smartcontractkit/libocr/offchainreporting2plus/types" +) + +// Test_Restart_WarmKVResumes checks that a restart which keeps the KV intact is +// invisible to the protocol: a plugin instance that replaces another mid warmup +// carries on from the persisted state instead of starting over. +func Test_Restart_WarmKVResumes(t *testing.T) { + ctx := tests.Context(t) + const depth = 3 + expression := fmt.Sprintf("Count(History(s100, %d))", depth) + + kv := newMemKV() + before := historyPlugin(t, expression) + bootstrapHistoryChannel(t, before, kv, expression) + + // Warm the window to one short of the required depth. + seqNr := uint64(3) + for round := 1; round < depth; round++ { + _, err := before.StateTransition(ctx, seqNr, ocrtypes.AttributedQuery{}, valueRound(t, uint64(round)*10_000, int64(round)), kv, testBlobs) + require.NoError(t, err) + seqNr++ + } + require.False(t, reportedFlag(t, kv, 1), "must not be reportable while warming up") + validAfterBefore := storedValidAfter(t, kv, 1) + + // Restart: a fresh instance with empty in-memory caches over the same KV. + after := historyPlugin(t, expression) + + // The channel definitions are read back from KV, not re-voted. + require.Contains(t, storedChannelDefinitions(t, kv), llotypes.ChannelID(1)) + + // The round that completes the window still lands on schedule, which only + // holds if the restarted instance saw the pre-restart records. + _, err := after.StateTransition(ctx, seqNr, ocrtypes.AttributedQuery{}, valueRound(t, uint64(depth)*10_000, depth), kv, testBlobs) + require.NoError(t, err) + + stored := readHistory(t, kv, 100, llotypes.AggregatorMedian) + require.NotNil(t, stored) + assert.Equal(t, depth, stored.Len(), "the restart must not drop persisted records") + assert.True(t, reportedFlag(t, kv, 1), "the window is deep enough, so the restarted instance must report") + // Coverage advances on the round following the one that emitted, as it does + // on an instance that never restarted. + _, err = after.StateTransition(ctx, seqNr+1, ocrtypes.AttributedQuery{}, valueRound(t, uint64(depth+1)*10_000, depth+1), kv, testBlobs) + require.NoError(t, err) + assert.Greater(t, storedValidAfter(t, kv, 1), validAfterBefore, "coverage must advance once the channel is reporting") +} + +// Test_Restart_ColdKVRebuilds checks the other half: a restart that also loses +// the KV comes up clean rather than wedged. Nothing is inherited, so the +// channel has to be re-voted and the window re-warmed from scratch. +func Test_Restart_ColdKVRebuilds(t *testing.T) { + ctx := tests.Context(t) + const depth = 3 + expression := fmt.Sprintf("Count(History(s100, %d))", depth) + + kv := newMemKV() + before := historyPlugin(t, expression) + bootstrapHistoryChannel(t, before, kv, expression) + for round := 1; round <= depth; round++ { + _, err := before.StateTransition(ctx, uint64(2+round), ocrtypes.AttributedQuery{}, valueRound(t, uint64(round)*10_000, int64(round)), kv, testBlobs) + require.NoError(t, err) + } + require.True(t, reportedFlag(t, kv, 1)) + + // Restart with a wiped store: new plugin, new KV. + cold := newMemKV() + after := historyPlugin(t, expression) + + require.Empty(t, storedChannelDefinitions(t, cold), "a wiped KV must not carry channels") + require.Nil(t, readHistory(t, cold, 100, llotypes.AggregatorMedian), "a wiped KV must not carry history") + + // Values arriving before the channel is re-voted are tolerated and produce + // nothing: there is no channel requiring the stream, so nothing is stored. + _, err := after.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, valueRound(t, 10_000, 1), cold, testBlobs) + require.NoError(t, err) + require.Nil(t, readHistory(t, cold, 100, llotypes.AggregatorMedian)) + require.False(t, reportedFlag(t, cold, 1)) + + // Re-vote the channel, then re-warm. Reportability returns only once the + // window is deep again, exactly as on a first start. + _, err = after.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, addChannelRound(t, 20_000, 1, historyExprChannel(expression)), cold, testBlobs) + require.NoError(t, err) + + for round := 1; round <= depth; round++ { + _, err := after.StateTransition(ctx, uint64(2+round), ocrtypes.AttributedQuery{}, valueRound(t, uint64(round+2)*10_000, int64(round)), cold, testBlobs) + require.NoError(t, err) + + if round < depth { + assert.False(t, reportedFlag(t, cold, 1), "round %d: must re-warm before reporting", round) + } else { + assert.True(t, reportedFlag(t, cold, 1), "round %d: must report once the window is deep again", round) + } + } + + stored := readHistory(t, cold, 100, llotypes.AggregatorMedian) + require.NotNil(t, stored) + assert.Equal(t, depth, stored.Len()) +} From d251d95a94ca29815aab905bf6ed9cfcf689da31 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 15:08:11 +0100 Subject: [PATCH 21/40] SPOR-0010 llo/dev/v31: agree on the predecessor signer set in c/pred Verifying an attested predecessor retirement report needs the predecessor signer set, which StateTransition read from the node-local retirement report cache. Vote the signer set into a new cold key, c/pred, and verify against that. Observation carries the local set until c/pred exists, which moves the node-local read to where oracles are allowed to differ. A report that fails verification against the agreed set now costs only the report. The observation timestamp, votes and stream values still count, where before one malformed field removed an oracle from the round. Return an observationTally from decodeObservations instead of eleven values. --- llo/dev/v31/doc.go | 5 + llo/dev/v31/flow_test.go | 253 +++++++++++++++++- llo/dev/v31/golden_test.go | 10 + llo/dev/v31/kv.go | 53 +++- llo/dev/v31/observation.go | 28 ++ llo/dev/v31/plugin.go | 39 +++ llo/dev/v31/statetransition.go | 205 +++++++++++--- .../testdata/golden/kv_predecessor_config.bin | 5 + llo/protocol/lifecycle.go | 19 +- llo/protocol/limits.go | 8 + llo/protocol/plugin_codecs.pb.go | 140 ++++++++-- llo/protocol/plugin_codecs.proto | 24 ++ .../plugin_scoped_retirement_report_cache.go | 23 +- llo/v30/plugin_observation_test.go | 6 + 14 files changed, 741 insertions(+), 77 deletions(-) create mode 100644 llo/dev/v31/testdata/golden/kv_predecessor_config.bin diff --git a/llo/dev/v31/doc.go b/llo/dev/v31/doc.go index c7a1cc0..f6e73bf 100644 --- a/llo/dev/v31/doc.go +++ b/llo/dev/v31/doc.go @@ -31,6 +31,11 @@ // timestamped aggregates — and is rewritten every round. // - c/defs holds every channel definition and is rewritten only when the // definitions change; c/seqnr records the sequence number of that write. +// - c/pred holds the predecessor instance's signer set and f, agreed by vote +// while staging and written at most once. Verifying an attested predecessor +// retirement report against it keeps the state transition reading only +// replicated state: the node-local retirement report cache is filled +// asynchronously, so reading it here would fork. // - c/lifecycle holds the lifecycle stage and is written only on change. // // Because c/defs is a pure function of c/seqnr, the plugin keeps the decoded diff --git a/llo/dev/v31/flow_test.go b/llo/dev/v31/flow_test.go index 96c4566..ecaaee4 100644 --- a/llo/dev/v31/flow_test.go +++ b/llo/dev/v31/flow_test.go @@ -2,6 +2,7 @@ package llo import ( "context" + "errors" "sync" "testing" "time" @@ -123,9 +124,28 @@ func (mockOnchainConfigCodec) Decode([]byte) (protocol.OnchainConfig, error) { } func (mockOnchainConfigCodec) Encode(protocol.OnchainConfig) ([]byte, error) { return nil, nil } +// testPredecessorSigners is the signer set a fixture node reads from its local +// retirement report cache and votes into c/pred. +var testPredecessorSigners = [][]byte{{0x01}, {0x02}, {0x03}, {0x04}} + type mockPredecessorRetirementReportCache struct { report protocol.RetirementReport err error + // noLocalConfig models a node whose config poller has not stored the + // predecessor config yet, so it cannot vote on c/pred. + noLocalConfig bool + // signers overrides testPredecessorSigners; f is the predecessor's f. + signers [][]byte + f uint8 + // verifyErr makes VerifyAttestedRetirementReport fail deterministically. + verifyErr error +} + +func (m *mockPredecessorRetirementReportCache) localSigners() [][]byte { + if m.signers != nil { + return m.signers + } + return testPredecessorSigners } func (m *mockPredecessorRetirementReportCache) AttestedRetirementReport(ocrtypes.ConfigDigest) ([]byte, error) { @@ -137,6 +157,33 @@ func (m *mockPredecessorRetirementReportCache) AttestedRetirementReport(ocrtypes func (m *mockPredecessorRetirementReportCache) CheckAttestedRetirementReport(ocrtypes.ConfigDigest, []byte) (protocol.RetirementReport, error) { return m.report, nil } +func (m *mockPredecessorRetirementReportCache) PredecessorConfig(ocrtypes.ConfigDigest) ([][]byte, uint8, bool) { + if m.noLocalConfig { + return nil, 0, false + } + return m.localSigners(), m.f, true +} +func (m *mockPredecessorRetirementReportCache) VerifyAttestedRetirementReport(_ ocrtypes.ConfigDigest, signers [][]byte, f uint8, _ []byte) (protocol.RetirementReport, error) { + if m.verifyErr != nil { + return protocol.RetirementReport{}, m.verifyErr + } + // Verification must run against the agreed set, never the local one. + if len(signers) == 0 { + return protocol.RetirementReport{}, errors.New("verify called with an empty signer set") + } + return m.report, nil +} + +// promotionObs is what a staging node observes once its predecessor has +// retired: the attested retirement report, plus a vote for the predecessor's +// signer set so the DON can agree on c/pred and verify the report against it. +func promotionObs(tsNanoseconds uint64) Observation { + return Observation{ + UnixTimestampNanoseconds: tsNanoseconds, + AttestedPredecessorRetirement: []byte("attested"), + PredecessorSigners: testPredecessorSigners, + } +} func jsonChannel() llotypes.ChannelDefinition { return llotypes.ChannelDefinition{ReportFormat: llotypes.ReportFormatJSON, Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}} @@ -333,7 +380,7 @@ func Test_StateTransition_Promotion(t *testing.T) { require.Equal(t, string(protocol.LifeCycleStageStaging), string(kv.m[string(keyLifecycle)])) // A round carrying a valid attested predecessor retirement report promotes to production. - promoObs := Observation{UnixTimestampNanoseconds: 1000, AttestedPredecessorRetirement: []byte("attested")} + promoObs := promotionObs(1000) aos := make([]ocrtypes.AttributedObservation, 0, 4) for i := 0; i < 4; i++ { aos = append(aos, ao(i, mustEncodeObs(t, promoObs))) @@ -386,7 +433,7 @@ func Test_StateTransition_Promotion_StagingOnlyChannelTreatedAsNew(t *testing.T) // Round 4 (ts=3000): a valid attested predecessor retirement report promotes // this instance to production. - promoObs := Observation{UnixTimestampNanoseconds: 3000, AttestedPredecessorRetirement: []byte("attested")} + promoObs := promotionObs(3000) aos := make([]ocrtypes.AttributedObservation, 0, 4) for i := 0; i < 4; i++ { aos = append(aos, ao(i, mustEncodeObs(t, promoObs))) @@ -498,3 +545,205 @@ func Test_ValidateObservation_RemoveAddSwapAtBudget(t *testing.T) { addOnly := Observation{UnixTimestampNanoseconds: 1_000, UpdateChannelDefinitions: replacement} require.Error(t, p.ValidateObservation(ctx, 2, ocrtypes.AttributedQuery{}, ao(0, mustEncodeObs(t, addOnly)), kv, testBlobs)) } + +// stagedPlugin returns a staging plugin with a predecessor configured, plus its +// bootstrapped KV. +func stagedPlugin(t *testing.T, prrc *mockPredecessorRetirementReportCache) (*Plugin, *memKV) { + t.Helper() + ctx := tests.Context(t) + p := testPlugin(t) + predecessor := ocrtypes.ConfigDigest{0xAB} + p.PredecessorConfigDigest = &predecessor + p.PredecessorRetirementReportCache = prrc + kv := newMemKV() + _, err := p.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, nil), ao(1, nil), ao(2, nil)}, kv, testBlobs) + require.NoError(t, err) + require.Equal(t, string(protocol.LifeCycleStageStaging), string(kv.m[string(keyLifecycle)])) + return p, kv +} + +// promotionRoundAOs builds a round of four observations with distinct +// timestamps, all carrying an attested retirement report, of which the first +// voters many vote for the predecessor's signer set. The timestamps are chosen +// so the median moves if the last observation is dropped: kept gives 3000, +// dropped gives 2000. +func promotionRoundAOs(t *testing.T, voters int) []ocrtypes.AttributedObservation { + t.Helper() + aos := make([]ocrtypes.AttributedObservation, 0, 4) + for i := 0; i < 4; i++ { + obs := promotionObs(uint64(1000 * (i + 1))) //nolint:gosec // small test constant + if i >= voters { + // This node's config poller has not caught up, so it abstains from + // the signer-set vote but observes everything else as usual. + obs.PredecessorSigners = nil + } + aos = append(aos, ao(i, mustEncodeObs(t, obs))) + } + return aos +} + +// Test_StateTransition_PredecessorConfigAgreedAndUsedSameRound covers +// the signer set needed to verify an attested retirement report is agreed by +// vote into c/pred rather than read from the node-local cache, and a set agreed +// this round is usable this round, so the handover costs no extra round. +func Test_StateTransition_PredecessorConfigAgreedAndUsedSameRound(t *testing.T) { + ctx := tests.Context(t) + p, kv := stagedPlugin(t, &mockPredecessorRetirementReportCache{ + report: protocol.RetirementReport{ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{1: 500}}, + }) + + _, err := p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, promotionRoundAOs(t, 4), kv, testBlobs) + require.NoError(t, err) + + require.Equal(t, string(protocol.LifeCycleStageProduction), string(kv.m[string(keyLifecycle)])) + stored, err := readPredecessorConfig(kv) + require.NoError(t, err) + require.NotNil(t, stored, "the agreed signer set must be replicated in c/pred") + require.Equal(t, testPredecessorSigners, stored.signers) +} + +// Test_StateTransition_LaggingPollersDoNotFork is the finding itself: nodes +// whose config poller has not stored the predecessor config cannot verify +// locally. Their observations must still count in full, and the round must +// produce the same state as one where every node was caught up, since f+1 +// voters are enough to agree on c/pred. +func Test_StateTransition_LaggingPollersDoNotFork(t *testing.T) { + ctx := tests.Context(t) + report := protocol.RetirementReport{ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{1: 500}} + + pAll, kvAll := stagedPlugin(t, &mockPredecessorRetirementReportCache{report: report}) + precursorAll, err := pAll.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, promotionRoundAOs(t, 4), kvAll, testBlobs) + require.NoError(t, err) + + // Only f+1 = 2 of the 4 nodes had the config; the other two abstained. + pLagging, kvLagging := stagedPlugin(t, &mockPredecessorRetirementReportCache{report: report}) + precursorLagging, err := pLagging.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, promotionRoundAOs(t, 2), kvLagging, testBlobs) + require.NoError(t, err) + + require.Equal(t, precursorAll, precursorLagging, "a lagging poller must not change the state transition") + require.Equal(t, string(protocol.LifeCycleStageProduction), string(kvLagging.m[string(keyLifecycle)])) + require.Equal(t, kvAll.m[string(keyPredecessorConfig)], kvLagging.m[string(keyPredecessorConfig)]) + + // The abstaining nodes' observations still counted: the median timestamp is + // over all four, not just the two that voted. + out, err := decodePrecursor(precursorLagging) + require.NoError(t, err) + require.Equal(t, uint64(3000), out.ObservationTimestampNanoseconds) +} + +// Test_StateTransition_PredecessorConfigNeedsQuorum checks the vote threshold: +// a single voter is not enough to install a signer set, so nothing is written +// and no report is verified. The round itself still completes. +func Test_StateTransition_PredecessorConfigNeedsQuorum(t *testing.T) { + ctx := tests.Context(t) + p, kv := stagedPlugin(t, &mockPredecessorRetirementReportCache{ + report: protocol.RetirementReport{ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{1: 500}}, + }) + + precursor, err := p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, promotionRoundAOs(t, 1), kv, testBlobs) + require.NoError(t, err) + + require.Equal(t, string(protocol.LifeCycleStageStaging), string(kv.m[string(keyLifecycle)])) + stored, err := readPredecessorConfig(kv) + require.NoError(t, err) + require.Nil(t, stored, "one vote is not more than f") + + out, err := decodePrecursor(precursor) + require.NoError(t, err) + require.Equal(t, uint64(3000), out.ObservationTimestampNanoseconds) +} + +// Test_StateTransition_PredecessorConfigIsWriteOnce guards the forgery path: if +// the agreed signer set could be revoted, a coalition that later reaches f+1 +// could install its own signers and attest a handover that never happened. +func Test_StateTransition_PredecessorConfigIsWriteOnce(t *testing.T) { + ctx := tests.Context(t) + prrc := &mockPredecessorRetirementReportCache{report: protocol.RetirementReport{}} + p, kv := stagedPlugin(t, prrc) + + // Round 2 agrees on the signer set but carries no retirement report, so the + // instance stays in staging and keeps voting. + noReport := make([]ocrtypes.AttributedObservation, 0, 4) + for i := 0; i < 4; i++ { + noReport = append(noReport, ao(i, mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 1000, PredecessorSigners: testPredecessorSigners}))) + } + _, err := p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, noReport, kv, testBlobs) + require.NoError(t, err) + agreed := kv.m[string(keyPredecessorConfig)] + require.NotEmpty(t, agreed) + + // Round 3: every node votes for a different signer set. + attacker := [][]byte{{0xFF}, {0xFE}} + revote := make([]ocrtypes.AttributedObservation, 0, 4) + for i := 0; i < 4; i++ { + revote = append(revote, ao(i, mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 2000, PredecessorSigners: attacker}))) + } + _, err = p.StateTransition(ctx, 3, ocrtypes.AttributedQuery{}, revote, kv, testBlobs) + require.NoError(t, err) + require.Equal(t, agreed, kv.m[string(keyPredecessorConfig)], "c/pred must be written at most once") +} + +// Test_StateTransition_InvalidRetirementReport_KeepsObservation covers the +// verification against the agreed signer set fails identically on every oracle, +// so only the retirement report is ignored: the observation's timestamp, +// votes and stream values still count. Otherwise one malformed field would +// silently remove an oracle from the round. +func Test_StateTransition_InvalidRetirementReport_KeepsObservation(t *testing.T) { + ctx := tests.Context(t) + p, kv := stagedPlugin(t, &mockPredecessorRetirementReportCache{verifyErr: errors.New("Verify failed; not enough valid signatures")}) + + precursor, err := p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, promotionRoundAOs(t, 4), kv, testBlobs) + require.NoError(t, err, "a deterministic verification failure must not fail the round") + require.Equal(t, string(protocol.LifeCycleStageStaging), string(kv.m[string(keyLifecycle)]), + "an unverifiable retirement report must not promote") + + // Every observation still counted: the median is over all four timestamps. + out, err := decodePrecursor(precursor) + require.NoError(t, err) + require.Equal(t, uint64(3000), out.ObservationTimestampNanoseconds) +} + +// Test_Observation_PredecessorConfigVote covers the observation side: a staging +// node votes its local signer set until c/pred exists, abstains when its poller +// has nothing, and stops voting once the set is replicated. +func Test_Observation_PredecessorConfigVote(t *testing.T) { + ctx := tests.Context(t) + + // Observation reads the node-local caches; the vote is the only one under + // test, so the rest just answer. + stagedObserver := func(t *testing.T, prrc *mockPredecessorRetirementReportCache) (*Plugin, *memKV) { + t.Helper() + p, kv := stagedPlugin(t, prrc) + p.ShouldRetireCache = &mockShouldRetireCache{} + p.ChannelDefinitionCache = &mockChannelDefinitionCache{defs: llotypes.ChannelDefinitions{}} + return p, kv + } + + t.Run("votes while c/pred is absent", func(t *testing.T) { + p, kv := stagedObserver(t, &mockPredecessorRetirementReportCache{}) + obsBytes, err := p.Observation(ctx, 2, ocrtypes.AttributedQuery{}, kv, nil) + require.NoError(t, err) + obs, err := decodeObservation(ctx, obsBytes, testBlobs, nil) + require.NoError(t, err) + require.Equal(t, testPredecessorSigners, obs.PredecessorSigners) + }) + + t.Run("abstains when the poller has not caught up", func(t *testing.T) { + p, kv := stagedObserver(t, &mockPredecessorRetirementReportCache{noLocalConfig: true}) + obsBytes, err := p.Observation(ctx, 2, ocrtypes.AttributedQuery{}, kv, nil) + require.NoError(t, err, "a lagging poller must not fail Observation") + obs, err := decodeObservation(ctx, obsBytes, testBlobs, nil) + require.NoError(t, err) + require.Empty(t, obs.PredecessorSigners) + }) + + t.Run("stops voting once c/pred is agreed", func(t *testing.T) { + p, kv := stagedObserver(t, &mockPredecessorRetirementReportCache{}) + require.NoError(t, writePredecessorConfig(kv, predecessorConfig{signers: testPredecessorSigners})) + obsBytes, err := p.Observation(ctx, 2, ocrtypes.AttributedQuery{}, kv, nil) + require.NoError(t, err) + obs, err := decodeObservation(ctx, obsBytes, testBlobs, nil) + require.NoError(t, err) + require.Empty(t, obs.PredecessorSigners) + }) +} diff --git a/llo/dev/v31/golden_test.go b/llo/dev/v31/golden_test.go index 566f302..d786a54 100644 --- a/llo/dev/v31/golden_test.go +++ b/llo/dev/v31/golden_test.go @@ -138,6 +138,10 @@ func Test_Golden_KVRecords(t *testing.T) { }, logger.Test(t), )) + require.NoError(t, writePredecessorConfig(kv, predecessorConfig{ + signers: [][]byte{{0xAA, 0xBB}, {0xCC}, {0xDD}, {0xEE}}, + f: 1, + })) require.NoError(t, writeHistoryLayoutVersion(kv)) require.NoError(t, writeHistoryIndex(kv, []histKey{ {streamID: 100, aggregator: llotypes.AggregatorMedian}, @@ -152,6 +156,7 @@ func Test_Golden_KVRecords(t *testing.T) { {"kv_channel_state.bin", keyChannelState}, {"kv_channel_seqnr.bin", keyChannelSeqNr}, {"kv_hot_state.bin", keyHotState}, + {"kv_predecessor_config.bin", keyPredecessorConfig}, {"kv_history_version.bin", keyHistoryVersion}, {"kv_history_index.bin", keyHistoryIndex}, } { @@ -174,6 +179,10 @@ func Test_Golden_KVRecords(t *testing.T) { require.Equal(t, map[llotypes.ChannelID]bool{3: true, 2: true}, s.reportedLastRound) require.Len(t, s.carryForward, 2) + pc, err := readPredecessorConfig(kv) + require.NoError(t, err) + require.Equal(t, &predecessorConfig{signers: [][]byte{{0xAA, 0xBB}, {0xCC}, {0xDD}, {0xEE}}, f: 1}, pc) + version, err := readHistoryLayoutVersion(kv) require.NoError(t, err) require.Equal(t, historyLayoutVersion, version) @@ -240,6 +249,7 @@ func Test_Golden_KVKeys(t *testing.T) { {"c/defs", string(keyChannelState)}, {"c/seqnr", string(keyChannelSeqNr)}, {"r/agg", string(keyHotState)}, + {"c/pred", string(keyPredecessorConfig)}, {"hidx", string(keyHistoryIndex)}, {"hv", string(keyHistoryVersion)}, } { diff --git a/llo/dev/v31/kv.go b/llo/dev/v31/kv.go index 0badbfe..5ff7a84 100644 --- a/llo/dev/v31/kv.go +++ b/llo/dev/v31/kv.go @@ -22,6 +22,9 @@ import ( // c/defs -> LLOChannelStateProto: every live channel definition // (written only when the definitions change) // c/seqnr -> uint64 BE seqNr of the last c/defs write +// c/pred -> LLOPredecessorConfigProto: the predecessor's signer set and +// f, agreed by vote while staging (written at most once, and +// only by an instance that has a predecessor) // r/agg -> LLOHotStateProto: observation timestamp, validAfter // watermarks, per-channel reportability, and carry-forward // timestamped aggregates (written every round) @@ -42,10 +45,11 @@ import ( // a read cost of depth/chunkSize point reads instead of one. See // protocol.RingWindow. var ( - keyLifecycle = []byte("c/lifecycle") - keyChannelState = []byte("c/defs") - keyChannelSeqNr = []byte("c/seqnr") - keyHotState = []byte("r/agg") + keyLifecycle = []byte("c/lifecycle") + keyChannelState = []byte("c/defs") + keyChannelSeqNr = []byte("c/seqnr") + keyPredecessorConfig = []byte("c/pred") + keyHotState = []byte("r/agg") keyHistoryIndex = []byte("hidx") keyHistoryVersion = []byte("hv") @@ -204,6 +208,47 @@ func readChannelState(r ocr3_1types.KeyValueStateReader) (llotypes.ChannelDefini return defs, nil } +// predecessorConfig is the decoded c/pred record: the signer set and f of the +// predecessor instance, which is what verifying an attested predecessor +// retirement report needs. +type predecessorConfig struct { + signers [][]byte + f uint8 +} + +// readPredecessorConfig reads and decodes the c/pred record, returning nil when +// it has not been agreed yet. +// +// Only a staging instance that has a predecessor ever reads or writes this key, +// so every other instance pays nothing for it. +func readPredecessorConfig(r ocr3_1types.KeyValueStateReader) (*predecessorConfig, error) { + b, err := r.Read(keyPredecessorConfig) + if err != nil { + return nil, fmt.Errorf("read predecessor config: %w", err) + } + if len(b) == 0 { + return nil, nil + } + pb := &protocol.LLOPredecessorConfigProto{} + if err := proto.Unmarshal(b, pb); err != nil { + return nil, fmt.Errorf("unmarshal predecessor config: %w", err) + } + if pb.F > 255 { + return nil, fmt.Errorf("predecessor config has f out of range: %d", pb.F) + } + return &predecessorConfig{signers: pb.Signers, f: uint8(pb.F)}, nil +} + +// writePredecessorConfig persists the agreed c/pred record. Signer order is +// preserved: a signature names its signer by index into the set. +func writePredecessorConfig(w ocr3_1types.KeyValueStateReadWriter, pc predecessorConfig) error { + b, err := deterministicMarshal.Marshal(&protocol.LLOPredecessorConfigProto{Signers: pc.signers, F: uint32(pc.f)}) + if err != nil { + return fmt.Errorf("marshal predecessor config: %w", err) + } + return w.Write(keyPredecessorConfig, b) +} + // readHotState reads and decodes the r/agg record into s. func readHotState(r ocr3_1types.KeyValueStateReader, s *kvState) error { b, err := r.Read(keyHotState) diff --git a/llo/dev/v31/observation.go b/llo/dev/v31/observation.go index 81cb16a..03b5146 100644 --- a/llo/dev/v31/observation.go +++ b/llo/dev/v31/observation.go @@ -30,6 +30,15 @@ type Observation struct { // read without forking; advertising it here turns it into a replicated fact // that reportability can gate on (see isReportable). SupportedReportFormats []llotypes.ReportFormat + // PredecessorSigners and PredecessorF are the predecessor instance's signer + // set and f, read from the node-local retirement report cache. A staging + // instance carries them until c/pred is agreed, which turns them into a + // replicated fact the state transition can verify retirement reports + // against; see readPredecessorConfig. + // + // Signer order is significant: a signature names its signer by index. + PredecessorSigners [][]byte + PredecessorF uint8 } // observationWireVersion is the leading byte of the v31 observation framing. @@ -59,6 +68,8 @@ func encodeObservation(obs Observation, handles [][]byte) (ocrtypes.Observation, AttestedPredecessorRetirement: obs.AttestedPredecessorRetirement, ShouldRetire: obs.ShouldRetire, UnixTimestampNanoseconds: obs.UnixTimestampNanoseconds, + PredecessorSigners: obs.PredecessorSigners, + PredecessorF: uint32(obs.PredecessorF), } for id := range obs.RemoveChannelIDs { main.RemoveChannelIDs = append(main.RemoveChannelIDs, id) @@ -259,6 +270,23 @@ func observationFromProto(main *protocol.LLOObservationProto) (Observation, erro return Observation{}, fmt.Errorf("observation advertises too many report formats: %d (max %d)", len(main.SupportedReportFormats), protocol.MaxObservationSupportedReportFormatsLength) } + if len(main.PredecessorSigners) > protocol.MaxObservationPredecessorSignersLength { + return Observation{}, fmt.Errorf("observation carries too many predecessor signers: %d (max %d)", len(main.PredecessorSigners), protocol.MaxObservationPredecessorSignersLength) + } + for i, signer := range main.PredecessorSigners { + if len(signer) == 0 || len(signer) > protocol.MaxPredecessorSignerBytes { + return Observation{}, fmt.Errorf("observation carries predecessor signer %d of invalid length %d (max %d)", i, len(signer), protocol.MaxPredecessorSignerBytes) + } + } + // f indexes nothing, but a set that cannot reach f+1 valid signatures could + // never verify a report, so treat it as malformed rather than carrying it + // into the vote. + if main.PredecessorF > 0 && int(main.PredecessorF) >= len(main.PredecessorSigners) { + return Observation{}, fmt.Errorf("observation carries predecessor f=%d for a signer set of %d", main.PredecessorF, len(main.PredecessorSigners)) + } + obs.PredecessorSigners = main.PredecessorSigners + obs.PredecessorF = uint8(main.PredecessorF) + obs.SupportedReportFormats = sortedUniqueFormatsFromWire(main.SupportedReportFormats) return obs, nil } diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index dc9f565..207b5e4 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -145,6 +145,7 @@ func (p *Plugin) Observation(_ context.Context, seqNr uint64, _ ocrtypes.Attribu obs.AttestedPredecessorRetirement = nil p.Logger.Errorw("Failed to fetch attested retirement report from cache, omitting it from this observation", "stage", "Observation", "seqNr", seqNr, "err", err) } + p.voteOnPredecessorConfig(&obs, kvReader, seqNr) } obs.ShouldRetire, err = p.ShouldRetireCache.ShouldRetire(p.ConfigDigest) @@ -260,6 +261,41 @@ func sortedChannelIDSet(set map[llotypes.ChannelID]struct{}) []llotypes.ChannelI return ids } +// voteOnPredecessorConfig populates obs.PredecessorSigners / obs.PredecessorF +// with the predecessor's signer set from the node-local retirement report +// cache, so the DON can agree on it and store it in c/pred. +// +// Verifying an attested predecessor retirement report needs that signer set. +// Reading it from the local cache inside the state transition would fork the +// state, because the config poller fills the cache asynchronously and a lagging +// node reaches a different verdict from one that is caught up. Voting on it +// here moves the node-local read into the observation, where oracles are +// allowed to differ, and leaves the state transition reading only replicated +// state. +func (p *Plugin) voteOnPredecessorConfig(obs *Observation, kvReader ocr3_1types.KeyValueStateReader, seqNr uint64) { + agreed, err := readPredecessorConfig(kvReader) + if err != nil { + p.Logger.Errorw("Failed to read agreed predecessor config, not voting on it this round", "stage", "Observation", "seqNr", seqNr, "err", err) + return + } + if agreed != nil { + // Already replicated, so the vote would be dead weight on every + // observation for the rest of the staging period. + return + } + signers, f, exists := p.PredecessorRetirementReportCache.PredecessorConfig(*p.PredecessorConfigDigest) + if !exists { + p.Logger.Warnw("Predecessor config not in the local cache yet, not voting on it this round", "stage", "Observation", "seqNr", seqNr, "predecessorConfigDigest", *p.PredecessorConfigDigest) + return + } + if len(signers) == 0 || len(signers) > protocol.MaxObservationPredecessorSignersLength { + p.Logger.Errorw("Local predecessor config has an unusable signer set, not voting on it", "stage", "Observation", "seqNr", seqNr, "signers", len(signers)) + return + } + obs.PredecessorSigners = signers + obs.PredecessorF = f +} + // voteOnChannels populates obs.RemoveChannelIDs / obs.UpdateChannelDefinitions // by comparing the desired channel definitions against current KV state. func (p *Plugin) voteOnChannels(obs *Observation, state *kvState) { @@ -317,6 +353,9 @@ func (p *Plugin) ValidateObservation(ctx context.Context, seqNr uint64, _ ocrtyp if p.PredecessorConfigDigest == nil && len(observation.AttestedPredecessorRetirement) != 0 { return errors.New("AttestedPredecessorRetirement is not empty even though this instance has no predecessor") } + if p.PredecessorConfigDigest == nil && len(observation.PredecessorSigners) != 0 { + return errors.New("PredecessorSigners is not empty even though this instance has no predecessor") + } if len(observation.UpdateChannelDefinitions) > protocol.MaxObservationUpdateChannelDefinitionsLength { return fmt.Errorf("UpdateChannelDefinitions is too long: %v vs %v", len(observation.UpdateChannelDefinitions), protocol.MaxObservationUpdateChannelDefinitionsLength) } diff --git a/llo/dev/v31/statetransition.go b/llo/dev/v31/statetransition.go index 39b2e84..a5ae3f8 100644 --- a/llo/dev/v31/statetransition.go +++ b/llo/dev/v31/statetransition.go @@ -3,6 +3,8 @@ package llo import ( "bytes" "context" + "crypto/sha256" + "encoding/binary" "errors" "fmt" "sort" @@ -70,26 +72,36 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A return nil, fmt.Errorf("failed to load KV state: %w", err) } - timestamps, validPredecessorRetirementReport, shouldRetireVotes, removeChannelVotesByID, updateDefsByHash, updateVotesByHash, supportByFormat, streamObservations, err := p.decodeObservations(ctx, aos, bf, p.BlobPayloads.round(seqNr)) + tally, err := p.decodeObservations(ctx, aos, bf, p.BlobPayloads.round(seqNr)) if err != nil { return nil, err } - if len(timestamps) == 0 { + if len(tally.timestampsNanoseconds) == 0 { return nil, fmt.Errorf("no valid observations") } + // Verifying an attested predecessor retirement report needs the + // predecessor's signer set, which is node-local until the DON agrees on it. + // Agree first, then verify against the agreed set, so every oracle reaches + // the same verdict. A set agreed this round is usable this round, so a + // handover normally costs no extra round. + validPredecessorRetirementReport, err := p.resolvePredecessorRetirement(kvRW, seqNr, prev.lifeCycleStage, tally) + if err != nil { + return nil, err + } + // The definitions in effect for this round are the ones Observation read. // Changes agreed below land in pending and take effect next round. effective := cloneChannelDefinitions(prev.channelDefinitions) pending := cloneChannelDefinitions(prev.channelDefinitions) out := precursor{ - ObservationTimestampNanoseconds: medianTimestamp(timestamps), + ObservationTimestampNanoseconds: medianTimestamp(tally.timestampsNanoseconds), ChannelDefinitions: effective, ChannelStateSeqNr: prev.channelStateSeqNr, ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{}, StreamAggregates: protocol.StreamAggregates{}, - SupportByFormat: supportByFormat, + SupportByFormat: tally.supportVotesByFormat, } // Lifecycle stage & promotion. @@ -101,7 +113,7 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A } else { out.LifeCycleStage = prev.lifeCycleStage } - if out.LifeCycleStage == protocol.LifeCycleStageProduction && shouldRetireVotes > p.F { + if out.LifeCycleStage == protocol.LifeCycleStageProduction && tally.shouldRetireVotes > p.F { p.Logger.Infow("Retiring production protocol instance âš°ï¸�", "seqNr", seqNr) out.LifeCycleStage = protocol.LifeCycleStageRetired } @@ -109,7 +121,7 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A // Channel definition changes (skipped once retired). These apply to pending // only: they take effect next round. if out.LifeCycleStage != protocol.LifeCycleStageRetired { - applyChannelVotes(pending, removeChannelVotesByID, updateDefsByHash, updateVotesByHash, p.F) + applyChannelVotes(pending, tally.removeChannelVotesByID, tally.updateChannelDefinitionsByHash, tally.updateChannelVotesByHash, p.F) } // validAfter. @@ -190,7 +202,7 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A // next round. Runs over the effective set, which is what was observed. The // agreed value of every pair history requires is recorded as it is computed. carryForward := map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue{} - if err := p.aggregate(prev.carryForward, carryForward, effective, streamObservations, out.StreamAggregates, + if err := p.aggregate(prev.carryForward, carryForward, effective, tally.streamObservations, out.StreamAggregates, history, requirements, out.ObservationTimestampNanoseconds); err != nil { return nil, err } @@ -223,22 +235,35 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A return encodePrecursor(out) } -func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.AttributedObservation, bf ocr3_1types.BlobFetcher, memo *roundBlobPayloads) ( - timestampsNanoseconds []uint64, - validPredecessorRetirementReport *protocol.RetirementReport, - shouldRetireVotes int, - removeChannelVotesByID map[llotypes.ChannelID]int, - updateChannelDefinitionsByHash map[[32]byte]protocol.ChannelDefinitionWithID, - updateChannelVotesByHash map[[32]byte]int, - supportVotesByFormat map[llotypes.ReportFormat]int, - streamObservations map[llotypes.StreamID][]protocol.StreamValue, - err error, -) { - removeChannelVotesByID = make(map[llotypes.ChannelID]int) - supportVotesByFormat = make(map[llotypes.ReportFormat]int) - updateChannelDefinitionsByHash = make(map[[32]byte]protocol.ChannelDefinitionWithID) - updateChannelVotesByHash = make(map[[32]byte]int) - streamObservations = make(map[llotypes.StreamID][]protocol.StreamValue) +// observationTally is what the round's observations add up to: the votes, +// timestamps and stream values the state transition works from. Every field is +// accumulated in aos order, so it does not depend on which observation decoded +// first. +type observationTally struct { + timestampsNanoseconds []uint64 + // attestedRetirements are collected but not verified: verification needs + // the agreed predecessor signer set, which the votes below decide. + attestedRetirements [][]byte + predConfigsByHash map[[32]byte]predecessorConfig + predConfigVotesByHash map[[32]byte]int + shouldRetireVotes int + removeChannelVotesByID map[llotypes.ChannelID]int + updateChannelDefinitionsByHash map[[32]byte]protocol.ChannelDefinitionWithID + updateChannelVotesByHash map[[32]byte]int + supportVotesByFormat map[llotypes.ReportFormat]int + streamObservations map[llotypes.StreamID][]protocol.StreamValue +} + +func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.AttributedObservation, bf ocr3_1types.BlobFetcher, memo *roundBlobPayloads) (observationTally, error) { + tally := observationTally{ + predConfigsByHash: make(map[[32]byte]predecessorConfig), + predConfigVotesByHash: make(map[[32]byte]int), + removeChannelVotesByID: make(map[llotypes.ChannelID]int), + updateChannelDefinitionsByHash: make(map[[32]byte]protocol.ChannelDefinitionWithID), + updateChannelVotesByHash: make(map[[32]byte]int), + supportVotesByFormat: make(map[llotypes.ReportFormat]int), + streamObservations: make(map[llotypes.StreamID][]protocol.StreamValue), + } // Decode concurrently: each observation may reference blobs that are not yet // assembled locally, and waiting for one serially delays every other. The @@ -266,8 +291,7 @@ func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.Attribut // non-deterministic; instead abort the round so every oracle // retries uniformly. Determinism is not required when returning // an error (see the ReportingPlugin contract). - err = fmt.Errorf("failed to fetch blob for observation from oracle %v: %w", ao.Observer, derr) - return + return observationTally{}, fmt.Errorf("failed to fetch blob for observation from oracle %v: %w", ao.Observer, derr) } // Deterministic decode failure (same bytes on every oracle): safe to // drop just this observation. @@ -275,43 +299,45 @@ func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.Attribut continue } - if len(observation.AttestedPredecessorRetirement) != 0 && validPredecessorRetirementReport == nil && p.PredecessorConfigDigest != nil { - pcd := *p.PredecessorConfigDigest - retirementReport, cerr := p.PredecessorRetirementReportCache.CheckAttestedRetirementReport(pcd, observation.AttestedPredecessorRetirement) - if cerr != nil { - p.Logger.Warnw("ignoring observation with invalid attested predecessor retirement", "oracleID", ao.Observer, "error", cerr, "predecessorConfigDigest", pcd) - continue + if p.PredecessorConfigDigest != nil { + if len(observation.AttestedPredecessorRetirement) != 0 { + tally.attestedRetirements = append(tally.attestedRetirements, observation.AttestedPredecessorRetirement) + } + if len(observation.PredecessorSigners) > 0 { + pc := predecessorConfig{signers: observation.PredecessorSigners, f: observation.PredecessorF} + h := hashPredecessorConfig(pc) + tally.predConfigVotesByHash[h]++ + tally.predConfigsByHash[h] = pc } - validPredecessorRetirementReport = &retirementReport } if observation.ShouldRetire { - shouldRetireVotes++ + tally.shouldRetireVotes++ } - timestampsNanoseconds = append(timestampsNanoseconds, observation.UnixTimestampNanoseconds) + tally.timestampsNanoseconds = append(tally.timestampsNanoseconds, observation.UnixTimestampNanoseconds) // Deduped by decodeObservation, so one oracle contributes at most one // vote per format. for _, format := range observation.SupportedReportFormats { - supportVotesByFormat[format]++ + tally.supportVotesByFormat[format]++ } for channelID := range observation.RemoveChannelIDs { - removeChannelVotesByID[channelID]++ + tally.removeChannelVotesByID[channelID]++ } for channelID, channelDefinition := range observation.UpdateChannelDefinitions { defWithID := protocol.ChannelDefinitionWithID{ChannelDefinition: channelDefinition, ChannelID: channelID} h := makeChannelHash(defWithID) - updateChannelVotesByHash[h]++ - updateChannelDefinitionsByHash[h] = defWithID + tally.updateChannelVotesByHash[h]++ + tally.updateChannelDefinitionsByHash[h] = defWithID } for id, sv := range observation.StreamValues { if sv == nil { continue } - streamObservations[id] = append(streamObservations[id], sv) + tally.streamObservations[id] = append(tally.streamObservations[id], sv) } } - return + return tally, nil } // applyChannelVotes applies remove/add votes with a >F threshold, in ascending @@ -597,6 +623,105 @@ func medianTimestamp(timestampsNanoseconds []uint64) uint64 { // makeChannelHash delegates to the shared implementation so that v3.0 running // protocol version 2 and v3.1 cannot drift apart on channel identity. +// hashPredecessorConfig identifies a candidate predecessor config so votes for +// the same one can be tallied. Signer order is part of the identity: a +// signature names its signer by index, so two sets differing only in order are +// different configs. +func hashPredecessorConfig(pc predecessorConfig) [32]byte { + h := sha256.New() + var buf [8]byte + binary.BigEndian.PutUint64(buf[:], uint64(pc.f)) + h.Write(buf[:]) + for _, signer := range pc.signers { + binary.BigEndian.PutUint64(buf[:], uint64(len(signer))) + h.Write(buf[:]) + h.Write(signer) + } + var out [32]byte + copy(out[:], h.Sum(nil)) + return out +} + +// resolvePredecessorRetirement agrees on the predecessor's signer set, then +// verifies this round's attested retirement reports against it. +// +// The signer set is written to c/pred once more than f oracles vote for the +// same one, so at least one honest oracle vouches for it. It is written at most +// once: if it could be revoted, a coalition that later reaches f+1 could swap +// in a signer set of its own and forge a retirement report, promoting this +// instance on a handover that never happened. +// +// Everything here reads replicated state only, so every oracle reaches the same +// verdict. A report that fails verification is ignored, not fatal: the bytes +// are the same everywhere, so ignoring them is deterministic too. +func (p *Plugin) resolvePredecessorRetirement( + kvRW ocr3_1types.KeyValueStateReadWriter, + seqNr uint64, + stage llotypes.LifeCycleStage, + tally observationTally, +) (*protocol.RetirementReport, error) { + // Only a staging instance with a predecessor has a handover to complete. + if p.PredecessorConfigDigest == nil || stage != protocol.LifeCycleStageStaging { + return nil, nil + } + + agreed, err := readPredecessorConfig(kvRW) + if err != nil { + return nil, err + } + if agreed == nil { + if elected := electPredecessorConfig(tally.predConfigsByHash, tally.predConfigVotesByHash, p.F); elected != nil { + if err := writePredecessorConfig(kvRW, *elected); err != nil { + return nil, err + } + p.Logger.Infow("Agreed on predecessor config", "seqNr", seqNr, "signers", len(elected.signers), "f", elected.f, "predecessorConfigDigest", *p.PredecessorConfigDigest) + agreed = elected + } + } + if agreed == nil { + if len(tally.attestedRetirements) > 0 { + p.Logger.Warnw("Ignoring attested predecessor retirement reports: the predecessor config is not agreed yet", "seqNr", seqNr, "reports", len(tally.attestedRetirements)) + } + return nil, nil + } + + for _, attested := range tally.attestedRetirements { + retirementReport, verr := p.PredecessorRetirementReportCache.VerifyAttestedRetirementReport(*p.PredecessorConfigDigest, agreed.signers, agreed.f, attested) + if verr != nil { + p.Logger.Warnw("Ignoring invalid attested predecessor retirement", "seqNr", seqNr, "error", verr, "predecessorConfigDigest", *p.PredecessorConfigDigest) + continue + } + return &retirementReport, nil + } + return nil, nil +} + +// electPredecessorConfig returns the candidate with more than f votes, or nil. +// +// More than f votes means at least one honest oracle voted for the winner, and +// honest oracles read the set from the predecessor's onchain config, which is +// immutable for a given config digest. So the winner is always the real signer +// set: the f byzantine oracles cannot reach the threshold on their own, and +// there is no honest set for them to outvote. +// +// Candidates are still considered in hash order, so that a tie, which needs +// honest oracles to disagree and therefore should not happen, resolves the same +// way on every oracle instead of following map iteration. +func electPredecessorConfig(byHash map[[32]byte]predecessorConfig, votesByHash map[[32]byte]int, f int) *predecessorConfig { + hashes := make([][32]byte, 0, len(byHash)) + for h := range byHash { + hashes = append(hashes, h) + } + sort.Slice(hashes, func(i, j int) bool { return bytes.Compare(hashes[i][:], hashes[j][:]) < 0 }) + for _, h := range hashes { + if votesByHash[h] > f { + pc := byHash[h] + return &pc + } + } + return nil +} + func makeChannelHash(cd protocol.ChannelDefinitionWithID) [32]byte { return protocol.ChannelHashV2(cd) } diff --git a/llo/dev/v31/testdata/golden/kv_predecessor_config.bin b/llo/dev/v31/testdata/golden/kv_predecessor_config.bin new file mode 100644 index 0000000..ec9517b --- /dev/null +++ b/llo/dev/v31/testdata/golden/kv_predecessor_config.bin @@ -0,0 +1,5 @@ + +ª» +Ì +Ý +î \ No newline at end of file diff --git a/llo/protocol/lifecycle.go b/llo/protocol/lifecycle.go index adf0fe0..c50e8dc 100644 --- a/llo/protocol/lifecycle.go +++ b/llo/protocol/lifecycle.go @@ -45,6 +45,23 @@ type PredecessorRetirementReportCache interface { AttestedRetirementReport(predecessorConfigDigest ocr2types.ConfigDigest) ([]byte, error) // CheckAttestedRetirementReport verifies that an attested retirement // report, which may have come from another node, is valid (signed) with - // signers corresponding to the given config digest + // signers corresponding to the given config digest. + // + // The signer set is read from the local cache, which the config poller + // fills asynchronously, so the verdict is node-local: a caller that must + // reach the same verdict on every node should agree on the signer set + // first and use VerifyAttestedRetirementReport instead. CheckAttestedRetirementReport(predecessorConfigDigest ocr2types.ConfigDigest, attestedRetirementReport []byte) (RetirementReport, error) + // PredecessorConfig returns the predecessor's signer set and f from the + // local cache. exists is false while the config poller has not stored the + // config yet. + // + // The order of signers is significant: a signature in an attested + // retirement report names its signer by index into this slice. + PredecessorConfig(predecessorConfigDigest ocr2types.ConfigDigest) (signers [][]byte, f uint8, exists bool) + // VerifyAttestedRetirementReport verifies an attested retirement report + // against an explicitly supplied signer set, reading no local state. Given + // the same arguments it returns the same result on every node, so callers + // that have agreed on the signer set can use it inside a state transition. + VerifyAttestedRetirementReport(predecessorConfigDigest ocr2types.ConfigDigest, signers [][]byte, f uint8, attestedRetirementReport []byte) (RetirementReport, error) } diff --git a/llo/protocol/limits.go b/llo/protocol/limits.go index f6ce5f2..e3c89d8 100644 --- a/llo/protocol/limits.go +++ b/llo/protocol/limits.go @@ -32,6 +32,14 @@ const ( // of entries; the headroom keeps the bound from needing revision while // still stopping a peer from padding its observation. MaxObservationSupportedReportFormatsLength = 32 + // MaxObservationPredecessorSignersLength bounds the predecessor signer set + // an observation may carry. libocr allows at most types.MaxOracles signers + // in a config, so anything longer cannot be a real predecessor config. + MaxObservationPredecessorSignersLength = 31 + // MaxPredecessorSignerBytes bounds one onchain public key in that set. The + // largest key any supported chain uses is well under this; the headroom + // stops a peer from padding its observation with oversized entries. + MaxPredecessorSignerBytes = 128 // Maximum allowed number of streams per channel MaxStreamsPerChannel = 10_000 // MaxDecimalExponent bounds the absolute value of the base-10 exponent of diff --git a/llo/protocol/plugin_codecs.pb.go b/llo/protocol/plugin_codecs.pb.go index 1b85958..4524712 100644 --- a/llo/protocol/plugin_codecs.pb.go +++ b/llo/protocol/plugin_codecs.pb.go @@ -95,8 +95,15 @@ type LLOObservationProto struct { // cannot be read in the state transition; this carries it as a replicated // fact instead. SupportedReportFormats []uint32 `protobuf:"varint,8,rep,packed,name=supportedReportFormats,proto3" json:"supportedReportFormats,omitempty"` - unknownFields protoimpl.UnknownFields - sizeCache protoimpl.SizeCache + // The predecessor instance's signer set and f, read from the node-local + // retirement report cache. Carried only by a v31 staging instance that has + // not yet agreed on c/pred, so the state transition can verify attested + // retirement reports against replicated state instead of node-local state. + // Order is significant: a signature names its signer by index into it. + PredecessorSigners [][]byte `protobuf:"bytes,9,rep,name=predecessorSigners,proto3" json:"predecessorSigners,omitempty"` + PredecessorF uint32 `protobuf:"varint,10,opt,name=predecessorF,proto3" json:"predecessorF,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache } func (x *LLOObservationProto) Reset() { @@ -185,6 +192,20 @@ func (x *LLOObservationProto) GetSupportedReportFormats() []uint32 { return nil } +func (x *LLOObservationProto) GetPredecessorSigners() [][]byte { + if x != nil { + return x.PredecessorSigners + } + return nil +} + +func (x *LLOObservationProto) GetPredecessorF() uint32 { + if x != nil { + return x.PredecessorF + } + return 0 +} + type LLOStreamValue struct { state protoimpl.MessageState `protogen:"open.v1"` Type LLOStreamValue_Type `protobuf:"varint,1,opt,name=type,proto3,enum=v1.LLOStreamValue_Type" json:"type,omitempty"` @@ -1170,6 +1191,70 @@ func (x *LLOChannelStateProto) GetChannelDefinitions() []*LLOChannelIDAndDefinit return nil } +// LLOPredecessorConfigProto is the v31 KeyValueState record holding the +// predecessor instance's signer set and f under a single key (c/pred), agreed +// by vote while staging and written at most once. +// +// It exists so that verifying an attested predecessor retirement report reads +// only replicated state. The same data is available node-locally from the +// retirement report cache, but that cache is filled asynchronously by the +// config poller, so oracles reach different verdicts from it and the state +// transition would fork. +// +// NOTE: must serialize deterministically. signers MUST keep the order the +// predecessor's config gives them, since a signature names its signer by index. +type LLOPredecessorConfigProto struct { + state protoimpl.MessageState `protogen:"open.v1"` + Signers [][]byte `protobuf:"bytes,1,rep,name=signers,proto3" json:"signers,omitempty"` + F uint32 `protobuf:"varint,2,opt,name=f,proto3" json:"f,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache +} + +func (x *LLOPredecessorConfigProto) Reset() { + *x = LLOPredecessorConfigProto{} + mi := &file_plugin_codecs_proto_msgTypes[17] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) +} + +func (x *LLOPredecessorConfigProto) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*LLOPredecessorConfigProto) ProtoMessage() {} + +func (x *LLOPredecessorConfigProto) ProtoReflect() protoreflect.Message { + mi := &file_plugin_codecs_proto_msgTypes[17] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use LLOPredecessorConfigProto.ProtoReflect.Descriptor instead. +func (*LLOPredecessorConfigProto) Descriptor() ([]byte, []int) { + return file_plugin_codecs_proto_rawDescGZIP(), []int{17} +} + +func (x *LLOPredecessorConfigProto) GetSigners() [][]byte { + if x != nil { + return x.Signers + } + return nil +} + +func (x *LLOPredecessorConfigProto) GetF() uint32 { + if x != nil { + return x.F + } + return 0 +} + // LLOHotStateProto is the v31 KeyValueState record holding the per-round // ("hot") state under a single key (r/agg): the state that changes on // essentially every round. @@ -1194,7 +1279,7 @@ type LLOHotStateProto struct { func (x *LLOHotStateProto) Reset() { *x = LLOHotStateProto{} - mi := &file_plugin_codecs_proto_msgTypes[17] + mi := &file_plugin_codecs_proto_msgTypes[18] ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) ms.StoreMessageInfo(mi) } @@ -1206,7 +1291,7 @@ func (x *LLOHotStateProto) String() string { func (*LLOHotStateProto) ProtoMessage() {} func (x *LLOHotStateProto) ProtoReflect() protoreflect.Message { - mi := &file_plugin_codecs_proto_msgTypes[17] + mi := &file_plugin_codecs_proto_msgTypes[18] if x != nil { ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) if ms.LoadMessageInfo() == nil { @@ -1219,7 +1304,7 @@ func (x *LLOHotStateProto) ProtoReflect() protoreflect.Message { // Deprecated: Use LLOHotStateProto.ProtoReflect.Descriptor instead. func (*LLOHotStateProto) Descriptor() ([]byte, []int) { - return file_plugin_codecs_proto_rawDescGZIP(), []int{17} + return file_plugin_codecs_proto_rawDescGZIP(), []int{18} } func (x *LLOHotStateProto) GetObservationTimestampNanoseconds() uint64 { @@ -1276,7 +1361,7 @@ type LLOPrecursorProto struct { func (x *LLOPrecursorProto) Reset() { *x = LLOPrecursorProto{} - mi := &file_plugin_codecs_proto_msgTypes[18] + mi := &file_plugin_codecs_proto_msgTypes[19] ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) ms.StoreMessageInfo(mi) } @@ -1288,7 +1373,7 @@ func (x *LLOPrecursorProto) String() string { func (*LLOPrecursorProto) ProtoMessage() {} func (x *LLOPrecursorProto) ProtoReflect() protoreflect.Message { - mi := &file_plugin_codecs_proto_msgTypes[18] + mi := &file_plugin_codecs_proto_msgTypes[19] if x != nil { ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) if ms.LoadMessageInfo() == nil { @@ -1301,7 +1386,7 @@ func (x *LLOPrecursorProto) ProtoReflect() protoreflect.Message { // Deprecated: Use LLOPrecursorProto.ProtoReflect.Descriptor instead. func (*LLOPrecursorProto) Descriptor() ([]byte, []int) { - return file_plugin_codecs_proto_rawDescGZIP(), []int{18} + return file_plugin_codecs_proto_rawDescGZIP(), []int{19} } func (x *LLOPrecursorProto) GetLifeCycleStage() string { @@ -1365,7 +1450,7 @@ type LLOReportFormatSupportProto struct { func (x *LLOReportFormatSupportProto) Reset() { *x = LLOReportFormatSupportProto{} - mi := &file_plugin_codecs_proto_msgTypes[19] + mi := &file_plugin_codecs_proto_msgTypes[20] ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) ms.StoreMessageInfo(mi) } @@ -1377,7 +1462,7 @@ func (x *LLOReportFormatSupportProto) String() string { func (*LLOReportFormatSupportProto) ProtoMessage() {} func (x *LLOReportFormatSupportProto) ProtoReflect() protoreflect.Message { - mi := &file_plugin_codecs_proto_msgTypes[19] + mi := &file_plugin_codecs_proto_msgTypes[20] if x != nil { ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) if ms.LoadMessageInfo() == nil { @@ -1390,7 +1475,7 @@ func (x *LLOReportFormatSupportProto) ProtoReflect() protoreflect.Message { // Deprecated: Use LLOReportFormatSupportProto.ProtoReflect.Descriptor instead. func (*LLOReportFormatSupportProto) Descriptor() ([]byte, []int) { - return file_plugin_codecs_proto_rawDescGZIP(), []int{19} + return file_plugin_codecs_proto_rawDescGZIP(), []int{20} } func (x *LLOReportFormatSupportProto) GetReportFormat() uint32 { @@ -1411,7 +1496,7 @@ var File_plugin_codecs_proto protoreflect.FileDescriptor const file_plugin_codecs_proto_rawDesc = "" + "\n" + - "\x13plugin_codecs.proto\x12\x02v1\"\xea\x05\n" + + "\x13plugin_codecs.proto\x12\x02v1\"\xbe\x06\n" + "\x13LLOObservationProto\x12D\n" + "\x1dattestedPredecessorRetirement\x18\x01 \x01(\fR\x1dattestedPredecessorRetirement\x12\"\n" + "\fshouldRetire\x18\x02 \x01(\bR\fshouldRetire\x12F\n" + @@ -1420,7 +1505,10 @@ const file_plugin_codecs_proto_rawDesc = "" + "\x10removeChannelIDs\x18\x04 \x03(\rR\x10removeChannelIDs\x12q\n" + "\x18updateChannelDefinitions\x18\x05 \x03(\v25.v1.LLOObservationProto.UpdateChannelDefinitionsEntryR\x18updateChannelDefinitions\x12M\n" + "\fstreamValues\x18\x06 \x03(\v2).v1.LLOObservationProto.StreamValuesEntryR\fstreamValues\x126\n" + - "\x16supportedReportFormats\x18\b \x03(\rR\x16supportedReportFormats\x1aj\n" + + "\x16supportedReportFormats\x18\b \x03(\rR\x16supportedReportFormats\x12.\n" + + "\x12predecessorSigners\x18\t \x03(\fR\x12predecessorSigners\x12\"\n" + + "\fpredecessorF\x18\n" + + " \x01(\rR\fpredecessorF\x1aj\n" + "\x1dUpdateChannelDefinitionsEntry\x12\x10\n" + "\x03key\x18\x01 \x01(\rR\x03key\x123\n" + "\x05value\x18\x02 \x01(\v2\x1d.v1.LLOChannelDefinitionProtoR\x05value:\x028\x01\x1aS\n" + @@ -1496,7 +1584,10 @@ const file_plugin_codecs_proto_rawDesc = "" + "aggregator\x18\x03 \x01(\rR\n" + "aggregator\"j\n" + "\x14LLOChannelStateProto\x12R\n" + - "\x12channelDefinitions\x18\x01 \x03(\v2\".v1.LLOChannelIDAndDefinitionProtoR\x12channelDefinitions\"\xb9\x02\n" + + "\x12channelDefinitions\x18\x01 \x03(\v2\".v1.LLOChannelIDAndDefinitionProtoR\x12channelDefinitions\"C\n" + + "\x19LLOPredecessorConfigProto\x12\x18\n" + + "\asigners\x18\x01 \x03(\fR\asigners\x12\f\n" + + "\x01f\x18\x02 \x01(\rR\x01f\"\xb9\x02\n" + "\x10LLOHotStateProto\x12H\n" + "\x1fobservationTimestampNanoseconds\x18\x01 \x01(\x04R\x1fobservationTimestampNanoseconds\x12c\n" + "\x15validAfterNanoseconds\x18\x02 \x03(\v2-.v1.LLOChannelIDAndValidAfterNanosecondsProtoR\x15validAfterNanoseconds\x122\n" + @@ -1528,7 +1619,7 @@ func file_plugin_codecs_proto_rawDescGZIP() []byte { } var file_plugin_codecs_proto_enumTypes = make([]protoimpl.EnumInfo, 1) -var file_plugin_codecs_proto_msgTypes = make([]protoimpl.MessageInfo, 22) +var file_plugin_codecs_proto_msgTypes = make([]protoimpl.MessageInfo, 23) var file_plugin_codecs_proto_goTypes = []any{ (LLOStreamValue_Type)(0), // 0: v1.LLOStreamValue.Type (*LLOObservationProto)(nil), // 1: v1.LLOObservationProto @@ -1548,15 +1639,16 @@ var file_plugin_codecs_proto_goTypes = []any{ (*LLOChannelIDAndValidAfterNanosecondsProto)(nil), // 15: v1.LLOChannelIDAndValidAfterNanosecondsProto (*LLOStreamAggregate)(nil), // 16: v1.LLOStreamAggregate (*LLOChannelStateProto)(nil), // 17: v1.LLOChannelStateProto - (*LLOHotStateProto)(nil), // 18: v1.LLOHotStateProto - (*LLOPrecursorProto)(nil), // 19: v1.LLOPrecursorProto - (*LLOReportFormatSupportProto)(nil), // 20: v1.LLOReportFormatSupportProto - nil, // 21: v1.LLOObservationProto.UpdateChannelDefinitionsEntry - nil, // 22: v1.LLOObservationProto.StreamValuesEntry + (*LLOPredecessorConfigProto)(nil), // 18: v1.LLOPredecessorConfigProto + (*LLOHotStateProto)(nil), // 19: v1.LLOHotStateProto + (*LLOPrecursorProto)(nil), // 20: v1.LLOPrecursorProto + (*LLOReportFormatSupportProto)(nil), // 21: v1.LLOReportFormatSupportProto + nil, // 22: v1.LLOObservationProto.UpdateChannelDefinitionsEntry + nil, // 23: v1.LLOObservationProto.StreamValuesEntry } var file_plugin_codecs_proto_depIdxs = []int32{ - 21, // 0: v1.LLOObservationProto.updateChannelDefinitions:type_name -> v1.LLOObservationProto.UpdateChannelDefinitionsEntry - 22, // 1: v1.LLOObservationProto.streamValues:type_name -> v1.LLOObservationProto.StreamValuesEntry + 22, // 0: v1.LLOObservationProto.updateChannelDefinitions:type_name -> v1.LLOObservationProto.UpdateChannelDefinitionsEntry + 23, // 1: v1.LLOObservationProto.streamValues:type_name -> v1.LLOObservationProto.StreamValuesEntry 0, // 2: v1.LLOStreamValue.type:type_name -> v1.LLOStreamValue.Type 2, // 3: v1.LLOTimestampedStreamValue.streamValue:type_name -> v1.LLOStreamValue 2, // 4: v1.LLOStreamHistoryRecord.value:type_name -> v1.LLOStreamValue @@ -1576,7 +1668,7 @@ var file_plugin_codecs_proto_depIdxs = []int32{ 13, // 18: v1.LLOPrecursorProto.channelDefinitions:type_name -> v1.LLOChannelIDAndDefinitionProto 15, // 19: v1.LLOPrecursorProto.validAfterNanoseconds:type_name -> v1.LLOChannelIDAndValidAfterNanosecondsProto 16, // 20: v1.LLOPrecursorProto.streamAggregates:type_name -> v1.LLOStreamAggregate - 20, // 21: v1.LLOPrecursorProto.supportByFormat:type_name -> v1.LLOReportFormatSupportProto + 21, // 21: v1.LLOPrecursorProto.supportByFormat:type_name -> v1.LLOReportFormatSupportProto 8, // 22: v1.LLOObservationProto.UpdateChannelDefinitionsEntry.value:type_name -> v1.LLOChannelDefinitionProto 2, // 23: v1.LLOObservationProto.StreamValuesEntry.value:type_name -> v1.LLOStreamValue 24, // [24:24] is the sub-list for method output_type @@ -1597,7 +1689,7 @@ func file_plugin_codecs_proto_init() { GoPackagePath: reflect.TypeOf(x{}).PkgPath(), RawDescriptor: unsafe.Slice(unsafe.StringData(file_plugin_codecs_proto_rawDesc), len(file_plugin_codecs_proto_rawDesc)), NumEnums: 1, - NumMessages: 22, + NumMessages: 23, NumExtensions: 0, NumServices: 0, }, diff --git a/llo/protocol/plugin_codecs.proto b/llo/protocol/plugin_codecs.proto index c0e4a40..0048f8c 100644 --- a/llo/protocol/plugin_codecs.proto +++ b/llo/protocol/plugin_codecs.proto @@ -33,6 +33,13 @@ message LLOObservationProto { // cannot be read in the state transition; this carries it as a replicated // fact instead. repeated uint32 supportedReportFormats = 8; + // The predecessor instance's signer set and f, read from the node-local + // retirement report cache. Carried only by a v31 staging instance that has + // not yet agreed on c/pred, so the state transition can verify attested + // retirement reports against replicated state instead of node-local state. + // Order is significant: a signature names its signer by index into it. + repeated bytes predecessorSigners = 9; + uint32 predecessorF = 10; } message LLOStreamValue { @@ -181,6 +188,23 @@ message LLOChannelStateProto { repeated LLOChannelIDAndDefinitionProto channelDefinitions = 1; } +// LLOPredecessorConfigProto is the v31 KeyValueState record holding the +// predecessor instance's signer set and f under a single key (c/pred), agreed +// by vote while staging and written at most once. +// +// It exists so that verifying an attested predecessor retirement report reads +// only replicated state. The same data is available node-locally from the +// retirement report cache, but that cache is filled asynchronously by the +// config poller, so oracles reach different verdicts from it and the state +// transition would fork. +// +// NOTE: must serialize deterministically. signers MUST keep the order the +// predecessor's config gives them, since a signature names its signer by index. +message LLOPredecessorConfigProto { + repeated bytes signers = 1; + uint32 f = 2; +} + // LLOHotStateProto is the v31 KeyValueState record holding the per-round // ("hot") state under a single key (r/agg): the state that changes on // essentially every round. diff --git a/llo/retirement/plugin_scoped_retirement_report_cache.go b/llo/retirement/plugin_scoped_retirement_report_cache.go index fc8c30b..f76e4dd 100644 --- a/llo/retirement/plugin_scoped_retirement_report_cache.go +++ b/llo/retirement/plugin_scoped_retirement_report_cache.go @@ -38,23 +38,34 @@ func NewPluginScopedRetirementReportCache(rrc RetirementReportCacheReader, verif } } +func (pr *pluginScopedRetirementReportCache) PredecessorConfig(predecessorConfigDigest ocr2types.ConfigDigest) ([][]byte, uint8, bool) { + config, exists := pr.rrc.Config(predecessorConfigDigest) + if !exists { + return nil, 0, false + } + return config.Signers, config.F, true +} + func (pr *pluginScopedRetirementReportCache) CheckAttestedRetirementReport(predecessorConfigDigest ocr2types.ConfigDigest, serializedAttestedRetirementReport []byte) (protocol.RetirementReport, error) { config, exists := pr.rrc.Config(predecessorConfigDigest) if !exists { return protocol.RetirementReport{}, fmt.Errorf("Verify failed; predecessor config not found for config digest %x", predecessorConfigDigest[:]) } + return pr.VerifyAttestedRetirementReport(predecessorConfigDigest, config.Signers, config.F, serializedAttestedRetirementReport) +} +func (pr *pluginScopedRetirementReportCache) VerifyAttestedRetirementReport(predecessorConfigDigest ocr2types.ConfigDigest, signers [][]byte, f uint8, serializedAttestedRetirementReport []byte) (protocol.RetirementReport, error) { var arr protocol.AttestedRetirementReport if err := proto.Unmarshal(serializedAttestedRetirementReport, &arr); err != nil { return protocol.RetirementReport{}, fmt.Errorf("Verify failed; failed to unmarshal protobuf: %w", err) } validSigs := 0 - seenSigners := make(map[uint32]struct{}, len(config.Signers)) + seenSigners := make(map[uint32]struct{}, len(signers)) for _, sig := range arr.Sigs { // #nosec G115 - if sig.Signer >= uint32(len(config.Signers)) { - return protocol.RetirementReport{}, fmt.Errorf("Verify failed; attested report signer index out of bounds (got: %d, max: %d)", sig.Signer, len(config.Signers)-1) + if sig.Signer >= uint32(len(signers)) { + return protocol.RetirementReport{}, fmt.Errorf("Verify failed; attested report signer index out of bounds (got: %d, max: %d)", sig.Signer, len(signers)-1) } // ensure we have unique signatures @@ -63,7 +74,7 @@ func (pr *pluginScopedRetirementReportCache) CheckAttestedRetirementReport(prede } seenSigners[sig.Signer] = struct{}{} - signer := config.Signers[sig.Signer] + signer := signers[sig.Signer] valid := pr.verifier.Verify(types.OnchainPublicKey(signer), predecessorConfigDigest, arr.SeqNr, ocr3types.ReportWithInfo[llotypes.ReportInfo]{ Report: arr.RetirementReport, Info: llotypes.ReportInfo{ReportFormat: llotypes.ReportFormatRetirement}, @@ -73,8 +84,8 @@ func (pr *pluginScopedRetirementReportCache) CheckAttestedRetirementReport(prede } validSigs++ } - if validSigs <= int(config.F) { - return protocol.RetirementReport{}, fmt.Errorf("Verify failed; not enough valid signatures (got: %d, need: %d)", validSigs, config.F+1) + if validSigs <= int(f) { + return protocol.RetirementReport{}, fmt.Errorf("Verify failed; not enough valid signatures (got: %d, need: %d)", validSigs, f+1) } decoded, err := pr.codec.Decode(arr.RetirementReport) if err != nil { diff --git a/llo/v30/plugin_observation_test.go b/llo/v30/plugin_observation_test.go index e2cd4d9..4a68470 100644 --- a/llo/v30/plugin_observation_test.go +++ b/llo/v30/plugin_observation_test.go @@ -35,6 +35,12 @@ func (p *mockPredecessorRetirementReportCache) AttestedRetirementReport(predeces func (p *mockPredecessorRetirementReportCache) CheckAttestedRetirementReport(predecessorConfigDigest ocr2types.ConfigDigest, attestedRetirementReport []byte) (protocol.RetirementReport, error) { panic("not implemented") } +func (p *mockPredecessorRetirementReportCache) PredecessorConfig(predecessorConfigDigest ocr2types.ConfigDigest) ([][]byte, uint8, bool) { + panic("not implemented") +} +func (p *mockPredecessorRetirementReportCache) VerifyAttestedRetirementReport(predecessorConfigDigest ocr2types.ConfigDigest, signers [][]byte, f uint8, attestedRetirementReport []byte) (protocol.RetirementReport, error) { + panic("not implemented") +} func Test_Observation(t *testing.T) { for _, codec := range []OutcomeCodec{protoOutcomeCodecV0{}, protoOutcomeCodecV1{}} { From 1ea53a85f7e0afa8ef08ece0a1526300824ae33a Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 15:37:52 +0100 Subject: [PATCH 22/40] SPOR-0015 llo/dev/v31: encode observations and blob payloads deterministically --- llo/dev/v31/blobpump.go | 4 ++-- llo/dev/v31/flow_test.go | 45 ++++++++++++++++++++++++++++++++++++++ llo/dev/v31/observation.go | 11 +++++++--- 3 files changed, 55 insertions(+), 5 deletions(-) diff --git a/llo/dev/v31/blobpump.go b/llo/dev/v31/blobpump.go index ee2df25..1c8d504 100644 --- a/llo/dev/v31/blobpump.go +++ b/llo/dev/v31/blobpump.go @@ -16,7 +16,6 @@ import ( "github.com/smartcontractkit/libocr/offchainreporting2plus/ocr3_1types" ocrtypes "github.com/smartcontractkit/libocr/offchainreporting2plus/types" - "google.golang.org/protobuf/proto" ) // Defaults for the blob pump. See PluginFactoryParams for the overrides. @@ -495,7 +494,8 @@ func marshalStreamValues(sv protocol.StreamValues) ([]byte, error) { if len(pb) == 0 { return nil, nil } - raw, err := proto.Marshal(&protocol.LLOObservationProto{StreamValues: pb}) + + raw, err := deterministicMarshal.Marshal(&protocol.LLOObservationProto{StreamValues: pb}) if err != nil { return nil, fmt.Errorf("marshal stream values: %w", err) } diff --git a/llo/dev/v31/flow_test.go b/llo/dev/v31/flow_test.go index ecaaee4..b5c8198 100644 --- a/llo/dev/v31/flow_test.go +++ b/llo/dev/v31/flow_test.go @@ -747,3 +747,48 @@ func Test_Observation_PredecessorConfigVote(t *testing.T) { require.Empty(t, obs.PredecessorSigners) }) } + +// Test_EncodeObservation_IsDeterministic covers the encoder builds repeated fields +// and proto maps from Go maps, so without deterministic marshaling two oracles could +// produce different bytes for the same logical observation. +// Nothing compares observation bytes today and this keeps a future path that does from being silently wrong. +func Test_EncodeObservation_IsDeterministic(t *testing.T) { + obs := Observation{ + UnixTimestampNanoseconds: 1234, + AttestedPredecessorRetirement: []byte("attested"), + PredecessorSigners: testPredecessorSigners, + PredecessorF: 1, + RemoveChannelIDs: map[llotypes.ChannelID]struct{}{7: {}, 1: {}, 4: {}, 2: {}, 9: {}}, + UpdateChannelDefinitions: llotypes.ChannelDefinitions{ + 5: jsonChannel(), 3: jsonChannel(), 8: jsonChannel(), 1: jsonChannel(), + }, + SupportedReportFormats: testSupportedReportFormats, + } + + want, err := encodeObservation(obs, [][]byte{[]byte("handle")}) + require.NoError(t, err) + for range 64 { + got, err := encodeObservation(obs, [][]byte{[]byte("handle")}) + require.NoError(t, err) + require.Equal(t, want, got, "the same logical observation must encode to the same bytes") + } +} + +// Test_MarshalStreamValues_IsDeterministic pins the same property for the blob +// payload, which is content-addressed: the same values must hash to the same +// handle on every oracle, or the round's fetch memo cannot dedupe. +func Test_MarshalStreamValues_IsDeterministic(t *testing.T) { + sv := protocol.StreamValues{} + for id := llotypes.StreamID(1); id <= 32; id++ { + sv[id] = protocol.ToDecimal(decimal.NewFromInt(int64(id))) + } + + want, err := marshalStreamValues(sv) + require.NoError(t, err) + require.NotEmpty(t, want) + for range 64 { + got, err := marshalStreamValues(sv) + require.NoError(t, err) + require.Equal(t, want, got, "the same stream values must marshal to the same payload") + } +} diff --git a/llo/dev/v31/observation.go b/llo/dev/v31/observation.go index 03b5146..177c67e 100644 --- a/llo/dev/v31/observation.go +++ b/llo/dev/v31/observation.go @@ -74,6 +74,9 @@ func encodeObservation(obs Observation, handles [][]byte) (ocrtypes.Observation, for id := range obs.RemoveChannelIDs { main.RemoveChannelIDs = append(main.RemoveChannelIDs, id) } + // Map iteration order, so sort: Deterministic marshaling canonicalizes + // proto map fields, not a repeated field built from a Go map. + sortChannelIDs(main.RemoveChannelIDs) if len(obs.UpdateChannelDefinitions) > 0 { main.UpdateChannelDefinitions = make(map[uint32]*protocol.LLOChannelDefinitionProto, len(obs.UpdateChannelDefinitions)) for id, cd := range obs.UpdateChannelDefinitions { @@ -81,11 +84,13 @@ func encodeObservation(obs Observation, handles [][]byte) (ocrtypes.Observation, } } - // Sorted and deduped: the wire bytes need not be deterministic, but a - // canonical list keeps goldens stable and matches what decode enforces. + // Sorted and deduped, matching what decode enforces. main.SupportedReportFormats = sortedUniqueFormats(obs.SupportedReportFormats) - mainBytes, err := proto.Marshal(main) + // Deterministic even though nothing compares observation bytes today + // A future path which does compare or hash the value cannot be + // silently wrong. + mainBytes, err := deterministicMarshal.Marshal(main) if err != nil { return nil, fmt.Errorf("marshal observation: %w", err) } From 3b6f0e71c10e636dc86caabbcf71120ced1cfa24 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 15:44:53 +0100 Subject: [PATCH 23/40] SPOR-0017 llo/dev/v31: cover reporting from an evicted channel generation --- llo/dev/v31/concurrency_test.go | 83 +++++++++++++++++++++++++++++++++ 1 file changed, 83 insertions(+) diff --git a/llo/dev/v31/concurrency_test.go b/llo/dev/v31/concurrency_test.go index 91b11fc..7e8c817 100644 --- a/llo/dev/v31/concurrency_test.go +++ b/llo/dev/v31/concurrency_test.go @@ -1,6 +1,7 @@ package llo import ( + "errors" "fmt" "sync" "testing" @@ -141,3 +142,85 @@ func Test_Reports_StateTransition_Concurrent_OptsIsolation(t *testing.T) { require.Equal(t, "v=1", got, "Reports must encode with the opts of the record its precursor was built from, whatever a concurrent StateTransition loads") } } + +// Test_Reports_RebuildsEvictedGeneration covers ChannelCache retains +// only channelGenerationsRetained generations, so a precursor whose record has +// since been evicted must be reported from the definitions the precursor +// carries. The rebuild path is only reachable once enough newer records exist, +// which the overlap test above never produces. +// +// The report must be identical either way: a generation is a pure function of +// its sequence number, so rebuilding one is not a degraded path. +func Test_Reports_RebuildsEvictedGeneration(t *testing.T) { + ctx := tests.Context(t) + + withOpts := func(v int) llotypes.ChannelDefinitions { + return llotypes.ChannelDefinitions{1: { + ReportFormat: llotypes.ReportFormatJSON, + Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}, + Opts: llotypes.ChannelOpts(fmt.Sprintf(`{"v":%d}`, v)), + }} + } + obsRound := func(ts uint64, defs llotypes.ChannelDefinitions) []ocrtypes.AttributedObservation { + obs := Observation{ + UnixTimestampNanoseconds: ts, + UpdateChannelDefinitions: defs, + StreamValues: protocol.StreamValues{100: protocol.ToDecimal(decimal.NewFromInt(5))}, + } + aos := make([]ocrtypes.AttributedObservation, 0, 4) + for i := 0; i < 4; i++ { + aos = append(aos, ao(i, mustEncodeObs(t, obs))) + } + return aos + } + + p := testPlugin(t) + p.ReportCodecs = map[llotypes.ReportFormat]protocol.ReportCodec{llotypes.ReportFormatJSON: optsEchoCodec{}} + kv := newMemKV() + + _, err := p.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, obsRound(1_000, nil), kv, testBlobs) + require.NoError(t, err) + // Round 2 agrees the channel; it takes effect in round 3, which seeds its + // watermark, and round 4 is the first that reports it. No round after 2 + // changes the definitions, so round 4's precursor is still built from the + // record round 2 wrote (opts v=1). + _, err = p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, obsRound(2_000, withOpts(1)), kv, testBlobs) + require.NoError(t, err) + _, err = p.StateTransition(ctx, 3, ocrtypes.AttributedQuery{}, obsRound(3_000, nil), kv, testBlobs) + require.NoError(t, err) + prec4, err := p.StateTransition(ctx, 4, ocrtypes.AttributedQuery{}, obsRound(4_000, nil), kv, testBlobs) + require.NoError(t, err) + + decoded, err := decodePrecursor(prec4) + require.NoError(t, err) + require.Equal(t, uint64(2), decoded.ChannelStateSeqNr) + + // The generation is cached at this point, so this is the un-evicted baseline. + baseline, err := p.Reports(ctx, 4, prec4) + require.NoError(t, err) + require.Len(t, baseline, 1) + require.Equal(t, "v=1", string(baseline[0].ReportWithInfo.Report)) + + // Write a new record every round until the round-2 record is evicted. Each + // round changes the opts, so each one writes a record of its own. + for seqNr, v := uint64(5), 2; seqNr < 20 && cachedGeneration(p.ChannelCache, 2); seqNr, v = seqNr+1, v+1 { + _, err = p.StateTransition(ctx, seqNr, ocrtypes.AttributedQuery{}, obsRound(seqNr*1_000, withOpts(v)), kv, testBlobs) + require.NoError(t, err) + } + require.False(t, cachedGeneration(p.ChannelCache, 2), "the round-2 record must be evicted for this test to mean anything") + + // Same precursor, same reports, now off the rebuild path. + rebuilt, err := p.Reports(ctx, 4, prec4) + require.NoError(t, err) + require.Equal(t, baseline, rebuilt, "a rebuilt generation must report exactly what the cached one did") +} + +// cachedGeneration reports whether the cache still holds the generation for +// seqNr, without inserting one: Load only calls build on a miss, and a build +// that fails stores nothing. +func cachedGeneration(c *protocol.ChannelCache, seqNr uint64) bool { + _, err := c.Load(seqNr, func() (llotypes.ChannelDefinitions, error) { + return nil, errors.New("not cached") + }) + return err == nil +} From 811a21d6be2bcc1a9c96ceab4dfbb1f43726e48d Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 17:06:26 +0100 Subject: [PATCH 24/40] SPOR-0008 llo/dev/v31: do not read an empty desired set as a vote to remove --- llo/dev/v31/flow_test.go | 82 ++++++++++++++++++++++++++++++++++++++++ llo/dev/v31/plugin.go | 51 ++++++++++++++++++++++++- 2 files changed, 132 insertions(+), 1 deletion(-) diff --git a/llo/dev/v31/flow_test.go b/llo/dev/v31/flow_test.go index b5c8198..31faee4 100644 --- a/llo/dev/v31/flow_test.go +++ b/llo/dev/v31/flow_test.go @@ -792,3 +792,85 @@ func Test_MarshalStreamValues_IsDeterministic(t *testing.T) { require.Equal(t, want, got, "the same stream values must marshal to the same payload") } } + +// Test_Observation_EmptyDesiredSetRemovesOnlyTombstones covers that +// the onchain cache reaps a tombstone by having the owner omit it, +// so an all-tombstoned committed set legitimately merges to empty, +// and abstaining there would strand those channels forever. +func Test_Observation_EmptyDesiredSetRemovesOnlyTombstones(t *testing.T) { + ctx := tests.Context(t) + + live := jsonChannel() + tombstoned := jsonChannel() + tombstoned.Tombstone = true + + // observeRemovals commits defs, then asks what the node votes to remove + // once the definitions source has gone empty. + observeRemovals := func(t *testing.T, defs llotypes.ChannelDefinitions) map[llotypes.ChannelID]struct{} { + t.Helper() + p := testPlugin(t) + p.ShouldRetireCache = &mockShouldRetireCache{} + cdc := &mockChannelDefinitionCache{defs: defs} + p.ChannelDefinitionCache = cdc + kv := newMemKV() + + _, err := p.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, nil), ao(1, nil), ao(2, nil)}, kv, testBlobs) + require.NoError(t, err) + agree := make([]ocrtypes.AttributedObservation, 0, 4) + for i := 0; i < 4; i++ { + agree = append(agree, ao(i, mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 1000, UpdateChannelDefinitions: defs}))) + } + _, err = p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, agree, kv, testBlobs) + require.NoError(t, err) + require.Len(t, kvChannelDefs(t, kv), len(defs)) + + cdc.defs = llotypes.ChannelDefinitions{} + obsBytes, err := p.Observation(ctx, 3, ocrtypes.AttributedQuery{}, kv, nil) + require.NoError(t, err) + obs, err := decodeObservation(ctx, obsBytes, testBlobs, nil) + require.NoError(t, err) + return obs.RemoveChannelIDs + } + + t.Run("live channels are never removed by an empty set", func(t *testing.T) { + require.Empty(t, observeRemovals(t, llotypes.ChannelDefinitions{1: live, 2: live})) + }) + + t.Run("tombstoned channels still are", func(t *testing.T) { + require.Equal(t, + map[llotypes.ChannelID]struct{}{2: {}}, + observeRemovals(t, llotypes.ChannelDefinitions{1: live, 2: tombstoned})) + }) + + t.Run("an all-tombstoned set can be reaped to empty", func(t *testing.T) { + require.Equal(t, + map[llotypes.ChannelID]struct{}{1: {}, 2: {}}, + observeRemovals(t, llotypes.ChannelDefinitions{1: tombstoned, 2: tombstoned})) + }) + + t.Run("a non-empty desired set is still a genuine opinion", func(t *testing.T) { + p := testPlugin(t) + p.ShouldRetireCache = &mockShouldRetireCache{} + cdc := &mockChannelDefinitionCache{defs: llotypes.ChannelDefinitions{1: live, 2: live}} + p.ChannelDefinitionCache = cdc + kv := newMemKV() + + _, err := p.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, nil), ao(1, nil), ao(2, nil)}, kv, testBlobs) + require.NoError(t, err) + agree := make([]ocrtypes.AttributedObservation, 0, 4) + for i := 0; i < 4; i++ { + agree = append(agree, ao(i, mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 1000, UpdateChannelDefinitions: cdc.defs}))) + } + _, err = p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, agree, kv, testBlobs) + require.NoError(t, err) + + // Dropping one channel while still reporting the other is an opinion, + // so the dropped one is voted off even though it is live. + cdc.defs = llotypes.ChannelDefinitions{1: live} + obsBytes, err := p.Observation(ctx, 3, ocrtypes.AttributedQuery{}, kv, nil) + require.NoError(t, err) + obs, err := decodeObservation(ctx, obsBytes, testBlobs, nil) + require.NoError(t, err) + require.Equal(t, map[llotypes.ChannelID]struct{}{2: {}}, obs.RemoveChannelIDs) + }) +} diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index 207b5e4..27f676a 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -298,6 +298,30 @@ func (p *Plugin) voteOnPredecessorConfig(obs *Observation, kvReader ocr3_1types. // voteOnChannels populates obs.RemoveChannelIDs / obs.UpdateChannelDefinitions // by comparing the desired channel definitions against current KV state. +// +// ChannelDefinitionCache.Definitions(committed) is a reconciliation, not a +// snapshot of a file: the shipped onchain cache merges what it has fetched into +// the committed set it is handed and returns the result, so the desired set is +// normally the committed one plus additions and changes. It deletes nothing +// implicitly. +// +// - before anything has been fetched, and after a fetch error, a poll that +// saw nothing, or a stale event, it returns the committed set unchanged. A +// source it cannot read produces no opinion rather than a removal; +// - the merge is upsert-only. A channel missing from a newly fetched file is +// preserved, not dropped. Removal is explicit: the owner marks the channel +// with Tombstone, which is a definition change like any other; +// - the single deletion path is reaping an already tombstoned channel, once +// the owner omits it from a later file. So a channel leaves the committed +// set only after the DON has already agreed it is a tombstone. +// +// This function must not assume that, because the contract does not require it. +// Definitions may be implemented by anything, and the static cache shipped for +// benchmarks and the dummy relayer ignores the committed set entirely and +// returns its configured JSON verbatim. Everything below is therefore written +// against the weaker guarantee: the desired set is one node's opinion, votes +// decide, and an absent channel means "not mentioned", which is only treated as +// "remove" when the set as a whole is credible. See the empty-set case below. func (p *Plugin) voteOnChannels(obs *Observation, state *kvState) { obs.RemoveChannelIDs = map[llotypes.ChannelID]struct{}{} @@ -312,7 +336,32 @@ func (p *Plugin) voteOnChannels(obs *Observation, state *kvState) { return } - removeChannelDefinitions := protocol.SubtractChannelDefinitions(state.channelDefinitions, expectedChannelDefs, protocol.MaxObservationRemoveChannelIDsLength) + // An empty desired set against a non-empty committed one is not read as + // "remove every channel". Under the reconciliation above the onchain cache + // cannot produce that for live channels, but nothing in the interface says + // so, and an implementation that returns nothing before it has loaded + // anything, or on a source it could not read, would make every live channel + // a removal candidate. Every node would do it from the same input, so the + // channels would really go. + // + // Tombstoned channels stay removable. An empty desired set is exactly what + // the onchain cache produces on the last reap: the owner omits the + // tombstones it wants dropped, and when every committed channel is a + // tombstone the merge comes back empty. Abstaining there would stop the + // removal the DON has already agreed to and strand those channels in the + // definitions permanently. + removable := state.channelDefinitions + if len(expectedChannelDefs) == 0 && len(state.channelDefinitions) > 0 { + removable = llotypes.ChannelDefinitions{} + for channelID, cd := range state.channelDefinitions { + if cd.Tombstone { + removable[channelID] = cd + } + } + p.Logger.Warnw("ChannelDefinitionCache.Definitions is empty while channels are committed; voting to remove only the tombstoned ones", "committed", len(state.channelDefinitions), "removable", len(removable)) + } + + removeChannelDefinitions := protocol.SubtractChannelDefinitions(removable, expectedChannelDefs, protocol.MaxObservationRemoveChannelIDsLength) for channelID := range removeChannelDefinitions { obs.RemoveChannelIDs[channelID] = struct{}{} } From 47d41b07e740c7d8e5add7836d70aae7fe335b99 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 17:24:49 +0100 Subject: [PATCH 25/40] SPOR-0012 llo/dev/v31: saturate the report cadence deadline --- llo/dev/v31/flow_test.go | 40 ++++++++++++++++++++++++++++++++++++++++ llo/dev/v31/reports.go | 19 ++++++++++++++++++- 2 files changed, 58 insertions(+), 1 deletion(-) diff --git a/llo/dev/v31/flow_test.go b/llo/dev/v31/flow_test.go index 31faee4..bc3a94b 100644 --- a/llo/dev/v31/flow_test.go +++ b/llo/dev/v31/flow_test.go @@ -3,6 +3,7 @@ package llo import ( "context" "errors" + "math" "sync" "testing" "time" @@ -874,3 +875,42 @@ func Test_Observation_EmptyDesiredSetRemovesOnlyTombstones(t *testing.T) { require.Equal(t, map[llotypes.ChannelID]struct{}{2: {}}, obs.RemoveChannelIDs) }) } + +// Test_IsReportable_MinReportIntervalDoesNotOverflow +// validAfter is a nanosecond wall-clock timestamp and the offchain config +// bounds DefaultMinReportIntervalNanoseconds only away from zero, so a large +// enough interval used to wrap validAfter+minReportInterval to a small number. +// The cadence comparison then passed for every channel and the interval +// silently stopped gating anything. +func Test_IsReportable_MinReportIntervalDoesNotOverflow(t *testing.T) { + const validAfter = uint64(1_700_000_000_000_000_000) + + out := precursor{ + LifeCycleStage: protocol.LifeCycleStageProduction, + ObservationTimestampNanoseconds: validAfter + 1, + ChannelDefinitions: llotypes.ChannelDefinitions{1: jsonChannel()}, + ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{1: validAfter}, + StreamAggregates: protocol.StreamAggregates{ + 100: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(1))}, + }, + SupportByFormat: map[llotypes.ReportFormat]int{llotypes.ReportFormatJSON: 4}, + } + gen, err := protocol.NewChannelCache().Load(1, func() (llotypes.ChannelDefinitions, error) { + return out.ChannelDefinitions, nil + }) + require.NoError(t, err) + optsCache := gen.Opts() + + // One nanosecond past validAfter, so only the interval can hold it back. + require.True(t, out.isReportable(1, 1, 1, optsCache, logger.Test(t)), + "a one nanosecond interval must not gate a report one nanosecond late") + + // An interval that overflows the sum must gate, not wrap into passing. + for _, interval := range []uint64{math.MaxUint64, math.MaxUint64 - validAfter + 1} { + require.False(t, out.isReportable(1, interval, 1, optsCache, logger.Test(t)), + "interval %d must gate the report, not wrap", interval) + } + + require.Equal(t, uint64(math.MaxUint64), saturatingAdd(validAfter, math.MaxUint64)) + require.Equal(t, uint64(9), saturatingAdd(4, 5)) +} diff --git a/llo/dev/v31/reports.go b/llo/dev/v31/reports.go index c8f6e4b..568c843 100644 --- a/llo/dev/v31/reports.go +++ b/llo/dev/v31/reports.go @@ -3,6 +3,7 @@ package llo import ( "context" "fmt" + "math" "sort" "github.com/smartcontractkit/chainlink-common/pkg/logger" @@ -194,6 +195,22 @@ func (o precursor) formatIsEncodable(format llotypes.ReportFormat, f int, channe // reportableChannels returns the sorted set of channels reportable in this // (current) round (see isReportable). +// saturatingAdd returns a+b, clamped to MaxUint64 instead of wrapping. +// +// validAfter is a nanosecond wall-clock timestamp, so it already sits around +// 1.7e18, and the offchain config bounds DefaultMinReportIntervalNanoseconds +// only away from zero (see protocol.OffchainConfig.Validate). A large enough +// interval would wrap the sum to a small number, the cadence comparison would +// then always pass, and the interval would silently stop gating anything. The +// config is replicated, so every oracle would do it identically: a silent loss +// of the cadence, not a fork. +func saturatingAdd(a, b uint64) uint64 { + if sum := a + b; sum >= a { + return sum + } + return math.MaxUint64 +} + func (o precursor) reportableChannels(minReportInterval uint64, f int, optsCache *protocol.OptsCache, lggr logger.Logger) []llotypes.ChannelID { reportable := make([]llotypes.ChannelID, 0, len(o.ChannelDefinitions)) for channelID := range o.ChannelDefinitions { @@ -268,7 +285,7 @@ func (o precursor) isReportable(channelID llotypes.ChannelID, minReportInterval if !ok { return false } - if o.ObservationTimestampNanoseconds < validAfter+minReportInterval || o.ObservationTimestampNanoseconds <= validAfter { + if o.ObservationTimestampNanoseconds < saturatingAdd(validAfter, minReportInterval) || o.ObservationTimestampNanoseconds <= validAfter { return false } // For seconds-resolution report formats, also require a full second between From 7afa96c0a4aa68481e234ebb2c6cca6a9e089caf Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 18 Sep 2026 19:16:33 +0100 Subject: [PATCH 26/40] llo/transmitter/dataengine: wait on the queue instead of sleeping llo/reportcodec: revert test error requirement --- llo/transmitter/dataengine/transmitter_test.go | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/llo/transmitter/dataengine/transmitter_test.go b/llo/transmitter/dataengine/transmitter_test.go index 729e7f9..131c762 100644 --- a/llo/transmitter/dataengine/transmitter_test.go +++ b/llo/transmitter/dataengine/transmitter_test.go @@ -123,8 +123,16 @@ func Test_Transmitter_Transmit(t *testing.T) { err = mt.Transmit(t.Context(), digest, seqNr, report, sigs) require.NoError(t, err) - // wait for the commit loop to run - time.Sleep(2 * commitInterval) + // Wait for the commit loops to pick the transmissions up rather + // than assuming a fixed delay. A loop batches until its ticker + // fires, and transmit then inserts into the database before + // pushing, so the enqueue lands an unbounded time after Transmit + // returns. + require.Eventually(t, func() bool { + return mt.servers[sURL].q.(*transmitQueue).Len() == 1 && + mt.servers[sURL2].q.(*transmitQueue).Len() == 1 && + mt.servers[sURL3].q.(*transmitQueue).Len() == 1 + }, tests.WaitTimeout(t), commitInterval/5, "all three servers must have the transmission enqueued") // ensure it was added to the queue require.Equal(t, 1, mt.servers[sURL].q.(*transmitQueue).Len()) From 79cb24a99cf35704d565f069eefb7ae65b63e055 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Tue, 22 Sep 2026 11:15:21 +0100 Subject: [PATCH 27/40] SPOR-0003 llo/protocol: bound stream value nesting limit during unmarshal --- llo/protocol/limits.go | 5 +- llo/protocol/stream_value.go | 101 +++++++++++++----- llo/protocol/stream_value_coefficient_test.go | 32 +++++- llo/protocol/stream_value_fuzz_test.go | 2 +- 4 files changed, 108 insertions(+), 32 deletions(-) diff --git a/llo/protocol/limits.go b/llo/protocol/limits.go index e3c89d8..090df36 100644 --- a/llo/protocol/limits.go +++ b/llo/protocol/limits.go @@ -70,8 +70,9 @@ const ( MaxDecimalCoefficientBits = 192 // MaxStreamValueNesting bounds how deeply a stream value may nest another. // Only TimestampedStreamValue nests, and only one level is meaningful, so - // this exists to keep the bounds check over untrusted bytes from recursing - // on a value crafted to nest. + // this exists to stop untrusted bytes crafted to nest from driving + // unbounded recursion. Enforced during unmarshal, which is where the + // recursion lives (see unmarshalProtoStreamValue). MaxStreamValueNesting = 4 // MaxOutcomeChannelDefinitionsLength is the maximum number of channels that diff --git a/llo/protocol/stream_value.go b/llo/protocol/stream_value.go index 94e7f16..51c5375 100644 --- a/llo/protocol/stream_value.go +++ b/llo/protocol/stream_value.go @@ -65,7 +65,7 @@ func UnmarshalObservedProtoStreamValue(enc *LLOStreamValue) (StreamValue, error) if err != nil { return nil, err } - if err := checkObservedStreamValue(sv, 0); err != nil { + if err := checkObservedStreamValue(sv); err != nil { return nil, err } return sv, nil @@ -73,10 +73,11 @@ func UnmarshalObservedProtoStreamValue(enc *LLOStreamValue) (StreamValue, error) // checkObservedStreamValue applies the observation-only bounds to every decimal // a stream value carries, at any nesting depth. -func checkObservedStreamValue(sv StreamValue, depth int) error { - if depth > MaxStreamValueNesting { - return fmt.Errorf("%w: got more than %d levels", ErrStreamValueNestingTooDeep, MaxStreamValueNesting) - } +// +// Nesting itself is bounded during unmarshal, not here: the recursion that has +// to be bounded is the one inside unmarshalling, so a value deep enough to +// matter never reaches this function. +func checkObservedStreamValue(sv StreamValue) error { switch v := sv.(type) { case nil: return nil @@ -96,7 +97,7 @@ func checkObservedStreamValue(sv StreamValue, depth int) error { if v == nil { return nil } - return checkObservedStreamValue(v.StreamValue, depth+1) + return checkObservedStreamValue(v.StreamValue) default: // An unknown type carries no decimal this function knows how to reach. // UnmarshalProtoStreamValue rejects types it does not recognize, so this @@ -152,25 +153,49 @@ func unmarshalTextDecimal(d *decimal.Decimal, data []byte) error { return nil } -func UnmarshalProtoStreamValue(enc *LLOStreamValue) (sv StreamValue, err error) { +func UnmarshalProtoStreamValue(enc *LLOStreamValue) (StreamValue, error) { + return unmarshalProtoStreamValue(enc, 0) +} + +// unmarshalProtoStreamValue decodes a stream value, refusing to descend past +// MaxStreamValueNesting. +// +// The bound has to live here rather than in a check over the decoded value: +// only TimestampedStreamValue nests, and it nests by calling back into this +// function from its UnmarshalBinary, so the recursion a crafted value drives is +// the unmarshalling itself. Each level also re-slices the nested Value bytes, +// making the work quadratic in the payload, so an unbounded descent burns +// minutes of CPU on every honest node before any post-hoc check could run. +func unmarshalProtoStreamValue(enc *LLOStreamValue, depth int) (StreamValue, error) { if enc == nil { // Shouldn't ever happen except from byzantine node, but we must not panic return nil, ErrNilStreamValue } + if depth > MaxStreamValueNesting { + return nil, fmt.Errorf("%w: got more than %d levels", ErrStreamValueNestingTooDeep, MaxStreamValueNesting) + } switch enc.Type { case LLOStreamValue_Quote: - sv = new(Quote) + sv := new(Quote) + if err := sv.UnmarshalBinary(enc.Value); err != nil { + return nil, err + } + return sv, nil case LLOStreamValue_Decimal: - sv = new(Decimal) + sv := new(Decimal) + if err := sv.UnmarshalBinary(enc.Value); err != nil { + return nil, err + } + return sv, nil case LLOStreamValue_TimestampedStreamValue: - sv = new(TimestampedStreamValue) + sv := new(TimestampedStreamValue) + if err := sv.unmarshalBinary(enc.Value, depth+1); err != nil { + return nil, err + } + return sv, nil default: return nil, fmt.Errorf("cannot unmarshal protobuf stream value; unknown StreamValueType %d", enc.Type) } - if err := sv.UnmarshalBinary(enc.Value); err != nil { - return nil, err - } - return sv, nil } func NewTypedTextStreamValue(sv StreamValue) (TypedTextStreamValue, error) { @@ -193,25 +218,41 @@ type TypedTextStreamValue struct { } func UnmarshalTypedTextStreamValue(enc *TypedTextStreamValue) (StreamValue, error) { + return unmarshalTypedTextStreamValue(enc, 0) +} + +// unmarshalTypedTextStreamValue is the text counterpart of +// unmarshalProtoStreamValue and bounds the same recursion. +func unmarshalTypedTextStreamValue(enc *TypedTextStreamValue, depth int) (StreamValue, error) { if enc == nil { // Shouldn't ever happen except from byzantine node, but we must not panic return nil, ErrNilStreamValue } - var sv StreamValue + if depth > MaxStreamValueNesting { + return nil, fmt.Errorf("%w: got more than %d levels", ErrStreamValueNestingTooDeep, MaxStreamValueNesting) + } switch enc.Type { case LLOStreamValue_Decimal: - sv = new(Decimal) + sv := new(Decimal) + if err := sv.UnmarshalText([]byte(enc.SerializedStreamValue)); err != nil { + return nil, err + } + return sv, nil case LLOStreamValue_Quote: - sv = new(Quote) + sv := new(Quote) + if err := sv.UnmarshalText([]byte(enc.SerializedStreamValue)); err != nil { + return nil, err + } + return sv, nil case LLOStreamValue_TimestampedStreamValue: - sv = new(TimestampedStreamValue) + sv := new(TimestampedStreamValue) + if err := sv.unmarshalText([]byte(enc.SerializedStreamValue), depth+1); err != nil { + return nil, err + } + return sv, nil default: return nil, fmt.Errorf("unknown StreamValueType %d", enc.Type) } - if err := (sv).UnmarshalText([]byte(enc.SerializedStreamValue)); err != nil { - return nil, err - } - return sv, nil } func Decode(value StreamValue, data []byte) error { @@ -390,12 +431,18 @@ func (v *TimestampedStreamValue) MarshalBinary() ([]byte, error) { } func (v *TimestampedStreamValue) UnmarshalBinary(data []byte) error { + return v.unmarshalBinary(data, 0) +} + +// unmarshalBinary carries the nesting depth reached so far. See +// unmarshalProtoStreamValue. +func (v *TimestampedStreamValue) unmarshalBinary(data []byte, depth int) error { t := new(LLOTimestampedStreamValue) if err := proto.Unmarshal(data, t); err != nil { return err } v.ObservedAtNanoseconds = t.ObservedAtNanoseconds - sv, err := UnmarshalProtoStreamValue(t.StreamValue) + sv, err := unmarshalProtoStreamValue(t.StreamValue, depth) if err != nil { return err } @@ -425,6 +472,12 @@ func (v *TimestampedStreamValue) MarshalText() ([]byte, error) { var timestampedStreamValueRegex = regexp.MustCompile(`^TSV\{ObservedAtNanoseconds: ([0-9]+), StreamValue: (.+)\}$`) func (v *TimestampedStreamValue) UnmarshalText(data []byte) error { + return v.unmarshalText(data, 0) +} + +// unmarshalText carries the nesting depth reached so far. See +// unmarshalProtoStreamValue. +func (v *TimestampedStreamValue) unmarshalText(data []byte, depth int) error { if v == nil { return ErrNilStreamValue } @@ -445,7 +498,7 @@ func (v *TimestampedStreamValue) UnmarshalText(data []byte) error { return fmt.Errorf("failed to unmarshal text stream value: %w", err) } - sv, err := UnmarshalTypedTextStreamValue(tSv) + sv, err := unmarshalTypedTextStreamValue(tSv, depth) if err != nil { return fmt.Errorf("failed to unmarshal text stream value: %w", err) } diff --git a/llo/protocol/stream_value_coefficient_test.go b/llo/protocol/stream_value_coefficient_test.go index 637912d..99ba866 100644 --- a/llo/protocol/stream_value_coefficient_test.go +++ b/llo/protocol/stream_value_coefficient_test.go @@ -68,12 +68,34 @@ func Test_UnmarshalObservedProtoStreamValue_CoefficientBound(t *testing.T) { require.NoError(t, err) }) - t.Run("nesting is bounded", func(t *testing.T) { - var sv StreamValue = ToDecimal(decimal.NewFromInt(1)) - for range MaxStreamValueNesting + 1 { - sv = &TimestampedStreamValue{ObservedAtNanoseconds: 1, StreamValue: sv} + t.Run("nesting is bounded during unmarshal", func(t *testing.T) { + nested := func(levels int) *LLOStreamValue { + var sv StreamValue = ToDecimal(decimal.NewFromInt(1)) + for range levels { + sv = &TimestampedStreamValue{ObservedAtNanoseconds: 1, StreamValue: sv} + } + return protoOf(t, sv) } - require.ErrorIs(t, checkObservedStreamValue(sv, 0), ErrStreamValueNestingTooDeep) + + // The bound is enforced by unmarshal itself, not by a check over the + // decoded value: the recursion is inside unmarshalling, and each level + // re-slices the nested bytes, so a post-hoc check runs too late. + _, err := UnmarshalProtoStreamValue(nested(MaxStreamValueNesting + 1)) + require.ErrorIs(t, err, ErrStreamValueNestingTooDeep) + _, err = UnmarshalObservedProtoStreamValue(nested(MaxStreamValueNesting + 1)) + require.ErrorIs(t, err, ErrStreamValueNestingTooDeep) + _, err = UnmarshalProtoStreamValue(nested(MaxStreamValueNesting)) + require.NoError(t, err) + + // Same recursion, text encoding. + tsv := &TimestampedStreamValue{ObservedAtNanoseconds: 1, StreamValue: ToDecimal(decimal.NewFromInt(1))} + for range MaxStreamValueNesting { + tsv = &TimestampedStreamValue{ObservedAtNanoseconds: 1, StreamValue: tsv} + } + ttsv, err := NewTypedTextStreamValue(tsv) + require.NoError(t, err) + _, err = UnmarshalTypedTextStreamValue(&ttsv) + require.ErrorIs(t, err, ErrStreamValueNestingTooDeep) }) t.Run("stored-state decode is deliberately unchecked", func(t *testing.T) { diff --git a/llo/protocol/stream_value_fuzz_test.go b/llo/protocol/stream_value_fuzz_test.go index 5f15aa6..79d6538 100644 --- a/llo/protocol/stream_value_fuzz_test.go +++ b/llo/protocol/stream_value_fuzz_test.go @@ -39,7 +39,7 @@ func FuzzUnmarshalObservedProtoStreamValue(f *testing.F) { } // An accepted value must satisfy the bounds the decoder claims to // enforce, at every nesting level. - if err := checkObservedStreamValue(sv, 0); err != nil { + if err := checkObservedStreamValue(sv); err != nil { t.Fatalf("accepted a value that violates its own bounds: %v", err) } }) From 4abc319cd694deffdc22bb6fb1257877ef2289fa Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Tue, 22 Sep 2026 11:06:31 +0100 Subject: [PATCH 28/40] SPOR-0004 llo/dev/v31: accumulate report codec support across rounds Persist what each oracle last advertised in a new c/codecs record and count supporters over that. The count increments to n across rounds while an oracle still speaks only for itself, so f byzantine oracles move it by at most f and 2f+1 remains reachable. An empty advertisement is recorded rather than ignored, so an oracle that loses a codec stops counting for it. The record is rewritten only when some oracle's advertised set changes, so the per-round write cost is unchanged. --- llo/dev/v31/kv.go | 77 ++++++++++++++++ llo/dev/v31/plugin_test.go | 79 +++++++++++++++++ llo/dev/v31/precursor.go | 16 +++- llo/dev/v31/statetransition.go | 100 ++++++++++++++++++--- llo/protocol/plugin_codecs.pb.go | 148 +++++++++++++++++++++++++++---- llo/protocol/plugin_codecs.proto | 15 ++++ 6 files changed, 406 insertions(+), 29 deletions(-) diff --git a/llo/dev/v31/kv.go b/llo/dev/v31/kv.go index 5ff7a84..006cbd0 100644 --- a/llo/dev/v31/kv.go +++ b/llo/dev/v31/kv.go @@ -10,7 +10,9 @@ import ( protocol "github.com/smartcontractkit/chainlink-data-streams/llo/protocol" + "github.com/smartcontractkit/libocr/commontypes" "github.com/smartcontractkit/libocr/offchainreporting2plus/ocr3_1types" + ocrtypes "github.com/smartcontractkit/libocr/offchainreporting2plus/types" "google.golang.org/protobuf/proto" ) @@ -25,6 +27,9 @@ import ( // c/pred -> LLOPredecessorConfigProto: the predecessor's signer set and // f, agreed by vote while staging (written at most once, and // only by an instance that has a predecessor) +// c/codecs -> LLOCodecSupportProto: the report formats each oracle last +// advertised a codec for (written only when some oracle's +// advertised set changes) // r/agg -> LLOHotStateProto: observation timestamp, validAfter // watermarks, per-channel reportability, and carry-forward // timestamped aggregates (written every round) @@ -49,6 +54,7 @@ var ( keyChannelState = []byte("c/defs") keyChannelSeqNr = []byte("c/seqnr") keyPredecessorConfig = []byte("c/pred") + keyCodecSupport = []byte("c/codecs") keyHotState = []byte("r/agg") keyHistoryIndex = []byte("hidx") @@ -111,6 +117,10 @@ type kvState struct { // concurrently-running round from swapping decoded opts out from under this // one; always read opts from here rather than from plugin-wide state. opts *protocol.OptsCache + // codecSupport[oracleID] is the report formats that oracle last advertised + // a codec for. See writeCodecSupport for why it is remembered per oracle + // instead of being counted per round. + codecSupport map[commontypes.OracleID][]llotypes.ReportFormat // channelStateSeqNr is the seqNr at which channelDefinitions were written. channelStateSeqNr uint64 validAfterNanoseconds map[llotypes.ChannelID]uint64 @@ -154,6 +164,7 @@ func loadKVState(r ocr3_1types.KeyValueStateReader, cache *protocol.ChannelCache func loadColdKVState(r ocr3_1types.KeyValueStateReader, cache *protocol.ChannelCache) (*kvState, error) { s := &kvState{ channelDefinitions: llotypes.ChannelDefinitions{}, + codecSupport: map[commontypes.OracleID][]llotypes.ReportFormat{}, validAfterNanoseconds: map[llotypes.ChannelID]uint64{}, reportedLastRound: map[llotypes.ChannelID]bool{}, carryForward: map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue{}, @@ -182,9 +193,75 @@ func loadColdKVState(r ocr3_1types.KeyValueStateReader, cache *protocol.ChannelC s.channelDefinitions = gen.Definitions() s.opts = gen.Opts() + support, err := readCodecSupport(r) + if err != nil { + return nil, err + } + s.codecSupport = support + return s, nil } +// readCodecSupport reads and decodes the c/codecs record. +func readCodecSupport(r ocr3_1types.KeyValueStateReader) (map[commontypes.OracleID][]llotypes.ReportFormat, error) { + support := map[commontypes.OracleID][]llotypes.ReportFormat{} + b, err := r.Read(keyCodecSupport) + if err != nil { + return nil, fmt.Errorf("read codec support: %w", err) + } + if len(b) == 0 { + return support, nil + } + pb := &protocol.LLOCodecSupportProto{} + if err := proto.Unmarshal(b, pb); err != nil { + return nil, fmt.Errorf("unmarshal codec support: %w", err) + } + if len(pb.Oracles) > ocrtypes.MaxOracles { + return nil, fmt.Errorf("codec support names too many oracles: %d (max %d)", len(pb.Oracles), ocrtypes.MaxOracles) + } + for _, entry := range pb.Oracles { + if len(entry.ReportFormats) > protocol.MaxObservationSupportedReportFormatsLength { + return nil, fmt.Errorf("oracle %d advertises too many report formats: %d (max %d)", entry.OracleID, len(entry.ReportFormats), protocol.MaxObservationSupportedReportFormatsLength) + } + formats := make([]llotypes.ReportFormat, 0, len(entry.ReportFormats)) + for _, f := range entry.ReportFormats { + formats = append(formats, llotypes.ReportFormat(f)) + } + support[commontypes.OracleID(entry.OracleID)] = formats + } + return support, nil +} + +// writeCodecSupport persists the report formats each oracle last advertised a +// codec for. +// +// Support is remembered per oracle rather than counted per round because the +// observation quorum is 2f+1. A count taken from one round can never exceed +// 2f+1, so the supporter threshold reportability would demand that every observation +// in a minimal quorum advertise the format. +func writeCodecSupport(w ocr3_1types.KeyValueStateReadWriter, support map[commontypes.OracleID][]llotypes.ReportFormat) error { + pb := &protocol.LLOCodecSupportProto{ + Oracles: make([]*protocol.LLOOracleCodecSupportProto, 0, len(support)), + } + for oracleID, formats := range support { + encoded := make([]uint32, 0, len(formats)) + for _, f := range formats { + encoded = append(encoded, uint32(f)) + } + sort.Slice(encoded, func(i, j int) bool { return encoded[i] < encoded[j] }) + pb.Oracles = append(pb.Oracles, &protocol.LLOOracleCodecSupportProto{ + OracleID: uint32(oracleID), + ReportFormats: encoded, + }) + } + sort.Slice(pb.Oracles, func(i, j int) bool { return pb.Oracles[i].OracleID < pb.Oracles[j].OracleID }) + b, err := deterministicMarshal.Marshal(pb) + if err != nil { + return fmt.Errorf("marshal codec support: %w", err) + } + return w.Write(keyCodecSupport, b) +} + // readChannelState reads and decodes the c/defs record. func readChannelState(r ocr3_1types.KeyValueStateReader) (llotypes.ChannelDefinitions, error) { b, err := r.Read(keyChannelState) diff --git a/llo/dev/v31/plugin_test.go b/llo/dev/v31/plugin_test.go index 73ab5bc..13bf6ca 100644 --- a/llo/dev/v31/plugin_test.go +++ b/llo/dev/v31/plugin_test.go @@ -1270,6 +1270,9 @@ func Test_StateTransition_TalliesReportFormatSupport(t *testing.T) { _, err := p.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, nil), ao(1, nil), ao(2, nil)}, kv, testBlobs) require.NoError(t, err) + require.NoError(t, writeChannelState(kv, 1, llotypes.ChannelDefinitions{ + 1: {ReportFormat: llotypes.ReportFormatJSON, Streams: []llotypes.Stream{{StreamID: 1, Aggregator: llotypes.AggregatorMedian}}}, + })) // Two oracles advertise JSON, one advertises nothing: the tally is a count // of advertisements, and an oracle that advertises nothing is not counted. @@ -1283,6 +1286,82 @@ func Test_StateTransition_TalliesReportFormatSupport(t *testing.T) { require.Equal(t, 2, prec.SupportByFormat[llotypes.ReportFormatJSON]) } +// Codec support accumulates across rounds, so f oracles omitting their +// advertisement cannot drop a format below the 2f+1 threshold. +// +// Counted per round, support could never exceed the 2f+1 observation quorum, so +// the threshold would demand that every observation in a minimal quorum +// advertise the format and any single omission would make every channel of that +// format unreportable. +func Test_StateTransition_CodecSupportAccumulatesAcrossRounds(t *testing.T) { + ctx := tests.Context(t) + p := testPlugin(t) + kv := newMemKV() + + _, err := p.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, nil), ao(1, nil), ao(2, nil)}, kv, testBlobs) + require.NoError(t, err) + require.NoError(t, writeChannelState(kv, 1, llotypes.ChannelDefinitions{ + 1: {ReportFormat: llotypes.ReportFormatJSON, Streams: []llotypes.Stream{{StreamID: 1, Aggregator: llotypes.AggregatorMedian}}}, + })) + + obsJSON := func(ts uint64) []byte { + return mustEncodeObs(t, Observation{UnixTimestampNanoseconds: ts, SupportedReportFormats: []llotypes.ReportFormat{llotypes.ReportFormatJSON}}) + } + obsNone := func(ts uint64) []byte { + return mustEncodeObs(t, Observation{UnixTimestampNanoseconds: ts, SupportedReportFormats: []llotypes.ReportFormat{}}) + } + + // All N oracles advertise JSON. + precBytes, err := p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, obsJSON(10)), ao(1, obsJSON(10)), ao(2, obsJSON(10)), ao(3, obsJSON(10))}, kv, testBlobs) + require.NoError(t, err) + prec, err := decodePrecursor(precBytes) + require.NoError(t, err) + require.Equal(t, p.N, prec.SupportByFormat[llotypes.ReportFormatJSON]) + + // A minimal quorum in which one oracle now omits its advertisement. Oracle + // 3 contributes nothing this round, but its last advertisement still counts, + // so the format keeps 2f+1 supporters and the channel stays reportable. + precBytes, err = p.StateTransition(ctx, 3, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, obsJSON(20)), ao(1, obsJSON(20)), ao(2, obsNone(20))}, kv, testBlobs) + require.NoError(t, err) + prec, err = decodePrecursor(precBytes) + require.NoError(t, err) + require.Equal(t, 2*p.F+1, prec.SupportByFormat[llotypes.ReportFormatJSON]) + require.Equal(t, []llotypes.ChannelID{1}, prec.reportableChannels(0, p.F, nil, logger.Test(t))) +} + +// One oracle padding its observation with unused report formats must not be +// able to grow the precursor tally: the encoded map is keyed by oracle-chosen +// values, and an oversized one fails to decode on every oracle, which would +// stop reporting DON-wide while StateTransition keeps succeeding. +func Test_StateTransition_PrunesUnusedReportFormatSupport(t *testing.T) { + ctx := tests.Context(t) + p := testPlugin(t) + kv := newMemKV() + + _, err := p.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, nil), ao(1, nil), ao(2, nil)}, kv, testBlobs) + require.NoError(t, err) + require.NoError(t, writeChannelState(kv, 1, llotypes.ChannelDefinitions{ + 1: {ReportFormat: llotypes.ReportFormatJSON, Streams: []llotypes.Stream{{StreamID: 1, Aggregator: llotypes.AggregatorMedian}}}, + })) + + // Each oracle advertises JSON plus a disjoint block of junk formats, so an + // unpruned tally would hold 3*MaxObservationSupportedReportFormatsLength-2 + // entries. + padded := func(oracle int) []byte { + formats := []llotypes.ReportFormat{llotypes.ReportFormatJSON} + for i := 1; i < protocol.MaxObservationSupportedReportFormatsLength; i++ { + formats = append(formats, llotypes.ReportFormat(1_000_000+oracle*1_000+i)) + } + return mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 1, SupportedReportFormats: formats}) + } + precBytes, err := p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, padded(0)), ao(1, padded(1)), ao(2, padded(2))}, kv, testBlobs) + require.NoError(t, err) + + prec, err := decodePrecursor(precBytes) + require.NoError(t, err) + require.Equal(t, map[llotypes.ReportFormat]int{llotypes.ReportFormatJSON: 3}, prec.SupportByFormat) +} + // strictJSONCodec is reportcodec.JSONReportCodec with an extra Verify rule this // build has and the build that admitted the definition did not: the version // skew that makes a baseline failure on committed state reachable. diff --git a/llo/dev/v31/precursor.go b/llo/dev/v31/precursor.go index d632750..b3363aa 100644 --- a/llo/dev/v31/precursor.go +++ b/llo/dev/v31/precursor.go @@ -26,8 +26,9 @@ type precursor struct { // ChannelDefinitions came from. It lets Reports tell whether the decoded-opts // cache already matches these definitions without walking every channel. ChannelStateSeqNr uint64 - // SupportByFormat is the number of oracles that advertised a report codec - // for each report format in the round that produced this precursor. + // SupportByFormat is the number of oracles whose last advertisement named + // each report format, counted over the cumulative c/codecs record rather + // than over this round's observations alone (see writeCodecSupport). // // Snapshotted rather than recomputed so that reportability and the // validAfter advance for a round read the identical number: Reports runs @@ -91,6 +92,9 @@ func encodePrecursor(p precursor) (ocr3_1types.ReportsPlusPrecursor, error) { }) } + if len(p.SupportByFormat) > protocol.MaxOutcomeChannelDefinitionsLength { + return nil, fmt.Errorf("too many report format support entries: %d (max %d)", len(p.SupportByFormat), protocol.MaxOutcomeChannelDefinitionsLength) + } if len(p.SupportByFormat) > 0 { pb.SupportByFormat = make([]*protocol.LLOReportFormatSupportProto, 0, len(p.SupportByFormat)) for format, count := range p.SupportByFormat { @@ -136,8 +140,12 @@ func decodePrecursor(b ocr3_1types.ReportsPlusPrecursor) (precursor, error) { for _, va := range pb.ValidAfterNanoseconds { p.ValidAfterNanoseconds[va.ChannelID] = va.ValidAfterNanoseconds } - if len(pb.SupportByFormat) > protocol.MaxObservationSupportedReportFormatsLength { - return precursor{}, fmt.Errorf("precursor carries too many report format support entries: %d (max %d)", len(pb.SupportByFormat), protocol.MaxObservationSupportedReportFormatsLength) + // One entry per report format in the effective channel set, so the channel + // cap is the bound here. The observation-side cap does not apply: the tally + // is pruned to effective formats before encoding, and the effective set may + // legitimately use more distinct formats than one observation may advertise. + if len(pb.SupportByFormat) > protocol.MaxOutcomeChannelDefinitionsLength { + return precursor{}, fmt.Errorf("precursor carries too many report format support entries: %d (max %d)", len(pb.SupportByFormat), protocol.MaxOutcomeChannelDefinitionsLength) } if len(pb.SupportByFormat) > 0 { p.SupportByFormat = make(map[llotypes.ReportFormat]int, len(pb.SupportByFormat)) diff --git a/llo/dev/v31/statetransition.go b/llo/dev/v31/statetransition.go index a5ae3f8..00d394b 100644 --- a/llo/dev/v31/statetransition.go +++ b/llo/dev/v31/statetransition.go @@ -16,6 +16,7 @@ import ( protocol "github.com/smartcontractkit/chainlink-data-streams/llo/protocol" "github.com/smartcontractkit/chainlink-data-streams/llo/protocol/calculated" + "github.com/smartcontractkit/libocr/commontypes" "github.com/smartcontractkit/libocr/offchainreporting2plus/ocr3_1types" ocrtypes "github.com/smartcontractkit/libocr/offchainreporting2plus/types" ) @@ -90,6 +91,10 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A return nil, err } + // Codec coverage is cumulative across rounds: merge this round's + // advertisements over the persisted ones before counting supporters. + codecSupport := mergeCodecSupport(prev.codecSupport, tally.supportedFormatsByOracle) + // The definitions in effect for this round are the ones Observation read. // Changes agreed below land in pending and take effect next round. effective := cloneChannelDefinitions(prev.channelDefinitions) @@ -101,7 +106,7 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A ChannelStateSeqNr: prev.channelStateSeqNr, ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{}, StreamAggregates: protocol.StreamAggregates{}, - SupportByFormat: tally.supportVotesByFormat, + SupportByFormat: supportVotesForEffectiveFormats(countSupportByFormat(codecSupport), effective), } // Lifecycle stage & promotion. @@ -220,7 +225,7 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A calculated.ProcessCalculatedStreams(p.Logger, effective, out.StreamAggregates, out.ObservationTimestampNanoseconds, prev.opts, history) // Flush KV mutations. - if err := p.flushKV(kvRW, seqNr, prev, out, pending, carryForward, history); err != nil { + if err := p.flushKV(kvRW, seqNr, prev, out, pending, codecSupport, carryForward, history); err != nil { return nil, err } @@ -250,8 +255,11 @@ type observationTally struct { removeChannelVotesByID map[llotypes.ChannelID]int updateChannelDefinitionsByHash map[[32]byte]protocol.ChannelDefinitionWithID updateChannelVotesByHash map[[32]byte]int - supportVotesByFormat map[llotypes.ReportFormat]int - streamObservations map[llotypes.StreamID][]protocol.StreamValue + // supportedFormatsByOracle[oracleID] is the formats that oracle advertised + // this round, deduped by decodeObservation. It is merged into the persisted + // per-oracle record rather than counted here: see writeCodecSupport. + supportedFormatsByOracle map[commontypes.OracleID][]llotypes.ReportFormat + streamObservations map[llotypes.StreamID][]protocol.StreamValue } func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.AttributedObservation, bf ocr3_1types.BlobFetcher, memo *roundBlobPayloads) (observationTally, error) { @@ -261,7 +269,7 @@ func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.Attribut removeChannelVotesByID: make(map[llotypes.ChannelID]int), updateChannelDefinitionsByHash: make(map[[32]byte]protocol.ChannelDefinitionWithID), updateChannelVotesByHash: make(map[[32]byte]int), - supportVotesByFormat: make(map[llotypes.ReportFormat]int), + supportedFormatsByOracle: make(map[commontypes.OracleID][]llotypes.ReportFormat), streamObservations: make(map[llotypes.StreamID][]protocol.StreamValue), } @@ -316,11 +324,11 @@ func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.Attribut } tally.timestampsNanoseconds = append(tally.timestampsNanoseconds, observation.UnixTimestampNanoseconds) - // Deduped by decodeObservation, so one oracle contributes at most one - // vote per format. - for _, format := range observation.SupportedReportFormats { - tally.supportVotesByFormat[format]++ - } + // Deduped by decodeObservation, so an oracle names each format at most + // once. Recording the whole advertised set (including an empty one) + // makes this round's advertisement replace that oracle's last, so an + // oracle that loses a codec stops counting for it. + tally.supportedFormatsByOracle[ao.Observer] = observation.SupportedReportFormats for channelID := range observation.RemoveChannelIDs { tally.removeChannelVotesByID[channelID]++ } @@ -340,6 +348,69 @@ func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.Attribut return tally, nil } +// mergeCodecSupport overlays this round's advertisements on the persisted ones, +// replacing the entry of every oracle that contributed an observation and +// leaving the rest untouched. +func mergeCodecSupport(persisted, thisRound map[commontypes.OracleID][]llotypes.ReportFormat) map[commontypes.OracleID][]llotypes.ReportFormat { + merged := make(map[commontypes.OracleID][]llotypes.ReportFormat, len(persisted)+len(thisRound)) + for oracleID, formats := range persisted { + merged[oracleID] = formats + } + for oracleID, formats := range thisRound { + merged[oracleID] = formats + } + return merged +} + +// codecSupportChanged reports whether any oracle's advertised set differs. Both +// sides hold deduped sets, so comparing as sets (not slices) is what matters: +// order must not trigger a rewrite. +func codecSupportChanged(prev, next map[commontypes.OracleID][]llotypes.ReportFormat) bool { + if len(prev) != len(next) { + return true + } + for oracleID, nextFormats := range next { + prevFormats, ok := prev[oracleID] + if !ok || len(prevFormats) != len(nextFormats) { + return true + } + seen := make(map[llotypes.ReportFormat]struct{}, len(prevFormats)) + for _, f := range prevFormats { + seen[f] = struct{}{} + } + for _, f := range nextFormats { + if _, ok := seen[f]; !ok { + return true + } + } + } + return false +} + +// countSupportByFormat counts, per report format, the oracles whose last +// advertisement named it. +func countSupportByFormat(support map[commontypes.OracleID][]llotypes.ReportFormat) map[llotypes.ReportFormat]int { + counts := make(map[llotypes.ReportFormat]int) + for _, formats := range support { + for _, format := range formats { + counts[format]++ + } + } + return counts +} + +// supportVotesForEffectiveFormats restricts the codec support tally to the +// report formats this round can actually needs. +func supportVotesForEffectiveFormats(votes map[llotypes.ReportFormat]int, effective llotypes.ChannelDefinitions) map[llotypes.ReportFormat]int { + pruned := make(map[llotypes.ReportFormat]int, len(votes)) + for _, cd := range effective { + if n, ok := votes[cd.ReportFormat]; ok { + pruned[cd.ReportFormat] = n + } + } + return pruned +} + // applyChannelVotes applies remove/add votes with a >F threshold, in ascending // channelID order, respecting MaxOutcomeChannelDefinitionsLength. // @@ -553,6 +624,7 @@ func (p *Plugin) flushKV( prev *kvState, out precursor, pending llotypes.ChannelDefinitions, + codecSupport map[commontypes.OracleID][]llotypes.ReportFormat, carryForward map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue, history *historyStore, ) error { @@ -574,6 +646,14 @@ func (p *Plugin) flushKV( // reloads. } + // Codec coverage: rewrite the record only when an oracle's advertised set + // actually changed, which is rare outside a rollout. + if codecSupportChanged(prev.codecSupport, codecSupport) { + if err := writeCodecSupport(kvRW, codecSupport); err != nil { + return err + } + } + // Reportability: persist this round's decision for each channel so the next // round can advance validAfter faithfully (see prevReportable). reportable := make(map[llotypes.ChannelID]bool, len(out.ChannelDefinitions)) diff --git a/llo/protocol/plugin_codecs.pb.go b/llo/protocol/plugin_codecs.pb.go index 4524712..04caf82 100644 --- a/llo/protocol/plugin_codecs.pb.go +++ b/llo/protocol/plugin_codecs.pb.go @@ -1,7 +1,7 @@ // Code generated by protoc-gen-go. DO NOT EDIT. // versions: // protoc-gen-go v1.36.12 -// protoc v7.36.1 +// protoc v7.36.2 // source: plugin_codecs.proto package protocol @@ -1492,6 +1492,116 @@ func (x *LLOReportFormatSupportProto) GetOracleCount() uint32 { return 0 } +// LLOCodecSupportProto is the v31 KeyValueState record (c/codecs) holding the +// report formats each oracle last advertised a codec for. +// +// Kept per oracle rather than as a per-round count because the round's +// observation quorum is only 2f+1: a count taken from a single round can never +// exceed that, so requiring 2f+1 supporters would demand unanimity and f +// oracles omitting their advertisement would make every format unreportable. +// Remembering each oracle's last advertisement lets the supporter count reach +// n over successive rounds, and an oracle can still only speak for itself. +// +// NOTE: must serialize deterministically. oracles MUST be sorted ascending by +// oracleID, and each oracle's reportFormats MUST be sorted ascending. +type LLOCodecSupportProto struct { + state protoimpl.MessageState `protogen:"open.v1"` + Oracles []*LLOOracleCodecSupportProto `protobuf:"bytes,1,rep,name=oracles,proto3" json:"oracles,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache +} + +func (x *LLOCodecSupportProto) Reset() { + *x = LLOCodecSupportProto{} + mi := &file_plugin_codecs_proto_msgTypes[21] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) +} + +func (x *LLOCodecSupportProto) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*LLOCodecSupportProto) ProtoMessage() {} + +func (x *LLOCodecSupportProto) ProtoReflect() protoreflect.Message { + mi := &file_plugin_codecs_proto_msgTypes[21] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use LLOCodecSupportProto.ProtoReflect.Descriptor instead. +func (*LLOCodecSupportProto) Descriptor() ([]byte, []int) { + return file_plugin_codecs_proto_rawDescGZIP(), []int{21} +} + +func (x *LLOCodecSupportProto) GetOracles() []*LLOOracleCodecSupportProto { + if x != nil { + return x.Oracles + } + return nil +} + +// LLOOracleCodecSupportProto is one oracle's last advertised set of report +// formats. +type LLOOracleCodecSupportProto struct { + state protoimpl.MessageState `protogen:"open.v1"` + OracleID uint32 `protobuf:"varint,1,opt,name=oracleID,proto3" json:"oracleID,omitempty"` + ReportFormats []uint32 `protobuf:"varint,2,rep,packed,name=reportFormats,proto3" json:"reportFormats,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache +} + +func (x *LLOOracleCodecSupportProto) Reset() { + *x = LLOOracleCodecSupportProto{} + mi := &file_plugin_codecs_proto_msgTypes[22] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) +} + +func (x *LLOOracleCodecSupportProto) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*LLOOracleCodecSupportProto) ProtoMessage() {} + +func (x *LLOOracleCodecSupportProto) ProtoReflect() protoreflect.Message { + mi := &file_plugin_codecs_proto_msgTypes[22] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use LLOOracleCodecSupportProto.ProtoReflect.Descriptor instead. +func (*LLOOracleCodecSupportProto) Descriptor() ([]byte, []int) { + return file_plugin_codecs_proto_rawDescGZIP(), []int{22} +} + +func (x *LLOOracleCodecSupportProto) GetOracleID() uint32 { + if x != nil { + return x.OracleID + } + return 0 +} + +func (x *LLOOracleCodecSupportProto) GetReportFormats() []uint32 { + if x != nil { + return x.ReportFormats + } + return nil +} + var File_plugin_codecs_proto protoreflect.FileDescriptor const file_plugin_codecs_proto_rawDesc = "" + @@ -1603,7 +1713,12 @@ const file_plugin_codecs_proto_rawDesc = "" + "\x0fsupportByFormat\x18\a \x03(\v2\x1f.v1.LLOReportFormatSupportProtoR\x0fsupportByFormat\"c\n" + "\x1bLLOReportFormatSupportProto\x12\"\n" + "\freportFormat\x18\x01 \x01(\rR\freportFormat\x12 \n" + - "\voracleCount\x18\x02 \x01(\rR\voracleCountB\fZ\n" + + "\voracleCount\x18\x02 \x01(\rR\voracleCount\"P\n" + + "\x14LLOCodecSupportProto\x128\n" + + "\aoracles\x18\x01 \x03(\v2\x1e.v1.LLOOracleCodecSupportProtoR\aoracles\"^\n" + + "\x1aLLOOracleCodecSupportProto\x12\x1a\n" + + "\boracleID\x18\x01 \x01(\rR\boracleID\x12$\n" + + "\rreportFormats\x18\x02 \x03(\rR\rreportFormatsB\fZ\n" + ".;protocolb\x06proto3" var ( @@ -1619,7 +1734,7 @@ func file_plugin_codecs_proto_rawDescGZIP() []byte { } var file_plugin_codecs_proto_enumTypes = make([]protoimpl.EnumInfo, 1) -var file_plugin_codecs_proto_msgTypes = make([]protoimpl.MessageInfo, 23) +var file_plugin_codecs_proto_msgTypes = make([]protoimpl.MessageInfo, 25) var file_plugin_codecs_proto_goTypes = []any{ (LLOStreamValue_Type)(0), // 0: v1.LLOStreamValue.Type (*LLOObservationProto)(nil), // 1: v1.LLOObservationProto @@ -1643,12 +1758,14 @@ var file_plugin_codecs_proto_goTypes = []any{ (*LLOHotStateProto)(nil), // 19: v1.LLOHotStateProto (*LLOPrecursorProto)(nil), // 20: v1.LLOPrecursorProto (*LLOReportFormatSupportProto)(nil), // 21: v1.LLOReportFormatSupportProto - nil, // 22: v1.LLOObservationProto.UpdateChannelDefinitionsEntry - nil, // 23: v1.LLOObservationProto.StreamValuesEntry + (*LLOCodecSupportProto)(nil), // 22: v1.LLOCodecSupportProto + (*LLOOracleCodecSupportProto)(nil), // 23: v1.LLOOracleCodecSupportProto + nil, // 24: v1.LLOObservationProto.UpdateChannelDefinitionsEntry + nil, // 25: v1.LLOObservationProto.StreamValuesEntry } var file_plugin_codecs_proto_depIdxs = []int32{ - 22, // 0: v1.LLOObservationProto.updateChannelDefinitions:type_name -> v1.LLOObservationProto.UpdateChannelDefinitionsEntry - 23, // 1: v1.LLOObservationProto.streamValues:type_name -> v1.LLOObservationProto.StreamValuesEntry + 24, // 0: v1.LLOObservationProto.updateChannelDefinitions:type_name -> v1.LLOObservationProto.UpdateChannelDefinitionsEntry + 25, // 1: v1.LLOObservationProto.streamValues:type_name -> v1.LLOObservationProto.StreamValuesEntry 0, // 2: v1.LLOStreamValue.type:type_name -> v1.LLOStreamValue.Type 2, // 3: v1.LLOTimestampedStreamValue.streamValue:type_name -> v1.LLOStreamValue 2, // 4: v1.LLOStreamHistoryRecord.value:type_name -> v1.LLOStreamValue @@ -1669,13 +1786,14 @@ var file_plugin_codecs_proto_depIdxs = []int32{ 15, // 19: v1.LLOPrecursorProto.validAfterNanoseconds:type_name -> v1.LLOChannelIDAndValidAfterNanosecondsProto 16, // 20: v1.LLOPrecursorProto.streamAggregates:type_name -> v1.LLOStreamAggregate 21, // 21: v1.LLOPrecursorProto.supportByFormat:type_name -> v1.LLOReportFormatSupportProto - 8, // 22: v1.LLOObservationProto.UpdateChannelDefinitionsEntry.value:type_name -> v1.LLOChannelDefinitionProto - 2, // 23: v1.LLOObservationProto.StreamValuesEntry.value:type_name -> v1.LLOStreamValue - 24, // [24:24] is the sub-list for method output_type - 24, // [24:24] is the sub-list for method input_type - 24, // [24:24] is the sub-list for extension type_name - 24, // [24:24] is the sub-list for extension extendee - 0, // [0:24] is the sub-list for field type_name + 23, // 22: v1.LLOCodecSupportProto.oracles:type_name -> v1.LLOOracleCodecSupportProto + 8, // 23: v1.LLOObservationProto.UpdateChannelDefinitionsEntry.value:type_name -> v1.LLOChannelDefinitionProto + 2, // 24: v1.LLOObservationProto.StreamValuesEntry.value:type_name -> v1.LLOStreamValue + 25, // [25:25] is the sub-list for method output_type + 25, // [25:25] is the sub-list for method input_type + 25, // [25:25] is the sub-list for extension type_name + 25, // [25:25] is the sub-list for extension extendee + 0, // [0:25] is the sub-list for field type_name } func init() { file_plugin_codecs_proto_init() } @@ -1689,7 +1807,7 @@ func file_plugin_codecs_proto_init() { GoPackagePath: reflect.TypeOf(x{}).PkgPath(), RawDescriptor: unsafe.Slice(unsafe.StringData(file_plugin_codecs_proto_rawDesc), len(file_plugin_codecs_proto_rawDesc)), NumEnums: 1, - NumMessages: 23, + NumMessages: 25, NumExtensions: 0, NumServices: 0, }, diff --git a/llo/protocol/plugin_codecs.proto b/llo/protocol/plugin_codecs.proto index 0048f8c..14cda94 100644 --- a/llo/protocol/plugin_codecs.proto +++ b/llo/protocol/plugin_codecs.proto @@ -251,3 +251,18 @@ message LLOReportFormatSupportProto { uint32 reportFormat = 1; uint32 oracleCount = 2; } + +// LLOCodecSupportProto is the v31 KeyValueState record (c/codecs) holding the +// report formats each oracle last advertised a codec for. +// NOTE: must serialize deterministically. oracles MUST be sorted ascending by +// oracleID, and each oracle's reportFormats MUST be sorted ascending. +message LLOCodecSupportProto { + repeated LLOOracleCodecSupportProto oracles = 1; +} + +// LLOOracleCodecSupportProto is one oracle's last advertised set of report +// formats. +message LLOOracleCodecSupportProto { + uint32 oracleID = 1; + repeated uint32 reportFormats = 2; +} From e13406b8ef05239b3d6d2c07db569c7aa8a83cf9 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Tue, 22 Sep 2026 12:33:53 +0100 Subject: [PATCH 29/40] SPOR-0013 llo/dev/v31: use observation timestamp and ensure monotonicity --- llo/dev/v31/plugin.go | 11 +------ llo/dev/v31/plugin_test.go | 54 ++++++++++++++++++++++++++++++++++ llo/dev/v31/statetransition.go | 19 +++++++++++- 3 files changed, 73 insertions(+), 11 deletions(-) diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index 27f676a..e6d62be 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -183,16 +183,7 @@ func (p *Plugin) Observation(_ context.Context, seqNr uint64, _ ocrtypes.Attribu } } - // Timestamp the data, not the round. The pump gathers stream values off the - // critical path, so they were read before this round started; stamping - // time.Now() would have the report claim the values are newer than they are - // for every aggregate that does not carry its own timestamp. A round with no - // snapshot carries only votes, for which the round time is the right stamp. - obsTime := time.Now() - if snap != nil { - obsTime = snap.observedAt - } - obsTSNanos := obsTime.UnixNano() + obsTSNanos := time.Now().UnixNano() if obsTSNanos < 0 { return nil, fmt.Errorf("negative observation timestamps are not supported, got: %d", obsTSNanos) } diff --git a/llo/dev/v31/plugin_test.go b/llo/dev/v31/plugin_test.go index 13bf6ca..2afb0c5 100644 --- a/llo/dev/v31/plugin_test.go +++ b/llo/dev/v31/plugin_test.go @@ -1286,6 +1286,60 @@ func Test_StateTransition_TalliesReportFormatSupport(t *testing.T) { require.Equal(t, 2, prec.SupportByFormat[llotypes.ReportFormatJSON]) } +func Test_StateTransition_ObservationTimestampDoesNotRegress(t *testing.T) { + ctx := tests.Context(t) + p := testPlugin(t) + kv := newMemKV() + + _, err := p.StateTransition(ctx, 1, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, nil), ao(1, nil), ao(2, nil)}, kv, testBlobs) + require.NoError(t, err) + + obsAt := func(ts uint64) []byte { + return mustEncodeObs(t, Observation{UnixTimestampNanoseconds: ts}) + } + agreedAt := func(seqNr uint64, aos ...ocrtypes.AttributedObservation) uint64 { + precBytes, err := p.StateTransition(ctx, seqNr, ocrtypes.AttributedQuery{}, aos, kv, testBlobs) + require.NoError(t, err) + prec, err := decodePrecursor(precBytes) + require.NoError(t, err) + return prec.ObservationTimestampNanoseconds + } + + require.Equal(t, uint64(1_000), agreedAt(2, ao(0, obsAt(1_000)), ao(1, obsAt(1_000)), ao(2, obsAt(1_000)))) + + // Node clocks disagree and the quorum that carried the higher stamp is gone: + // hold the previous timestamp rather than dating a report before a watermark + // already in use. + require.Equal(t, uint64(1_000), agreedAt(3, ao(0, obsAt(500)), ao(1, obsAt(500)), ao(2, obsAt(500)))) + + // Forward again once the clock passes the floor. + require.Equal(t, uint64(2_000), agreedAt(4, ao(0, obsAt(2_000)), ao(1, obsAt(2_000)), ao(2, obsAt(2_000)))) +} + +func Test_Observation_StampsTheRoundNotTheSnapshot(t *testing.T) { + ctx := tests.Context(t) + p := testPlugin(t) + kv := newMemKV() + require.NoError(t, writeChannelState(kv, 1, llotypes.ChannelDefinitions{ + 1: {ReportFormat: llotypes.ReportFormatJSON, Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}}, + })) + p.ChannelDefinitionCache = &mockChannelDefinitionCache{} + p.ShouldRetireCache = &mockShouldRetireCache{} + bc := newFakeBroadcaster() + attachPump(t, p, &recordingDataSource{}, bc) + + // Whether or not the pump has a snapshot ready, the stamp is the round time. + for _, round := range []uint64{2, 3} { + before := uint64(time.Now().UnixNano()) //nolint:gosec // G115 test clock is positive + obsBytes, err := p.Observation(ctx, round, ocrtypes.AttributedQuery{}, kv, nil) + require.NoError(t, err) + obs, err := decodeObservation(ctx, obsBytes, bc, nil) + require.NoError(t, err) + require.GreaterOrEqual(t, obs.UnixTimestampNanoseconds, before) + require.Eventually(t, func() bool { return p.pump.Cycles() >= 1 }, tests.WaitTimeout(t), 10*time.Millisecond) + } +} + // Codec support accumulates across rounds, so f oracles omitting their // advertisement cannot drop a format below the 2f+1 threshold. // diff --git a/llo/dev/v31/statetransition.go b/llo/dev/v31/statetransition.go index 00d394b..49ec823 100644 --- a/llo/dev/v31/statetransition.go +++ b/llo/dev/v31/statetransition.go @@ -101,7 +101,7 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A pending := cloneChannelDefinitions(prev.channelDefinitions) out := precursor{ - ObservationTimestampNanoseconds: medianTimestamp(tally.timestampsNanoseconds), + ObservationTimestampNanoseconds: p.agreedObservationTimestamp(tally.timestampsNanoseconds, prev.observationTimestampNs, seqNr), ChannelDefinitions: effective, ChannelStateSeqNr: prev.channelStateSeqNr, ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{}, @@ -696,6 +696,23 @@ func prevReportable(prev *kvState, channelID llotypes.ChannelID) bool { return prev.reportedLastRound[channelID] } +// agreedObservationTimestamp is the round observation timestamp, the median of +// the observed timestamps, held to the previous round's value as a floor. +// +// Monotonically increasing, validAfter advances to it, reportability requires the +// next round to exceed that watermark and appendHistory only records a value +// strictly newer than the newest stored one. A regression leaves every channel +// unreportable and silently drops history appends until the clock catches back up. +func (p *Plugin) agreedObservationTimestamp(timestampsNanoseconds []uint64, prevObservationTimestampNs uint64, seqNr uint64) uint64 { + median := medianTimestamp(timestampsNanoseconds) + if median < prevObservationTimestampNs { + p.Logger.Warnw("Observation timestamp median regressed; holding the previous round's timestamp", + "seqNr", seqNr, "median", median, "prev", prevObservationTimestampNs, "contributors", len(timestampsNanoseconds)) + return prevObservationTimestampNs + } + return median +} + func medianTimestamp(timestampsNanoseconds []uint64) uint64 { sort.Slice(timestampsNanoseconds, func(i, j int) bool { return timestampsNanoseconds[i] < timestampsNanoseconds[j] }) return timestampsNanoseconds[len(timestampsNanoseconds)/2] From b6fa0583707d0a433765f4ffc2f3645d5949506b Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Tue, 22 Sep 2026 13:37:20 +0100 Subject: [PATCH 30/40] SPOR-0019 llo/dev/v31: bound the blob payload memo to a byte budget --- llo/dev/v31/blobmemo.go | 40 +++++++++++++++++++++++++++++++--- llo/dev/v31/blobmemo_test.go | 42 ++++++++++++++++++++++++++++++++++++ 2 files changed, 79 insertions(+), 3 deletions(-) diff --git a/llo/dev/v31/blobmemo.go b/llo/dev/v31/blobmemo.go index 9f0cea9..b8e0cbf 100644 --- a/llo/dev/v31/blobmemo.go +++ b/llo/dev/v31/blobmemo.go @@ -4,8 +4,21 @@ import ( "sync" protocol "github.com/smartcontractkit/chainlink-data-streams/llo/protocol" + + ocrtypes "github.com/smartcontractkit/libocr/offchainreporting2plus/types" ) +// maxMemoizedPayloadBytes bounds the decompressed bytes one round's memo +// may hold across all of its entries.The budget is a small multiple of what +// one observation may carry, which covers honest traffic while keeping the +// worst case independent of N. +const maxMemoizedPayloadBytes = 4 * maxObservationDecompressedBytes + +// maxMemoizedPayloads bounds the entry count, which the byte budget alone does +// not: an empty payload costs no bytes. No round can legitimately reference more +// handles than every oracle naming the most it is allowed. +const maxMemoizedPayloads = ocrtypes.MaxOracles * maxObservationBlobHandles + // blobPayloadCache memoizes the stream values decoded from blob payloads within // one sequence number. ValidateObservation and StateTransition decode the same // observations in the same round, and every FetchBlob re-verifies the blob's @@ -23,9 +36,16 @@ import ( // // Decoded stream values are treated as immutable: a hit copies the map entries // into the observation rather than handing out the memoized map. +// +// The memo enforces its own budget (maxMemoizedPayloadBytes) and simply declines +// to store beyond it. Declining is safe because memoization is an optimization, +// and a miss decodes and costs exactly what a hit would have. type blobPayloadCache struct { - mu sync.Mutex - seqNr uint64 + mu sync.Mutex + seqNr uint64 + // bytes is the sum of entries' sizes, tracked so the budget does not have + // to walk the map on every write. + bytes int entries map[string]blobPayloadEntry } @@ -54,6 +74,7 @@ func (c *blobPayloadCache) round(seqNr uint64) *roundBlobPayloads { if c.seqNr != seqNr || c.entries == nil { c.seqNr = seqNr c.entries = make(map[string]blobPayloadEntry) + c.bytes = 0 } return &roundBlobPayloads{cache: c, seqNr: seqNr} } @@ -88,5 +109,18 @@ func (r *roundBlobPayloads) put(handle []byte, entry blobPayloadEntry) { if r.cache.seqNr != r.seqNr { return } - r.cache.entries[string(handle)] = entry + key := string(handle) + prev, replacing := r.cache.entries[key] + if !replacing && len(r.cache.entries) >= maxMemoizedPayloads { + return + } + bytes := r.cache.bytes + entry.size + if replacing { + bytes -= prev.size + } + if bytes > maxMemoizedPayloadBytes { + return + } + r.cache.entries[key] = entry + r.cache.bytes = bytes } diff --git a/llo/dev/v31/blobmemo_test.go b/llo/dev/v31/blobmemo_test.go index b306df5..b6e000e 100644 --- a/llo/dev/v31/blobmemo_test.go +++ b/llo/dev/v31/blobmemo_test.go @@ -145,3 +145,45 @@ func Test_BlobPayloadCache_BudgetChargedOnHit(t *testing.T) { require.NoError(t, memoErr) require.Len(t, withMemo.StreamValues, len(withoutMemo.StreamValues)) } + +// The memo declines to grow past its budget. +func Test_BlobPayloadCache_BoundedByByteBudget(t *testing.T) { + c := newBlobPayloadCache() + r := c.round(7) + + // Each entry claims a quarter of the budget, so the fifth does not fit. + size := maxMemoizedPayloadBytes / 4 + for i := 0; i < 4; i++ { + r.put([]byte{byte(i)}, blobPayloadEntry{values: testStreamValues(1), size: size}) + } + require.Len(t, c.entries, 4) + require.Equal(t, maxMemoizedPayloadBytes, c.bytes) + + r.put([]byte{4}, blobPayloadEntry{values: testStreamValues(1), size: 1}) + require.Len(t, c.entries, 4, "an entry that does not fit the budget is not memoized") + require.Equal(t, maxMemoizedPayloadBytes, c.bytes) + + // Replacing an entry is charged as a delta, not as an addition. + r.put([]byte{0}, blobPayloadEntry{values: testStreamValues(1), size: size - 1}) + require.Len(t, c.entries, 4) + require.Equal(t, maxMemoizedPayloadBytes-1, c.bytes) + + // The budget is per round, so a new sequence number starts empty. + r = c.round(8) + require.Empty(t, c.entries) + require.Zero(t, c.bytes) + r.put([]byte{0}, blobPayloadEntry{values: testStreamValues(1), size: size}) + require.Len(t, c.entries, 1) +} + +// The byte budget alone does not bound the entry count, because an empty +// payload costs no bytes. +func Test_BlobPayloadCache_BoundedByEntryCount(t *testing.T) { + c := newBlobPayloadCache() + r := c.round(7) + + for i := 0; i < maxMemoizedPayloads+10; i++ { + r.put([]byte{byte(i % 256), byte(i / 256)}, blobPayloadEntry{size: 0}) + } + require.Len(t, c.entries, maxMemoizedPayloads) +} From b4025e256534a4eebac08ae82f164a9ecc9db095 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Tue, 22 Sep 2026 14:01:22 +0100 Subject: [PATCH 31/40] llo/dev/v31: colapse the unreportable channel warnings --- llo/dev/v31/flow_test.go | 4 +- llo/dev/v31/plugin_test.go | 79 +++++++++++++++++++++++---------- llo/dev/v31/reports.go | 81 +++++++++++++++++++++++++++++----- llo/dev/v31/statetransition.go | 3 +- 4 files changed, 130 insertions(+), 37 deletions(-) diff --git a/llo/dev/v31/flow_test.go b/llo/dev/v31/flow_test.go index bc3a94b..300259a 100644 --- a/llo/dev/v31/flow_test.go +++ b/llo/dev/v31/flow_test.go @@ -902,12 +902,12 @@ func Test_IsReportable_MinReportIntervalDoesNotOverflow(t *testing.T) { optsCache := gen.Opts() // One nanosecond past validAfter, so only the interval can hold it back. - require.True(t, out.isReportable(1, 1, 1, optsCache, logger.Test(t)), + require.True(t, out.isReportable(1, 1, 1, optsCache, nil), "a one nanosecond interval must not gate a report one nanosecond late") // An interval that overflows the sum must gate, not wrap into passing. for _, interval := range []uint64{math.MaxUint64, math.MaxUint64 - validAfter + 1} { - require.False(t, out.isReportable(1, interval, 1, optsCache, logger.Test(t)), + require.False(t, out.isReportable(1, interval, 1, optsCache, nil), "interval %d must gate the report, not wrap", interval) } diff --git a/llo/dev/v31/plugin_test.go b/llo/dev/v31/plugin_test.go index 2afb0c5..5c72f45 100644 --- a/llo/dev/v31/plugin_test.go +++ b/llo/dev/v31/plugin_test.go @@ -517,7 +517,7 @@ func Test_SecondsResolutionOverlap(t *testing.T) { for _, tc := range tests { t.Run(tc.name, func(t *testing.T) { p := mkPrec(tc.format, tc.opts, tc.validAfter, tc.obsTs) - got := p.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), logger.Test(t)) + got := p.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), nil) if tc.reportable { require.Equal(t, []llotypes.ChannelID{1}, got) } else { @@ -545,14 +545,14 @@ func Test_DisableNilStreamValues(t *testing.T) { // Missing stream 200 -> not reportable. missing := base(protocol.StreamAggregates{100: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(1))}}) - require.Empty(t, missing.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), logger.Test(t))) + require.Empty(t, missing.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), nil)) // Both streams present -> reportable. full := base(protocol.StreamAggregates{ 100: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(1))}, 200: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(2))}, }) - require.Equal(t, []llotypes.ChannelID{1}, full.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), logger.Test(t))) + require.Equal(t, []llotypes.ChannelID{1}, full.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), nil)) } func Test_DisableNilStreamValues_CalculatedStreams(t *testing.T) { @@ -609,17 +609,17 @@ func Test_DisableNilStreamValues_CalculatedStreams(t *testing.T) { // ProcessCalculatedStreams bailed before writing the calculated // aggregate; the definition alone looks complete. o := mkPrec(true, validOpts, baseStreams, baseAggregates()) - require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), logger.Test(t))) + require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), nil)) }) t.Run("inline calculated stream but nil aggregate -> not reportable", func(t *testing.T) { o := mkPrec(true, validOpts, withCalculated, baseAggregates()) - require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), logger.Test(t))) + require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), nil)) }) t.Run("fully evaluated -> reportable", func(t *testing.T) { o := mkPrec(true, validOpts, withCalculated, evaluatedAggregates()) - require.Equal(t, []llotypes.ChannelID{1}, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), logger.Test(t))) + require.Equal(t, []llotypes.ChannelID{1}, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), nil)) }) t.Run("DisableNilStreamValues=false, evaluation failed -> not reportable", func(t *testing.T) { @@ -629,32 +629,32 @@ func Test_DisableNilStreamValues_CalculatedStreams(t *testing.T) { // report. Treating the channel as reportable would advance validAfter // over a round that emitted nothing. o := mkPrec(false, validOpts, baseStreams, baseAggregates()) - require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), logger.Test(t))) + require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), nil)) }) t.Run("DisableNilStreamValues=false, fully evaluated -> reportable", func(t *testing.T) { o := mkPrec(false, validOpts, withCalculated, evaluatedAggregates()) - require.Equal(t, []llotypes.ChannelID{1}, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), logger.Test(t))) + require.Equal(t, []llotypes.ChannelID{1}, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), nil)) }) t.Run("malformed opts -> not reportable", func(t *testing.T) { o := mkPrec(true, []byte(`{"abi":`), withCalculated, evaluatedAggregates()) - require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), logger.Test(t))) + require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), nil)) }) t.Run("opts declare no expressions -> not reportable", func(t *testing.T) { o := mkPrec(true, []byte(`{"abi":[]}`), withCalculated, evaluatedAggregates()) - require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), logger.Test(t))) + require.Empty(t, o.withSupport(1).reportableChannels(0, 0, populatedCache(o), nil)) }) t.Run("cache miss falls back to channel opts -> reportable", func(t *testing.T) { o := mkPrec(true, validOpts, withCalculated, evaluatedAggregates()) - require.Equal(t, []llotypes.ChannelID{1}, o.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), logger.Test(t))) + require.Equal(t, []llotypes.ChannelID{1}, o.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), nil)) }) t.Run("cache miss falls back to channel opts -> not reportable when unevaluated", func(t *testing.T) { o := mkPrec(true, validOpts, baseStreams, baseAggregates()) - require.Empty(t, o.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), logger.Test(t))) + require.Empty(t, o.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), nil)) }) } @@ -1058,7 +1058,7 @@ func Test_IsReportable_EffectiveStreamsFailure(t *testing.T) { 100: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(1))}, }, } - require.Empty(t, prec.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), logger.Test(t))) + require.Empty(t, prec.withSupport(1).reportableChannels(0, 0, protocol.NewOptsCache(), nil)) } func Test_SelectBackfillCandidate_UnemittableRow(t *testing.T) { @@ -1124,21 +1124,21 @@ func Test_ReportFormatSupportGate(t *testing.T) { // f=1 requires 2f+1 = 3 advertised supporters: 2f would only guarantee f+1 // real encoders if none of the advertisements were lies. - require.Equal(t, []llotypes.ChannelID{cid}, base(3).reportableChannels(0, 1, protocol.NewOptsCache(), logger.Test(t))) - require.Empty(t, base(2).reportableChannels(0, 1, protocol.NewOptsCache(), logger.Test(t))) - require.Empty(t, base(0).reportableChannels(0, 1, protocol.NewOptsCache(), logger.Test(t))) + require.Equal(t, []llotypes.ChannelID{cid}, base(3).reportableChannels(0, 1, protocol.NewOptsCache(), nil)) + require.Empty(t, base(2).reportableChannels(0, 1, protocol.NewOptsCache(), nil)) + require.Empty(t, base(0).reportableChannels(0, 1, protocol.NewOptsCache(), nil)) // A format no oracle advertises is never reportable, however healthy the // channel otherwise is. noEntry := base(3) noEntry.SupportByFormat = map[llotypes.ReportFormat]int{} - require.Empty(t, noEntry.reportableChannels(0, 1, protocol.NewOptsCache(), logger.Test(t))) + require.Empty(t, noEntry.reportableChannels(0, 1, protocol.NewOptsCache(), nil)) // Support is keyed by format, not channel: an unrelated format's coverage // does not carry the channel. wrongFormat := base(0) wrongFormat.SupportByFormat = map[llotypes.ReportFormat]int{llotypes.ReportFormatEVMPremiumLegacy: 4} - require.Empty(t, wrongFormat.reportableChannels(0, 1, protocol.NewOptsCache(), logger.Test(t))) + require.Empty(t, wrongFormat.reportableChannels(0, 1, protocol.NewOptsCache(), nil)) } func Test_ReportFormatSupportGate_Backfill(t *testing.T) { @@ -1169,10 +1169,10 @@ func Test_ReportFormatSupportGate_Backfill(t *testing.T) { // format is what must be covered. Coverage of history_backfill itself is // irrelevant: no codec encodes it. targetCovered := base(map[llotypes.ReportFormat]int{llotypes.ReportFormatJSON: 3}) - require.Equal(t, []llotypes.ChannelID{backfillCID}, targetCovered.reportableChannels(0, 1, protocol.NewOptsCache(), logger.Test(t))) + require.Equal(t, []llotypes.ChannelID{backfillCID}, targetCovered.reportableChannels(0, 1, protocol.NewOptsCache(), nil)) backfillCoveredOnly := base(map[llotypes.ReportFormat]int{llotypes.ReportFormatHistoryBackfill: 4}) - require.Empty(t, backfillCoveredOnly.reportableChannels(0, 1, protocol.NewOptsCache(), logger.Test(t))) + require.Empty(t, backfillCoveredOnly.reportableChannels(0, 1, protocol.NewOptsCache(), nil)) } func Test_ReportFormatSupportGate_StopsValidAfterAdvance(t *testing.T) { @@ -1196,11 +1196,11 @@ func Test_ReportFormatSupportGate_StopsValidAfterAdvance(t *testing.T) { covered := prec covered.SupportByFormat = map[llotypes.ReportFormat]int{llotypes.ReportFormatJSON: 3} - require.True(t, covered.isReportable(cid, 0, 1, protocol.NewOptsCache(), logger.Test(t))) + require.True(t, covered.isReportable(cid, 0, 1, protocol.NewOptsCache(), nil)) uncovered := prec uncovered.SupportByFormat = map[llotypes.ReportFormat]int{llotypes.ReportFormatJSON: 2} - require.False(t, uncovered.isReportable(cid, 0, 1, protocol.NewOptsCache(), logger.Test(t))) + require.False(t, uncovered.isReportable(cid, 0, 1, protocol.NewOptsCache(), nil)) } func Test_Observation_SupportedReportFormats_RoundTrip(t *testing.T) { @@ -1380,7 +1380,7 @@ func Test_StateTransition_CodecSupportAccumulatesAcrossRounds(t *testing.T) { prec, err = decodePrecursor(precBytes) require.NoError(t, err) require.Equal(t, 2*p.F+1, prec.SupportByFormat[llotypes.ReportFormatJSON]) - require.Equal(t, []llotypes.ChannelID{1}, prec.reportableChannels(0, p.F, nil, logger.Test(t))) + require.Equal(t, []llotypes.ChannelID{1}, prec.reportableChannels(0, p.F, nil, nil)) } // One oracle padding its observation with unused report formats must not be @@ -1607,3 +1607,36 @@ func Test_Observation_RetirementCacheErrorsAreNotFatal(t *testing.T) { require.Empty(t, obs.AttestedPredecessorRetirement) require.False(t, obs.ShouldRetire) } + +func Test_UnreportableTally_CollapsesPerChannelWarnings(t *testing.T) { + const channels = 50 + out := precursor{ + LifeCycleStage: protocol.LifeCycleStageProduction, + ObservationTimestampNanoseconds: 1_000, + ChannelDefinitions: llotypes.ChannelDefinitions{}, + ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{}, + StreamAggregates: protocol.StreamAggregates{}, + } + for i := 1; i <= channels; i++ { + cid := llotypes.ChannelID(i) //nolint:gosec // G115 bounded by the loop + out.ChannelDefinitions[cid] = llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatJSON, + Streams: []llotypes.Stream{{StreamID: 1, Aggregator: llotypes.AggregatorMedian}}, + } + out.ValidAfterNanoseconds[cid] = 1 + } + + // No oracle advertises the format, so every channel fails the same check. + tally := &unreportableTally{} + require.Empty(t, out.withSupport(0).reportableChannels(0, 1, protocol.NewOptsCache(), tally)) + + require.Len(t, tally.reasons, 1, "one reason, not one entry per channel") + reason := tally.reasons["too few oracles advertise a report codec for this format"] + require.NotNil(t, reason) + require.Equal(t, channels, reason.count) + require.Len(t, reason.channels, maxUnreportableSamples, "the sample is bounded") + require.Equal(t, []any{"reportFormat", llotypes.ReportFormatJSON, "supporters", 0, "required", 3}, reason.detail) + + // A nil tally is the no-op the predicate-only callers pass. + require.Empty(t, out.withSupport(0).reportableChannels(0, 1, protocol.NewOptsCache(), nil)) +} diff --git a/llo/dev/v31/reports.go b/llo/dev/v31/reports.go index 568c843..f4a9098 100644 --- a/llo/dev/v31/reports.go +++ b/llo/dev/v31/reports.go @@ -63,7 +63,9 @@ func (p *Plugin) Reports(ctx context.Context, seqNr uint64, rawPrecursor ocr3_1t }) } - for _, cid := range out.reportableChannels(p.DefaultMinReportIntervalNanoseconds, p.F, channelOpts, p.Logger) { + unreportable := &unreportableTally{} + defer unreportable.log(p.Logger, "Report", seqNr) + for _, cid := range out.reportableChannels(p.DefaultMinReportIntervalNanoseconds, p.F, channelOpts, unreportable) { cd := out.ChannelDefinitions[cid] if cd.ReportFormat == llotypes.ReportFormatHistoryBackfill { @@ -178,16 +180,73 @@ func (p *Plugin) Reports(ctx context.Context, seqNr uint64, rawPrecursor ocr3_1t return rwis, nil } +// maxUnreportableSamples bounds the channel IDs carried per reason, so one +// summary line stays readable on a DON with hundreds of channels. +const maxUnreportableSamples = 10 + +// unreportableTally aggregates the reasons a round found channels unreportable. +// A nil tally records nothing. flushKV passes one: it runs the same predicate +// over the same precursor to persist reportedLastRound, so letting it log too +// would only report every reason twice per round. +type unreportableTally struct { + reasons map[string]*unreportableReason +} + +// unreportableReason is one reason's aggregate: how many channels hit it, a +// bounded sample of which, and the structured fields of the first occurrence as +// a worked example. +type unreportableReason struct { + count int + channels []llotypes.ChannelID + detail []any +} + +func (t *unreportableTally) note(reason string, channelID llotypes.ChannelID, detail ...any) { + if t == nil { + return + } + if t.reasons == nil { + t.reasons = map[string]*unreportableReason{} + } + r := t.reasons[reason] + if r == nil { + r = &unreportableReason{detail: detail} + t.reasons[reason] = r + } + r.count++ + if len(r.channels) < maxUnreportableSamples { + r.channels = append(r.channels, channelID) + } +} + +// log emits one warning per distinct reason, in a stable order. +func (t *unreportableTally) log(lggr logger.Logger, stage string, seqNr uint64) { + if t == nil || len(t.reasons) == 0 { + return + } + reasons := make([]string, 0, len(t.reasons)) + for reason := range t.reasons { + reasons = append(reasons, reason) + } + sort.Strings(reasons) + for _, reason := range reasons { + r := t.reasons[reason] + kv := []any{"stage", stage, "seqNr", seqNr, "channels", r.count, "sampleChannelIDs", r.channels} + kv = append(kv, r.detail...) + lggr.Warnw("IsReportable=false; "+reason, kv...) + } +} + // formatIsEncodable reports whether enough oracles advertised a report codec // for format that a report encoded with it would actually be certified. // // The tally is read from the precursor, not from p.ReportCodecs, so this stays // a pure function of replicated state. An oracle that cannot encode still // computes reportable=true when enough peers can. -func (o precursor) formatIsEncodable(format llotypes.ReportFormat, f int, channelID llotypes.ChannelID, lggr logger.Logger) bool { +func (o precursor) formatIsEncodable(format llotypes.ReportFormat, f int, channelID llotypes.ChannelID, tally *unreportableTally) bool { if supporters := o.SupportByFormat[format]; supporters < 2*f+1 { - lggr.Warnw("IsReportable=false; too few oracles advertise a report codec for this format", - "channelID", channelID, "reportFormat", format, "supporters", supporters, "required", 2*f+1) + tally.note("too few oracles advertise a report codec for this format", channelID, + "reportFormat", format, "supporters", supporters, "required", 2*f+1) return false } return true @@ -211,10 +270,10 @@ func saturatingAdd(a, b uint64) uint64 { return math.MaxUint64 } -func (o precursor) reportableChannels(minReportInterval uint64, f int, optsCache *protocol.OptsCache, lggr logger.Logger) []llotypes.ChannelID { +func (o precursor) reportableChannels(minReportInterval uint64, f int, optsCache *protocol.OptsCache, tally *unreportableTally) []llotypes.ChannelID { reportable := make([]llotypes.ChannelID, 0, len(o.ChannelDefinitions)) for channelID := range o.ChannelDefinitions { - if o.isReportable(channelID, minReportInterval, f, optsCache, lggr) { + if o.isReportable(channelID, minReportInterval, f, optsCache, tally) { reportable = append(reportable, channelID) } } @@ -222,7 +281,7 @@ func (o precursor) reportableChannels(minReportInterval uint64, f int, optsCache return reportable } -func (o precursor) isReportable(channelID llotypes.ChannelID, minReportInterval uint64, f int, optsCache *protocol.OptsCache, lggr logger.Logger) bool { +func (o precursor) isReportable(channelID llotypes.ChannelID, minReportInterval uint64, f int, optsCache *protocol.OptsCache, tally *unreportableTally) bool { if o.LifeCycleStage == protocol.LifeCycleStageRetired { return false } @@ -238,7 +297,7 @@ func (o precursor) isReportable(channelID llotypes.ChannelID, minReportInterval // Backfill reports are encoded with the target channel's codec, so the // target's format is the one that must be encodable DON-wide. Selection // above already established the target exists. - return o.formatIsEncodable(o.ChannelDefinitions[opts.TargetChannelID].ReportFormat, f, channelID, lggr) + return o.formatIsEncodable(o.ChannelDefinitions[opts.TargetChannelID].ReportFormat, f, channelID, tally) } // When DisableNilStreamValues is set, every stream must have a (non-nil) // aggregate value for the channel to be reportable. @@ -261,7 +320,7 @@ func (o precursor) isReportable(channelID llotypes.ChannelID, minReportInterval // node-dependent. Those failure modes remain outside this predicate. streams, err := protocol.EffectiveStreams(optsCache, cd, channelID) if err != nil { - lggr.Warnw("IsReportable=false; cannot derive effective streams", "channelID", channelID, "err", err) + tally.note("cannot derive effective streams", channelID, "err", err) return false } // Calculated streams are derived state, and unlike observed streams a @@ -274,11 +333,11 @@ func (o precursor) isReportable(channelID llotypes.ChannelID, minReportInterval continue } if o.StreamAggregates[strm.StreamID][strm.Aggregator] == nil { - lggr.Warnw("IsReportable=false; nil calculated stream value", "channelID", channelID, "streamID", strm.StreamID) + tally.note("nil calculated stream value", channelID, "streamID", strm.StreamID) return false } } - if !o.formatIsEncodable(cd.ReportFormat, f, channelID, lggr) { + if !o.formatIsEncodable(cd.ReportFormat, f, channelID, tally) { return false } validAfter, ok := o.ValidAfterNanoseconds[channelID] diff --git a/llo/dev/v31/statetransition.go b/llo/dev/v31/statetransition.go index 49ec823..f36b157 100644 --- a/llo/dev/v31/statetransition.go +++ b/llo/dev/v31/statetransition.go @@ -658,7 +658,8 @@ func (p *Plugin) flushKV( // round can advance validAfter faithfully (see prevReportable). reportable := make(map[llotypes.ChannelID]bool, len(out.ChannelDefinitions)) for id := range out.ChannelDefinitions { - reportable[id] = out.isReportable(id, p.DefaultMinReportIntervalNanoseconds, p.F, prev.opts, p.Logger) + // nill tally, Reports() will handle the logging + reportable[id] = out.isReportable(id, p.DefaultMinReportIntervalNanoseconds, p.F, prev.opts, nil) } // Stream history: write modified windows, delete pairs no live channel From 2b4483c09b13ee953ddb8ab38f854c5bedcd1aad Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Wed, 23 Sep 2026 13:39:49 +0100 Subject: [PATCH 32/40] SPOR-0010 llo/dev/v31: agree on the predecessor signer set per round Get the predecessor signer set from the round's votes and verify the round attested retirement reports against it, without persisting. --- llo/dev/v31/doc.go | 5 - llo/dev/v31/flow_test.go | 104 ++++++------- llo/dev/v31/golden_test.go | 10 -- llo/dev/v31/kv.go | 55 +------ llo/dev/v31/observation.go | 6 +- llo/dev/v31/plugin.go | 30 ++-- llo/dev/v31/statetransition.go | 68 ++++----- .../testdata/golden/kv_predecessor_config.bin | 5 - llo/protocol/plugin_codecs.pb.go | 142 ++++-------------- llo/protocol/plugin_codecs.proto | 25 +-- 10 files changed, 142 insertions(+), 308 deletions(-) delete mode 100644 llo/dev/v31/testdata/golden/kv_predecessor_config.bin diff --git a/llo/dev/v31/doc.go b/llo/dev/v31/doc.go index f6e73bf..c7a1cc0 100644 --- a/llo/dev/v31/doc.go +++ b/llo/dev/v31/doc.go @@ -31,11 +31,6 @@ // timestamped aggregates — and is rewritten every round. // - c/defs holds every channel definition and is rewritten only when the // definitions change; c/seqnr records the sequence number of that write. -// - c/pred holds the predecessor instance's signer set and f, agreed by vote -// while staging and written at most once. Verifying an attested predecessor -// retirement report against it keeps the state transition reading only -// replicated state: the node-local retirement report cache is filled -// asynchronously, so reading it here would fork. // - c/lifecycle holds the lifecycle stage and is written only on change. // // Because c/defs is a pure function of c/seqnr, the plugin keeps the decoded diff --git a/llo/dev/v31/flow_test.go b/llo/dev/v31/flow_test.go index 300259a..ea372d9 100644 --- a/llo/dev/v31/flow_test.go +++ b/llo/dev/v31/flow_test.go @@ -126,14 +126,14 @@ func (mockOnchainConfigCodec) Decode([]byte) (protocol.OnchainConfig, error) { func (mockOnchainConfigCodec) Encode(protocol.OnchainConfig) ([]byte, error) { return nil, nil } // testPredecessorSigners is the signer set a fixture node reads from its local -// retirement report cache and votes into c/pred. +// retirement report cache and votes on. var testPredecessorSigners = [][]byte{{0x01}, {0x02}, {0x03}, {0x04}} type mockPredecessorRetirementReportCache struct { report protocol.RetirementReport err error // noLocalConfig models a node whose config poller has not stored the - // predecessor config yet, so it cannot vote on c/pred. + // predecessor config yet, so it cannot vote on the signer set. noLocalConfig bool // signers overrides testPredecessorSigners; f is the predecessor's f. signers [][]byte @@ -177,7 +177,7 @@ func (m *mockPredecessorRetirementReportCache) VerifyAttestedRetirementReport(_ // promotionObs is what a staging node observes once its predecessor has // retired: the attested retirement report, plus a vote for the predecessor's -// signer set so the DON can agree on c/pred and verify the report against it. +// signer set so the DON can agree on it and verify the report against it. func promotionObs(tsNanoseconds uint64) Observation { return Observation{ UnixTimestampNanoseconds: tsNanoseconds, @@ -584,9 +584,9 @@ func promotionRoundAOs(t *testing.T, voters int) []ocrtypes.AttributedObservatio } // Test_StateTransition_PredecessorConfigAgreedAndUsedSameRound covers -// the signer set needed to verify an attested retirement report is agreed by -// vote into c/pred rather than read from the node-local cache, and a set agreed -// this round is usable this round, so the handover costs no extra round. +// the signer set needed to verify an attested retirement report being agreed by +// vote rather than read from the node-local cache, within the round that +// carries the report, so the handover costs no extra round. func Test_StateTransition_PredecessorConfigAgreedAndUsedSameRound(t *testing.T) { ctx := tests.Context(t) p, kv := stagedPlugin(t, &mockPredecessorRetirementReportCache{ @@ -597,17 +597,13 @@ func Test_StateTransition_PredecessorConfigAgreedAndUsedSameRound(t *testing.T) require.NoError(t, err) require.Equal(t, string(protocol.LifeCycleStageProduction), string(kv.m[string(keyLifecycle)])) - stored, err := readPredecessorConfig(kv) - require.NoError(t, err) - require.NotNil(t, stored, "the agreed signer set must be replicated in c/pred") - require.Equal(t, testPredecessorSigners, stored.signers) } // Test_StateTransition_LaggingPollersDoNotFork is the finding itself: nodes // whose config poller has not stored the predecessor config cannot verify // locally. Their observations must still count in full, and the round must // produce the same state as one where every node was caught up, since f+1 -// voters are enough to agree on c/pred. +// voters are enough to agree on the signer set. func Test_StateTransition_LaggingPollersDoNotFork(t *testing.T) { ctx := tests.Context(t) report := protocol.RetirementReport{ValidAfterNanoseconds: map[llotypes.ChannelID]uint64{1: 500}} @@ -623,7 +619,6 @@ func Test_StateTransition_LaggingPollersDoNotFork(t *testing.T) { require.Equal(t, precursorAll, precursorLagging, "a lagging poller must not change the state transition") require.Equal(t, string(protocol.LifeCycleStageProduction), string(kvLagging.m[string(keyLifecycle)])) - require.Equal(t, kvAll.m[string(keyPredecessorConfig)], kvLagging.m[string(keyPredecessorConfig)]) // The abstaining nodes' observations still counted: the median timestamp is // over all four, not just the two that voted. @@ -633,8 +628,8 @@ func Test_StateTransition_LaggingPollersDoNotFork(t *testing.T) { } // Test_StateTransition_PredecessorConfigNeedsQuorum checks the vote threshold: -// a single voter is not enough to install a signer set, so nothing is written -// and no report is verified. The round itself still completes. +// a single voter is not enough to elect a signer set, so no report is verified +// and the instance stays in staging. The round itself still completes. func Test_StateTransition_PredecessorConfigNeedsQuorum(t *testing.T) { ctx := tests.Context(t) p, kv := stagedPlugin(t, &mockPredecessorRetirementReportCache{ @@ -645,43 +640,47 @@ func Test_StateTransition_PredecessorConfigNeedsQuorum(t *testing.T) { require.NoError(t, err) require.Equal(t, string(protocol.LifeCycleStageStaging), string(kv.m[string(keyLifecycle)])) - stored, err := readPredecessorConfig(kv) - require.NoError(t, err) - require.Nil(t, stored, "one vote is not more than f") out, err := decodePrecursor(precursor) require.NoError(t, err) require.Equal(t, uint64(3000), out.ObservationTimestampNanoseconds) } -// Test_StateTransition_PredecessorConfigIsWriteOnce guards the forgery path: if -// the agreed signer set could be revoted, a coalition that later reaches f+1 -// could install its own signers and attest a handover that never happened. -func Test_StateTransition_PredecessorConfigIsWriteOnce(t *testing.T) { +// Test_StateTransition_PredecessorConfigAgreementIsPerRound guards the forgery +// path: agreement is scoped to the round that uses it and nothing carries over, +// so a coalition of f can never build on an earlier round to install a signer +// set of its own and attest a handover that never happened. +func Test_StateTransition_PredecessorConfigAgreementIsPerRound(t *testing.T) { ctx := tests.Context(t) - prrc := &mockPredecessorRetirementReportCache{report: protocol.RetirementReport{}} - p, kv := stagedPlugin(t, prrc) + p, kv := stagedPlugin(t, &mockPredecessorRetirementReportCache{report: protocol.RetirementReport{}}) - // Round 2 agrees on the signer set but carries no retirement report, so the - // instance stays in staging and keeps voting. - noReport := make([]ocrtypes.AttributedObservation, 0, 4) + // Round 2 reaches quorum on the real signer set but carries no retirement + // report, so the instance stays in staging and the agreement is discarded. + votesOnly := make([]ocrtypes.AttributedObservation, 0, 4) for i := 0; i < 4; i++ { - noReport = append(noReport, ao(i, mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 1000, PredecessorSigners: testPredecessorSigners}))) + votesOnly = append(votesOnly, ao(i, mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 1000, PredecessorSigners: testPredecessorSigners}))) } - _, err := p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, noReport, kv, testBlobs) + _, err := p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, votesOnly, kv, testBlobs) require.NoError(t, err) - agreed := kv.m[string(keyPredecessorConfig)] - require.NotEmpty(t, agreed) + require.Equal(t, string(protocol.LifeCycleStageStaging), string(kv.m[string(keyLifecycle)])) - // Round 3: every node votes for a different signer set. + // Round 3: a single byzantine oracle presents its own signer set and a + // report attested by it. One vote is not more than f, and round 2 left + // nothing behind to lean on, so nothing is elected and nothing is verified. attacker := [][]byte{{0xFF}, {0xFE}} - revote := make([]ocrtypes.AttributedObservation, 0, 4) + forged := make([]ocrtypes.AttributedObservation, 0, 4) for i := 0; i < 4; i++ { - revote = append(revote, ao(i, mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 2000, PredecessorSigners: attacker}))) + obs := Observation{UnixTimestampNanoseconds: 2000} + if i == 0 { + obs.AttestedPredecessorRetirement = []byte("attested") + obs.PredecessorSigners = attacker + } + forged = append(forged, ao(i, mustEncodeObs(t, obs))) } - _, err = p.StateTransition(ctx, 3, ocrtypes.AttributedQuery{}, revote, kv, testBlobs) + _, err = p.StateTransition(ctx, 3, ocrtypes.AttributedQuery{}, forged, kv, testBlobs) require.NoError(t, err) - require.Equal(t, agreed, kv.m[string(keyPredecessorConfig)], "c/pred must be written at most once") + require.Equal(t, string(protocol.LifeCycleStageStaging), string(kv.m[string(keyLifecycle)]), + "a coalition of f must not promote the instance") } // Test_StateTransition_InvalidRetirementReport_KeepsObservation covers the @@ -705,8 +704,9 @@ func Test_StateTransition_InvalidRetirementReport_KeepsObservation(t *testing.T) } // Test_Observation_PredecessorConfigVote covers the observation side: a staging -// node votes its local signer set until c/pred exists, abstains when its poller -// has nothing, and stops voting once the set is replicated. +// node votes its local signer set alongside an attested retirement report, +// abstains when its poller has nothing, and never votes without a report to +// verify. func Test_Observation_PredecessorConfigVote(t *testing.T) { ctx := tests.Context(t) @@ -720,32 +720,32 @@ func Test_Observation_PredecessorConfigVote(t *testing.T) { return p, kv } - t.Run("votes while c/pred is absent", func(t *testing.T) { - p, kv := stagedObserver(t, &mockPredecessorRetirementReportCache{}) + observe := func(t *testing.T, p *Plugin, kv *memKV) Observation { + t.Helper() obsBytes, err := p.Observation(ctx, 2, ocrtypes.AttributedQuery{}, kv, nil) require.NoError(t, err) obs, err := decodeObservation(ctx, obsBytes, testBlobs, nil) require.NoError(t, err) - require.Equal(t, testPredecessorSigners, obs.PredecessorSigners) + return obs + } + + t.Run("votes alongside an attested retirement report", func(t *testing.T) { + p, kv := stagedObserver(t, &mockPredecessorRetirementReportCache{}) + require.Equal(t, testPredecessorSigners, observe(t, p, kv).PredecessorSigners) }) t.Run("abstains when the poller has not caught up", func(t *testing.T) { p, kv := stagedObserver(t, &mockPredecessorRetirementReportCache{noLocalConfig: true}) - obsBytes, err := p.Observation(ctx, 2, ocrtypes.AttributedQuery{}, kv, nil) - require.NoError(t, err, "a lagging poller must not fail Observation") - obs, err := decodeObservation(ctx, obsBytes, testBlobs, nil) - require.NoError(t, err) + obs := observe(t, p, kv) + require.NotEmpty(t, obs.AttestedPredecessorRetirement, "a lagging poller must not drop the report") require.Empty(t, obs.PredecessorSigners) }) - t.Run("stops voting once c/pred is agreed", func(t *testing.T) { - p, kv := stagedObserver(t, &mockPredecessorRetirementReportCache{}) - require.NoError(t, writePredecessorConfig(kv, predecessorConfig{signers: testPredecessorSigners})) - obsBytes, err := p.Observation(ctx, 2, ocrtypes.AttributedQuery{}, kv, nil) - require.NoError(t, err) - obs, err := decodeObservation(ctx, obsBytes, testBlobs, nil) - require.NoError(t, err) - require.Empty(t, obs.PredecessorSigners) + t.Run("does not vote without a report to verify", func(t *testing.T) { + p, kv := stagedObserver(t, &mockPredecessorRetirementReportCache{err: errors.New("no attested retirement report yet")}) + obs := observe(t, p, kv) + require.Empty(t, obs.AttestedPredecessorRetirement) + require.Empty(t, obs.PredecessorSigners, "the signer set is dead weight without a report") }) } diff --git a/llo/dev/v31/golden_test.go b/llo/dev/v31/golden_test.go index d786a54..566f302 100644 --- a/llo/dev/v31/golden_test.go +++ b/llo/dev/v31/golden_test.go @@ -138,10 +138,6 @@ func Test_Golden_KVRecords(t *testing.T) { }, logger.Test(t), )) - require.NoError(t, writePredecessorConfig(kv, predecessorConfig{ - signers: [][]byte{{0xAA, 0xBB}, {0xCC}, {0xDD}, {0xEE}}, - f: 1, - })) require.NoError(t, writeHistoryLayoutVersion(kv)) require.NoError(t, writeHistoryIndex(kv, []histKey{ {streamID: 100, aggregator: llotypes.AggregatorMedian}, @@ -156,7 +152,6 @@ func Test_Golden_KVRecords(t *testing.T) { {"kv_channel_state.bin", keyChannelState}, {"kv_channel_seqnr.bin", keyChannelSeqNr}, {"kv_hot_state.bin", keyHotState}, - {"kv_predecessor_config.bin", keyPredecessorConfig}, {"kv_history_version.bin", keyHistoryVersion}, {"kv_history_index.bin", keyHistoryIndex}, } { @@ -179,10 +174,6 @@ func Test_Golden_KVRecords(t *testing.T) { require.Equal(t, map[llotypes.ChannelID]bool{3: true, 2: true}, s.reportedLastRound) require.Len(t, s.carryForward, 2) - pc, err := readPredecessorConfig(kv) - require.NoError(t, err) - require.Equal(t, &predecessorConfig{signers: [][]byte{{0xAA, 0xBB}, {0xCC}, {0xDD}, {0xEE}}, f: 1}, pc) - version, err := readHistoryLayoutVersion(kv) require.NoError(t, err) require.Equal(t, historyLayoutVersion, version) @@ -249,7 +240,6 @@ func Test_Golden_KVKeys(t *testing.T) { {"c/defs", string(keyChannelState)}, {"c/seqnr", string(keyChannelSeqNr)}, {"r/agg", string(keyHotState)}, - {"c/pred", string(keyPredecessorConfig)}, {"hidx", string(keyHistoryIndex)}, {"hv", string(keyHistoryVersion)}, } { diff --git a/llo/dev/v31/kv.go b/llo/dev/v31/kv.go index 006cbd0..35cab31 100644 --- a/llo/dev/v31/kv.go +++ b/llo/dev/v31/kv.go @@ -24,9 +24,6 @@ import ( // c/defs -> LLOChannelStateProto: every live channel definition // (written only when the definitions change) // c/seqnr -> uint64 BE seqNr of the last c/defs write -// c/pred -> LLOPredecessorConfigProto: the predecessor's signer set and -// f, agreed by vote while staging (written at most once, and -// only by an instance that has a predecessor) // c/codecs -> LLOCodecSupportProto: the report formats each oracle last // advertised a codec for (written only when some oracle's // advertised set changes) @@ -50,12 +47,11 @@ import ( // a read cost of depth/chunkSize point reads instead of one. See // protocol.RingWindow. var ( - keyLifecycle = []byte("c/lifecycle") - keyChannelState = []byte("c/defs") - keyChannelSeqNr = []byte("c/seqnr") - keyPredecessorConfig = []byte("c/pred") - keyCodecSupport = []byte("c/codecs") - keyHotState = []byte("r/agg") + keyLifecycle = []byte("c/lifecycle") + keyChannelState = []byte("c/defs") + keyChannelSeqNr = []byte("c/seqnr") + keyCodecSupport = []byte("c/codecs") + keyHotState = []byte("r/agg") keyHistoryIndex = []byte("hidx") keyHistoryVersion = []byte("hv") @@ -285,47 +281,6 @@ func readChannelState(r ocr3_1types.KeyValueStateReader) (llotypes.ChannelDefini return defs, nil } -// predecessorConfig is the decoded c/pred record: the signer set and f of the -// predecessor instance, which is what verifying an attested predecessor -// retirement report needs. -type predecessorConfig struct { - signers [][]byte - f uint8 -} - -// readPredecessorConfig reads and decodes the c/pred record, returning nil when -// it has not been agreed yet. -// -// Only a staging instance that has a predecessor ever reads or writes this key, -// so every other instance pays nothing for it. -func readPredecessorConfig(r ocr3_1types.KeyValueStateReader) (*predecessorConfig, error) { - b, err := r.Read(keyPredecessorConfig) - if err != nil { - return nil, fmt.Errorf("read predecessor config: %w", err) - } - if len(b) == 0 { - return nil, nil - } - pb := &protocol.LLOPredecessorConfigProto{} - if err := proto.Unmarshal(b, pb); err != nil { - return nil, fmt.Errorf("unmarshal predecessor config: %w", err) - } - if pb.F > 255 { - return nil, fmt.Errorf("predecessor config has f out of range: %d", pb.F) - } - return &predecessorConfig{signers: pb.Signers, f: uint8(pb.F)}, nil -} - -// writePredecessorConfig persists the agreed c/pred record. Signer order is -// preserved: a signature names its signer by index into the set. -func writePredecessorConfig(w ocr3_1types.KeyValueStateReadWriter, pc predecessorConfig) error { - b, err := deterministicMarshal.Marshal(&protocol.LLOPredecessorConfigProto{Signers: pc.signers, F: uint32(pc.f)}) - if err != nil { - return fmt.Errorf("marshal predecessor config: %w", err) - } - return w.Write(keyPredecessorConfig, b) -} - // readHotState reads and decodes the r/agg record into s. func readHotState(r ocr3_1types.KeyValueStateReader, s *kvState) error { b, err := r.Read(keyHotState) diff --git a/llo/dev/v31/observation.go b/llo/dev/v31/observation.go index 177c67e..c0f4e9b 100644 --- a/llo/dev/v31/observation.go +++ b/llo/dev/v31/observation.go @@ -32,9 +32,9 @@ type Observation struct { SupportedReportFormats []llotypes.ReportFormat // PredecessorSigners and PredecessorF are the predecessor instance's signer // set and f, read from the node-local retirement report cache. A staging - // instance carries them until c/pred is agreed, which turns them into a - // replicated fact the state transition can verify retirement reports - // against; see readPredecessorConfig. + // instance carries them alongside an attested retirement report, so the + // state transition can agree on the set by vote and verify the report + // against it; see resolvePredecessorRetirement. // // Signer order is significant: a signature names its signer by index. PredecessorSigners [][]byte diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index e6d62be..873cd11 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -145,7 +145,9 @@ func (p *Plugin) Observation(_ context.Context, seqNr uint64, _ ocrtypes.Attribu obs.AttestedPredecessorRetirement = nil p.Logger.Errorw("Failed to fetch attested retirement report from cache, omitting it from this observation", "stage", "Observation", "seqNr", seqNr, "err", err) } - p.voteOnPredecessorConfig(&obs, kvReader, seqNr) + if len(obs.AttestedPredecessorRetirement) != 0 { + p.voteOnPredecessorConfig(&obs, seqNr) + } } obs.ShouldRetire, err = p.ShouldRetireCache.ShouldRetire(p.ConfigDigest) @@ -254,26 +256,19 @@ func sortedChannelIDSet(set map[llotypes.ChannelID]struct{}) []llotypes.ChannelI // voteOnPredecessorConfig populates obs.PredecessorSigners / obs.PredecessorF // with the predecessor's signer set from the node-local retirement report -// cache, so the DON can agree on it and store it in c/pred. +// cache, so the DON can agree on it for the round. // // Verifying an attested predecessor retirement report needs that signer set. // Reading it from the local cache inside the state transition would fork the // state, because the config poller fills the cache asynchronously and a lagging // node reaches a different verdict from one that is caught up. Voting on it // here moves the node-local read into the observation, where oracles are -// allowed to differ, and leaves the state transition reading only replicated -// state. -func (p *Plugin) voteOnPredecessorConfig(obs *Observation, kvReader ocr3_1types.KeyValueStateReader, seqNr uint64) { - agreed, err := readPredecessorConfig(kvReader) - if err != nil { - p.Logger.Errorw("Failed to read agreed predecessor config, not voting on it this round", "stage", "Observation", "seqNr", seqNr, "err", err) - return - } - if agreed != nil { - // Already replicated, so the vote would be dead weight on every - // observation for the rest of the staging period. - return - } +// allowed to differ, and leaves the state transition reading only the round's +// replicated observations. +// +// Only called when this observation carries an attested retirement report, so +// the vote rides only the rounds where a promotion is possible. +func (p *Plugin) voteOnPredecessorConfig(obs *Observation, seqNr uint64) { signers, f, exists := p.PredecessorRetirementReportCache.PredecessorConfig(*p.PredecessorConfigDigest) if !exists { p.Logger.Warnw("Predecessor config not in the local cache yet, not voting on it this round", "stage", "Observation", "seqNr", seqNr, "predecessorConfigDigest", *p.PredecessorConfigDigest) @@ -396,6 +391,11 @@ func (p *Plugin) ValidateObservation(ctx context.Context, seqNr uint64, _ ocrtyp if p.PredecessorConfigDigest == nil && len(observation.PredecessorSigners) != 0 { return errors.New("PredecessorSigners is not empty even though this instance has no predecessor") } + // The signer set is only ever needed to verify a report carried alongside + // it, so a set without one is dead weight on the wire. + if len(observation.PredecessorSigners) != 0 && len(observation.AttestedPredecessorRetirement) == 0 { + return errors.New("PredecessorSigners is not empty even though this observation carries no AttestedPredecessorRetirement") + } if len(observation.UpdateChannelDefinitions) > protocol.MaxObservationUpdateChannelDefinitionsLength { return fmt.Errorf("UpdateChannelDefinitions is too long: %v vs %v", len(observation.UpdateChannelDefinitions), protocol.MaxObservationUpdateChannelDefinitionsLength) } diff --git a/llo/dev/v31/statetransition.go b/llo/dev/v31/statetransition.go index f36b157..5a952d3 100644 --- a/llo/dev/v31/statetransition.go +++ b/llo/dev/v31/statetransition.go @@ -82,14 +82,10 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A } // Verifying an attested predecessor retirement report needs the - // predecessor's signer set, which is node-local until the DON agrees on it. - // Agree first, then verify against the agreed set, so every oracle reaches - // the same verdict. A set agreed this round is usable this round, so a - // handover normally costs no extra round. - validPredecessorRetirementReport, err := p.resolvePredecessorRetirement(kvRW, seqNr, prev.lifeCycleStage, tally) - if err != nil { - return nil, err - } + // predecessor's signer set, which is node-local. Agree on it from this + // round's votes, then verify against the agreed set, so every oracle + // reaches the same verdict. + validPredecessorRetirementReport := p.resolvePredecessorRetirement(seqNr, prev.lifeCycleStage, tally) // Codec coverage is cumulative across rounds: merge this round's // advertisements over the persisted ones before counting supporters. @@ -721,6 +717,14 @@ func medianTimestamp(timestampsNanoseconds []uint64) uint64 { // makeChannelHash delegates to the shared implementation so that v3.0 running // protocol version 2 and v3.1 cannot drift apart on channel identity. +// predecessorConfig is a candidate predecessor instance's signer set and f, +// which is what verifying an attested predecessor retirement report needs. It +// is agreed by vote within a round and never persisted. +type predecessorConfig struct { + signers [][]byte + f uint8 +} + // hashPredecessorConfig identifies a candidate predecessor config so votes for // the same one can be tallied. Signer order is part of the identity: a // signature names its signer by index, so two sets differing only in order are @@ -740,47 +744,35 @@ func hashPredecessorConfig(pc predecessorConfig) [32]byte { return out } -// resolvePredecessorRetirement agrees on the predecessor's signer set, then -// verifies this round's attested retirement reports against it. +// resolvePredecessorRetirement agrees on the predecessor's signer set from +// this round's votes, then verifies this round's attested retirement reports +// against it. // -// The signer set is written to c/pred once more than f oracles vote for the -// same one, so at least one honest oracle vouches for it. It is written at most -// once: if it could be revoted, a coalition that later reaches f+1 could swap -// in a signer set of its own and forge a retirement report, promoting this -// instance on a handover that never happened. +// Agreement holds for this round only and is never stored. More than f votes +// for the same set means at least one honest oracle vouches for it, and that +// argument is per round: a coalition of f can never elect a set of its own, in +// this round or any later one, and nothing accumulates between rounds for it +// to build on. // -// Everything here reads replicated state only, so every oracle reaches the same -// verdict. A report that fails verification is ignored, not fatal: the bytes -// are the same everywhere, so ignoring them is deterministic too. +// Everything here reads the round's observations only, so every oracle reaches +// the same verdict. A report that fails verification is ignored, not fatal: +// the bytes are the same everywhere, so ignoring them is deterministic too. func (p *Plugin) resolvePredecessorRetirement( - kvRW ocr3_1types.KeyValueStateReadWriter, seqNr uint64, stage llotypes.LifeCycleStage, tally observationTally, -) (*protocol.RetirementReport, error) { +) *protocol.RetirementReport { // Only a staging instance with a predecessor has a handover to complete. if p.PredecessorConfigDigest == nil || stage != protocol.LifeCycleStageStaging { - return nil, nil + return nil } - agreed, err := readPredecessorConfig(kvRW) - if err != nil { - return nil, err - } - if agreed == nil { - if elected := electPredecessorConfig(tally.predConfigsByHash, tally.predConfigVotesByHash, p.F); elected != nil { - if err := writePredecessorConfig(kvRW, *elected); err != nil { - return nil, err - } - p.Logger.Infow("Agreed on predecessor config", "seqNr", seqNr, "signers", len(elected.signers), "f", elected.f, "predecessorConfigDigest", *p.PredecessorConfigDigest) - agreed = elected - } - } + agreed := electPredecessorConfig(tally.predConfigsByHash, tally.predConfigVotesByHash, p.F) if agreed == nil { if len(tally.attestedRetirements) > 0 { - p.Logger.Warnw("Ignoring attested predecessor retirement reports: the predecessor config is not agreed yet", "seqNr", seqNr, "reports", len(tally.attestedRetirements)) + p.Logger.Warnw("Ignoring attested predecessor retirement reports: the predecessor config is not agreed this round", "seqNr", seqNr, "reports", len(tally.attestedRetirements)) } - return nil, nil + return nil } for _, attested := range tally.attestedRetirements { @@ -789,9 +781,9 @@ func (p *Plugin) resolvePredecessorRetirement( p.Logger.Warnw("Ignoring invalid attested predecessor retirement", "seqNr", seqNr, "error", verr, "predecessorConfigDigest", *p.PredecessorConfigDigest) continue } - return &retirementReport, nil + return &retirementReport } - return nil, nil + return nil } // electPredecessorConfig returns the candidate with more than f votes, or nil. diff --git a/llo/dev/v31/testdata/golden/kv_predecessor_config.bin b/llo/dev/v31/testdata/golden/kv_predecessor_config.bin deleted file mode 100644 index ec9517b..0000000 --- a/llo/dev/v31/testdata/golden/kv_predecessor_config.bin +++ /dev/null @@ -1,5 +0,0 @@ - -ª» -Ì -Ý -î \ No newline at end of file diff --git a/llo/protocol/plugin_codecs.pb.go b/llo/protocol/plugin_codecs.pb.go index 04caf82..28bf1b3 100644 --- a/llo/protocol/plugin_codecs.pb.go +++ b/llo/protocol/plugin_codecs.pb.go @@ -96,10 +96,10 @@ type LLOObservationProto struct { // fact instead. SupportedReportFormats []uint32 `protobuf:"varint,8,rep,packed,name=supportedReportFormats,proto3" json:"supportedReportFormats,omitempty"` // The predecessor instance's signer set and f, read from the node-local - // retirement report cache. Carried only by a v31 staging instance that has - // not yet agreed on c/pred, so the state transition can verify attested - // retirement reports against replicated state instead of node-local state. - // Order is significant: a signature names its signer by index into it. + // retirement report cache. Carried by a v31 staging instance alongside an + // attested retirement report, so the state transition can agree on the set + // by vote and verify the report against it instead of reading node-local + // state. Order is significant: a signature names its signer by index. PredecessorSigners [][]byte `protobuf:"bytes,9,rep,name=predecessorSigners,proto3" json:"predecessorSigners,omitempty"` PredecessorF uint32 `protobuf:"varint,10,opt,name=predecessorF,proto3" json:"predecessorF,omitempty"` unknownFields protoimpl.UnknownFields @@ -1191,70 +1191,6 @@ func (x *LLOChannelStateProto) GetChannelDefinitions() []*LLOChannelIDAndDefinit return nil } -// LLOPredecessorConfigProto is the v31 KeyValueState record holding the -// predecessor instance's signer set and f under a single key (c/pred), agreed -// by vote while staging and written at most once. -// -// It exists so that verifying an attested predecessor retirement report reads -// only replicated state. The same data is available node-locally from the -// retirement report cache, but that cache is filled asynchronously by the -// config poller, so oracles reach different verdicts from it and the state -// transition would fork. -// -// NOTE: must serialize deterministically. signers MUST keep the order the -// predecessor's config gives them, since a signature names its signer by index. -type LLOPredecessorConfigProto struct { - state protoimpl.MessageState `protogen:"open.v1"` - Signers [][]byte `protobuf:"bytes,1,rep,name=signers,proto3" json:"signers,omitempty"` - F uint32 `protobuf:"varint,2,opt,name=f,proto3" json:"f,omitempty"` - unknownFields protoimpl.UnknownFields - sizeCache protoimpl.SizeCache -} - -func (x *LLOPredecessorConfigProto) Reset() { - *x = LLOPredecessorConfigProto{} - mi := &file_plugin_codecs_proto_msgTypes[17] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) -} - -func (x *LLOPredecessorConfigProto) String() string { - return protoimpl.X.MessageStringOf(x) -} - -func (*LLOPredecessorConfigProto) ProtoMessage() {} - -func (x *LLOPredecessorConfigProto) ProtoReflect() protoreflect.Message { - mi := &file_plugin_codecs_proto_msgTypes[17] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { - ms.StoreMessageInfo(mi) - } - return ms - } - return mi.MessageOf(x) -} - -// Deprecated: Use LLOPredecessorConfigProto.ProtoReflect.Descriptor instead. -func (*LLOPredecessorConfigProto) Descriptor() ([]byte, []int) { - return file_plugin_codecs_proto_rawDescGZIP(), []int{17} -} - -func (x *LLOPredecessorConfigProto) GetSigners() [][]byte { - if x != nil { - return x.Signers - } - return nil -} - -func (x *LLOPredecessorConfigProto) GetF() uint32 { - if x != nil { - return x.F - } - return 0 -} - // LLOHotStateProto is the v31 KeyValueState record holding the per-round // ("hot") state under a single key (r/agg): the state that changes on // essentially every round. @@ -1279,7 +1215,7 @@ type LLOHotStateProto struct { func (x *LLOHotStateProto) Reset() { *x = LLOHotStateProto{} - mi := &file_plugin_codecs_proto_msgTypes[18] + mi := &file_plugin_codecs_proto_msgTypes[17] ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) ms.StoreMessageInfo(mi) } @@ -1291,7 +1227,7 @@ func (x *LLOHotStateProto) String() string { func (*LLOHotStateProto) ProtoMessage() {} func (x *LLOHotStateProto) ProtoReflect() protoreflect.Message { - mi := &file_plugin_codecs_proto_msgTypes[18] + mi := &file_plugin_codecs_proto_msgTypes[17] if x != nil { ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) if ms.LoadMessageInfo() == nil { @@ -1304,7 +1240,7 @@ func (x *LLOHotStateProto) ProtoReflect() protoreflect.Message { // Deprecated: Use LLOHotStateProto.ProtoReflect.Descriptor instead. func (*LLOHotStateProto) Descriptor() ([]byte, []int) { - return file_plugin_codecs_proto_rawDescGZIP(), []int{18} + return file_plugin_codecs_proto_rawDescGZIP(), []int{17} } func (x *LLOHotStateProto) GetObservationTimestampNanoseconds() uint64 { @@ -1361,7 +1297,7 @@ type LLOPrecursorProto struct { func (x *LLOPrecursorProto) Reset() { *x = LLOPrecursorProto{} - mi := &file_plugin_codecs_proto_msgTypes[19] + mi := &file_plugin_codecs_proto_msgTypes[18] ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) ms.StoreMessageInfo(mi) } @@ -1373,7 +1309,7 @@ func (x *LLOPrecursorProto) String() string { func (*LLOPrecursorProto) ProtoMessage() {} func (x *LLOPrecursorProto) ProtoReflect() protoreflect.Message { - mi := &file_plugin_codecs_proto_msgTypes[19] + mi := &file_plugin_codecs_proto_msgTypes[18] if x != nil { ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) if ms.LoadMessageInfo() == nil { @@ -1386,7 +1322,7 @@ func (x *LLOPrecursorProto) ProtoReflect() protoreflect.Message { // Deprecated: Use LLOPrecursorProto.ProtoReflect.Descriptor instead. func (*LLOPrecursorProto) Descriptor() ([]byte, []int) { - return file_plugin_codecs_proto_rawDescGZIP(), []int{19} + return file_plugin_codecs_proto_rawDescGZIP(), []int{18} } func (x *LLOPrecursorProto) GetLifeCycleStage() string { @@ -1450,7 +1386,7 @@ type LLOReportFormatSupportProto struct { func (x *LLOReportFormatSupportProto) Reset() { *x = LLOReportFormatSupportProto{} - mi := &file_plugin_codecs_proto_msgTypes[20] + mi := &file_plugin_codecs_proto_msgTypes[19] ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) ms.StoreMessageInfo(mi) } @@ -1462,7 +1398,7 @@ func (x *LLOReportFormatSupportProto) String() string { func (*LLOReportFormatSupportProto) ProtoMessage() {} func (x *LLOReportFormatSupportProto) ProtoReflect() protoreflect.Message { - mi := &file_plugin_codecs_proto_msgTypes[20] + mi := &file_plugin_codecs_proto_msgTypes[19] if x != nil { ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) if ms.LoadMessageInfo() == nil { @@ -1475,7 +1411,7 @@ func (x *LLOReportFormatSupportProto) ProtoReflect() protoreflect.Message { // Deprecated: Use LLOReportFormatSupportProto.ProtoReflect.Descriptor instead. func (*LLOReportFormatSupportProto) Descriptor() ([]byte, []int) { - return file_plugin_codecs_proto_rawDescGZIP(), []int{20} + return file_plugin_codecs_proto_rawDescGZIP(), []int{19} } func (x *LLOReportFormatSupportProto) GetReportFormat() uint32 { @@ -1494,14 +1430,6 @@ func (x *LLOReportFormatSupportProto) GetOracleCount() uint32 { // LLOCodecSupportProto is the v31 KeyValueState record (c/codecs) holding the // report formats each oracle last advertised a codec for. -// -// Kept per oracle rather than as a per-round count because the round's -// observation quorum is only 2f+1: a count taken from a single round can never -// exceed that, so requiring 2f+1 supporters would demand unanimity and f -// oracles omitting their advertisement would make every format unreportable. -// Remembering each oracle's last advertisement lets the supporter count reach -// n over successive rounds, and an oracle can still only speak for itself. -// // NOTE: must serialize deterministically. oracles MUST be sorted ascending by // oracleID, and each oracle's reportFormats MUST be sorted ascending. type LLOCodecSupportProto struct { @@ -1513,7 +1441,7 @@ type LLOCodecSupportProto struct { func (x *LLOCodecSupportProto) Reset() { *x = LLOCodecSupportProto{} - mi := &file_plugin_codecs_proto_msgTypes[21] + mi := &file_plugin_codecs_proto_msgTypes[20] ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) ms.StoreMessageInfo(mi) } @@ -1525,7 +1453,7 @@ func (x *LLOCodecSupportProto) String() string { func (*LLOCodecSupportProto) ProtoMessage() {} func (x *LLOCodecSupportProto) ProtoReflect() protoreflect.Message { - mi := &file_plugin_codecs_proto_msgTypes[21] + mi := &file_plugin_codecs_proto_msgTypes[20] if x != nil { ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) if ms.LoadMessageInfo() == nil { @@ -1538,7 +1466,7 @@ func (x *LLOCodecSupportProto) ProtoReflect() protoreflect.Message { // Deprecated: Use LLOCodecSupportProto.ProtoReflect.Descriptor instead. func (*LLOCodecSupportProto) Descriptor() ([]byte, []int) { - return file_plugin_codecs_proto_rawDescGZIP(), []int{21} + return file_plugin_codecs_proto_rawDescGZIP(), []int{20} } func (x *LLOCodecSupportProto) GetOracles() []*LLOOracleCodecSupportProto { @@ -1560,7 +1488,7 @@ type LLOOracleCodecSupportProto struct { func (x *LLOOracleCodecSupportProto) Reset() { *x = LLOOracleCodecSupportProto{} - mi := &file_plugin_codecs_proto_msgTypes[22] + mi := &file_plugin_codecs_proto_msgTypes[21] ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) ms.StoreMessageInfo(mi) } @@ -1572,7 +1500,7 @@ func (x *LLOOracleCodecSupportProto) String() string { func (*LLOOracleCodecSupportProto) ProtoMessage() {} func (x *LLOOracleCodecSupportProto) ProtoReflect() protoreflect.Message { - mi := &file_plugin_codecs_proto_msgTypes[22] + mi := &file_plugin_codecs_proto_msgTypes[21] if x != nil { ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) if ms.LoadMessageInfo() == nil { @@ -1585,7 +1513,7 @@ func (x *LLOOracleCodecSupportProto) ProtoReflect() protoreflect.Message { // Deprecated: Use LLOOracleCodecSupportProto.ProtoReflect.Descriptor instead. func (*LLOOracleCodecSupportProto) Descriptor() ([]byte, []int) { - return file_plugin_codecs_proto_rawDescGZIP(), []int{22} + return file_plugin_codecs_proto_rawDescGZIP(), []int{21} } func (x *LLOOracleCodecSupportProto) GetOracleID() uint32 { @@ -1694,10 +1622,7 @@ const file_plugin_codecs_proto_rawDesc = "" + "aggregator\x18\x03 \x01(\rR\n" + "aggregator\"j\n" + "\x14LLOChannelStateProto\x12R\n" + - "\x12channelDefinitions\x18\x01 \x03(\v2\".v1.LLOChannelIDAndDefinitionProtoR\x12channelDefinitions\"C\n" + - "\x19LLOPredecessorConfigProto\x12\x18\n" + - "\asigners\x18\x01 \x03(\fR\asigners\x12\f\n" + - "\x01f\x18\x02 \x01(\rR\x01f\"\xb9\x02\n" + + "\x12channelDefinitions\x18\x01 \x03(\v2\".v1.LLOChannelIDAndDefinitionProtoR\x12channelDefinitions\"\xb9\x02\n" + "\x10LLOHotStateProto\x12H\n" + "\x1fobservationTimestampNanoseconds\x18\x01 \x01(\x04R\x1fobservationTimestampNanoseconds\x12c\n" + "\x15validAfterNanoseconds\x18\x02 \x03(\v2-.v1.LLOChannelIDAndValidAfterNanosecondsProtoR\x15validAfterNanoseconds\x122\n" + @@ -1734,7 +1659,7 @@ func file_plugin_codecs_proto_rawDescGZIP() []byte { } var file_plugin_codecs_proto_enumTypes = make([]protoimpl.EnumInfo, 1) -var file_plugin_codecs_proto_msgTypes = make([]protoimpl.MessageInfo, 25) +var file_plugin_codecs_proto_msgTypes = make([]protoimpl.MessageInfo, 24) var file_plugin_codecs_proto_goTypes = []any{ (LLOStreamValue_Type)(0), // 0: v1.LLOStreamValue.Type (*LLOObservationProto)(nil), // 1: v1.LLOObservationProto @@ -1754,18 +1679,17 @@ var file_plugin_codecs_proto_goTypes = []any{ (*LLOChannelIDAndValidAfterNanosecondsProto)(nil), // 15: v1.LLOChannelIDAndValidAfterNanosecondsProto (*LLOStreamAggregate)(nil), // 16: v1.LLOStreamAggregate (*LLOChannelStateProto)(nil), // 17: v1.LLOChannelStateProto - (*LLOPredecessorConfigProto)(nil), // 18: v1.LLOPredecessorConfigProto - (*LLOHotStateProto)(nil), // 19: v1.LLOHotStateProto - (*LLOPrecursorProto)(nil), // 20: v1.LLOPrecursorProto - (*LLOReportFormatSupportProto)(nil), // 21: v1.LLOReportFormatSupportProto - (*LLOCodecSupportProto)(nil), // 22: v1.LLOCodecSupportProto - (*LLOOracleCodecSupportProto)(nil), // 23: v1.LLOOracleCodecSupportProto - nil, // 24: v1.LLOObservationProto.UpdateChannelDefinitionsEntry - nil, // 25: v1.LLOObservationProto.StreamValuesEntry + (*LLOHotStateProto)(nil), // 18: v1.LLOHotStateProto + (*LLOPrecursorProto)(nil), // 19: v1.LLOPrecursorProto + (*LLOReportFormatSupportProto)(nil), // 20: v1.LLOReportFormatSupportProto + (*LLOCodecSupportProto)(nil), // 21: v1.LLOCodecSupportProto + (*LLOOracleCodecSupportProto)(nil), // 22: v1.LLOOracleCodecSupportProto + nil, // 23: v1.LLOObservationProto.UpdateChannelDefinitionsEntry + nil, // 24: v1.LLOObservationProto.StreamValuesEntry } var file_plugin_codecs_proto_depIdxs = []int32{ - 24, // 0: v1.LLOObservationProto.updateChannelDefinitions:type_name -> v1.LLOObservationProto.UpdateChannelDefinitionsEntry - 25, // 1: v1.LLOObservationProto.streamValues:type_name -> v1.LLOObservationProto.StreamValuesEntry + 23, // 0: v1.LLOObservationProto.updateChannelDefinitions:type_name -> v1.LLOObservationProto.UpdateChannelDefinitionsEntry + 24, // 1: v1.LLOObservationProto.streamValues:type_name -> v1.LLOObservationProto.StreamValuesEntry 0, // 2: v1.LLOStreamValue.type:type_name -> v1.LLOStreamValue.Type 2, // 3: v1.LLOTimestampedStreamValue.streamValue:type_name -> v1.LLOStreamValue 2, // 4: v1.LLOStreamHistoryRecord.value:type_name -> v1.LLOStreamValue @@ -1785,8 +1709,8 @@ var file_plugin_codecs_proto_depIdxs = []int32{ 13, // 18: v1.LLOPrecursorProto.channelDefinitions:type_name -> v1.LLOChannelIDAndDefinitionProto 15, // 19: v1.LLOPrecursorProto.validAfterNanoseconds:type_name -> v1.LLOChannelIDAndValidAfterNanosecondsProto 16, // 20: v1.LLOPrecursorProto.streamAggregates:type_name -> v1.LLOStreamAggregate - 21, // 21: v1.LLOPrecursorProto.supportByFormat:type_name -> v1.LLOReportFormatSupportProto - 23, // 22: v1.LLOCodecSupportProto.oracles:type_name -> v1.LLOOracleCodecSupportProto + 20, // 21: v1.LLOPrecursorProto.supportByFormat:type_name -> v1.LLOReportFormatSupportProto + 22, // 22: v1.LLOCodecSupportProto.oracles:type_name -> v1.LLOOracleCodecSupportProto 8, // 23: v1.LLOObservationProto.UpdateChannelDefinitionsEntry.value:type_name -> v1.LLOChannelDefinitionProto 2, // 24: v1.LLOObservationProto.StreamValuesEntry.value:type_name -> v1.LLOStreamValue 25, // [25:25] is the sub-list for method output_type @@ -1807,7 +1731,7 @@ func file_plugin_codecs_proto_init() { GoPackagePath: reflect.TypeOf(x{}).PkgPath(), RawDescriptor: unsafe.Slice(unsafe.StringData(file_plugin_codecs_proto_rawDesc), len(file_plugin_codecs_proto_rawDesc)), NumEnums: 1, - NumMessages: 25, + NumMessages: 24, NumExtensions: 0, NumServices: 0, }, diff --git a/llo/protocol/plugin_codecs.proto b/llo/protocol/plugin_codecs.proto index 14cda94..0d0919c 100644 --- a/llo/protocol/plugin_codecs.proto +++ b/llo/protocol/plugin_codecs.proto @@ -34,10 +34,10 @@ message LLOObservationProto { // fact instead. repeated uint32 supportedReportFormats = 8; // The predecessor instance's signer set and f, read from the node-local - // retirement report cache. Carried only by a v31 staging instance that has - // not yet agreed on c/pred, so the state transition can verify attested - // retirement reports against replicated state instead of node-local state. - // Order is significant: a signature names its signer by index into it. + // retirement report cache. Carried by a v31 staging instance alongside an + // attested retirement report, so the state transition can agree on the set + // by vote and verify the report against it instead of reading node-local + // state. Order is significant: a signature names its signer by index. repeated bytes predecessorSigners = 9; uint32 predecessorF = 10; } @@ -188,23 +188,6 @@ message LLOChannelStateProto { repeated LLOChannelIDAndDefinitionProto channelDefinitions = 1; } -// LLOPredecessorConfigProto is the v31 KeyValueState record holding the -// predecessor instance's signer set and f under a single key (c/pred), agreed -// by vote while staging and written at most once. -// -// It exists so that verifying an attested predecessor retirement report reads -// only replicated state. The same data is available node-locally from the -// retirement report cache, but that cache is filled asynchronously by the -// config poller, so oracles reach different verdicts from it and the state -// transition would fork. -// -// NOTE: must serialize deterministically. signers MUST keep the order the -// predecessor's config gives them, since a signature names its signer by index. -message LLOPredecessorConfigProto { - repeated bytes signers = 1; - uint32 f = 2; -} - // LLOHotStateProto is the v31 KeyValueState record holding the per-round // ("hot") state under a single key (r/agg): the state that changes on // essentially every round. From 45f59ee846af111007fc41ceb5b25f763805418c Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Wed, 23 Sep 2026 15:05:15 +0100 Subject: [PATCH 33/40] llo/dev/v31: gate the backfill watermark on prevReportable --- llo/dev/v31/plugin_test.go | 134 +++++++++++++++++++++++++++++++++ llo/dev/v31/statetransition.go | 12 +-- 2 files changed, 140 insertions(+), 6 deletions(-) diff --git a/llo/dev/v31/plugin_test.go b/llo/dev/v31/plugin_test.go index 5c72f45..b82e564 100644 --- a/llo/dev/v31/plugin_test.go +++ b/llo/dev/v31/plugin_test.go @@ -1640,3 +1640,137 @@ func Test_UnreportableTally_CollapsesPerChannelWarnings(t *testing.T) { // A nil tally is the no-op the predicate-only callers pass. require.Empty(t, out.withSupport(0).reportableChannels(0, 1, protocol.NewOptsCache(), nil)) } + +// Test_StateTransition_BackfillValidAfterRequiresPreviousReport covers the +// backfill watermark advancing only over a row the previous round actually +// reported. +func Test_StateTransition_BackfillValidAfterRequiresPreviousReport(t *testing.T) { + const ( + targetCID = llotypes.ChannelID(10) + backfillCID = llotypes.ChannelID(20) + fiveSec = uint64(5_000_000_000) + tenSec = uint64(10_000_000_000) + ) + targetCD := llotypes.ChannelDefinition{ReportFormat: llotypes.ReportFormatJSON, Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}} + backfillCD := llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatHistoryBackfill, + Opts: []byte(`{"targetChannelId":10,"observations":{"5":{"100":"1.5"},"8":{"100":"2.5"}}}`), + } + + // round builds four identical observations at ts. + round := func(t *testing.T, ts uint64, shape func(*Observation)) []ocrtypes.AttributedObservation { + t.Helper() + aos := make([]ocrtypes.AttributedObservation, 0, 4) + for i := 0; i < 4; i++ { + obs := Observation{UnixTimestampNanoseconds: ts} + if shape != nil { + shape(&obs) + } + aos = append(aos, ao(i, mustEncodeObs(t, obs))) + } + return aos + } + + for _, tc := range []struct { + name string + // defs is what the verdict round starts from; shape is how its + // observations differ from a healthy round. + defs llotypes.ChannelDefinitions + shape func(*Observation) + want uint64 + }{ + { + // Guard 1: a retired instance emits nothing, and retirement is + // terminal, so an unguarded advance drains the whole backfill. + name: "retired instance", + defs: llotypes.ChannelDefinitions{targetCID: targetCD, backfillCID: backfillCD}, + shape: func(obs *Observation) { obs.ShouldRetire = true }, + want: 0, + }, + { + // Guard 2: selection checks that the channel exists but not that + // it is live, so a tombstone is invisible to it. + name: "tombstoned backfill channel", + defs: func() llotypes.ChannelDefinitions { + tombstoned := backfillCD + tombstoned.Tombstone = true + return llotypes.ChannelDefinitions{targetCID: targetCD, backfillCID: tombstoned} + }(), + want: 0, + }, + { + // The support gate: no oracle advertises a codec for the target's + // format, so the verdict round could not certify a report. Codec + // coverage is not an input to selection. Coverage returns in the + // next round, which must still not skip the row. + name: "target format not encodable", + defs: llotypes.ChannelDefinitions{targetCID: targetCD, backfillCID: backfillCD}, + shape: func(obs *Observation) { + obs.SupportedReportFormats = []llotypes.ReportFormat{llotypes.ReportFormatHistoryBackfill} + }, + want: 0, + }, + { + // Selection reads this round's definitions against the previous + // round's watermark and timestamp. The verdict round has no target + // channel, so nothing was selectable and nothing was reported; the + // vote it agrees adds one, which makes the same row selectable in + // the round that decides the advance. + name: "target channel added since the verdict", + defs: llotypes.ChannelDefinitions{backfillCID: backfillCD}, + shape: func(obs *Observation) { + obs.UpdateChannelDefinitions = llotypes.ChannelDefinitions{targetCID: targetCD} + }, + want: 0, + }, + { + // Control: the verdict round did report, so the watermark moves to + // the row it emitted. + name: "previous round reported", + defs: llotypes.ChannelDefinitions{targetCID: targetCD, backfillCID: backfillCD}, + want: fiveSec, + }, + } { + t.Run(tc.name, func(t *testing.T) { + ctx := tests.Context(t) + p := testPlugin(t) + kv := newMemKV() + + // State the verdict round starts from. A candidate is selectable + // throughout (watermark 0, observations at 5s and 8s), so the + // advance turns entirely on the verdict. + require.NoError(t, writeLifecycle(kv, protocol.LifeCycleStageProduction)) + require.NoError(t, writeChannelState(kv, 1, tc.defs)) + require.NoError(t, writeHotState(kv, tenSec, + map[llotypes.ChannelID]uint64{targetCID: tenSec, backfillCID: 0}, + map[llotypes.ChannelID]bool{targetCID: false, backfillCID: false}, + nil, logger.Test(t))) + + // The verdict round. + _, err := p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, round(t, tenSec+1, tc.shape), kv, testBlobs) + require.NoError(t, err) + require.Equal(t, uint64(0), readHotStateForTest(t, kv).validAfterNanoseconds[backfillCID], + "the verdict round must not have advanced the watermark itself") + + // The round that decides the advance. + precursorBytes, err := p.StateTransition(ctx, 3, ocrtypes.AttributedQuery{}, round(t, tenSec+2, nil), kv, testBlobs) + require.NoError(t, err) + + out, err := decodePrecursor(precursorBytes) + require.NoError(t, err) + require.Equal(t, tc.want, out.ValidAfterNanoseconds[backfillCID]) + }) + } +} + +// readHotStateForTest decodes the r/agg record written by the last round. +func readHotStateForTest(t *testing.T, kv *memKV) *kvState { + t.Helper() + s := &kvState{ + validAfterNanoseconds: map[llotypes.ChannelID]uint64{}, + reportedLastRound: map[llotypes.ChannelID]bool{}, + carryForward: map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue{}, + } + require.NoError(t, readHotState(kv, s)) + return s +} diff --git a/llo/dev/v31/statetransition.go b/llo/dev/v31/statetransition.go index 5a952d3..8615ca8 100644 --- a/llo/dev/v31/statetransition.go +++ b/llo/dev/v31/statetransition.go @@ -149,12 +149,12 @@ func (p *Plugin) StateTransition(ctx context.Context, seqNr uint64, _ ocrtypes.A continue } if cd.ReportFormat == llotypes.ReportFormatHistoryBackfill { - // Backfill: the watermark advances to whatever observation the - // previous round would have selected (and emitted), or stays put. - if tsNanos, _, _, found := selectBackfillCandidate(effective, prev.validAfterNanoseconds, prev.observationTimestampNs, channelID, prev.opts); found { - out.ValidAfterNanoseconds[channelID] = tsNanos - } else { - out.ValidAfterNanoseconds[channelID] = prevValidAfter + // Backfill: prevReportable and selection conditions must be met, or stays put. + out.ValidAfterNanoseconds[channelID] = prevValidAfter + if prevReportable(prev, channelID) { + if tsNanos, _, _, found := selectBackfillCandidate(effective, prev.validAfterNanoseconds, prev.observationTimestampNs, channelID, prev.opts); found { + out.ValidAfterNanoseconds[channelID] = tsNanos + } } continue } From 0d067e755def96d6bbeaf35f7e89c0c42b35249f Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Wed, 23 Sep 2026 15:18:44 +0100 Subject: [PATCH 34/40] llo/dev/v31: hold advertised report formats as a set, add golden test for c/codecs --- llo/dev/v31/decode_fuzz_test.go | 10 ++-- llo/dev/v31/golden_test.go | 21 +++++++++ llo/dev/v31/kv.go | 16 +++---- llo/dev/v31/observation.go | 40 ++++++---------- llo/dev/v31/plugin.go | 18 ++++--- llo/dev/v31/plugin_test.go | 47 +++++++++++++------ llo/dev/v31/statetransition.go | 35 +++++--------- .../v31/testdata/golden/kv_codec_support.bin | 3 ++ 8 files changed, 110 insertions(+), 80 deletions(-) create mode 100644 llo/dev/v31/testdata/golden/kv_codec_support.bin diff --git a/llo/dev/v31/decode_fuzz_test.go b/llo/dev/v31/decode_fuzz_test.go index 186e3ce..7194689 100644 --- a/llo/dev/v31/decode_fuzz_test.go +++ b/llo/dev/v31/decode_fuzz_test.go @@ -41,7 +41,7 @@ func FuzzDecodeObservation(f *testing.F) { UnixTimestampNanoseconds: 1_700_000_000_000_000_000, RemoveChannelIDs: map[llotypes.ChannelID]struct{}{7: {}}, UpdateChannelDefinitions: llotypes.ChannelDefinitions{1: jsonChannel()}, - SupportedReportFormats: []llotypes.ReportFormat{llotypes.ReportFormatJSON, llotypes.ReportFormatEVMPremiumLegacy}, + SupportedReportFormats: formatSet(llotypes.ReportFormatJSON, llotypes.ReportFormatEVMPremiumLegacy), } encoded, err := encodeObservation(obs, nil) if err != nil { @@ -76,9 +76,11 @@ func FuzzDecodeObservation(f *testing.F) { if len(decoded.SupportedReportFormats) > protocol.MaxObservationSupportedReportFormatsLength { t.Fatalf("decoded %d report formats, max %d", len(decoded.SupportedReportFormats), protocol.MaxObservationSupportedReportFormatsLength) } - for i := 1; i < len(decoded.SupportedReportFormats); i++ { - if decoded.SupportedReportFormats[i-1] >= decoded.SupportedReportFormats[i] { - t.Fatalf("report formats are not strictly ascending: %v", decoded.SupportedReportFormats) + // Re-encoding the decoded set is what the wire has to be canonical in. + wire := sortedFormatsToWire(decoded.SupportedReportFormats) + for i := 1; i < len(wire); i++ { + if wire[i-1] >= wire[i] { + t.Fatalf("report formats are not strictly ascending: %v", wire) } } for id, cd := range decoded.UpdateChannelDefinitions { diff --git a/llo/dev/v31/golden_test.go b/llo/dev/v31/golden_test.go index 566f302..3431c75 100644 --- a/llo/dev/v31/golden_test.go +++ b/llo/dev/v31/golden_test.go @@ -13,6 +13,8 @@ import ( llotypes "github.com/smartcontractkit/chainlink-common/pkg/types/llo" protocol "github.com/smartcontractkit/chainlink-data-streams/llo/protocol" + + "github.com/smartcontractkit/libocr/commontypes" ) // Golden tests freeze the wire format of the two records the plugin cannot @@ -138,6 +140,9 @@ func Test_Golden_KVRecords(t *testing.T) { }, logger.Test(t), )) + // Out-of-order oracles and formats, so the sorting the writer does is part + // of what is frozen. + require.NoError(t, writeCodecSupport(kv, goldenCodecSupport())) require.NoError(t, writeHistoryLayoutVersion(kv)) require.NoError(t, writeHistoryIndex(kv, []histKey{ {streamID: 100, aggregator: llotypes.AggregatorMedian}, @@ -152,6 +157,7 @@ func Test_Golden_KVRecords(t *testing.T) { {"kv_channel_state.bin", keyChannelState}, {"kv_channel_seqnr.bin", keyChannelSeqNr}, {"kv_hot_state.bin", keyHotState}, + {"kv_codec_support.bin", keyCodecSupport}, {"kv_history_version.bin", keyHistoryVersion}, {"kv_history_index.bin", keyHistoryIndex}, } { @@ -173,6 +179,7 @@ func Test_Golden_KVRecords(t *testing.T) { require.Equal(t, p.ValidAfterNanoseconds, s.validAfterNanoseconds) require.Equal(t, map[llotypes.ChannelID]bool{3: true, 2: true}, s.reportedLastRound) require.Len(t, s.carryForward, 2) + require.Equal(t, goldenCodecSupport(), s.codecSupport) version, err := readHistoryLayoutVersion(kv) require.NoError(t, err) @@ -185,6 +192,20 @@ func Test_Golden_KVRecords(t *testing.T) { }, index) } +// goldenCodecSupport is the codec support record the golden cases freeze: two +// oracles, disjoint sets, and neither oracles nor formats in ascending order. +func goldenCodecSupport() map[commontypes.OracleID]map[llotypes.ReportFormat]struct{} { + return map[commontypes.OracleID]map[llotypes.ReportFormat]struct{}{ + 3: { + llotypes.ReportFormatJSON: {}, + llotypes.ReportFormatEVMPremiumLegacy: {}, + }, + 1: { + llotypes.ReportFormatEVMPremiumLegacy: {}, + }, + } +} + func Test_Golden_KVHistoryRecords(t *testing.T) { const ( sid = llotypes.StreamID(100) diff --git a/llo/dev/v31/kv.go b/llo/dev/v31/kv.go index 35cab31..de4d3e2 100644 --- a/llo/dev/v31/kv.go +++ b/llo/dev/v31/kv.go @@ -116,7 +116,7 @@ type kvState struct { // codecSupport[oracleID] is the report formats that oracle last advertised // a codec for. See writeCodecSupport for why it is remembered per oracle // instead of being counted per round. - codecSupport map[commontypes.OracleID][]llotypes.ReportFormat + codecSupport map[commontypes.OracleID]map[llotypes.ReportFormat]struct{} // channelStateSeqNr is the seqNr at which channelDefinitions were written. channelStateSeqNr uint64 validAfterNanoseconds map[llotypes.ChannelID]uint64 @@ -160,7 +160,7 @@ func loadKVState(r ocr3_1types.KeyValueStateReader, cache *protocol.ChannelCache func loadColdKVState(r ocr3_1types.KeyValueStateReader, cache *protocol.ChannelCache) (*kvState, error) { s := &kvState{ channelDefinitions: llotypes.ChannelDefinitions{}, - codecSupport: map[commontypes.OracleID][]llotypes.ReportFormat{}, + codecSupport: map[commontypes.OracleID]map[llotypes.ReportFormat]struct{}{}, validAfterNanoseconds: map[llotypes.ChannelID]uint64{}, reportedLastRound: map[llotypes.ChannelID]bool{}, carryForward: map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue{}, @@ -199,8 +199,8 @@ func loadColdKVState(r ocr3_1types.KeyValueStateReader, cache *protocol.ChannelC } // readCodecSupport reads and decodes the c/codecs record. -func readCodecSupport(r ocr3_1types.KeyValueStateReader) (map[commontypes.OracleID][]llotypes.ReportFormat, error) { - support := map[commontypes.OracleID][]llotypes.ReportFormat{} +func readCodecSupport(r ocr3_1types.KeyValueStateReader) (map[commontypes.OracleID]map[llotypes.ReportFormat]struct{}, error) { + support := map[commontypes.OracleID]map[llotypes.ReportFormat]struct{}{} b, err := r.Read(keyCodecSupport) if err != nil { return nil, fmt.Errorf("read codec support: %w", err) @@ -219,9 +219,9 @@ func readCodecSupport(r ocr3_1types.KeyValueStateReader) (map[commontypes.Oracle if len(entry.ReportFormats) > protocol.MaxObservationSupportedReportFormatsLength { return nil, fmt.Errorf("oracle %d advertises too many report formats: %d (max %d)", entry.OracleID, len(entry.ReportFormats), protocol.MaxObservationSupportedReportFormatsLength) } - formats := make([]llotypes.ReportFormat, 0, len(entry.ReportFormats)) + formats := make(map[llotypes.ReportFormat]struct{}, len(entry.ReportFormats)) for _, f := range entry.ReportFormats { - formats = append(formats, llotypes.ReportFormat(f)) + formats[llotypes.ReportFormat(f)] = struct{}{} } support[commontypes.OracleID(entry.OracleID)] = formats } @@ -235,13 +235,13 @@ func readCodecSupport(r ocr3_1types.KeyValueStateReader) (map[commontypes.Oracle // observation quorum is 2f+1. A count taken from one round can never exceed // 2f+1, so the supporter threshold reportability would demand that every observation // in a minimal quorum advertise the format. -func writeCodecSupport(w ocr3_1types.KeyValueStateReadWriter, support map[commontypes.OracleID][]llotypes.ReportFormat) error { +func writeCodecSupport(w ocr3_1types.KeyValueStateReadWriter, support map[commontypes.OracleID]map[llotypes.ReportFormat]struct{}) error { pb := &protocol.LLOCodecSupportProto{ Oracles: make([]*protocol.LLOOracleCodecSupportProto, 0, len(support)), } for oracleID, formats := range support { encoded := make([]uint32, 0, len(formats)) - for _, f := range formats { + for f := range formats { encoded = append(encoded, uint32(f)) } sort.Slice(encoded, func(i, j int) bool { return encoded[i] < encoded[j] }) diff --git a/llo/dev/v31/observation.go b/llo/dev/v31/observation.go index c0f4e9b..0bfe164 100644 --- a/llo/dev/v31/observation.go +++ b/llo/dev/v31/observation.go @@ -29,7 +29,7 @@ type Observation struct { // codec for. Encoding is node-local state that the state transition cannot // read without forking; advertising it here turns it into a replicated fact // that reportability can gate on (see isReportable). - SupportedReportFormats []llotypes.ReportFormat + SupportedReportFormats map[llotypes.ReportFormat]struct{} // PredecessorSigners and PredecessorF are the predecessor instance's signer // set and f, read from the node-local retirement report cache. A staging // instance carries them alongside an attested retirement report, so the @@ -84,8 +84,9 @@ func encodeObservation(obs Observation, handles [][]byte) (ocrtypes.Observation, } } - // Sorted and deduped, matching what decode enforces. - main.SupportedReportFormats = sortedUniqueFormats(obs.SupportedReportFormats) + // Map iteration order, so sort: the wire form is a repeated field, which + // deterministic marshaling does not canonicalize. + main.SupportedReportFormats = sortedFormatsToWire(obs.SupportedReportFormats) // Deterministic even though nothing compares observation bytes today // A future path which does compare or hash the value cannot be @@ -292,46 +293,35 @@ func observationFromProto(main *protocol.LLOObservationProto) (Observation, erro obs.PredecessorSigners = main.PredecessorSigners obs.PredecessorF = uint8(main.PredecessorF) - obs.SupportedReportFormats = sortedUniqueFormatsFromWire(main.SupportedReportFormats) + obs.SupportedReportFormats = formatsFromWire(main.SupportedReportFormats) return obs, nil } -// sortedUniqueFormats returns the formats sorted ascending with duplicates -// removed. nil in, nil out, so an oracle advertising nothing stays absent from -// the wire rather than carrying an empty list. -func sortedUniqueFormats(in []llotypes.ReportFormat) []uint32 { +// sortedFormatsToWire returns the formats sorted ascending. nil in, nil out, so +// an oracle advertising nothing stays absent from the wire rather than carrying +// an empty list. +func sortedFormatsToWire(in map[llotypes.ReportFormat]struct{}) []uint32 { if len(in) == 0 { return nil } - seen := make(map[llotypes.ReportFormat]struct{}, len(in)) out := make([]uint32, 0, len(in)) - for _, f := range in { - if _, dup := seen[f]; dup { - continue - } - seen[f] = struct{}{} + for f := range in { out = append(out, uint32(f)) } sort.Slice(out, func(i, j int) bool { return out[i] < out[j] }) return out } -// sortedUniqueFormatsFromWire is sortedUniqueFormats for the wire -// representation: sorted ascending, duplicates removed. -func sortedUniqueFormatsFromWire(in []uint32) []llotypes.ReportFormat { +// formatsFromWire is sortedFormatsToWire inverted. Duplicate wire entries +// collapse, so the set an oracle advertises never depends on repetition. +func formatsFromWire(in []uint32) map[llotypes.ReportFormat]struct{} { if len(in) == 0 { return nil } - seen := make(map[uint32]struct{}, len(in)) - out := make([]llotypes.ReportFormat, 0, len(in)) + out := make(map[llotypes.ReportFormat]struct{}, len(in)) for _, f := range in { - if _, dup := seen[f]; dup { - continue - } - seen[f] = struct{}{} - out = append(out, llotypes.ReportFormat(f)) + out[llotypes.ReportFormat(f)] = struct{}{} } - sort.Slice(out, func(i, j int) bool { return out[i] < out[j] }) return out } diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index 873cd11..1c275fd 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -205,14 +205,20 @@ func (p *Plugin) Observation(_ context.Context, seqNr uint64, _ ocrtypes.Attribu // from the codecs it was constructed with. Truncated to the advertisable bound // so the observation stays within its size budget; a real codec map is far // smaller than the bound, so this never fires in practice. -func supportedReportFormats(codecs map[llotypes.ReportFormat]protocol.ReportCodec) []llotypes.ReportFormat { - out := make([]llotypes.ReportFormat, 0, len(codecs)) +func supportedReportFormats(codecs map[llotypes.ReportFormat]protocol.ReportCodec) map[llotypes.ReportFormat]struct{} { + sorted := make([]llotypes.ReportFormat, 0, len(codecs)) for format := range codecs { - out = append(out, format) + sorted = append(sorted, format) } - sort.Slice(out, func(i, j int) bool { return out[i] < out[j] }) - if len(out) > protocol.MaxObservationSupportedReportFormatsLength { - out = out[:protocol.MaxObservationSupportedReportFormatsLength] + // Truncation must not depend on map iteration order: every node with the + // same codecs has to advertise the same set. + sort.Slice(sorted, func(i, j int) bool { return sorted[i] < sorted[j] }) + if len(sorted) > protocol.MaxObservationSupportedReportFormatsLength { + sorted = sorted[:protocol.MaxObservationSupportedReportFormatsLength] + } + out := make(map[llotypes.ReportFormat]struct{}, len(sorted)) + for _, format := range sorted { + out[format] = struct{}{} } return out } diff --git a/llo/dev/v31/plugin_test.go b/llo/dev/v31/plugin_test.go index b82e564..017955d 100644 --- a/llo/dev/v31/plugin_test.go +++ b/llo/dev/v31/plugin_test.go @@ -146,12 +146,21 @@ func ao(observer int, obsBytes []byte) ocrtypes.AttributedObservation { // testSupportedReportFormats is the codec coverage a fixture oracle advertises // by default: every format the tests in this package build channels for. -var testSupportedReportFormats = []llotypes.ReportFormat{ +var testSupportedReportFormats = formatSet( llotypes.ReportFormatJSON, llotypes.ReportFormatEVMPremiumLegacy, llotypes.ReportFormatEVMABIEncodeUnpacked, llotypes.ReportFormatEVMABIEncodeUnpackedExpr, llotypes.ReportFormatHistoryBackfill, +) + +// formatSet builds the advertised-format set a fixture oracle carries. +func formatSet(formats ...llotypes.ReportFormat) map[llotypes.ReportFormat]struct{} { + out := make(map[llotypes.ReportFormat]struct{}, len(formats)) + for _, f := range formats { + out[f] = struct{}{} + } + return out } // testBlobs is the shared in-memory blob store used by StateTransition @@ -1206,21 +1215,31 @@ func Test_ReportFormatSupportGate_StopsValidAfterAdvance(t *testing.T) { func Test_Observation_SupportedReportFormats_RoundTrip(t *testing.T) { ctx := tests.Context(t) - // Duplicates are collapsed and the list is sorted, so one oracle can only - // ever contribute one vote per format to the tally. + // The set is sorted on the wire and collapses back to a set on decode, so + // one oracle can only ever contribute one vote per format to the tally. obs := Observation{ UnixTimestampNanoseconds: 1, - SupportedReportFormats: []llotypes.ReportFormat{ + SupportedReportFormats: formatSet( llotypes.ReportFormatJSON, llotypes.ReportFormatEVMPremiumLegacy, - llotypes.ReportFormatJSON, - }, + ), } b, err := encodeObservation(obs, nil) require.NoError(t, err) got, err := decodeObservation(ctx, b, nil, nil) require.NoError(t, err) - require.Equal(t, []llotypes.ReportFormat{llotypes.ReportFormatEVMPremiumLegacy, llotypes.ReportFormatJSON}, got.SupportedReportFormats) + require.Equal(t, formatSet(llotypes.ReportFormatEVMPremiumLegacy, llotypes.ReportFormatJSON), got.SupportedReportFormats) + require.Equal(t, []uint32{uint32(llotypes.ReportFormatEVMPremiumLegacy), uint32(llotypes.ReportFormatJSON)}, sortedFormatsToWire(got.SupportedReportFormats)) + + // Duplicate wire entries collapse rather than double counting. + dup, err := proto.Marshal(&protocol.LLOObservationProto{ + UnixTimestampNanoseconds: 1, + SupportedReportFormats: []uint32{uint32(llotypes.ReportFormatJSON), uint32(llotypes.ReportFormatJSON)}, + }) + require.NoError(t, err) + got, err = decodeObservation(ctx, frameObservation(nil, dup), nil, nil) + require.NoError(t, err) + require.Equal(t, formatSet(llotypes.ReportFormatJSON), got.SupportedReportFormats) // An oracle advertising nothing decodes as advertising nothing, rather than // as an empty-but-present list. @@ -1276,8 +1295,8 @@ func Test_StateTransition_TalliesReportFormatSupport(t *testing.T) { // Two oracles advertise JSON, one advertises nothing: the tally is a count // of advertisements, and an oracle that advertises nothing is not counted. - obsJSON := mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 1, SupportedReportFormats: []llotypes.ReportFormat{llotypes.ReportFormatJSON}}) - obsNone := mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 1, SupportedReportFormats: []llotypes.ReportFormat{}}) + obsJSON := mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 1, SupportedReportFormats: formatSet(llotypes.ReportFormatJSON)}) + obsNone := mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 1, SupportedReportFormats: formatSet()}) precBytes, err := p.StateTransition(ctx, 2, ocrtypes.AttributedQuery{}, []ocrtypes.AttributedObservation{ao(0, obsJSON), ao(1, obsJSON), ao(2, obsNone)}, kv, testBlobs) require.NoError(t, err) @@ -1359,10 +1378,10 @@ func Test_StateTransition_CodecSupportAccumulatesAcrossRounds(t *testing.T) { })) obsJSON := func(ts uint64) []byte { - return mustEncodeObs(t, Observation{UnixTimestampNanoseconds: ts, SupportedReportFormats: []llotypes.ReportFormat{llotypes.ReportFormatJSON}}) + return mustEncodeObs(t, Observation{UnixTimestampNanoseconds: ts, SupportedReportFormats: formatSet(llotypes.ReportFormatJSON)}) } obsNone := func(ts uint64) []byte { - return mustEncodeObs(t, Observation{UnixTimestampNanoseconds: ts, SupportedReportFormats: []llotypes.ReportFormat{}}) + return mustEncodeObs(t, Observation{UnixTimestampNanoseconds: ts, SupportedReportFormats: formatSet()}) } // All N oracles advertise JSON. @@ -1402,9 +1421,9 @@ func Test_StateTransition_PrunesUnusedReportFormatSupport(t *testing.T) { // unpruned tally would hold 3*MaxObservationSupportedReportFormatsLength-2 // entries. padded := func(oracle int) []byte { - formats := []llotypes.ReportFormat{llotypes.ReportFormatJSON} + formats := formatSet(llotypes.ReportFormatJSON) for i := 1; i < protocol.MaxObservationSupportedReportFormatsLength; i++ { - formats = append(formats, llotypes.ReportFormat(1_000_000+oracle*1_000+i)) + formats[llotypes.ReportFormat(1_000_000+oracle*1_000+i)] = struct{}{} } return mustEncodeObs(t, Observation{UnixTimestampNanoseconds: 1, SupportedReportFormats: formats}) } @@ -1706,7 +1725,7 @@ func Test_StateTransition_BackfillValidAfterRequiresPreviousReport(t *testing.T) name: "target format not encodable", defs: llotypes.ChannelDefinitions{targetCID: targetCD, backfillCID: backfillCD}, shape: func(obs *Observation) { - obs.SupportedReportFormats = []llotypes.ReportFormat{llotypes.ReportFormatHistoryBackfill} + obs.SupportedReportFormats = formatSet(llotypes.ReportFormatHistoryBackfill) }, want: 0, }, diff --git a/llo/dev/v31/statetransition.go b/llo/dev/v31/statetransition.go index 8615ca8..f659ef2 100644 --- a/llo/dev/v31/statetransition.go +++ b/llo/dev/v31/statetransition.go @@ -7,6 +7,7 @@ import ( "encoding/binary" "errors" "fmt" + "maps" "sort" "sync" @@ -254,7 +255,7 @@ type observationTally struct { // supportedFormatsByOracle[oracleID] is the formats that oracle advertised // this round, deduped by decodeObservation. It is merged into the persisted // per-oracle record rather than counted here: see writeCodecSupport. - supportedFormatsByOracle map[commontypes.OracleID][]llotypes.ReportFormat + supportedFormatsByOracle map[commontypes.OracleID]map[llotypes.ReportFormat]struct{} streamObservations map[llotypes.StreamID][]protocol.StreamValue } @@ -265,7 +266,7 @@ func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.Attribut removeChannelVotesByID: make(map[llotypes.ChannelID]int), updateChannelDefinitionsByHash: make(map[[32]byte]protocol.ChannelDefinitionWithID), updateChannelVotesByHash: make(map[[32]byte]int), - supportedFormatsByOracle: make(map[commontypes.OracleID][]llotypes.ReportFormat), + supportedFormatsByOracle: make(map[commontypes.OracleID]map[llotypes.ReportFormat]struct{}), streamObservations: make(map[llotypes.StreamID][]protocol.StreamValue), } @@ -320,8 +321,7 @@ func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.Attribut } tally.timestampsNanoseconds = append(tally.timestampsNanoseconds, observation.UnixTimestampNanoseconds) - // Deduped by decodeObservation, so an oracle names each format at most - // once. Recording the whole advertised set (including an empty one) + // Recording the whole advertised set (including an empty one) // makes this round's advertisement replace that oracle's last, so an // oracle that loses a codec stops counting for it. tally.supportedFormatsByOracle[ao.Observer] = observation.SupportedReportFormats @@ -347,8 +347,8 @@ func (p *Plugin) decodeObservations(ctx context.Context, aos []ocrtypes.Attribut // mergeCodecSupport overlays this round's advertisements on the persisted ones, // replacing the entry of every oracle that contributed an observation and // leaving the rest untouched. -func mergeCodecSupport(persisted, thisRound map[commontypes.OracleID][]llotypes.ReportFormat) map[commontypes.OracleID][]llotypes.ReportFormat { - merged := make(map[commontypes.OracleID][]llotypes.ReportFormat, len(persisted)+len(thisRound)) +func mergeCodecSupport(persisted, thisRound map[commontypes.OracleID]map[llotypes.ReportFormat]struct{}) map[commontypes.OracleID]map[llotypes.ReportFormat]struct{} { + merged := make(map[commontypes.OracleID]map[llotypes.ReportFormat]struct{}, len(persisted)+len(thisRound)) for oracleID, formats := range persisted { merged[oracleID] = formats } @@ -358,37 +358,26 @@ func mergeCodecSupport(persisted, thisRound map[commontypes.OracleID][]llotypes. return merged } -// codecSupportChanged reports whether any oracle's advertised set differs. Both -// sides hold deduped sets, so comparing as sets (not slices) is what matters: -// order must not trigger a rewrite. -func codecSupportChanged(prev, next map[commontypes.OracleID][]llotypes.ReportFormat) bool { +// codecSupportChanged reports whether any oracle's advertised set differs. +func codecSupportChanged(prev, next map[commontypes.OracleID]map[llotypes.ReportFormat]struct{}) bool { if len(prev) != len(next) { return true } for oracleID, nextFormats := range next { prevFormats, ok := prev[oracleID] - if !ok || len(prevFormats) != len(nextFormats) { + if !ok || !maps.Equal(prevFormats, nextFormats) { return true } - seen := make(map[llotypes.ReportFormat]struct{}, len(prevFormats)) - for _, f := range prevFormats { - seen[f] = struct{}{} - } - for _, f := range nextFormats { - if _, ok := seen[f]; !ok { - return true - } - } } return false } // countSupportByFormat counts, per report format, the oracles whose last // advertisement named it. -func countSupportByFormat(support map[commontypes.OracleID][]llotypes.ReportFormat) map[llotypes.ReportFormat]int { +func countSupportByFormat(support map[commontypes.OracleID]map[llotypes.ReportFormat]struct{}) map[llotypes.ReportFormat]int { counts := make(map[llotypes.ReportFormat]int) for _, formats := range support { - for _, format := range formats { + for format := range formats { counts[format]++ } } @@ -620,7 +609,7 @@ func (p *Plugin) flushKV( prev *kvState, out precursor, pending llotypes.ChannelDefinitions, - codecSupport map[commontypes.OracleID][]llotypes.ReportFormat, + codecSupport map[commontypes.OracleID]map[llotypes.ReportFormat]struct{}, carryForward map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue, history *historyStore, ) error { diff --git a/llo/dev/v31/testdata/golden/kv_codec_support.bin b/llo/dev/v31/testdata/golden/kv_codec_support.bin new file mode 100644 index 0000000..fcc69b6 --- /dev/null +++ b/llo/dev/v31/testdata/golden/kv_codec_support.bin @@ -0,0 +1,3 @@ + + + \ No newline at end of file From 5c19fc04e25690abaac743923a6fce1208b73125 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Wed, 23 Sep 2026 18:11:05 +0100 Subject: [PATCH 35/40] llo/dev/v31: BlombPump.Take waits if a cycle is currently in flight --- llo/dev/v31/blobpump.go | 65 +++++++++++----- llo/dev/v31/blobpump_test.go | 143 ++++++++++++++++++++++++++++++----- llo/dev/v31/factory.go | 1 + 3 files changed, 174 insertions(+), 35 deletions(-) diff --git a/llo/dev/v31/blobpump.go b/llo/dev/v31/blobpump.go index 1c8d504..3937467 100644 --- a/llo/dev/v31/blobpump.go +++ b/llo/dev/v31/blobpump.go @@ -80,6 +80,9 @@ const ( closeTimeoutSlackMultiplier = 5 // minCloseTimeout fixes the minimum time Close waits waits for an in-flight cycle. minCloseTimeout = 1 * time.Second + // BlobInFlightWaitDivisor scales MaxDurationObservation into how long Take + // waits for a cycle that is already in flight to park. + BlobInFlightWaitDivisor = 8 ) // perOracleUnexpiredBlobCount derives the per-oracle unexpired-blob budget from @@ -140,9 +143,11 @@ type blobPump struct { cycles atomic.Uint64 missStreak atomic.Uint64 + // ready holds the latest parked snapshot. + ready chan *blobSnapshot + mu sync.Mutex input pumpInput - ready *blobSnapshot // lastTakeAt and roundPeriod estimate the round cadence from the interval // between consecutive Take calls (one per round), which is the only signal // the plugin has: deltaRound is not part of the reporting plugin config. @@ -165,6 +170,8 @@ type blobPumpParams struct { maxSnapshotRounds uint64 // blobLifetimeRounds is the remote fetchability bound. See DefaultBlobLifetimeRounds. blobLifetimeRounds uint64 + // inFlightWait bounds how long Take waits for an in-flight cycle to park. + inFlightWait time.Duration } func newBlobPump(lggr logger.Logger, params blobPumpParams) *blobPump { @@ -173,6 +180,7 @@ func newBlobPump(lggr logger.Logger, params blobPumpParams) *blobPump { blobPumpParams: params, lggr: logger.Sugared(lggr).Named("BlobPump"), trigger: make(chan struct{}, 1), + ready: make(chan *blobSnapshot, 1), ctx: ctx, cancel: cancel, closeTimeout: max(params.observationTimeout*closeTimeoutSlackMultiplier, minCloseTimeout), @@ -232,35 +240,28 @@ func (p *blobPump) SetInput(in pumpInput) { // kicks the next cycle: kicking on a discard as well as on a hit is what stops // a single unusable snapshot from stalling the pump forever. The second return // value is the reason a snapshot was not returned, for logging. +// +// A round that finds nothing parked while a cycle is running waits for +// inFlightWait for the cycle to park. func (p *blobPump) Take(seqNr uint64) (*blobSnapshot, string) { if !p.enabled() { p.miss() return nil, "blob pump disabled" } - now := time.Now() - p.mu.Lock() - snap := p.ready - p.ready = nil - // Measure before resolving the limit, both under the same lock. On the very - // first Take there is no previous call to measure against, so roundPeriod is - // still zero and the derived age check is inert. That Take is also the one - // that kicks the first cycle, though, so nothing is parked yet and the round - // misses on "no snapshot parked" regardless. By the second Take, which is - // the first that can see a snapshot, the gap has been measured and the check - // is live. There is no round in which a snapshot exists and roundPeriod is - // still zero. - p.recordRoundLocked(now) + p.recordRoundLocked(time.Now()) ageLimit := p.snapshotAgeLimitLocked() p.mu.Unlock() defer p.kick() + snap, waited := p.takeReady(p.inFlightWait) + now := time.Now() switch { case snap == nil: p.miss() - if p.inFlight.Load() { + if waited || p.inFlight.Load() { return nil, "cycle in flight" } return nil, "no snapshot parked" @@ -276,6 +277,36 @@ func (p *blobPump) Take(seqNr uint64) (*blobSnapshot, string) { } } +// takeReady detaches the parked snapshot, waiting up to timeout for an +// in-flight cycle to park one. +func (p *blobPump) takeReady(timeout time.Duration) (*blobSnapshot, bool) { + waited := false + if timeout > 0 && p.inFlight.Load() { + waited = true + ctx, cancel := context.WithTimeout(p.ctx, timeout) + defer cancel() + select { + case snap := <-p.ready: + return snap, waited + case <-ctx.Done(): + } + } + + select { + case snap := <-p.ready: + return snap, waited + default: + return nil, waited + } +} + +// park makes a snapshot available to take, replacing if we have a snap +// that no round has taken. The newest snapshot has a higher priority. +func (p *blobPump) park(snap *blobSnapshot) { + p.takeReady(0) + p.ready <- snap +} + // miss records a round that found no usable snapshot. Sustained misses means // this node is not contributing at all. Record misses and log when above MissStreakLogThreshold. func (p *blobPump) miss() { @@ -386,9 +417,7 @@ func (p *blobPump) cycle() { } p.cycles.Add(1) - p.mu.Lock() - p.ready = snap - p.mu.Unlock() + p.park(snap) if p.verboseLogging { p.lggr.Debugw("Blob pump parked snapshot", "seqNr", in.seqNr, "usableBefore", snap.usableBefore, "expiresAt", snap.expiresAt, "streams", snap.streamCount, "handleBytes", len(snap.handleBytes)) diff --git a/llo/dev/v31/blobpump_test.go b/llo/dev/v31/blobpump_test.go index bbfb78b..8d1cf42 100644 --- a/llo/dev/v31/blobpump_test.go +++ b/llo/dev/v31/blobpump_test.go @@ -108,9 +108,7 @@ func Test_blobPump_TakeIsSingleUse(t *testing.T) { // The pump may have parked a fresh snapshot by now, so assert on the parked // slot directly rather than on a second Take. - p.mu.Lock() - p.ready = nil - p.mu.Unlock() + p.takeReady(0) snap, reason := p.Take(2) require.Nil(t, snap) require.NotEmpty(t, reason) @@ -122,9 +120,7 @@ func Test_blobPump_RejectsStaleSnapshots(t *testing.T) { // is still fetchable by peers. t.Run("too stale by sequence number", func(t *testing.T) { p := testPump(t, mockDS(), newFakeBroadcaster(), time.Minute) - p.mu.Lock() - p.ready = &blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now(), forSeqNr: 2, usableBefore: 4, expiresAt: 6} - p.mu.Unlock() + p.park(&blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now(), forSeqNr: 2, usableBefore: 4, expiresAt: 6}) snap, reason := p.Take(4) require.Nil(t, snap) @@ -134,9 +130,7 @@ func Test_blobPump_RejectsStaleSnapshots(t *testing.T) { t.Run("expired by wall clock", func(t *testing.T) { p := testPump(t, mockDS(), newFakeBroadcaster(), time.Nanosecond) - p.mu.Lock() - p.ready = &blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100} - p.mu.Unlock() + p.park(&blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100}) snap, reason := p.Take(3) require.Nil(t, snap) @@ -145,9 +139,7 @@ func Test_blobPump_RejectsStaleSnapshots(t *testing.T) { t.Run("age check disabled", func(t *testing.T) { p := testPump(t, mockDS(), newFakeBroadcaster(), -1) - p.mu.Lock() - p.ready = &blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100} - p.mu.Unlock() + p.park(&blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100}) snap, _ := p.Take(3) require.NotNil(t, snap, "with the age check disabled only maxSnapshotRounds bounds staleness") @@ -158,9 +150,7 @@ func Test_blobPump_RejectsStaleSnapshots(t *testing.T) { // cadence would silently stop the node contributing stream values. t.Run("derived age check is inert until a round period is measured", func(t *testing.T) { p := testPump(t, mockDS(), newFakeBroadcaster(), 0) - p.mu.Lock() - p.ready = &blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100} - p.mu.Unlock() + p.park(&blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100}) snap, reason := p.Take(3) require.NotNil(t, snap, "reason: %s", reason) @@ -170,8 +160,8 @@ func Test_blobPump_RejectsStaleSnapshots(t *testing.T) { p := testPump(t, mockDS(), newFakeBroadcaster(), 0) p.mu.Lock() p.roundPeriod = time.Millisecond - p.ready = &blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100} p.mu.Unlock() + p.park(&blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100}) snap, reason := p.Take(3) require.Nil(t, snap) @@ -252,6 +242,123 @@ func Test_blobPump_SingleFlight(t *testing.T) { close(release) } +// gatedDataSource blocks until released and then observes normally, modelling a +// cycle that is still gathering values when Take arrives. +type gatedDataSource struct { + release chan struct{} + entered chan struct{} + once sync.Once + // err, when set, fails the observation instead of producing values, so the + // cycle unwinds without parking anything. + err error +} + +func (g *gatedDataSource) Observe(ctx context.Context, sv protocol.StreamValues, opts DSOpts) error { + g.once.Do(func() { close(g.entered) }) + select { + case <-g.release: + case <-ctx.Done(): + return ctx.Err() + } + if g.err != nil { + return g.err + } + sv[100] = protocol.ToDecimal(decimal.NewFromInt(7)) + return nil +} + +// Test_blobPump_TakeWaitsForInFlightCycle covers the rescue path: a Take that +// finds nothing parked while a cycle is gathering waits for that cycle instead +// of missing the round outright. The data source is released only once the +// second Take is already waiting, so the snapshot can only have been served by +// the wait. +func Test_blobPump_TakeWaitsForInFlightCycle(t *testing.T) { + ds := &gatedDataSource{release: make(chan struct{}), entered: make(chan struct{})} + p := testPump(t, ds, newFakeBroadcaster(), time.Minute) + p.inFlightWait = tests.WaitTimeout(t) + + // The first Take only kicks the cycle: nothing is in flight yet, so there is + // nothing for it to wait on. + p.SetInput(pumpInputFor(2)) + snap, reason := p.Take(2) + require.Nil(t, snap) + require.Equal(t, "no snapshot parked", reason) + + select { + case <-ds.entered: + case <-time.After(tests.WaitTimeout(t)): + t.Fatal("DataSource.Observe was never called") + } + require.True(t, p.inFlight.Load()) + + go func() { + time.Sleep(50 * time.Millisecond) + close(ds.release) + }() + + snap, reason = p.Take(2) + require.NotNil(t, snap, "Take did not wait for the in-flight cycle: %s", reason) + require.Empty(t, reason) +} + +// Test_blobPump_TakeWaitFallsThroughOnTimeout asserts the wait is bounded and +// the miss path stays the fallback: a cycle that does not park in time still +// misses the round rather than holding up the observation. +func Test_blobPump_TakeWaitFallsThroughOnTimeout(t *testing.T) { + ds := &gatedDataSource{release: make(chan struct{}), entered: make(chan struct{})} + defer close(ds.release) + + p := testPump(t, ds, newFakeBroadcaster(), time.Minute) + p.inFlightWait = 50 * time.Millisecond + + p.SetInput(pumpInputFor(2)) + _, _ = p.Take(2) + select { + case <-ds.entered: + case <-time.After(tests.WaitTimeout(t)): + t.Fatal("DataSource.Observe was never called") + } + + start := time.Now() + snap, reason := p.Take(2) + elapsed := time.Since(start) + require.Nil(t, snap) + require.Equal(t, "cycle in flight", reason) + require.GreaterOrEqual(t, elapsed, p.inFlightWait, "Take returned before the wait elapsed") + require.Less(t, elapsed, 10*p.inFlightWait, "Take waited well past its bound") +} + +// Test_blobPump_TakeWaitReportsCycleAfterItEnds pins the miss reason for a +// round that waited: the cycle it waited on can finish empty and clear the +// in-flight flag before the wait expires, and the round still belongs to that +// cycle rather than to an absent snapshot. +func Test_blobPump_TakeWaitReportsCycleAfterItEnds(t *testing.T) { + ds := &gatedDataSource{release: make(chan struct{}), entered: make(chan struct{}), err: errors.New("boom")} + p := testPump(t, ds, newFakeBroadcaster(), time.Minute) + p.inFlightWait = 500 * time.Millisecond + + p.SetInput(pumpInputFor(2)) + _, _ = p.Take(2) + select { + case <-ds.entered: + case <-time.After(tests.WaitTimeout(t)): + t.Fatal("DataSource.Observe was never called") + } + + // Fail the cycle while the next Take is waiting on it. + go func() { + time.Sleep(50 * time.Millisecond) + close(ds.release) + }() + + // The reason is resolved inside Take, before its deferred kick starts the + // next cycle, so it is the only assertable evidence here: reading inFlight + // after Take returns would race that new cycle. + snap, reason := p.Take(2) + require.Nil(t, snap) + require.Equal(t, "cycle in flight", reason) +} + func Test_observableStreams(t *testing.T) { state := &kvState{channelDefinitions: llotypes.ChannelDefinitions{ 1: {ReportFormat: llotypes.ReportFormatJSON, Streams: []llotypes.Stream{ @@ -317,7 +424,9 @@ func Test_blobPump_SurvivesDataSourcePanic(t *testing.T) { p.SetInput(pumpInputFor(3)) _, _ = p.Take(3) require.Eventually(t, func() bool { return ds.calls.Load() >= 2 }, tests.WaitTimeout(t), 10*time.Millisecond) - require.False(t, p.inFlight.Load()) + // Take kicks before it returns, so the last kicked cycle may still be + // running; it must unwind rather than leave the flag stuck. + require.Eventually(t, func() bool { return !p.inFlight.Load() }, tests.WaitTimeout(t), 10*time.Millisecond) } // stuckDataSource ignores its context and blocks until released, modelling a diff --git a/llo/dev/v31/factory.go b/llo/dev/v31/factory.go index 9de991d..5286346 100644 --- a/llo/dev/v31/factory.go +++ b/llo/dev/v31/factory.go @@ -149,6 +149,7 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re configDigest: cfg.ConfigDigest, verboseLogging: f.Config.VerboseLogging, observationTimeout: blobObservationTimeout, + inFlightWait: cfg.MaxDurationObservation / BlobInFlightWaitDivisor, maxSnapshotAge: f.MaxBlobSnapshotAge, maxSnapshotRounds: maxSnapshotRounds, blobLifetimeRounds: blobLifetimeRounds, From 9287cfa664bef322032363c3d3ef09b87c9470e8 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Thu, 24 Sep 2026 14:35:33 +0100 Subject: [PATCH 36/40] llo: bound the precursor at admission rather than at write time Count the distinct aggregated (streamID, aggregator) pairs a definition set declares and reject it above MaxPersistedAggregates. The write-time truncation stays as a backstop for state committed before the rule existed. Cap calculated streams count with MaxTotalCalculatedStreams and bound each evaluated value coefficient. --- llo/dev/v31/factory.go | 16 +++--- llo/protocol/calculated/calculated.go | 8 +++ llo/protocol/calculated/series_test.go | 42 +++++++++++++++ llo/protocol/channel_definitions.go | 36 ++++++++++++- llo/protocol/channel_definitions_test.go | 66 ++++++++++++++++++++++++ llo/protocol/limits.go | 23 +++------ llo/protocol/stream_value.go | 6 +++ 7 files changed, 172 insertions(+), 25 deletions(-) diff --git a/llo/dev/v31/factory.go b/llo/dev/v31/factory.go index 5286346..bb3bc02 100644 --- a/llo/dev/v31/factory.go +++ b/llo/dev/v31/factory.go @@ -170,14 +170,18 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re // MaxObservationUpdateChannelDefinitionsLength // definitions. // MaxReportsPlusPrecursorBytes the precursor embeds every definition plus - // every stream aggregate. The aggregate count - // is bounded by protocol.MaxPersistedAggregates, - // and the definition set by channel count + // every stream aggregate. Admission bounds + // the aggregate count by + // protocol.MaxPersistedAggregates, and the + // definition set by channel count // (MaxOutcomeChannelDefinitionsLength), total // stream entries (MaxTotalStreamEntries) and - // opts bytes (MaxTotalOptsBytes) at admission. - // A stream value's decimal coefficient is bounded - // on observation decode. + // opts bytes (MaxTotalOptsBytes). Calculated + // streams are counted separately, by + // MaxTotalCalculatedStreams. A stream value's + // decimal coefficient is bounded on + // observation decode, and a calculated one + // where it is written into the aggregates. // MaxKeyValueModifiedKeys* the per-round write set is c/defs plus r/agg // plus the history windows, and history is // held to protocol.MaxHistoryTotalBytes so it diff --git a/llo/protocol/calculated/calculated.go b/llo/protocol/calculated/calculated.go index 1f7d004..1152509 100644 --- a/llo/protocol/calculated/calculated.go +++ b/llo/protocol/calculated/calculated.go @@ -863,6 +863,14 @@ func applyCalculatedStreams(lggr logger.Logger, channelDefinitions llotypes.Chan work.cid, abi.ExpressionStreamID, abi.Expression) break } + // A calculated value never passes through observation decode, so + // this is the only place its coefficient is bounded. + if cerr := protocol.CheckDecimalCoefficient(value); cerr != nil { + err = fmt.Errorf( + "calculated stream value out of range, channelID: %d, expressionStreamID: %d, expression: %s: %w", + work.cid, abi.ExpressionStreamID, abi.Expression, cerr) + break + } // update the aggregates with the new stream value if expression was successfully evaluated streamAggregates[abi.ExpressionStreamID] = map[llotypes.Aggregator]protocol.StreamValue{ llotypes.AggregatorCalculated: protocol.ToDecimal(value), diff --git a/llo/protocol/calculated/series_test.go b/llo/protocol/calculated/series_test.go index 074067d..8cc9f47 100644 --- a/llo/protocol/calculated/series_test.go +++ b/llo/protocol/calculated/series_test.go @@ -481,3 +481,45 @@ func TestProcessCalculatedStreamsDryRun_History(t *testing.T) { require.Error(t, ProcessCalculatedStreamsDryRun("Count(History(s1_timestamp, 10))")) require.Error(t, ProcessCalculatedStreamsDryRun(fmt.Sprintf("Count(History(s1, %d))", protocol.MaxHistoryRecordsPerPair+1))) } + +func TestProcessCalculatedStreams_CoefficientBound(t *testing.T) { + t.Parallel() + + // Two inputs at opposite ends of the exponent range. Each is one digit + // wide, but their sum spans both exponents, so addition alone produces a + // coefficient far past MaxDecimalCoefficientBits. + wide := protocol.StreamAggregates{ + 1: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.New(1, 600))}, + 2: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.New(1, -600))}, + } + + t.Run("an over-wide value writes no aggregate", func(t *testing.T) { + t.Parallel() + + defs := llotypes.ChannelDefinitions{ + 1: historyChannel(medianStreams(1, 2), "Add(s1, s2)"), + } + ProcessCalculatedStreams(logger.Test(t), defs, wide, 1_000, protocol.NewOptsCache(), newStubHistoryReader()) + + assert.Empty(t, wide[999], "an unbounded coefficient must not reach the aggregates") + }) + + t.Run("a value within the bound is written", func(t *testing.T) { + t.Parallel() + + defs := llotypes.ChannelDefinitions{ + 1: historyChannel(medianStreams(1, 2), "Add(s1, s2)"), + } + aggregates := protocol.StreamAggregates{ + 1: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(2))}, + 2: {llotypes.AggregatorMedian: protocol.ToDecimal(decimal.NewFromInt(3))}, + } + ProcessCalculatedStreams(logger.Test(t), defs, aggregates, 1_000, protocol.NewOptsCache(), newStubHistoryReader()) + + got := aggregates[999][llotypes.AggregatorCalculated] + require.NotNil(t, got) + value, ok := got.(*protocol.Decimal) + require.True(t, ok) + assert.True(t, decimal.NewFromInt(5).Equal(value.Decimal()), "got %s", value.Decimal()) + }) +} diff --git a/llo/protocol/channel_definitions.go b/llo/protocol/channel_definitions.go index ceee013..2df61eb 100644 --- a/llo/protocol/channel_definitions.go +++ b/llo/protocol/channel_definitions.go @@ -236,6 +236,16 @@ func analyzeChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, cha // reads) and checked once at the end. var totalStreamEntries, totalOptsBytes int + // Every (streamID, aggregator) pair that will hold an aggregate. Distinct + // pairs, not stream entries: several channels naming the same pair cost one + // aggregate between them. Bounded by MaxPersistedAggregates, which is what + // the r/agg record can carry. + aggregatedPairs := make(map[streamAggregatorPair]struct{}, len(channelDefs)) + // Calculated streams declared across the set. Counted separately from the + // pairs above because they are recomputed each round rather than carried + // forward, so they cost precursor bytes but no r/agg bytes. + var totalCalculatedStreams int + uniqueStreamIDs := make(map[llotypes.StreamID]struct{}, len(channelDefs)) // Owners of every stream ID that will hold an aggregate: observed streams // come from the definitions, calculated streams from the expressions their @@ -285,6 +295,12 @@ func analyzeChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, cha admit(fmt.Errorf("ChannelDefinition with ID %d has stream %d with unknown aggregator %d", channelID, strm.StreamID, strm.Aggregator), channelID) } uniqueStreamIDs[strm.StreamID] = struct{}{} + // Pairs that aggregation actually produces a value for: calculated + // streams are recomputed each round rather than aggregated, and + // history backfill channels are not aggregated at all. + if strm.Aggregator != llotypes.AggregatorCalculated && cd.ReportFormat != llotypes.ReportFormatHistoryBackfill { + aggregatedPairs[streamAggregatorPair{streamID: strm.StreamID, aggregator: strm.Aggregator}] = struct{}{} + } // Calculated streams are derived from the opts that declare them and // are not stored on the definition, so anything listed here is // observed. Definitions written by older code may still carry their @@ -302,6 +318,7 @@ func analyzeChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, cha if facts.calculatedErr != nil { admit(fmt.Errorf("invalid ChannelDefinition with ID %d: %w", channelID, facts.calculatedErr), channelID) } + totalCalculatedStreams += len(facts.calculatedIDs) for _, streamID := range facts.calculatedIDs { if owner, ok := calculatedBy[streamID]; ok { admit(fmt.Errorf("ChannelDefinition with ID %d declares calculated stream %d already declared by channel %d", channelID, streamID, owner), channelID, owner) @@ -349,11 +366,18 @@ func analyzeChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, cha } } - // Whole-set budgets. Both are what the sizes of the channel-definitions - // record and the precursor actually depend on; see the limits they name. + // Whole-set budgets. Each is what the sizes of the channel-definitions + // record, the precursor and the carry-forward record actually depend on; + // see the limits they name. if totalStreamEntries > MaxTotalStreamEntries { admitSet(fmt.Errorf("too many stream entries across all channels, got: %d/%d", totalStreamEntries, MaxTotalStreamEntries)) } + if len(aggregatedPairs) > MaxPersistedAggregates { + admitSet(fmt.Errorf("too many aggregated (stream, aggregator) pairs across all channels, got: %d/%d", len(aggregatedPairs), MaxPersistedAggregates)) + } + if totalCalculatedStreams > MaxTotalCalculatedStreams { + admitSet(fmt.Errorf("too many calculated streams across all channels, got: %d/%d", totalCalculatedStreams, MaxTotalCalculatedStreams)) + } if totalOptsBytes > MaxTotalOptsBytes { admitSet(fmt.Errorf("too many opts bytes across all channels, got: %d/%d", totalOptsBytes, MaxTotalOptsBytes)) } @@ -362,6 +386,14 @@ func analyzeChannelDefinitions(codecs map[llotypes.ReportFormat]ReportCodec, cha return res } +// streamAggregatorPair identifies one aggregate. The aggregator is part of the +// identity because the same stream can be aggregated differently by different +// channels, and each way costs its own carried-forward value. +type streamAggregatorPair struct { + streamID llotypes.StreamID + aggregator llotypes.Aggregator +} + func sortedStreamIDs(m map[llotypes.StreamID]llotypes.ChannelID) []llotypes.StreamID { ids := make([]llotypes.StreamID, 0, len(m)) for id := range m { diff --git a/llo/protocol/channel_definitions_test.go b/llo/protocol/channel_definitions_test.go index c4441f7..39af360 100644 --- a/llo/protocol/channel_definitions_test.go +++ b/llo/protocol/channel_definitions_test.go @@ -439,6 +439,72 @@ func Test_VerifyChannelDefinitions_SizeBudgets(t *testing.T) { fmt.Sprintf("too many stream entries across all channels, got: %d/%d", 11*(MaxTotalStreamEntries/10), MaxTotalStreamEntries)) }) + t.Run("aggregated pairs across the set", func(t *testing.T) { + // One channel per aggregator, each listing the same streams, so the + // pair count grows while the unique-stream-ID cap and the entry budget + // stay untouched. + perAggregator := func(aggregators ...llotypes.Aggregator) llotypes.ChannelDefinitions { + defs := llotypes.ChannelDefinitions{} + for c, agg := range aggregators { + cd := llotypes.ChannelDefinition{Streams: make([]llotypes.Stream, 0, MaxObservationStreamValuesLength)} + for i := range MaxObservationStreamValuesLength { + cd.Streams = append(cd.Streams, llotypes.Stream{StreamID: llotypes.StreamID(i + 1), Aggregator: agg}) + } + defs[llotypes.ChannelID(c+1)] = cd + } + return defs + } + + atLimit := perAggregator(llotypes.AggregatorMedian, llotypes.AggregatorMode) + require.NoError(t, verifyAdmittingAll(codecs, atLimit)) + + over := perAggregator(llotypes.AggregatorMedian, llotypes.AggregatorMode, llotypes.AggregatorQuote) + require.EqualError(t, verifyAdmittingAll(codecs, over), + fmt.Sprintf("too many aggregated (stream, aggregator) pairs across all channels, got: %d/%d", 3*MaxObservationStreamValuesLength, MaxPersistedAggregates)) + }) + + t.Run("a pair costs one aggregate however many channels name it", func(t *testing.T) { + // The same (stream, aggregator) pair repeated across channels is one + // carried-forward value, so the pair budget must count it once even + // where the entry budget counts it every time. + defs := sharedStream(4, MaxPersistedAggregates/2) + require.NoError(t, verifyAdmittingAll(codecs, defs)) + }) + + t.Run("calculated streams across the set", func(t *testing.T) { + // Expression channels declaring distinct calculated stream IDs, spread + // across enough channels to stay under MaxChannelOptsBytes. The entry + // and opts budgets stay well clear: a calculated stream costs an ABI + // entry, not a stream entry. + const perChannel = 500 + calculated := func(total int) llotypes.ChannelDefinitions { + defs := llotypes.ChannelDefinitions{} + next := llotypes.StreamID(1_000_000) + for c := 0; total > 0; c++ { + n := min(perChannel, total) + total -= n + abi := make([]string, 0, n) + for range n { + abi = append(abi, fmt.Sprintf(`{"expressionStreamID":%d}`, next)) + next++ + } + defs[llotypes.ChannelID(c+1)] = llotypes.ChannelDefinition{ + ReportFormat: llotypes.ReportFormatEVMABIEncodeUnpackedExpr, + Streams: []llotypes.Stream{{StreamID: llotypes.StreamID(c + 1), Aggregator: llotypes.AggregatorMedian}}, + Opts: llotypes.ChannelOpts(fmt.Sprintf(`{"abi":[%s]}`, strings.Join(abi, ","))), + } + } + return defs + } + + atLimit := calculated(MaxTotalCalculatedStreams) + require.NoError(t, verifyAdmittingAll(codecs, atLimit)) + + over := calculated(MaxTotalCalculatedStreams + 1) + require.EqualError(t, verifyAdmittingAll(codecs, over), + fmt.Sprintf("too many calculated streams across all channels, got: %d/%d", MaxTotalCalculatedStreams+1, MaxTotalCalculatedStreams)) + }) + t.Run("total opts bytes across the set", func(t *testing.T) { // Every channel individually within MaxChannelOptsBytes; only the sum // exceeds the budget. diff --git a/llo/protocol/limits.go b/llo/protocol/limits.go index 090df36..de7eade 100644 --- a/llo/protocol/limits.go +++ b/llo/protocol/limits.go @@ -118,6 +118,11 @@ const ( // (measured by TestLimits_ChannelStateWorstCaseFitsPerKeyLimit). MaxTotalOptsBytes = 1 << 20 + // MaxTotalCalculatedStreams bounds how many calculated streams the whole + // definition set may declare. One expression produces one value, so this is + // the expression count too. + MaxTotalCalculatedStreams = MaxPersistedAggregates + // Stream history limits. // // A history "pair" is a (streamID, aggregator) tuple: the identity of one @@ -263,23 +268,7 @@ const ( // MaxPersistedAggregates bounds how many (streamID, aggregator) pairs may // carry a timestamped aggregate forward across rounds in the v3.1 r/agg - // record. Pairs are ordered by (streamID, aggregator) and those beyond the - // cap are not persisted: their streams simply lose carry-forward and are - // re-aggregated from fresh observations each round, which is a degradation - // rather than a halt. - // - // Without it the count is bounded only indirectly, by - // MaxObservationStreamValuesLength unique streams times the number of - // aggregators each may be aggregated by, which reaches tens of thousands of - // pairs -- past libocr's 2 MiB per-key limit for a single record, at which - // point every oracle's write is rejected and the round fails. - // - // 20_000 is twice MaxObservationStreamValuesLength, so it accommodates every - // observed stream being aggregated two ways -- more than any real - // configuration -- while holding the record to the low MiB range. Note this - // caps the pair count, not the bytes: a single record is only bounded once - // the decimal coefficient is bounded too (see MaxHistoryRecordBytes for the - // same caveat). + // record. MaxPersistedAggregates = 2 * MaxObservationStreamValuesLength // MaxHistoryBackfillObservations bounds the maximum number of diff --git a/llo/protocol/stream_value.go b/llo/protocol/stream_value.go index 51c5375..ef645d9 100644 --- a/llo/protocol/stream_value.go +++ b/llo/protocol/stream_value.go @@ -106,6 +106,12 @@ func checkObservedStreamValue(sv StreamValue) error { } } +// CheckDecimalCoefficient bounds the coefficient length of a decimal that is +// about to be written into state. See MaxDecimalCoefficientBits. +func CheckDecimalCoefficient(d decimal.Decimal) error { + return checkDecimalCoefficient(d) +} + // checkDecimalCoefficient bounds the coefficient length of a decimal carried by // an observation. See MaxDecimalCoefficientBits. func checkDecimalCoefficient(d decimal.Decimal) error { From a96c43d448b52e335024b7005bb60faf35f7a052 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Thu, 24 Sep 2026 15:09:35 +0100 Subject: [PATCH 37/40] llo/dev/v31: make the blob in-flight wait factor configurable --- llo/dev/v31/blobpump.go | 7 ++++--- llo/dev/v31/factory.go | 12 +++++++++++- 2 files changed, 15 insertions(+), 4 deletions(-) diff --git a/llo/dev/v31/blobpump.go b/llo/dev/v31/blobpump.go index 3937467..c3e942f 100644 --- a/llo/dev/v31/blobpump.go +++ b/llo/dev/v31/blobpump.go @@ -80,9 +80,10 @@ const ( closeTimeoutSlackMultiplier = 5 // minCloseTimeout fixes the minimum time Close waits waits for an in-flight cycle. minCloseTimeout = 1 * time.Second - // BlobInFlightWaitDivisor scales MaxDurationObservation into how long Take - // waits for a cycle that is already in flight to park. - BlobInFlightWaitDivisor = 8 + // defaultBlobInFlightWaitFactor scales MaxDurationObservation into how long + // Take waits for a cycle that is already in flight to park. Overridden by + // PluginFactoryParams.BlobInFlightWaitFactor. + defaultBlobInFlightWaitFactor = 8 ) // perOracleUnexpiredBlobCount derives the per-oracle unexpired-blob budget from diff --git a/llo/dev/v31/factory.go b/llo/dev/v31/factory.go index bb3bc02..0deac5a 100644 --- a/llo/dev/v31/factory.go +++ b/llo/dev/v31/factory.go @@ -50,6 +50,11 @@ type PluginFactoryParams struct { // budget (default: DefaultBlobObservationDurationMultiplier * // cfg.MaxDurationObservation). MaxDurationBlobObservation time.Duration + // BlobInFlightWaitFactor overrides defaultBlobInFlightWaitFactor if + // non-zero. Divides cfg.MaxDurationObservation into how long Take waits for + // a cycle already in flight to park, so a larger factor waits less. Must + // be >0. + BlobInFlightWaitFactor uint64 // MaxBlobSnapshotAge pins the wall-clock age at which a parked snapshot is // discarded. Left at zero the pump derives it from the round period it // measures, which is the only safe default: any bound derived from @@ -110,6 +115,11 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re aggregationFaultTolerance, cfg.F, 2*aggregationFaultTolerance+1, 2*cfg.F+1) } + blobInFlightWaitFactor := f.BlobInFlightWaitFactor + if blobInFlightWaitFactor == 0 { + blobInFlightWaitFactor = defaultBlobInFlightWaitFactor + } + blobObservationTimeout := f.MaxDurationBlobObservation if blobObservationTimeout <= 0 { blobObservationTimeout = DefaultBlobObservationDurationMultiplier * cfg.MaxDurationObservation @@ -149,7 +159,7 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re configDigest: cfg.ConfigDigest, verboseLogging: f.Config.VerboseLogging, observationTimeout: blobObservationTimeout, - inFlightWait: cfg.MaxDurationObservation / BlobInFlightWaitDivisor, + inFlightWait: cfg.MaxDurationObservation / time.Duration(blobInFlightWaitFactor), maxSnapshotAge: f.MaxBlobSnapshotAge, maxSnapshotRounds: maxSnapshotRounds, blobLifetimeRounds: blobLifetimeRounds, From 01d29cdbc28d9e27edeff9d8fe46af5ef0f52610 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 25 Sep 2026 10:24:42 +0100 Subject: [PATCH 38/40] llo/dev/v31: ensure republish of the carried aggregate when aggregation fails --- llo/dev/v31/plugin_test.go | 45 ++++++++++++++++++++++++++++++++++ llo/dev/v31/statetransition.go | 38 ++++++++++++++-------------- 2 files changed, 65 insertions(+), 18 deletions(-) diff --git a/llo/dev/v31/plugin_test.go b/llo/dev/v31/plugin_test.go index 017955d..73878b7 100644 --- a/llo/dev/v31/plugin_test.go +++ b/llo/dev/v31/plugin_test.go @@ -699,6 +699,51 @@ func Test_TimestampedAggregate_CarryForward(t *testing.T) { require.Equal(t, uint64(200), persisted.ObservedAtNanoseconds) } +func Test_TimestampedAggregate_CarryForwardOnAggregationFailure(t *testing.T) { + p := testPlugin(t) // F=1, contribution floor of 3 + defs := llotypes.ChannelDefinitions{1: {ReportFormat: llotypes.ReportFormatJSON, Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}}} + + tsv := func(ts uint64, v int64) protocol.StreamValue { + return &protocol.TimestampedStreamValue{ObservedAtNanoseconds: ts, StreamValue: protocol.ToDecimal(decimal.NewFromInt(v))} + } + + // A single contribution is below the floor, so the median aggregator fails. + starved := map[llotypes.StreamID][]protocol.StreamValue{100: {tsv(200, 9)}} + + t.Run("carried value is republished into the precursor", func(t *testing.T) { + carry := map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue{} + next := map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue{} + out := protocol.StreamAggregates{} + + // Round 1: enough contributions, establishes ts=100. + healthy := map[llotypes.StreamID][]protocol.StreamValue{100: {tsv(100, 5), tsv(100, 5), tsv(100, 5)}} + require.NoError(t, p.aggregate(carry, next, defs, healthy, out, nil, historyRequirements{}, 100)) + // Round 2: aggregation fails, the carried value stands in for it. + carry = next + next = map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue{} + out = protocol.StreamAggregates{} + require.NoError(t, p.aggregate(carry, next, defs, starved, out, nil, historyRequirements{}, 200)) + + got, ok := out[100][llotypes.AggregatorMedian].(*protocol.TimestampedStreamValue) + require.True(t, ok, "failed aggregation must still publish the carried value") + require.Equal(t, uint64(100), got.ObservedAtNanoseconds) + + persisted := next[100][llotypes.AggregatorMedian] + require.NotNil(t, persisted, "the carry must survive into the next round") + require.Equal(t, uint64(100), persisted.ObservedAtNanoseconds) + }) + + t.Run("without a carry the pair is absent", func(t *testing.T) { + carry := map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue{} + next := map[llotypes.StreamID]map[llotypes.Aggregator]*protocol.TimestampedStreamValue{} + out := protocol.StreamAggregates{} + + require.NoError(t, p.aggregate(carry, next, defs, starved, out, nil, historyRequirements{}, 200)) + require.NotContains(t, out[100], llotypes.AggregatorMedian) + require.Empty(t, next) + }) +} + func Test_Telemetry(t *testing.T) { ctx := tests.Context(t) otCh := make(chan *protocol.LLOOutcomeTelemetry, 8) diff --git a/llo/dev/v31/statetransition.go b/llo/dev/v31/statetransition.go index f659ef2..3b2ec52 100644 --- a/llo/dev/v31/statetransition.go +++ b/llo/dev/v31/statetransition.go @@ -522,17 +522,29 @@ func (p *Plugin) aggregate( continue } result, aerr := aggF(streamObservations[sid], p.minContributions()) + if aerr != nil { + // Aggregation failed: republish and keep the carried-forward + // value (if any) so a transient failure does not discard it. + // Without a carry the pair is simply absent from the precursor. + if prevTSV != nil { + m[agg] = prevTSV + keep(sid, agg, prevTSV) + } + continue + } + if result == nil { + // An aggregator may agree on no value at all, e.g. mode with an + // empty bucket. Leave the pair absent rather than writing a nil + // into the precursor, and keep any carried value. + if prevTSV != nil { + m[agg] = prevTSV + keep(sid, agg, prevTSV) + } + continue + } switch v := result.(type) { case *protocol.TimestampedStreamValue: - if aerr != nil { - // Aggregation failed: keep the carried-forward value (if any). - if prevTSV != nil { - m[agg] = prevTSV - keep(sid, agg, prevTSV) - } - continue - } if prevTSV == nil || v.ObservedAtNanoseconds > prevTSV.ObservedAtNanoseconds { // Strictly newer: adopt and persist. m[agg] = v @@ -543,16 +555,6 @@ func (p *Plugin) aggregate( keep(sid, agg, prevTSV) } default: - if aerr != nil { - // Ignore streams that cannot be aggregated; absent from the - // precursor. A previously-carried value for this pair is - // preserved so a transient aggregation failure does not - // discard it. - if prevTSV != nil { - keep(sid, agg, prevTSV) - } - continue - } m[agg] = result // Defensive: if this pair was previously timestamped but now // yields a non-timestamped value, drop the stale carry-forward From 9b7805b8a813e5a2986998d9a20e84f6d5534911 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 25 Sep 2026 12:57:29 +0100 Subject: [PATCH 39/40] llo/dev/v31: observableStreams skip backfills --- llo/dev/v31/blobpump_test.go | 4 ++++ llo/dev/v31/plugin.go | 6 +++--- 2 files changed, 7 insertions(+), 3 deletions(-) diff --git a/llo/dev/v31/blobpump_test.go b/llo/dev/v31/blobpump_test.go index 8d1cf42..43bd056 100644 --- a/llo/dev/v31/blobpump_test.go +++ b/llo/dev/v31/blobpump_test.go @@ -368,6 +368,10 @@ func Test_observableStreams(t *testing.T) { // Duplicate stream across channels must be listed once. 2: {ReportFormat: llotypes.ReportFormatJSON, Streams: []llotypes.Stream{{StreamID: 100, Aggregator: llotypes.AggregatorMedian}}}, 3: {ReportFormat: llotypes.ReportFormatJSON, Tombstone: true, Streams: []llotypes.Stream{{StreamID: 102, Aggregator: llotypes.AggregatorMedian}}}, + // Backfill reports are built from the channel opts, so its streams are + // not observed. Here the target (3) is tombstoned, which is the case + // where the backfill channel would otherwise keep 102 observed. + 4: {ReportFormat: llotypes.ReportFormatHistoryBackfill, Streams: []llotypes.Stream{{StreamID: 102, Aggregator: llotypes.AggregatorMedian}}}, }} require.ElementsMatch(t, []llotypes.StreamID{100}, observableStreams(state)) require.Empty(t, observableStreams(&kvState{})) diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index 1c275fd..0685625 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -224,8 +224,8 @@ func supportedReportFormats(codecs map[llotypes.ReportFormat]protocol.ReportCode } // observableStreams lists the streams a round should observe: every stream of -// every live channel, minus calculated streams (which are derived in -// StateTransition rather than observed). +// every live channel, minus calculated streams (which are derived in StateTransition) +// and history backfill channels (values come from their opts, not observed). func observableStreams(state *kvState) []llotypes.StreamID { if len(state.channelDefinitions) == 0 { return nil @@ -233,7 +233,7 @@ func observableStreams(state *kvState) []llotypes.StreamID { seen := make(map[llotypes.StreamID]struct{}) streams := make([]llotypes.StreamID, 0, len(state.channelDefinitions)) for _, cd := range state.channelDefinitions { - if cd.Tombstone { + if cd.Tombstone || cd.ReportFormat == llotypes.ReportFormatHistoryBackfill { continue } for _, strm := range cd.Streams { From 020ffabd5bd6f0feb343977e78e0dd25b8009117 Mon Sep 17 00:00:00 2001 From: Bruno Moura Date: Fri, 25 Sep 2026 16:48:09 +0100 Subject: [PATCH 40/40] llo/dev/v31: honour the observation context and bound the round period estimate --- llo/dev/v31/blobpump.go | 78 +++++++++------ llo/dev/v31/blobpump_test.go | 180 ++++++++++++++++++++++++++++------- llo/dev/v31/factory.go | 10 ++ llo/dev/v31/flow_test.go | 25 +++++ llo/dev/v31/plugin.go | 12 ++- 5 files changed, 239 insertions(+), 66 deletions(-) diff --git a/llo/dev/v31/blobpump.go b/llo/dev/v31/blobpump.go index c3e942f..884c834 100644 --- a/llo/dev/v31/blobpump.go +++ b/llo/dev/v31/blobpump.go @@ -54,6 +54,11 @@ const ( // stream values. Slack makes this a jitter guard, not a second freshness // gate, since staleness in rounds is already bounded by MaxSnapshotRounds. SnapshotAgeSlack = 2 + // DefaultMaxRoundPeriod is the ceiling on the measured round period. The + // measurement is wall clock, so a stall or a pause folds a gap far wider + // than the protocol ever schedules and inflates the derived age bound for + // several rounds after. Not using DeltaRound which can be a valid zero + DefaultMaxRoundPeriod = 2 * time.Second // MaxBlobLifetimeRounds bounds BlobLifetimeRounds. The pump broadcasts // roughly one blob per round, so the unexpired-blob budget declared to // libocr grows with the lifetime; this keeps that budget sane. @@ -100,6 +105,13 @@ type pumpInput struct { lifeCycleStage llotypes.LifeCycleStage } +// observable reports whether a cycle fed this input can produce a snapshot. +// An unset input (no round has published one yet), a round with nothing to +// observe, and a retired instance all park nothing by design. +func (in pumpInput) observable() bool { + return in.seqNr != 0 && len(in.streams) > 0 && in.lifeCycleStage != protocol.LifeCycleStageRetired +} + // blobSnapshot is one completed pump cycle: stream values already serialized, // broadcast as a blob, and reduced to the marshaled handle that goes on the // wire. @@ -149,11 +161,12 @@ type blobPump struct { mu sync.Mutex input pumpInput - // lastTakeAt and roundPeriod estimate the round cadence from the interval - // between consecutive Take calls (one per round), which is the only signal - // the plugin has: deltaRound is not part of the reporting plugin config. - lastTakeAt time.Time - roundPeriod time.Duration + // lastTakeAt, lastTakeSeqNr and roundPeriod estimate the round cadence from + // the interval between Take calls for consecutive sequence numbers. The + // estimate is bounded by maxRoundPeriod. + lastTakeAt time.Time + lastTakeSeqNr uint64 + roundPeriod time.Duration } // blobPumpParams is the pump's configuration, resolved by the factory. @@ -169,6 +182,8 @@ type blobPumpParams struct { maxSnapshotAge time.Duration // maxSnapshotRounds is the local freshness gate. See DefaultMaxSnapshotRounds. maxSnapshotRounds uint64 + // maxRoundPeriod caps the measured round period. See DefaultMaxRoundPeriod. + maxRoundPeriod time.Duration // blobLifetimeRounds is the remote fetchability bound. See DefaultBlobLifetimeRounds. blobLifetimeRounds uint64 // inFlightWait bounds how long Take waits for an in-flight cycle to park. @@ -242,24 +257,26 @@ func (p *blobPump) SetInput(in pumpInput) { // a single unusable snapshot from stalling the pump forever. The second return // value is the reason a snapshot was not returned, for logging. // -// A round that finds nothing parked while a cycle is running waits for -// inFlightWait for the cycle to park. -func (p *blobPump) Take(seqNr uint64) (*blobSnapshot, string) { +// A round that finds nothing parked while a cycle is running waits up to +// inFlightWait for the cycle to park, or until ctx is done. +func (p *blobPump) Take(ctx context.Context, seqNr uint64) (*blobSnapshot, string) { if !p.enabled() { - p.miss() return nil, "blob pump disabled" } p.mu.Lock() - p.recordRoundLocked(time.Now()) + p.recordRoundLocked(time.Now(), seqNr) ageLimit := p.snapshotAgeLimitLocked() + in := p.input p.mu.Unlock() defer p.kick() - snap, waited := p.takeReady(p.inFlightWait) + snap, waited := p.takeReady(ctx, p.inFlightWait) now := time.Now() switch { + case snap == nil && !in.observable(): + return nil, "nothing to observe this round" case snap == nil: p.miss() if waited || p.inFlight.Load() { @@ -279,17 +296,19 @@ func (p *blobPump) Take(seqNr uint64) (*blobSnapshot, string) { } // takeReady detaches the parked snapshot, waiting up to timeout for an -// in-flight cycle to park one. -func (p *blobPump) takeReady(timeout time.Duration) (*blobSnapshot, bool) { +// in-flight cycle to park one. The wait ends early when ctx is done or the +// pump shuts down. +func (p *blobPump) takeReady(ctx context.Context, timeout time.Duration) (*blobSnapshot, bool) { waited := false if timeout > 0 && p.inFlight.Load() { waited = true - ctx, cancel := context.WithTimeout(p.ctx, timeout) + waitCtx, cancel := context.WithTimeout(ctx, timeout) defer cancel() select { case snap := <-p.ready: return snap, waited - case <-ctx.Done(): + case <-waitCtx.Done(): + case <-p.ctx.Done(): } } @@ -304,12 +323,13 @@ func (p *blobPump) takeReady(timeout time.Duration) (*blobSnapshot, bool) { // park makes a snapshot available to take, replacing if we have a snap // that no round has taken. The newest snapshot has a higher priority. func (p *blobPump) park(snap *blobSnapshot) { - p.takeReady(0) + p.takeReady(context.Background(), 0) p.ready <- snap } -// miss records a round that found no usable snapshot. Sustained misses means -// this node is not contributing at all. Record misses and log when above MissStreakLogThreshold. +// miss records a round that found no usable snapshot when one was expected. +// Sustained misses means this node is not contributing at all. Record misses +// and log when above MissStreakLogThreshold. func (p *blobPump) miss() { p.misses.Add(1) if streak := p.missStreak.Add(1); streak >= MissStreakLogThreshold && streak%MissStreakLogThreshold == 0 { @@ -318,18 +338,15 @@ func (p *blobPump) miss() { } } -// recordRoundLocked folds the interval since the previous Take into the round -// period estimate. Take is called once per round, so consecutive calls measure -// the round cadence. -// -// Rounds that observe no streams skip Take entirely, so the next gap spans several -// round periods and overestimates; a stall inflates one gap badly, and estimation -// takes about four rounds to decay it. -// Both leave the age bound too generous for a while rather than too tight, -// which is the direction that cannot silently stop this node contributing. -func (p *blobPump) recordRoundLocked(now time.Time) { - if !p.lastTakeAt.IsZero() { +// recordRoundLocked builds the interval since the previous Take into the round +// period estimate, but only for subsequent rounds. +// Gaps are capped at maxRoundPeriod, so a stall cannot overestimate. +func (p *blobPump) recordRoundLocked(now time.Time, seqNr uint64) { + if !p.lastTakeAt.IsZero() && seqNr == p.lastTakeSeqNr+1 { gap := now.Sub(p.lastTakeAt) + if p.maxRoundPeriod > 0 && gap > p.maxRoundPeriod { + gap = p.maxRoundPeriod + } if p.roundPeriod == 0 { p.roundPeriod = gap } else { @@ -337,6 +354,7 @@ func (p *blobPump) recordRoundLocked(now time.Time) { } } p.lastTakeAt = now + p.lastTakeSeqNr = seqNr } // snapshotAgeLimitLocked resolves the wall-clock bound for this round. Zero @@ -404,7 +422,7 @@ func (p *blobPump) cycle() { in := p.input p.mu.Unlock() - if !p.enabled() || in.seqNr == 0 || len(in.streams) == 0 || in.lifeCycleStage == protocol.LifeCycleStageRetired { + if !p.enabled() || !in.observable() { return } diff --git a/llo/dev/v31/blobpump_test.go b/llo/dev/v31/blobpump_test.go index 43bd056..a8eb73f 100644 --- a/llo/dev/v31/blobpump_test.go +++ b/llo/dev/v31/blobpump_test.go @@ -57,14 +57,14 @@ func Test_blobPump_TakeKicksNextCycle(t *testing.T) { p := testPump(t, ds, bc, time.Minute) p.SetInput(pumpInputFor(2)) - snap, reason := p.Take(2) + snap, reason := p.Take(t.Context(), 2) require.Nil(t, snap) require.NotEmpty(t, reason) require.Eventually(t, func() bool { return p.Cycles() >= 1 }, tests.WaitTimeout(t), 10*time.Millisecond) p.SetInput(pumpInputFor(3)) - snap, reason = p.Take(3) + snap, reason = p.Take(t.Context(), 3) require.NotNil(t, snap, "reason: %s", reason) require.Equal(t, uint64(2), snap.forSeqNr) require.Equal(t, uint64(2+DefaultMaxSnapshotRounds), snap.usableBefore) @@ -97,19 +97,19 @@ func Test_blobPump_TakeIsSingleUse(t *testing.T) { p := testPump(t, mockDS(), newFakeBroadcaster(), time.Minute) p.SetInput(pumpInputFor(2)) require.Eventually(t, func() bool { - _, _ = p.Take(2) + _, _ = p.Take(t.Context(), 2) return p.Cycles() >= 1 }, tests.WaitTimeout(t), 10*time.Millisecond) require.Eventually(t, func() bool { - snap, _ := p.Take(2) + snap, _ := p.Take(t.Context(), 2) return snap != nil }, tests.WaitTimeout(t), 10*time.Millisecond) // The pump may have parked a fresh snapshot by now, so assert on the parked // slot directly rather than on a second Take. - p.takeReady(0) - snap, reason := p.Take(2) + p.takeReady(t.Context(), 0) + snap, reason := p.Take(t.Context(), 2) require.Nil(t, snap) require.NotEmpty(t, reason) } @@ -122,7 +122,7 @@ func Test_blobPump_RejectsStaleSnapshots(t *testing.T) { p := testPump(t, mockDS(), newFakeBroadcaster(), time.Minute) p.park(&blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now(), forSeqNr: 2, usableBefore: 4, expiresAt: 6}) - snap, reason := p.Take(4) + snap, reason := p.Take(t.Context(), 4) require.Nil(t, snap) require.Contains(t, reason, "too stale") require.Equal(t, uint64(1), p.Misses()) @@ -132,7 +132,7 @@ func Test_blobPump_RejectsStaleSnapshots(t *testing.T) { p := testPump(t, mockDS(), newFakeBroadcaster(), time.Nanosecond) p.park(&blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100}) - snap, reason := p.Take(3) + snap, reason := p.Take(t.Context(), 3) require.Nil(t, snap) require.Contains(t, reason, "too old") }) @@ -141,7 +141,7 @@ func Test_blobPump_RejectsStaleSnapshots(t *testing.T) { p := testPump(t, mockDS(), newFakeBroadcaster(), -1) p.park(&blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100}) - snap, _ := p.Take(3) + snap, _ := p.Take(t.Context(), 3) require.NotNil(t, snap, "with the age check disabled only maxSnapshotRounds bounds staleness") }) @@ -152,7 +152,7 @@ func Test_blobPump_RejectsStaleSnapshots(t *testing.T) { p := testPump(t, mockDS(), newFakeBroadcaster(), 0) p.park(&blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100}) - snap, reason := p.Take(3) + snap, reason := p.Take(t.Context(), 3) require.NotNil(t, snap, "reason: %s", reason) }) @@ -163,7 +163,7 @@ func Test_blobPump_RejectsStaleSnapshots(t *testing.T) { p.mu.Unlock() p.park(&blobSnapshot{handleBytes: []byte{1}, observedAt: time.Now().Add(-time.Hour), forSeqNr: 2, usableBefore: 100, expiresAt: 100}) - snap, reason := p.Take(3) + snap, reason := p.Take(t.Context(), 3) require.Nil(t, snap) require.Contains(t, reason, "too old") }) @@ -178,11 +178,11 @@ func Test_blobPump_ParksNothingOnFailure(t *testing.T) { ds.err = errors.New("bridge down") p := testPump(t, ds, newFakeBroadcaster(), time.Minute) p.SetInput(pumpInputFor(2)) - _, _ = p.Take(2) + _, _ = p.Take(t.Context(), 2) require.Eventually(t, func() bool { return ds.observeCount() >= 1 }, tests.WaitTimeout(t), 10*time.Millisecond) require.Zero(t, p.Cycles()) - snap, _ := p.Take(3) + snap, _ := p.Take(t.Context(), 3) require.Nil(t, snap) }) @@ -194,11 +194,11 @@ func Test_blobPump_ParksNothingOnFailure(t *testing.T) { }() p := testPump(t, mockDS(), bc, time.Minute) p.SetInput(pumpInputFor(2)) - _, _ = p.Take(2) + _, _ = p.Take(t.Context(), 2) require.Eventually(t, func() bool { return bc.Broadcasts() >= 1 }, tests.WaitTimeout(t), 10*time.Millisecond) require.Zero(t, p.Cycles()) - snap, _ := p.Take(3) + snap, _ := p.Take(t.Context(), 3) require.Nil(t, snap) }) } @@ -217,7 +217,7 @@ func Test_blobPump_SkipsIdleInput(t *testing.T) { p := testPump(t, ds, bc, time.Minute) p.SetInput(in) for i := 0; i < 3; i++ { - _, _ = p.Take(2) + _, _ = p.Take(t.Context(), 2) } // Give the loop a chance to run the kicked cycles. require.Never(t, func() bool { return ds.observeCount() > 0 || bc.Broadcasts() > 0 }, 100*time.Millisecond, 10*time.Millisecond) @@ -235,7 +235,7 @@ func Test_blobPump_SingleFlight(t *testing.T) { p.SetInput(pumpInputFor(2)) for i := 0; i < 10; i++ { - _, _ = p.Take(2) + _, _ = p.Take(t.Context(), 2) } require.Eventually(t, func() bool { return ds.started() >= 1 }, tests.WaitTimeout(t), 10*time.Millisecond) require.Never(t, func() bool { return ds.concurrent() > 1 }, 100*time.Millisecond, 10*time.Millisecond) @@ -280,7 +280,7 @@ func Test_blobPump_TakeWaitsForInFlightCycle(t *testing.T) { // The first Take only kicks the cycle: nothing is in flight yet, so there is // nothing for it to wait on. p.SetInput(pumpInputFor(2)) - snap, reason := p.Take(2) + snap, reason := p.Take(t.Context(), 2) require.Nil(t, snap) require.Equal(t, "no snapshot parked", reason) @@ -296,7 +296,7 @@ func Test_blobPump_TakeWaitsForInFlightCycle(t *testing.T) { close(ds.release) }() - snap, reason = p.Take(2) + snap, reason = p.Take(t.Context(), 2) require.NotNil(t, snap, "Take did not wait for the in-flight cycle: %s", reason) require.Empty(t, reason) } @@ -312,7 +312,7 @@ func Test_blobPump_TakeWaitFallsThroughOnTimeout(t *testing.T) { p.inFlightWait = 50 * time.Millisecond p.SetInput(pumpInputFor(2)) - _, _ = p.Take(2) + _, _ = p.Take(t.Context(), 2) select { case <-ds.entered: case <-time.After(tests.WaitTimeout(t)): @@ -320,7 +320,7 @@ func Test_blobPump_TakeWaitFallsThroughOnTimeout(t *testing.T) { } start := time.Now() - snap, reason := p.Take(2) + snap, reason := p.Take(t.Context(), 2) elapsed := time.Since(start) require.Nil(t, snap) require.Equal(t, "cycle in flight", reason) @@ -328,6 +328,118 @@ func Test_blobPump_TakeWaitFallsThroughOnTimeout(t *testing.T) { require.Less(t, elapsed, 10*p.inFlightWait, "Take waited well past its bound") } +// Test_blobPump_RoundPeriodOnlyMeasuresConsecutiveRounds pins the cadence +// estimate to gaps that really are one round wide: rounds that observe no +// streams skip Take, and folding the wider gap they leave would inflate the +// derived age bound. +func Test_blobPump_RoundPeriodOnlyMeasuresConsecutiveRounds(t *testing.T) { + p := testPump(t, mockDS(), newFakeBroadcaster(), time.Minute) + + base := time.Now() + p.recordRoundLocked(base, 2) + require.Zero(t, p.roundPeriod, "first round has nothing to measure against") + + // Round 5 skips rounds 3 and 4, so its gap spans three round periods. + p.recordRoundLocked(base.Add(300*time.Millisecond), 5) + require.Zero(t, p.roundPeriod, "gap across skipped rounds was folded in") + + p.recordRoundLocked(base.Add(400*time.Millisecond), 6) + require.Equal(t, 100*time.Millisecond, p.roundPeriod) + + // A repeated sequence number is not a new round either. + p.recordRoundLocked(base.Add(900*time.Millisecond), 6) + require.Equal(t, 100*time.Millisecond, p.roundPeriod) +} + +// Test_blobPump_RoundPeriodIsCapped covers a stalled round: the gap is wall +// clock, so without the cap one stall would inflate the derived age bound for +// several rounds after it. +func Test_blobPump_RoundPeriodIsCapped(t *testing.T) { + p := testPump(t, mockDS(), newFakeBroadcaster(), time.Minute) + p.maxRoundPeriod = 400 * time.Millisecond + + base := time.Now() + p.recordRoundLocked(base, 2) + p.recordRoundLocked(base.Add(time.Hour), 3) + require.Equal(t, p.maxRoundPeriod, p.roundPeriod, "a stalled round was folded in uncapped") + + // Uncapped, the estimate stays free to follow a slow DON. + cap := p.maxRoundPeriod + p.maxRoundPeriod = 0 + p.recordRoundLocked(base.Add(2*time.Hour), 4) + require.Greater(t, p.roundPeriod, cap) +} + +// Test_blobPump_ExpectedMissesAreNotCounted covers the rounds that lose their +// stream values by design: a pump with no transport wired, and a round whose +// input gives a cycle nothing to do. Neither is a miss, so the counters stay +// clean and the streak error stays reserved for a pump that should be +// producing and is not. +func Test_blobPump_ExpectedMissesAreNotCounted(t *testing.T) { + t.Run("disabled pump", func(t *testing.T) { + p := testPump(t, nil, nil, time.Minute) + + for i := 0; i < 2*MissStreakLogThreshold; i++ { + snap, reason := p.Take(t.Context(), 2) + require.Nil(t, snap) + require.Equal(t, "blob pump disabled", reason) + } + require.Zero(t, p.Misses()) + require.Zero(t, p.missStreak.Load()) + }) + + t.Run("nothing to observe", func(t *testing.T) { + p := testPump(t, mockDS(), newFakeBroadcaster(), time.Minute) + p.SetInput(pumpInput{streams: nil, seqNr: 2, lifeCycleStage: protocol.LifeCycleStageProduction}) + + snap, reason := p.Take(t.Context(), 2) + require.Nil(t, snap) + require.Equal(t, "nothing to observe this round", reason) + require.Zero(t, p.Misses()) + require.Zero(t, p.missStreak.Load()) + }) + + t.Run("retired", func(t *testing.T) { + p := testPump(t, mockDS(), newFakeBroadcaster(), time.Minute) + p.SetInput(pumpInput{streams: []llotypes.StreamID{100}, seqNr: 2, lifeCycleStage: protocol.LifeCycleStageRetired}) + + snap, reason := p.Take(t.Context(), 2) + require.Nil(t, snap) + require.Equal(t, "nothing to observe this round", reason) + require.Zero(t, p.Misses()) + require.Zero(t, p.missStreak.Load()) + }) +} + +// Test_blobPump_TakeWaitHonoursContext asserts a canceled round context ends +// the in-flight wait early: OCR3.1 does not bound Observation, so cancellation +// is the only thing that can cut this wait short. +func Test_blobPump_TakeWaitHonoursContext(t *testing.T) { + ds := &gatedDataSource{release: make(chan struct{}), entered: make(chan struct{})} + defer close(ds.release) + + p := testPump(t, ds, newFakeBroadcaster(), time.Minute) + p.inFlightWait = tests.WaitTimeout(t) + + p.SetInput(pumpInputFor(2)) + _, _ = p.Take(t.Context(), 2) + select { + case <-ds.entered: + case <-time.After(tests.WaitTimeout(t)): + t.Fatal("DataSource.Observe was never called") + } + + ctx, cancel := context.WithCancel(t.Context()) + cancel() + + start := time.Now() + snap, reason := p.Take(ctx, 2) + elapsed := time.Since(start) + require.Nil(t, snap) + require.Equal(t, "cycle in flight", reason) + require.Less(t, elapsed, p.inFlightWait/2, "Take ignored the canceled context") +} + // Test_blobPump_TakeWaitReportsCycleAfterItEnds pins the miss reason for a // round that waited: the cycle it waited on can finish empty and clear the // in-flight flag before the wait expires, and the round still belongs to that @@ -338,7 +450,7 @@ func Test_blobPump_TakeWaitReportsCycleAfterItEnds(t *testing.T) { p.inFlightWait = 500 * time.Millisecond p.SetInput(pumpInputFor(2)) - _, _ = p.Take(2) + _, _ = p.Take(t.Context(), 2) select { case <-ds.entered: case <-time.After(tests.WaitTimeout(t)): @@ -354,7 +466,7 @@ func Test_blobPump_TakeWaitReportsCycleAfterItEnds(t *testing.T) { // The reason is resolved inside Take, before its deferred kick starts the // next cycle, so it is the only assertable evidence here: reading inFlight // after Take returns would race that new cycle. - snap, reason := p.Take(2) + snap, reason := p.Take(t.Context(), 2) require.Nil(t, snap) require.Equal(t, "cycle in flight", reason) } @@ -386,7 +498,7 @@ func Test_blobPump_DisabledIsInert(t *testing.T) { p := testPump(t, ds, nil, time.Minute) p.SetInput(pumpInputFor(2)) for i := 0; i < 3; i++ { - snap, reason := p.Take(2) + snap, reason := p.Take(t.Context(), 2) require.Nil(t, snap) require.Equal(t, "blob pump disabled", reason) } @@ -398,7 +510,7 @@ func Test_blobPump_DisabledIsInert(t *testing.T) { bc := newFakeBroadcaster() p := testPump(t, nil, bc, time.Minute) p.SetInput(pumpInputFor(2)) - snap, reason := p.Take(2) + snap, reason := p.Take(t.Context(), 2) require.Nil(t, snap) require.Equal(t, "blob pump disabled", reason) require.Zero(t, bc.Broadcasts()) @@ -420,13 +532,13 @@ func Test_blobPump_SurvivesDataSourcePanic(t *testing.T) { p := testPump(t, ds, newFakeBroadcaster(), time.Minute) p.SetInput(pumpInputFor(2)) - _, _ = p.Take(2) + _, _ = p.Take(t.Context(), 2) require.Eventually(t, func() bool { return ds.calls.Load() >= 1 }, tests.WaitTimeout(t), 10*time.Millisecond) require.Zero(t, p.Cycles()) // Pump goroutine is still alive: a second kick still reaches the DataSource. p.SetInput(pumpInputFor(3)) - _, _ = p.Take(3) + _, _ = p.Take(t.Context(), 3) require.Eventually(t, func() bool { return ds.calls.Load() >= 2 }, tests.WaitTimeout(t), 10*time.Millisecond) // Take kicks before it returns, so the last kicked cycle may still be // running; it must unwind rather than leave the flag stuck. @@ -468,7 +580,7 @@ func Test_blobPump_CloseDoesNotHangOnStuckDataSource(t *testing.T) { p.Start() p.SetInput(pumpInputFor(2)) - p.Take(2) + p.Take(t.Context(), 2) select { case <-ds.entered: case <-time.After(tests.WaitTimeout(t)): @@ -516,12 +628,12 @@ func Test_blobPump_RetriesFailedBroadcast(t *testing.T) { p := testPump(t, ds, bc, time.Minute) p.SetInput(pumpInputFor(2)) - _, _ = p.Take(2) + _, _ = p.Take(t.Context(), 2) require.Eventually(t, func() bool { return p.Cycles() >= 1 }, tests.WaitTimeout(t), 10*time.Millisecond) require.EqualValues(t, 2, bc.attempts.Load(), "the failed attempt must be retried, once") - snap, reason := p.Take(3) + snap, reason := p.Take(t.Context(), 3) require.NotNil(t, snap, reason) require.EqualValues(t, 2, snap.forSeqNr, "the retry does not change the round the values were gathered for") } @@ -544,7 +656,7 @@ func Test_blobPump_BroadcastRetryRefreshesExpiry(t *testing.T) { } p = testPump(t, ds, bc, time.Minute) p.SetInput(pumpInputFor(2)) - _, _ = p.Take(2) + _, _ = p.Take(t.Context(), 2) require.Eventually(t, func() bool { return p.Cycles() >= 1 }, tests.WaitTimeout(t), 10*time.Millisecond) hints := inner.Hints() @@ -552,7 +664,7 @@ func Test_blobPump_BroadcastRetryRefreshesExpiry(t *testing.T) { require.Equal(t, ocr3_1types.BlobExpirationHintSequenceNumber{SeqNr: 5 + DefaultBlobLifetimeRounds}, hints[0]) // Fetchability moved forward; local freshness did not. - snap, reason := p.Take(3) + snap, reason := p.Take(t.Context(), 3) require.NotNil(t, snap, reason) require.EqualValues(t, 2+DefaultMaxSnapshotRounds, snap.usableBefore) require.EqualValues(t, 5+DefaultBlobLifetimeRounds, snap.expiresAt) @@ -567,10 +679,10 @@ func Test_blobPump_BroadcastRetriesAreBounded(t *testing.T) { p := testPump(t, ds, bc, time.Minute) p.SetInput(pumpInputFor(2)) - _, _ = p.Take(2) + _, _ = p.Take(t.Context(), 2) require.Eventually(t, func() bool { return bc.attempts.Load() >= BlobBroadcastAttempts }, tests.WaitTimeout(t), 10*time.Millisecond) require.Zero(t, p.Cycles()) - snap, _ := p.Take(3) + snap, _ := p.Take(t.Context(), 3) require.Nil(t, snap) } diff --git a/llo/dev/v31/factory.go b/llo/dev/v31/factory.go index 0deac5a..58711dc 100644 --- a/llo/dev/v31/factory.go +++ b/llo/dev/v31/factory.go @@ -62,6 +62,10 @@ type PluginFactoryParams struct { // every snapshot. A negative value disables the check, leaving // MaxSnapshotRounds as the only staleness bound. MaxBlobSnapshotAge time.Duration + // MaxRoundPeriod overrides DefaultMaxRoundPeriod if non-zero. Bounds the + // round period the pump measures, so a stalled round cannot inflate the + // derived snapshot age bound. Must be set above the DON real round cadence. + MaxRoundPeriod time.Duration } func NewPluginFactory(p PluginFactoryParams) *PluginFactory { @@ -120,6 +124,11 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re blobInFlightWaitFactor = defaultBlobInFlightWaitFactor } + maxRoundPeriod := f.MaxRoundPeriod + if maxRoundPeriod <= 0 { + maxRoundPeriod = DefaultMaxRoundPeriod + } + blobObservationTimeout := f.MaxDurationBlobObservation if blobObservationTimeout <= 0 { blobObservationTimeout = DefaultBlobObservationDurationMultiplier * cfg.MaxDurationObservation @@ -162,6 +171,7 @@ func (f *PluginFactory) NewReportingPlugin(ctx context.Context, cfg ocr3types.Re inFlightWait: cfg.MaxDurationObservation / time.Duration(blobInFlightWaitFactor), maxSnapshotAge: f.MaxBlobSnapshotAge, maxSnapshotRounds: maxSnapshotRounds, + maxRoundPeriod: maxRoundPeriod, blobLifetimeRounds: blobLifetimeRounds, }) p.pump.Start() diff --git a/llo/dev/v31/flow_test.go b/llo/dev/v31/flow_test.go index ea372d9..e763c1a 100644 --- a/llo/dev/v31/flow_test.go +++ b/llo/dev/v31/flow_test.go @@ -236,6 +236,30 @@ func Test_Factory_NewReportingPlugin_aggregationFaultTolerance(t *testing.T) { }) } +func Test_Factory_MaxRoundPeriod(t *testing.T) { + ctx := tests.Context(t) + base := ocr3types.ReportingPluginConfig{N: 4, F: 1, ConfigDigest: ocrtypes.ConfigDigest{9}, OffchainConfig: mustEncodeOffchainConfig(t, 1)} + + newPump := func(t *testing.T, params PluginFactoryParams, cfg ocr3types.ReportingPluginConfig) *blobPump { + t.Helper() + params.OnchainConfigCodec = mockOnchainConfigCodec{} + params.Logger = logger.Test(t) + p, _, err := NewPluginFactory(params).NewReportingPlugin(ctx, cfg, nil) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, p.Close()) }) + return p.(*Plugin).pump + } + + t.Run("defaulted", func(t *testing.T) { + require.Equal(t, DefaultMaxRoundPeriod, newPump(t, PluginFactoryParams{}, base).maxRoundPeriod) + }) + + t.Run("override wins", func(t *testing.T) { + pump := newPump(t, PluginFactoryParams{MaxRoundPeriod: time.Second}, base) + require.Equal(t, time.Second, pump.maxRoundPeriod) + }) +} + func Test_Factory_NewReportingPlugin(t *testing.T) { ctx := tests.Context(t) f := NewPluginFactory(PluginFactoryParams{ @@ -259,6 +283,7 @@ func Test_Factory_NewReportingPlugin(t *testing.T) { require.NotNil(t, pl.pump) require.Equal(t, uint64(DefaultMaxSnapshotRounds), pl.pump.maxSnapshotRounds) require.Equal(t, uint64(DefaultBlobLifetimeRounds), pl.pump.blobLifetimeRounds) + require.Equal(t, DefaultMaxRoundPeriod, pl.pump.maxRoundPeriod) require.NoError(t, pl.Close()) } diff --git a/llo/dev/v31/plugin.go b/llo/dev/v31/plugin.go index 0685625..51a6ae2 100644 --- a/llo/dev/v31/plugin.go +++ b/llo/dev/v31/plugin.go @@ -102,7 +102,11 @@ func (p *Plugin) Query(ctx context.Context, seqNr uint64, _ ocr3_1types.KeyValue // values most recently gathered by the blob pump. // The per-round BlobBroadcastFetcher is unused: broadcasting happens in the blob // pump, which holds the identical fetcher handed to the factory. -func (p *Plugin) Observation(_ context.Context, seqNr uint64, _ ocrtypes.AttributedQuery, kvReader ocr3_1types.KeyValueStateReader, _ ocr3_1types.BlobBroadcastFetcher) (ocrtypes.Observation, error) { +// +// MaxDurationObservation is not enforced by OCR3.1 (it only logs a warning), so +// ctx carries no observation deadline; it is honoured for cancellation, which is +// what bounds the wait for an in-flight blob pump cycle. +func (p *Plugin) Observation(ctx context.Context, seqNr uint64, _ ocrtypes.AttributedQuery, kvReader ocr3_1types.KeyValueStateReader, _ ocr3_1types.BlobBroadcastFetcher) (ocrtypes.Observation, error) { if seqNr < 1 { return nil, fmt.Errorf("got invalid seqnr=%d, must be >=1", seqNr) } else if seqNr == 1 { @@ -110,6 +114,10 @@ func (p *Plugin) Observation(_ context.Context, seqNr uint64, _ ocrtypes.Attribu return nil, nil } + if err := ctx.Err(); err != nil { + return nil, fmt.Errorf("observation canceled: %w", err) + } + state, err := loadColdKVState(kvReader, p.ChannelCache) if err != nil { return nil, fmt.Errorf("failed to load KV state: %w", err) @@ -178,7 +186,7 @@ func (p *Plugin) Observation(_ context.Context, seqNr uint64, _ ocrtypes.Attribu p.pump.SetInput(pumpInput{streams: streams, seqNr: seqNr, lifeCycleStage: state.lifeCycleStage}) var reason string - if snap, reason = p.pump.Take(seqNr); snap != nil { + if snap, reason = p.pump.Take(ctx, seqNr); snap != nil { handles = append(handles, snap.handleBytes) } else { p.Logger.Debugw("No usable stream-value snapshot for this round", "stage", "Observation", "seqNr", seqNr, "reason", reason, "misses", p.pump.Misses(), "cycles", p.pump.Cycles())