11import { db } from '@sim/db'
2+ import { createLogger } from '@sim/logger'
3+ import { getErrorMessage } from '@sim/utils/errors'
4+ import { randomInt } from '@sim/utils/random'
25import { LRUCache } from 'lru-cache'
36import {
47 type BillingEntity ,
58 getBillingPeriodUsageCost ,
69 type UsageQueryPeriod ,
710} from '@/lib/billing/core/usage-log'
11+ import { getRedisClient } from '@/lib/core/config/redis'
12+ import { withinDeadline } from '@/lib/core/utils/deadline'
813import type { DbClient } from '@/lib/db/types'
914
15+ const logger = createLogger ( 'ReportingUsageCache' )
16+
1017/**
1118 * How long a reporting-window usage sum is served before it is summed again.
1219 *
@@ -18,23 +25,97 @@ import type { DbClient } from '@/lib/db/types'
1825 * an admission gate lets a payer run on for at most this long past their limit, and a sum at or
1926 * above the limit is a refusal the true sum would also give. Thirty seconds keeps that overrun
2027 * small against a year-long allowance while turning a per-event scan into one per window.
28+ *
29+ * A sum is held both in Redis, shared by every process, and in each process that reads it, so a
30+ * served sum can be up to twice this old (plus the Redis expiry's jitter).
2131 */
2232export const REPORTING_USAGE_CACHE_TTL_MS = 30_000
2333
34+ /** Redis expiry is jittered by up to this much, so payers summed together do not expire together. */
35+ const SHARED_TTL_JITTER_MS = 5_000
36+
37+ /**
38+ * How long a read waits on Redis before summing the ledger instead. The shared client queues
39+ * commands while disconnected and has long timeouts, so without this a Redis outage would stall
40+ * every gate behind it rather than cost one sum.
41+ */
42+ const SHARED_READ_TIMEOUT_MS = 250
43+
44+ /** Bump when a stored sum's meaning changes; old entries are then ignored. */
45+ const SHARED_KEY_VERSION = 'v1'
46+
2447/** A usage window known to be an enterprise reporting window — the only kind this cache serves. */
2548type ReportingQueryPeriod = UsageQueryPeriod & { source : 'reporting' }
2649
50+ function sharedReportingUsageKey ( key : string ) : string {
51+ return `usage:reporting:${ SHARED_KEY_VERSION } :${ key } `
52+ }
53+
54+ /**
55+ * A sum another process stored, or `undefined` when there is none to use. Redis being absent,
56+ * slow, or failing, and a value that is not a non-negative number, are all misses: the caller
57+ * sums the ledger, so the cache can cost a read its latency but never its answer.
58+ */
59+ async function readSharedReportingUsageCost ( key : string ) : Promise < number | undefined > {
60+ const redis = getRedisClient ( )
61+ if ( ! redis ) return undefined
62+ try {
63+ const stored = await withinDeadline (
64+ ( ) => redis . get ( sharedReportingUsageKey ( key ) ) ,
65+ Date . now ( ) + SHARED_READ_TIMEOUT_MS
66+ )
67+ if ( stored === null ) return undefined
68+ const cost = Number ( stored )
69+ if ( stored . trim ( ) !== '' && Number . isFinite ( cost ) && cost >= 0 ) return cost
70+ logger . warn ( 'Discarding unreadable shared reporting usage' , { key } )
71+ } catch ( error ) {
72+ logger . warn ( 'Shared reporting usage read failed; summing the ledger' , {
73+ error : getErrorMessage ( error ) ,
74+ } )
75+ }
76+ return undefined
77+ }
78+
79+ /** Fire-and-forget: a read never waits on, or fails because of, the shared write. */
80+ function writeSharedReportingUsageCost ( key : string , cost : number ) : void {
81+ const redis = getRedisClient ( )
82+ if ( ! redis ) return
83+ const ttlMs = REPORTING_USAGE_CACHE_TTL_MS + randomInt ( 0 , SHARED_TTL_JITTER_MS )
84+ redis . set ( sharedReportingUsageKey ( key ) , String ( cost ) , 'PX' , ttlMs ) . catch ( ( error : unknown ) => {
85+ logger . warn ( 'Shared reporting usage write failed' , { error : getErrorMessage ( error ) } )
86+ } )
87+ }
88+
89+ /**
90+ * The sum from Redis when another process stored one, else the ledger's exact sum, stored for
91+ * the others. Trigger.dev runs each task in a fresh process, so the in-process cache alone is
92+ * always cold there; the shared value is what spares those runs the scan.
93+ */
94+ async function sumReportingUsageCost (
95+ key : string ,
96+ entity : BillingEntity ,
97+ period : ReportingQueryPeriod
98+ ) : Promise < number > {
99+ const shared = await readSharedReportingUsageCost ( key )
100+ if ( shared !== undefined ) return shared
101+ const cost = await getBillingPeriodUsageCost ( entity , period )
102+ writeSharedReportingUsageCost ( key , cost )
103+ return cost
104+ }
105+
27106/**
28- * Sums shared across callers, one per payer and window. Every key is an enterprise payer's
29- * current window, a few dozen bytes each, so the ceiling sits far above any process's working
30- * set and only backstops memory; an eviction inside the TTL costs one extra sum.
107+ * Sums held by this process, one per payer and window, in front of the shared Redis value. Every
108+ * key is an enterprise payer's current window, a few dozen bytes each, so the ceiling sits far
109+ * above any process's working set and only backstops memory; an eviction inside the TTL costs
110+ * one extra read.
31111 *
32- * `fetchMethod` coalesces concurrent misses onto one sum . A rejected sum is evicted rather than
112+ * `fetchMethod` coalesces concurrent misses onto one read . A rejected sum is evicted rather than
33113 * stored (`noDeleteOnFetchRejection` and `allowStaleOnFetchRejection` stay off), so every caller
34114 * of that read sees the error it would have seen uncached and the next call sums again. There is
35115 * no settle deadline: the sum runs under the ledger's own `statement_timeout`, so the database
36- * ends a slow one. There is deliberately no invalidator either — usage is written by execution
37- * workers in other processes, so the TTL is the real bound.
116+ * ends a slow one. There is deliberately no invalidator or lock either — usage is written by
117+ * execution workers in other processes, so the TTL is the real bound, and concurrent misses in
118+ * different processes each sum once.
38119 */
39120const reportingUsageCache = new LRUCache <
40121 string ,
@@ -43,8 +124,8 @@ const reportingUsageCache = new LRUCache<
43124> ( {
44125 max : 1_000 ,
45126 ttl : REPORTING_USAGE_CACHE_TTL_MS ,
46- fetchMethod : ( _key , _stale , { context } ) =>
47- getBillingPeriodUsageCost ( context . entity , context . period ) ,
127+ fetchMethod : ( key , _stale , { context } ) =>
128+ sumReportingUsageCost ( key , context . entity , context . period ) ,
48129} )
49130
50131/**
@@ -72,7 +153,7 @@ async function readCachedReportingUsageCost(
72153/**
73154 * Period usage for a soft reader: an admission check, a display, or a level-triggered
74155 * notification that tolerates the cache's bounded under-count. Enterprise reporting windows are
75- * served from the shared cache for up to {@link REPORTING_USAGE_CACHE_TTL_MS}, since their
156+ * served from the shared cache for up to twice {@link REPORTING_USAGE_CACHE_TTL_MS}, since their
76157 * year-long sum is the expensive one; every other period is summed exactly, as before. A read on
77158 * a caller's own executor (a transaction or a replica) keeps its own snapshot and is never shared.
78159 * Never use it for invoicing, cycle close, an edge-triggered decision, or a read that must see its
0 commit comments