Skip to content

Commit fbc87b5

Browse files
committed
fix(billing): let a slow ledger read finish instead of blocking execution at the singleflight default
1 parent 3aa603f commit fbc87b5

2 files changed

Lines changed: 55 additions & 5 deletions

File tree

‎apps/sim/lib/billing/core/usage-gate-cache.test.ts‎

Lines changed: 35 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -16,6 +16,7 @@ import {
1616
checkIngestionUsageLimits,
1717
checkSearchUsageLimits,
1818
resetUsageGateCache,
19+
USAGE_GATE_SETTLE_TIMEOUT_MS,
1920
USAGE_GATE_TTL_MS,
2021
} from '@/lib/billing/core/usage-gate-cache'
2122

@@ -178,4 +179,38 @@ describe('checkExecutionUsageLimits', () => {
178179
await checkExecutionUsageLimits(ATTRIBUTION)
179180
expect(mockCheck).toHaveBeenCalledTimes(2)
180181
})
182+
183+
it('waits for a slow ledger read past the singleflight default instead of blocking', async () => {
184+
vi.useFakeTimers()
185+
try {
186+
mockCheck.mockReturnValueOnce(
187+
new Promise((resolve) => setTimeout(() => resolve({ isExceeded: false }), 45_000))
188+
)
189+
const pending = checkExecutionUsageLimits(ATTRIBUTION)
190+
await vi.advanceTimersByTimeAsync(45_000)
191+
await expect(pending).resolves.toEqual({ isExceeded: false })
192+
expect(mockCheck).toHaveBeenCalledTimes(1)
193+
} finally {
194+
vi.useRealTimers()
195+
}
196+
})
197+
198+
it('gives up on a read that never answers at the gate deadline, then reads fresh', async () => {
199+
vi.useFakeTimers()
200+
try {
201+
mockCheck.mockReturnValueOnce(new Promise(() => {}))
202+
const hung = checkExecutionUsageLimits(ATTRIBUTION)
203+
const rejection = expect(hung).rejects.toThrow(
204+
`did not settle within ${USAGE_GATE_SETTLE_TIMEOUT_MS}ms`
205+
)
206+
await vi.advanceTimersByTimeAsync(USAGE_GATE_SETTLE_TIMEOUT_MS)
207+
await rejection
208+
await expect(checkExecutionUsageLimits(ATTRIBUTION)).resolves.toEqual({
209+
isExceeded: false,
210+
})
211+
expect(mockCheck).toHaveBeenCalledTimes(2)
212+
} finally {
213+
vi.useRealTimers()
214+
}
215+
})
181216
})

‎apps/sim/lib/billing/core/usage-gate-cache.ts‎

Lines changed: 20 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -20,6 +20,19 @@ import { coalesceLocally } from '@/lib/concurrency/singleflight'
2020
*/
2121
export const USAGE_GATE_TTL_MS = 5 * 60 * 1000
2222

23+
/**
24+
* How long a coalesced ledger read may take before its callers give up on it. The read sums a
25+
* payer's ledger for the billing period, which for a large organization is millions of rows and,
26+
* from a cold cache or under heavy I/O, takes longer than the singleflight default of 30 s. That
27+
* default exists to bound a hung producer, and a slow read is not a hung one: the database bounds
28+
* every statement with its own timeout, after which the read fails on its own and the failure is
29+
* reported rather than cached. The deadline therefore sits above any statement ceiling the
30+
* deployment applies, so only a connection that never answers is given up on. A shorter deadline
31+
* fails the callers while the read is still running, and the next caller starts a second read
32+
* of the same ledger alongside it.
33+
*/
34+
export const USAGE_GATE_SETTLE_TIMEOUT_MS = 120_000
35+
2336
/**
2437
* Recent gate answers, admitted and refused, with `LRUCache` supplying the TTL
2538
* and the size bound. Each entry point decides which of them it may serve.
@@ -62,9 +75,9 @@ function gateKey(attribution: BillingAttributionSnapshot): string {
6275
* the cache. A read that throws writes nothing.
6376
*
6477
* `coalesceLocally` collapses concurrent misses onto one ledger read and bounds
65-
* a hung read at its settle deadline. The write stays on the value this caller
66-
* received, so a producer that timed out and later resolved cannot overwrite a
67-
* fresher answer.
78+
* a hung read at {@link USAGE_GATE_SETTLE_TIMEOUT_MS}. The write stays on the
79+
* value this caller received, so a producer that timed out and later resolved
80+
* cannot overwrite a fresher answer.
6881
*
6982
* There is deliberately no invalidator: usage and limit changes land in other
7083
* processes (execution workers, Stripe webhooks), so the TTL is the real bound.
@@ -77,8 +90,10 @@ async function checkUsageLimitsThroughCache(
7790
const cached = gateCache.get(key)
7891
if (cached !== undefined && (cacheRefusals || !cached.isExceeded)) return cached
7992

80-
const result = await coalesceLocally(`usage-gate:${key}`, () =>
81-
checkAttributedUsageLimits(attribution)
93+
const result = await coalesceLocally(
94+
`usage-gate:${key}`,
95+
() => checkAttributedUsageLimits(attribution),
96+
USAGE_GATE_SETTLE_TIMEOUT_MS
8297
)
8398
if (cacheRefusals || !result.isExceeded) gateCache.set(key, result)
8499
return result

0 commit comments

Comments
 (0)