vpp-ai-platform/packages/runtime/test/m4.test.ts
Thomas Bayes 381b6d3521
Some checks failed
ci / typescript (push) Has been cancelled
ci / python (push) Has been cancelled
ci / evals (push) Has been cancelled
M5: shadow run, kill-switch hierarchy L0–L4, KPI dashboard, shadow replay CLI
Shadow run (ROADMAP M5, phase-1 acceptance form): SHADOW runtime mode records
bids without submitting them, simulates the market answer from the actual
day-ahead clearing price, dispatches to the simulation gateway, runs the D+1
review, and scores every day shadow-vs-human-vs-hindsight (ShadowDayRecord)
with a lineage-completeness audit. KpiReport regenerated after every day per
docs/12 §4 definitions (C1/C2 placeholders as named config).

Kill switches (docs/13 §8): BreakerService with L0 permit revocation, L1
envelope suspension, L2 loss breaker (mark-to-market, reduce-only bids),
L3 channel breaker (bids fall back to the file channel, dispatch BLOCKED),
L4 AI-off (templates run, no Proposal created); abnormal-day protocol on
EXTREME situations; per-level authority (B8 placeholder); auditable drill.

Runtime: shadow-close workflow, shadow schedule entries, live-data ingestion
through the quality gate, human-bid ingestion, breaker/KPI/shadow endpoints
and insight cards, replay CLI (npm run shadow). Ledger, time series and
streak counters are file-backed so a multi-week run survives restarts.

Domain: HumanBidRecord, ShadowDayRecord, KpiReport, BreakerRecord, SHADOW
receipt channel; contracts, fixtures and pydantic models regenerated.
Services: L2 metrics moved from evals so the shadow run and the harness
share one implementation.

Fixes: envelope/permit validity compared ISO timestamps as strings
('…00Z' vs '…00.000Z'); L2 baseline was stale since M4 (skill_versions only,
metrics unchanged) — rewritten from the live service.

Docs: docs/14 shadow-run runbook (timeline, breaker trigger/authority/
recovery, KPI definitions as implemented); README and CLAUDE.md status.

Tests: 21 consecutive shadow days with complete lineage, KPI report, WIDEN
recommendation produced but not acted on; restart durability; drill; L2/L3/L4
and abnormal-day paths; API.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017wrZgPL9LoKaD69BpEQU4v
2026-09-02 22:54:14 -04:00

160 lines
9.9 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

import { describe, expect, it } from 'vitest'
import type { AwardNotice, Envelope } from '@vpp/domain'
import { insightCards } from '../src/cards.js'
import { TriggerService } from '../src/trigger.js'
import { MARKET_DATE, envelope, harness } from './helpers.js'
const APPROVER = { id: 'user-ops-lead', role: 'ops-lead' }
const flat = (v: string) => ({ interval_minutes: 15 as const, date: MARKET_DATE, values: Array(96).fill(v) as string[] })
const dispatchEnvelope = (bounds: Record<string, string>): Envelope =>
envelope(bounds, { id: 'env-dispatch-001', scope: { proposal_type: 'DISPATCH_PLAN', timescales: ['DAY_AHEAD'], resource_set: 'pool-hubei-01' } })
/** Runs D-1 08:00 (bid) and returns the bid's daily energy so the award can answer it. */
async function bidDay(h: Awaited<ReturnType<typeof harness>>) {
h.rt.ctx.envelopes.register(envelope({ max_energy_mwh: '1500.0' }))
const run = await h.rt.mastra.getWorkflow('dayAheadBid').createRun()
const res = await run.start({ inputData: { market_date: MARKET_DATE } })
if (res.status !== 'success') throw new Error(res.status)
return res.result
}
const awardFor = (digest: string, mwhPerInterval: string): AwardNotice => ({
id: `award-${MARKET_DATE}`,
market_date: MARKET_DATE,
bid_proposal_digest: digest,
awarded_mwh: flat(mwhPerInterval),
clearing_price_yuan_per_mwh: flat('418.30'),
received_at: '2026-03-14T08:00:00Z',
})
describe('docs/07 D-1 16:00: award → resource agent → dispatch plan through the safety chain', () => {
it('records the award in the ledger and releases a dispatch plan to the simulation gateway', async () => {
const h = await harness({ llm: null })
const bid = await bidDay(h)
h.rt.ctx.envelopes.register(dispatchEnvelope({ max_total_mw: '30.0' }))
const t = new TriggerService(h.rt)
const out = await t.onAward(awardFor(bid.proposal!.digest, '3.0')) // 12 MW target, resource offers 24 MW
expect(out.status).toBe('success')
const r = out.result!
expect(r.lifecycle?.outcome).toBe('RELEASED')
expect(r.proposal!.type).toBe('DISPATCH_PLAN')
expect(r.shortfall_mwh).toBe('0.000')
expect(h.rt.ctx.ledger.read().entries.map((e) => e.kind)).toEqual(['CONTRACT', 'BID_SUBMITTED', 'AWARD'])
expect(h.skills.calls.filter((c) => c === 'potential' || c === 'dispatch')).toEqual(['potential', 'dispatch'])
expect(h.rt.ctx.simulationGateway.receipts.list()).toHaveLength(1)
// P2 for dispatch numbers: allocations are byte-identical to the tool output.
const lin = h.rt.ctx.caseDesk.expandLineage(r.proposal_id!)
expect(lin.tool_calls.map((c) => c.tool)).toEqual(['potential-assessment', 'dispatch-optimization'])
expect((lin.tool_calls[1]!.outputs as { allocations: unknown }).allocations).toEqual(r.proposal!.type === 'DISPATCH_PLAN' ? r.proposal!.payload.allocations : null)
await h.rt.close()
})
it('a shortfall (award beyond assessed capacity) alerts in power-balance simulation and goes to a human', async () => {
const h = await harness({ llm: null })
const bid = await bidDay(h)
h.rt.ctx.envelopes.register(dispatchEnvelope({ max_total_mw: '100.0' }))
const out = await new TriggerService(h.rt).onAward(awardFor(bid.proposal!.digest, '10.0')) // 40 MW target > 24 MW
expect(out.status).toBe('success')
expect(out.result!.lifecycle_status).toBe('suspended')
expect(Number(out.result!.shortfall_mwh)).toBeGreaterThan(0)
expect(h.rt.ctx.caseDesk.inbox()[0]!.reasons[0]).toMatch(/short/)
await h.rt.close()
})
})
describe('docs/07 D+1: execution feedback → review → writebacks', () => {
async function throughExecution(h: Awaited<ReturnType<typeof harness>>, fulfilment: number) {
const bid = await bidDay(h)
h.rt.ctx.envelopes.register(dispatchEnvelope({ max_total_mw: '30.0' }))
const t = new TriggerService(h.rt)
const out = await t.onAward(awardFor(bid.proposal!.digest, '3.0'))
const digest = out.result!.proposal!.digest
t.onExecutionReports(h.rt.ctx.simulationGateway.execute(digest, { 'agg-unit-01': fulfilment }, '2026-03-16T01:00:00Z'))
const review = await t.onMetering({ id: `meter-${MARKET_DATE}`, market_date: MARKET_DATE, metered_mw: flat((12 * fulfilment).toFixed(3)), actual_price_yuan_per_mwh: flat('421.00'), received_at: '2026-03-16T02:00:00Z' })
return { t, review: review.result!, bid }
}
it('a ReviewFinding measurably updates the resource reliability score and writes semantic memory', async () => {
const h = await harness({ llm: null })
expect(h.rt.ctx.resources.get('res-storage-01')!.value.reliability_score).toBe('0.9')
const { review } = await throughExecution(h, 0.8)
expect(review.findings).toHaveLength(1)
const f = review.findings[0]!
expect(f.kind).toBe('RESPONSE')
expect(review.updated_profiles).toEqual([{ resource_id: 'res-storage-01', reliability_score: '0.880' }])
expect(h.rt.ctx.resources.get('res-storage-01')!.value.reliability_score).toBe('0.880')
expect(review.memory_entries).toBeGreaterThanOrEqual(1)
expect(h.rt.ctx.review.memory.list().some((m) => m.value.topic === 'response:agg-unit-01')).toBe(true)
expect(review.report_id).toMatch(/^rep-bid_backtest/)
expect(review.llm_used).toBe(false)
expect(h.rt.ctx.events.list({ event_type: 'ResourceProfileUpdated' })).toHaveLength(1)
// The next potential assessment cites this execution feedback (画像时效性).
const t = new TriggerService(h.rt)
h.rt.ctx.envelopes.register(dispatchEnvelope({ max_total_mw: '30.0' }))
const again = await t.onAward({ ...awardFor(h.rt.ctx.caseDesk.proposals.list()[0]!.value.digest, '2.0'), id: 'award-2', market_date: '2026-03-16' })
void again
const pot = h.rt.ctx.events.list({ event_type: 'PotentialAssessed' }).at(-1)!.payload as { units: Record<string, string[]> }
expect(pot.units['agg-unit-01']![0]).toBe('19.200') // 24 MW × 0.8 fulfilment
await h.rt.close()
})
it('envelope suspension triggers on a seeded deviation streak (N=3, threshold 10%)', async () => {
const h = await harness({ llm: null })
const { t } = await throughExecution(h, 0.8) // 20% deviation → streak 1
const digest = h.rt.ctx.caseDesk.proposals.list().find((p) => p.value.type === 'DISPATCH_PLAN')!.value.digest
for (let i = 0; i < 2; i++) {
t.onExecutionReports(h.rt.ctx.simulationGateway.execute(digest, { 'agg-unit-01': 0.8 }, '2026-03-16T01:00:00Z'))
await t.onMetering({ id: `meter-${i}`, market_date: MARKET_DATE, metered_mw: flat('9.600'), actual_price_yuan_per_mwh: flat('421.00'), received_at: '2026-03-16T02:00:00Z' })
}
expect(h.rt.ctx.envelopes.get('env-bid-001')!.status).toBe('SUSPENDED')
expect(h.rt.ctx.envelopes.get('env-dispatch-001')!.status).toBe('SUSPENDED')
expect(h.rt.ctx.events.list({ event_type: 'EnvelopeSuspended' })).toHaveLength(2)
// Autonomy has returned to humans: the next bid waits for approval.
const run = await h.rt.mastra.getWorkflow('dayAheadBid').createRun()
const res = await run.start({ inputData: { market_date: '2026-03-16' } })
expect(res.status === 'success' && res.result.lifecycle_status).toBe('suspended')
await h.rt.close()
})
it('a compliant streak produces a WIDEN request that only a human can apply (envelope-review)', async () => {
const h = await harness({ llm: null, config: { widenAfterCompliantDays: 2 } })
const { t, review } = await throughExecution(h, 1.0)
expect(review.envelope_requests).toHaveLength(0)
const digest = h.rt.ctx.caseDesk.proposals.list().find((p) => p.value.type === 'DISPATCH_PLAN')!.value.digest
t.onExecutionReports(h.rt.ctx.simulationGateway.execute(digest, {}, '2026-03-16T01:00:00Z'))
const second = (await t.onMetering({ id: 'meter-2', market_date: MARKET_DATE, metered_mw: flat('12.000'), actual_price_yuan_per_mwh: flat('421.00'), received_at: '2026-03-16T02:00:00Z' })).result!
expect(second.envelope_requests.map((r: { envelope_id: string }) => r.envelope_id).sort()).toEqual(['env-bid-001', 'env-dispatch-001'])
expect(second.envelope_review_run_ids).toHaveLength(2)
const inbox = h.rt.ctx.caseDesk.inbox().filter((p) => p.run_id !== '')
expect(inbox.every((p) => p.workflow === 'envelopeReview')).toBe(true)
expect(h.rt.ctx.envelopes.get('env-bid-001')!.bounds['price_deviation_pct']).toBeUndefined() // bid envelope had only max_energy
const bidReq = second.envelope_requests.find((r: { envelope_id: string }) => r.envelope_id === 'env-dispatch-001')!
const runId = inbox.find((p) => p.proposal_id === bidReq.id)!.run_id
const refused = await t.decide(runId, { decision: 'approve', approver: { id: 'resource-agent', role: 'ops-lead' } })
expect(refused.status).toBe('suspended')
const ok = await t.decide(runId, { decision: 'approve', approver: APPROVER, comment: 'shadow data supports widening' })
expect(ok.status).toBe('success')
expect(h.rt.ctx.envelopes.get('env-dispatch-001')!.bounds['max_total_mw']).toBe('36.00')
expect(h.rt.ctx.envelopes.get('env-dispatch-001')!.approval.approved_by).toEqual(['user-ops-lead'])
expect(h.rt.ctx.events.list({ event_type: 'EnvelopeChanged' })).toHaveLength(1)
await h.rt.close()
})
it('insight cards project situation, review and resource state with lineage refs', async () => {
const h = await harness({ llm: null })
const sit = await h.rt.mastra.getWorkflow('dayAheadSituation').createRun()
await sit.start({ inputData: { market_date: MARKET_DATE } })
await throughExecution(h, 0.8)
const trading = insightCards(h.rt.ctx, 'TRADING_DESK')
expect(trading.map((c) => c.title)).toEqual(['Day-ahead situation', 'D+1 review'])
expect(trading[0]!.metrics[0]!.ref?.path).toMatch(/^quantiles\.p50\.values\.\d+$/)
const pool = insightCards(h.rt.ctx, 'RESOURCE_POOL')
expect(pool[0]!.headline).toMatch(/Reliability 0\.900 → 0\.880/)
const ops = insightCards(h.rt.ctx, 'OPERATIONS_DASHBOARD')
expect(ops.find((c) => c.id === 'card-ops-cases')).toBeDefined()
await h.rt.close()
})
})