Shadow run (ROADMAP M5, phase-1 acceptance form): SHADOW runtime mode records
bids without submitting them, simulates the market answer from the actual
day-ahead clearing price, dispatches to the simulation gateway, runs the D+1
review, and scores every day shadow-vs-human-vs-hindsight (ShadowDayRecord)
with a lineage-completeness audit. KpiReport regenerated after every day per
docs/12 §4 definitions (C1/C2 placeholders as named config).
Kill switches (docs/13 §8): BreakerService with L0 permit revocation, L1
envelope suspension, L2 loss breaker (mark-to-market, reduce-only bids),
L3 channel breaker (bids fall back to the file channel, dispatch BLOCKED),
L4 AI-off (templates run, no Proposal created); abnormal-day protocol on
EXTREME situations; per-level authority (B8 placeholder); auditable drill.
Runtime: shadow-close workflow, shadow schedule entries, live-data ingestion
through the quality gate, human-bid ingestion, breaker/KPI/shadow endpoints
and insight cards, replay CLI (npm run shadow). Ledger, time series and
streak counters are file-backed so a multi-week run survives restarts.
Domain: HumanBidRecord, ShadowDayRecord, KpiReport, BreakerRecord, SHADOW
receipt channel; contracts, fixtures and pydantic models regenerated.
Services: L2 metrics moved from evals so the shadow run and the harness
share one implementation.
Fixes: envelope/permit validity compared ISO timestamps as strings
('…00Z' vs '…00.000Z'); L2 baseline was stale since M4 (skill_versions only,
metrics unchanged) — rewritten from the live service.
Docs: docs/14 shadow-run runbook (timeline, breaker trigger/authority/
recovery, KPI definitions as implemented); README and CLAUDE.md status.
Tests: 21 consecutive shadow days with complete lineage, KPI report, WIDEN
recommendation produced but not acted on; restart durability; drill; L2/L3/L4
and abnormal-day paths; API.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017wrZgPL9LoKaD69BpEQU4v
55 lines
3.0 KiB
TypeScript
55 lines
3.0 KiB
TypeScript
import type { AddressInfo } from 'node:net'
|
|
import { describe, expect, it } from 'vitest'
|
|
import { createApi } from '../src/api.js'
|
|
import { TriggerService } from '../src/trigger.js'
|
|
import { MARKET_DATE, harness } from './helpers.js'
|
|
|
|
async function serve(h: Awaited<ReturnType<typeof harness>>) {
|
|
const server = createApi(h.rt, new TriggerService(h.rt))
|
|
await new Promise<void>((r) => server.listen(0, '127.0.0.1', r))
|
|
const base = `http://127.0.0.1:${(server.address() as AddressInfo).port}`
|
|
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
const call = async (method: string, path: string, body?: unknown, headers: Record<string, string> = {}): Promise<{ status: number; body: any }> => {
|
|
const init: RequestInit = { method, headers: { 'content-type': 'application/json', ...headers } }
|
|
if (body !== undefined) init.body = JSON.stringify(body)
|
|
const res = await fetch(base + path, init)
|
|
return { status: res.status, body: await res.json() }
|
|
}
|
|
return { call, close: () => new Promise<void>((r) => server.close(() => r())) }
|
|
}
|
|
|
|
describe('Case Desk HTTP API', () => {
|
|
it('inbox → decide (approve) → case view + lineage expansion', async () => {
|
|
const h = await harness({ llm: null })
|
|
const { call, close } = await serve(h)
|
|
expect((await call('GET', '/health')).body).toMatchObject({ status: 'ok', llm: 'null', mode: 'FILE_EXPORT' })
|
|
|
|
const run = await h.rt.mastra.getWorkflow('dayAheadBid').createRun()
|
|
const out = await run.start({ inputData: { market_date: MARKET_DATE } })
|
|
if (out.status !== 'success') throw new Error(out.status)
|
|
|
|
const inbox = await call('GET', '/inbox')
|
|
expect(inbox.body).toHaveLength(1)
|
|
const { run_id, proposal_id, case_id } = inbox.body[0]
|
|
|
|
expect((await call('POST', `/approvals/${run_id}/decide`, { decision: 'approve' })).status).toBe(401)
|
|
const refused = await call('POST', `/approvals/${run_id}/decide`, { decision: 'approve' }, { 'x-user-id': 'trading-agent', 'x-user-role': 'senior-trader' })
|
|
expect(refused.body.status).toBe('suspended') // I1: still waiting for a legitimate human
|
|
|
|
const ok = await call('POST', `/approvals/${run_id}/decide`, { decision: 'approve', comment: 'fine' }, { 'x-user-id': 'user-trader-01', 'x-user-role': 'senior-trader' })
|
|
expect(ok.body).toMatchObject({ run_id, status: 'success', outcome: 'RELEASED' })
|
|
expect((await call('GET', '/inbox')).body).toHaveLength(0)
|
|
|
|
const view = await call('GET', `/cases/${case_id}`)
|
|
expect(view.body.case.status).toBe('CLOSED_DONE')
|
|
expect(view.body.approvals[0].approver.id).toBe('user-trader-01')
|
|
const lineage = await call('GET', `/proposals/${proposal_id}/lineage`)
|
|
expect(lineage.body.tool_calls.map((t: { tool: string }) => t.tool)).toContain('bid-optimization-milp')
|
|
expect((await call('GET', `/events?correlation_id=${case_id}`)).body.length).toBeGreaterThan(8)
|
|
expect((await call('GET', '/nope')).status).toBe(404)
|
|
expect((await call('POST', '/tasks', { message: 'hi' })).status).toBe(503) // router needs an LLM
|
|
await close()
|
|
await h.rt.close()
|
|
})
|
|
})
|