vpp-ai-platform/packages/runtime/test/m4.test.ts

160 lines
9.9 KiB
TypeScript
Raw Permalink Normal View History

M4: resource agent, envelopes live, review loop, insight cards - packages/domain: AwardNotice, DISPATCH_PLAN proposal payload, ExecutionReport, MeteringRecord, potential-assessment and dispatch-optimization contracts, ReviewFinding (+ typed writebacks), SemanticMemoryEntry, EnvelopeChangeRequest, InsightCard — exported with fixtures on both sides. - skills-py: potential-assessment (certified × rolling fulfilment, evidence days) and dispatch-optimization (per-interval LP on HiGHS, shortfall reported) skills + routes + tests. - packages/services: dispatch rules in the policy pack (over-allocation, award anchor, lineage integrity for allocations); PowerBalanceSimulator; SimulationGateway (permit-only, idempotent, seeded execute → ExecutionReports); envelope deviation-streak suspension + apply(); ReviewService (attribution, reliability EWMA writeback, semantic memory, envelope recommendations as change requests); dispatch assembler; skill client methods. - packages/runtime: resource agent; award-decomposition, review and envelope-review workflows; lifecycle selects simulator/gateway by proposal type; trigger hooks for awards, execution reports, metering; decide() resumes either lifecycle or envelope-review runs; insight cards API. - Tests: docs/07 D-1 16:00 and D+1 end to end; reliability score 0.9 → 0.880 and the next assessment de-rates capacity; envelope suspension on a seeded 3-day streak; WIDEN request applied only by a human. 184 TS + 80 Python. - docs/open-questions: B10 (reliability/potential parameters). README and CLAUDE.md status → M4 done, M5 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 19:31:12 -04:00
import { describe, expect, it } from 'vitest'
import type { AwardNotice, Envelope } from '@vpp/domain'
import { insightCards } from '../src/cards.js'
import { TriggerService } from '../src/trigger.js'
import { MARKET_DATE, envelope, harness } from './helpers.js'
const APPROVER = { id: 'user-ops-lead', role: 'ops-lead' }
const flat = (v: string) => ({ interval_minutes: 15 as const, date: MARKET_DATE, values: Array(96).fill(v) as string[] })
const dispatchEnvelope = (bounds: Record<string, string>): Envelope =>
envelope(bounds, { id: 'env-dispatch-001', scope: { proposal_type: 'DISPATCH_PLAN', timescales: ['DAY_AHEAD'], resource_set: 'pool-hubei-01' } })
/** Runs D-1 08:00 (bid) and returns the bid's daily energy so the award can answer it. */
async function bidDay(h: Awaited<ReturnType<typeof harness>>) {
h.rt.ctx.envelopes.register(envelope({ max_energy_mwh: '1500.0' }))
const run = await h.rt.mastra.getWorkflow('dayAheadBid').createRun()
const res = await run.start({ inputData: { market_date: MARKET_DATE } })
if (res.status !== 'success') throw new Error(res.status)
return res.result
}
const awardFor = (digest: string, mwhPerInterval: string): AwardNotice => ({
id: `award-${MARKET_DATE}`,
market_date: MARKET_DATE,
bid_proposal_digest: digest,
awarded_mwh: flat(mwhPerInterval),
clearing_price_yuan_per_mwh: flat('418.30'),
received_at: '2026-03-14T08:00:00Z',
})
describe('docs/07 D-1 16:00: award → resource agent → dispatch plan through the safety chain', () => {
it('records the award in the ledger and releases a dispatch plan to the simulation gateway', async () => {
const h = await harness({ llm: null })
const bid = await bidDay(h)
h.rt.ctx.envelopes.register(dispatchEnvelope({ max_total_mw: '30.0' }))
const t = new TriggerService(h.rt)
M5: shadow run, kill-switch hierarchy L0–L4, KPI dashboard, shadow replay CLI Shadow run (ROADMAP M5, phase-1 acceptance form): SHADOW runtime mode records bids without submitting them, simulates the market answer from the actual day-ahead clearing price, dispatches to the simulation gateway, runs the D+1 review, and scores every day shadow-vs-human-vs-hindsight (ShadowDayRecord) with a lineage-completeness audit. KpiReport regenerated after every day per docs/12 §4 definitions (C1/C2 placeholders as named config). Kill switches (docs/13 §8): BreakerService with L0 permit revocation, L1 envelope suspension, L2 loss breaker (mark-to-market, reduce-only bids), L3 channel breaker (bids fall back to the file channel, dispatch BLOCKED), L4 AI-off (templates run, no Proposal created); abnormal-day protocol on EXTREME situations; per-level authority (B8 placeholder); auditable drill. Runtime: shadow-close workflow, shadow schedule entries, live-data ingestion through the quality gate, human-bid ingestion, breaker/KPI/shadow endpoints and insight cards, replay CLI (npm run shadow). Ledger, time series and streak counters are file-backed so a multi-week run survives restarts. Domain: HumanBidRecord, ShadowDayRecord, KpiReport, BreakerRecord, SHADOW receipt channel; contracts, fixtures and pydantic models regenerated. Services: L2 metrics moved from evals so the shadow run and the harness share one implementation. Fixes: envelope/permit validity compared ISO timestamps as strings ('…00Z' vs '…00.000Z'); L2 baseline was stale since M4 (skill_versions only, metrics unchanged) — rewritten from the live service. Docs: docs/14 shadow-run runbook (timeline, breaker trigger/authority/ recovery, KPI definitions as implemented); README and CLAUDE.md status. Tests: 21 consecutive shadow days with complete lineage, KPI report, WIDEN recommendation produced but not acted on; restart durability; drill; L2/L3/L4 and abnormal-day paths; API. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017wrZgPL9LoKaD69BpEQU4v
2026-09-02 22:54:14 -04:00
const out = await t.onAward(awardFor(bid.proposal!.digest, '3.0')) // 12 MW target, resource offers 24 MW
M4: resource agent, envelopes live, review loop, insight cards - packages/domain: AwardNotice, DISPATCH_PLAN proposal payload, ExecutionReport, MeteringRecord, potential-assessment and dispatch-optimization contracts, ReviewFinding (+ typed writebacks), SemanticMemoryEntry, EnvelopeChangeRequest, InsightCard — exported with fixtures on both sides. - skills-py: potential-assessment (certified × rolling fulfilment, evidence days) and dispatch-optimization (per-interval LP on HiGHS, shortfall reported) skills + routes + tests. - packages/services: dispatch rules in the policy pack (over-allocation, award anchor, lineage integrity for allocations); PowerBalanceSimulator; SimulationGateway (permit-only, idempotent, seeded execute → ExecutionReports); envelope deviation-streak suspension + apply(); ReviewService (attribution, reliability EWMA writeback, semantic memory, envelope recommendations as change requests); dispatch assembler; skill client methods. - packages/runtime: resource agent; award-decomposition, review and envelope-review workflows; lifecycle selects simulator/gateway by proposal type; trigger hooks for awards, execution reports, metering; decide() resumes either lifecycle or envelope-review runs; insight cards API. - Tests: docs/07 D-1 16:00 and D+1 end to end; reliability score 0.9 → 0.880 and the next assessment de-rates capacity; envelope suspension on a seeded 3-day streak; WIDEN request applied only by a human. 184 TS + 80 Python. - docs/open-questions: B10 (reliability/potential parameters). README and CLAUDE.md status → M4 done, M5 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 19:31:12 -04:00
expect(out.status).toBe('success')
const r = out.result!
expect(r.lifecycle?.outcome).toBe('RELEASED')
M5: shadow run, kill-switch hierarchy L0–L4, KPI dashboard, shadow replay CLI Shadow run (ROADMAP M5, phase-1 acceptance form): SHADOW runtime mode records bids without submitting them, simulates the market answer from the actual day-ahead clearing price, dispatches to the simulation gateway, runs the D+1 review, and scores every day shadow-vs-human-vs-hindsight (ShadowDayRecord) with a lineage-completeness audit. KpiReport regenerated after every day per docs/12 §4 definitions (C1/C2 placeholders as named config). Kill switches (docs/13 §8): BreakerService with L0 permit revocation, L1 envelope suspension, L2 loss breaker (mark-to-market, reduce-only bids), L3 channel breaker (bids fall back to the file channel, dispatch BLOCKED), L4 AI-off (templates run, no Proposal created); abnormal-day protocol on EXTREME situations; per-level authority (B8 placeholder); auditable drill. Runtime: shadow-close workflow, shadow schedule entries, live-data ingestion through the quality gate, human-bid ingestion, breaker/KPI/shadow endpoints and insight cards, replay CLI (npm run shadow). Ledger, time series and streak counters are file-backed so a multi-week run survives restarts. Domain: HumanBidRecord, ShadowDayRecord, KpiReport, BreakerRecord, SHADOW receipt channel; contracts, fixtures and pydantic models regenerated. Services: L2 metrics moved from evals so the shadow run and the harness share one implementation. Fixes: envelope/permit validity compared ISO timestamps as strings ('…00Z' vs '…00.000Z'); L2 baseline was stale since M4 (skill_versions only, metrics unchanged) — rewritten from the live service. Docs: docs/14 shadow-run runbook (timeline, breaker trigger/authority/ recovery, KPI definitions as implemented); README and CLAUDE.md status. Tests: 21 consecutive shadow days with complete lineage, KPI report, WIDEN recommendation produced but not acted on; restart durability; drill; L2/L3/L4 and abnormal-day paths; API. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017wrZgPL9LoKaD69BpEQU4v
2026-09-02 22:54:14 -04:00
expect(r.proposal!.type).toBe('DISPATCH_PLAN')
M4: resource agent, envelopes live, review loop, insight cards - packages/domain: AwardNotice, DISPATCH_PLAN proposal payload, ExecutionReport, MeteringRecord, potential-assessment and dispatch-optimization contracts, ReviewFinding (+ typed writebacks), SemanticMemoryEntry, EnvelopeChangeRequest, InsightCard — exported with fixtures on both sides. - skills-py: potential-assessment (certified × rolling fulfilment, evidence days) and dispatch-optimization (per-interval LP on HiGHS, shortfall reported) skills + routes + tests. - packages/services: dispatch rules in the policy pack (over-allocation, award anchor, lineage integrity for allocations); PowerBalanceSimulator; SimulationGateway (permit-only, idempotent, seeded execute → ExecutionReports); envelope deviation-streak suspension + apply(); ReviewService (attribution, reliability EWMA writeback, semantic memory, envelope recommendations as change requests); dispatch assembler; skill client methods. - packages/runtime: resource agent; award-decomposition, review and envelope-review workflows; lifecycle selects simulator/gateway by proposal type; trigger hooks for awards, execution reports, metering; decide() resumes either lifecycle or envelope-review runs; insight cards API. - Tests: docs/07 D-1 16:00 and D+1 end to end; reliability score 0.9 → 0.880 and the next assessment de-rates capacity; envelope suspension on a seeded 3-day streak; WIDEN request applied only by a human. 184 TS + 80 Python. - docs/open-questions: B10 (reliability/potential parameters). README and CLAUDE.md status → M4 done, M5 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 19:31:12 -04:00
expect(r.shortfall_mwh).toBe('0.000')
expect(h.rt.ctx.ledger.read().entries.map((e) => e.kind)).toEqual(['CONTRACT', 'BID_SUBMITTED', 'AWARD'])
expect(h.skills.calls.filter((c) => c === 'potential' || c === 'dispatch')).toEqual(['potential', 'dispatch'])
expect(h.rt.ctx.simulationGateway.receipts.list()).toHaveLength(1)
// P2 for dispatch numbers: allocations are byte-identical to the tool output.
M5: shadow run, kill-switch hierarchy L0–L4, KPI dashboard, shadow replay CLI Shadow run (ROADMAP M5, phase-1 acceptance form): SHADOW runtime mode records bids without submitting them, simulates the market answer from the actual day-ahead clearing price, dispatches to the simulation gateway, runs the D+1 review, and scores every day shadow-vs-human-vs-hindsight (ShadowDayRecord) with a lineage-completeness audit. KpiReport regenerated after every day per docs/12 §4 definitions (C1/C2 placeholders as named config). Kill switches (docs/13 §8): BreakerService with L0 permit revocation, L1 envelope suspension, L2 loss breaker (mark-to-market, reduce-only bids), L3 channel breaker (bids fall back to the file channel, dispatch BLOCKED), L4 AI-off (templates run, no Proposal created); abnormal-day protocol on EXTREME situations; per-level authority (B8 placeholder); auditable drill. Runtime: shadow-close workflow, shadow schedule entries, live-data ingestion through the quality gate, human-bid ingestion, breaker/KPI/shadow endpoints and insight cards, replay CLI (npm run shadow). Ledger, time series and streak counters are file-backed so a multi-week run survives restarts. Domain: HumanBidRecord, ShadowDayRecord, KpiReport, BreakerRecord, SHADOW receipt channel; contracts, fixtures and pydantic models regenerated. Services: L2 metrics moved from evals so the shadow run and the harness share one implementation. Fixes: envelope/permit validity compared ISO timestamps as strings ('…00Z' vs '…00.000Z'); L2 baseline was stale since M4 (skill_versions only, metrics unchanged) — rewritten from the live service. Docs: docs/14 shadow-run runbook (timeline, breaker trigger/authority/ recovery, KPI definitions as implemented); README and CLAUDE.md status. Tests: 21 consecutive shadow days with complete lineage, KPI report, WIDEN recommendation produced but not acted on; restart durability; drill; L2/L3/L4 and abnormal-day paths; API. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017wrZgPL9LoKaD69BpEQU4v
2026-09-02 22:54:14 -04:00
const lin = h.rt.ctx.caseDesk.expandLineage(r.proposal_id!)
M4: resource agent, envelopes live, review loop, insight cards - packages/domain: AwardNotice, DISPATCH_PLAN proposal payload, ExecutionReport, MeteringRecord, potential-assessment and dispatch-optimization contracts, ReviewFinding (+ typed writebacks), SemanticMemoryEntry, EnvelopeChangeRequest, InsightCard — exported with fixtures on both sides. - skills-py: potential-assessment (certified × rolling fulfilment, evidence days) and dispatch-optimization (per-interval LP on HiGHS, shortfall reported) skills + routes + tests. - packages/services: dispatch rules in the policy pack (over-allocation, award anchor, lineage integrity for allocations); PowerBalanceSimulator; SimulationGateway (permit-only, idempotent, seeded execute → ExecutionReports); envelope deviation-streak suspension + apply(); ReviewService (attribution, reliability EWMA writeback, semantic memory, envelope recommendations as change requests); dispatch assembler; skill client methods. - packages/runtime: resource agent; award-decomposition, review and envelope-review workflows; lifecycle selects simulator/gateway by proposal type; trigger hooks for awards, execution reports, metering; decide() resumes either lifecycle or envelope-review runs; insight cards API. - Tests: docs/07 D-1 16:00 and D+1 end to end; reliability score 0.9 → 0.880 and the next assessment de-rates capacity; envelope suspension on a seeded 3-day streak; WIDEN request applied only by a human. 184 TS + 80 Python. - docs/open-questions: B10 (reliability/potential parameters). README and CLAUDE.md status → M4 done, M5 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 19:31:12 -04:00
expect(lin.tool_calls.map((c) => c.tool)).toEqual(['potential-assessment', 'dispatch-optimization'])
M5: shadow run, kill-switch hierarchy L0–L4, KPI dashboard, shadow replay CLI Shadow run (ROADMAP M5, phase-1 acceptance form): SHADOW runtime mode records bids without submitting them, simulates the market answer from the actual day-ahead clearing price, dispatches to the simulation gateway, runs the D+1 review, and scores every day shadow-vs-human-vs-hindsight (ShadowDayRecord) with a lineage-completeness audit. KpiReport regenerated after every day per docs/12 §4 definitions (C1/C2 placeholders as named config). Kill switches (docs/13 §8): BreakerService with L0 permit revocation, L1 envelope suspension, L2 loss breaker (mark-to-market, reduce-only bids), L3 channel breaker (bids fall back to the file channel, dispatch BLOCKED), L4 AI-off (templates run, no Proposal created); abnormal-day protocol on EXTREME situations; per-level authority (B8 placeholder); auditable drill. Runtime: shadow-close workflow, shadow schedule entries, live-data ingestion through the quality gate, human-bid ingestion, breaker/KPI/shadow endpoints and insight cards, replay CLI (npm run shadow). Ledger, time series and streak counters are file-backed so a multi-week run survives restarts. Domain: HumanBidRecord, ShadowDayRecord, KpiReport, BreakerRecord, SHADOW receipt channel; contracts, fixtures and pydantic models regenerated. Services: L2 metrics moved from evals so the shadow run and the harness share one implementation. Fixes: envelope/permit validity compared ISO timestamps as strings ('…00Z' vs '…00.000Z'); L2 baseline was stale since M4 (skill_versions only, metrics unchanged) — rewritten from the live service. Docs: docs/14 shadow-run runbook (timeline, breaker trigger/authority/ recovery, KPI definitions as implemented); README and CLAUDE.md status. Tests: 21 consecutive shadow days with complete lineage, KPI report, WIDEN recommendation produced but not acted on; restart durability; drill; L2/L3/L4 and abnormal-day paths; API. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017wrZgPL9LoKaD69BpEQU4v
2026-09-02 22:54:14 -04:00
expect((lin.tool_calls[1]!.outputs as { allocations: unknown }).allocations).toEqual(r.proposal!.type === 'DISPATCH_PLAN' ? r.proposal!.payload.allocations : null)
M4: resource agent, envelopes live, review loop, insight cards - packages/domain: AwardNotice, DISPATCH_PLAN proposal payload, ExecutionReport, MeteringRecord, potential-assessment and dispatch-optimization contracts, ReviewFinding (+ typed writebacks), SemanticMemoryEntry, EnvelopeChangeRequest, InsightCard — exported with fixtures on both sides. - skills-py: potential-assessment (certified × rolling fulfilment, evidence days) and dispatch-optimization (per-interval LP on HiGHS, shortfall reported) skills + routes + tests. - packages/services: dispatch rules in the policy pack (over-allocation, award anchor, lineage integrity for allocations); PowerBalanceSimulator; SimulationGateway (permit-only, idempotent, seeded execute → ExecutionReports); envelope deviation-streak suspension + apply(); ReviewService (attribution, reliability EWMA writeback, semantic memory, envelope recommendations as change requests); dispatch assembler; skill client methods. - packages/runtime: resource agent; award-decomposition, review and envelope-review workflows; lifecycle selects simulator/gateway by proposal type; trigger hooks for awards, execution reports, metering; decide() resumes either lifecycle or envelope-review runs; insight cards API. - Tests: docs/07 D-1 16:00 and D+1 end to end; reliability score 0.9 → 0.880 and the next assessment de-rates capacity; envelope suspension on a seeded 3-day streak; WIDEN request applied only by a human. 184 TS + 80 Python. - docs/open-questions: B10 (reliability/potential parameters). README and CLAUDE.md status → M4 done, M5 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 19:31:12 -04:00
await h.rt.close()
})
it('a shortfall (award beyond assessed capacity) alerts in power-balance simulation and goes to a human', async () => {
const h = await harness({ llm: null })
const bid = await bidDay(h)
h.rt.ctx.envelopes.register(dispatchEnvelope({ max_total_mw: '100.0' }))
M5: shadow run, kill-switch hierarchy L0–L4, KPI dashboard, shadow replay CLI Shadow run (ROADMAP M5, phase-1 acceptance form): SHADOW runtime mode records bids without submitting them, simulates the market answer from the actual day-ahead clearing price, dispatches to the simulation gateway, runs the D+1 review, and scores every day shadow-vs-human-vs-hindsight (ShadowDayRecord) with a lineage-completeness audit. KpiReport regenerated after every day per docs/12 §4 definitions (C1/C2 placeholders as named config). Kill switches (docs/13 §8): BreakerService with L0 permit revocation, L1 envelope suspension, L2 loss breaker (mark-to-market, reduce-only bids), L3 channel breaker (bids fall back to the file channel, dispatch BLOCKED), L4 AI-off (templates run, no Proposal created); abnormal-day protocol on EXTREME situations; per-level authority (B8 placeholder); auditable drill. Runtime: shadow-close workflow, shadow schedule entries, live-data ingestion through the quality gate, human-bid ingestion, breaker/KPI/shadow endpoints and insight cards, replay CLI (npm run shadow). Ledger, time series and streak counters are file-backed so a multi-week run survives restarts. Domain: HumanBidRecord, ShadowDayRecord, KpiReport, BreakerRecord, SHADOW receipt channel; contracts, fixtures and pydantic models regenerated. Services: L2 metrics moved from evals so the shadow run and the harness share one implementation. Fixes: envelope/permit validity compared ISO timestamps as strings ('…00Z' vs '…00.000Z'); L2 baseline was stale since M4 (skill_versions only, metrics unchanged) — rewritten from the live service. Docs: docs/14 shadow-run runbook (timeline, breaker trigger/authority/ recovery, KPI definitions as implemented); README and CLAUDE.md status. Tests: 21 consecutive shadow days with complete lineage, KPI report, WIDEN recommendation produced but not acted on; restart durability; drill; L2/L3/L4 and abnormal-day paths; API. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017wrZgPL9LoKaD69BpEQU4v
2026-09-02 22:54:14 -04:00
const out = await new TriggerService(h.rt).onAward(awardFor(bid.proposal!.digest, '10.0')) // 40 MW target > 24 MW
M4: resource agent, envelopes live, review loop, insight cards - packages/domain: AwardNotice, DISPATCH_PLAN proposal payload, ExecutionReport, MeteringRecord, potential-assessment and dispatch-optimization contracts, ReviewFinding (+ typed writebacks), SemanticMemoryEntry, EnvelopeChangeRequest, InsightCard — exported with fixtures on both sides. - skills-py: potential-assessment (certified × rolling fulfilment, evidence days) and dispatch-optimization (per-interval LP on HiGHS, shortfall reported) skills + routes + tests. - packages/services: dispatch rules in the policy pack (over-allocation, award anchor, lineage integrity for allocations); PowerBalanceSimulator; SimulationGateway (permit-only, idempotent, seeded execute → ExecutionReports); envelope deviation-streak suspension + apply(); ReviewService (attribution, reliability EWMA writeback, semantic memory, envelope recommendations as change requests); dispatch assembler; skill client methods. - packages/runtime: resource agent; award-decomposition, review and envelope-review workflows; lifecycle selects simulator/gateway by proposal type; trigger hooks for awards, execution reports, metering; decide() resumes either lifecycle or envelope-review runs; insight cards API. - Tests: docs/07 D-1 16:00 and D+1 end to end; reliability score 0.9 → 0.880 and the next assessment de-rates capacity; envelope suspension on a seeded 3-day streak; WIDEN request applied only by a human. 184 TS + 80 Python. - docs/open-questions: B10 (reliability/potential parameters). README and CLAUDE.md status → M4 done, M5 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 19:31:12 -04:00
expect(out.status).toBe('success')
expect(out.result!.lifecycle_status).toBe('suspended')
expect(Number(out.result!.shortfall_mwh)).toBeGreaterThan(0)
expect(h.rt.ctx.caseDesk.inbox()[0]!.reasons[0]).toMatch(/short/)
await h.rt.close()
})
})
describe('docs/07 D+1: execution feedback → review → writebacks', () => {
async function throughExecution(h: Awaited<ReturnType<typeof harness>>, fulfilment: number) {
const bid = await bidDay(h)
h.rt.ctx.envelopes.register(dispatchEnvelope({ max_total_mw: '30.0' }))
const t = new TriggerService(h.rt)
M5: shadow run, kill-switch hierarchy L0–L4, KPI dashboard, shadow replay CLI Shadow run (ROADMAP M5, phase-1 acceptance form): SHADOW runtime mode records bids without submitting them, simulates the market answer from the actual day-ahead clearing price, dispatches to the simulation gateway, runs the D+1 review, and scores every day shadow-vs-human-vs-hindsight (ShadowDayRecord) with a lineage-completeness audit. KpiReport regenerated after every day per docs/12 §4 definitions (C1/C2 placeholders as named config). Kill switches (docs/13 §8): BreakerService with L0 permit revocation, L1 envelope suspension, L2 loss breaker (mark-to-market, reduce-only bids), L3 channel breaker (bids fall back to the file channel, dispatch BLOCKED), L4 AI-off (templates run, no Proposal created); abnormal-day protocol on EXTREME situations; per-level authority (B8 placeholder); auditable drill. Runtime: shadow-close workflow, shadow schedule entries, live-data ingestion through the quality gate, human-bid ingestion, breaker/KPI/shadow endpoints and insight cards, replay CLI (npm run shadow). Ledger, time series and streak counters are file-backed so a multi-week run survives restarts. Domain: HumanBidRecord, ShadowDayRecord, KpiReport, BreakerRecord, SHADOW receipt channel; contracts, fixtures and pydantic models regenerated. Services: L2 metrics moved from evals so the shadow run and the harness share one implementation. Fixes: envelope/permit validity compared ISO timestamps as strings ('…00Z' vs '…00.000Z'); L2 baseline was stale since M4 (skill_versions only, metrics unchanged) — rewritten from the live service. Docs: docs/14 shadow-run runbook (timeline, breaker trigger/authority/ recovery, KPI definitions as implemented); README and CLAUDE.md status. Tests: 21 consecutive shadow days with complete lineage, KPI report, WIDEN recommendation produced but not acted on; restart durability; drill; L2/L3/L4 and abnormal-day paths; API. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017wrZgPL9LoKaD69BpEQU4v
2026-09-02 22:54:14 -04:00
const out = await t.onAward(awardFor(bid.proposal!.digest, '3.0'))
const digest = out.result!.proposal!.digest
M4: resource agent, envelopes live, review loop, insight cards - packages/domain: AwardNotice, DISPATCH_PLAN proposal payload, ExecutionReport, MeteringRecord, potential-assessment and dispatch-optimization contracts, ReviewFinding (+ typed writebacks), SemanticMemoryEntry, EnvelopeChangeRequest, InsightCard — exported with fixtures on both sides. - skills-py: potential-assessment (certified × rolling fulfilment, evidence days) and dispatch-optimization (per-interval LP on HiGHS, shortfall reported) skills + routes + tests. - packages/services: dispatch rules in the policy pack (over-allocation, award anchor, lineage integrity for allocations); PowerBalanceSimulator; SimulationGateway (permit-only, idempotent, seeded execute → ExecutionReports); envelope deviation-streak suspension + apply(); ReviewService (attribution, reliability EWMA writeback, semantic memory, envelope recommendations as change requests); dispatch assembler; skill client methods. - packages/runtime: resource agent; award-decomposition, review and envelope-review workflows; lifecycle selects simulator/gateway by proposal type; trigger hooks for awards, execution reports, metering; decide() resumes either lifecycle or envelope-review runs; insight cards API. - Tests: docs/07 D-1 16:00 and D+1 end to end; reliability score 0.9 → 0.880 and the next assessment de-rates capacity; envelope suspension on a seeded 3-day streak; WIDEN request applied only by a human. 184 TS + 80 Python. - docs/open-questions: B10 (reliability/potential parameters). README and CLAUDE.md status → M4 done, M5 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 19:31:12 -04:00
t.onExecutionReports(h.rt.ctx.simulationGateway.execute(digest, { 'agg-unit-01': fulfilment }, '2026-03-16T01:00:00Z'))
const review = await t.onMetering({ id: `meter-${MARKET_DATE}`, market_date: MARKET_DATE, metered_mw: flat((12 * fulfilment).toFixed(3)), actual_price_yuan_per_mwh: flat('421.00'), received_at: '2026-03-16T02:00:00Z' })
return { t, review: review.result!, bid }
}
it('a ReviewFinding measurably updates the resource reliability score and writes semantic memory', async () => {
const h = await harness({ llm: null })
expect(h.rt.ctx.resources.get('res-storage-01')!.value.reliability_score).toBe('0.9')
const { review } = await throughExecution(h, 0.8)
expect(review.findings).toHaveLength(1)
const f = review.findings[0]!
expect(f.kind).toBe('RESPONSE')
expect(review.updated_profiles).toEqual([{ resource_id: 'res-storage-01', reliability_score: '0.880' }])
expect(h.rt.ctx.resources.get('res-storage-01')!.value.reliability_score).toBe('0.880')
expect(review.memory_entries).toBeGreaterThanOrEqual(1)
expect(h.rt.ctx.review.memory.list().some((m) => m.value.topic === 'response:agg-unit-01')).toBe(true)
expect(review.report_id).toMatch(/^rep-bid_backtest/)
expect(review.llm_used).toBe(false)
expect(h.rt.ctx.events.list({ event_type: 'ResourceProfileUpdated' })).toHaveLength(1)
// The next potential assessment cites this execution feedback (画像时效性).
const t = new TriggerService(h.rt)
h.rt.ctx.envelopes.register(dispatchEnvelope({ max_total_mw: '30.0' }))
const again = await t.onAward({ ...awardFor(h.rt.ctx.caseDesk.proposals.list()[0]!.value.digest, '2.0'), id: 'award-2', market_date: '2026-03-16' })
void again
const pot = h.rt.ctx.events.list({ event_type: 'PotentialAssessed' }).at(-1)!.payload as { units: Record<string, string[]> }
expect(pot.units['agg-unit-01']![0]).toBe('19.200') // 24 MW × 0.8 fulfilment
await h.rt.close()
})
it('envelope suspension triggers on a seeded deviation streak (N=3, threshold 10%)', async () => {
const h = await harness({ llm: null })
const { t } = await throughExecution(h, 0.8) // 20% deviation → streak 1
const digest = h.rt.ctx.caseDesk.proposals.list().find((p) => p.value.type === 'DISPATCH_PLAN')!.value.digest
for (let i = 0; i < 2; i++) {
t.onExecutionReports(h.rt.ctx.simulationGateway.execute(digest, { 'agg-unit-01': 0.8 }, '2026-03-16T01:00:00Z'))
await t.onMetering({ id: `meter-${i}`, market_date: MARKET_DATE, metered_mw: flat('9.600'), actual_price_yuan_per_mwh: flat('421.00'), received_at: '2026-03-16T02:00:00Z' })
}
expect(h.rt.ctx.envelopes.get('env-bid-001')!.status).toBe('SUSPENDED')
expect(h.rt.ctx.envelopes.get('env-dispatch-001')!.status).toBe('SUSPENDED')
expect(h.rt.ctx.events.list({ event_type: 'EnvelopeSuspended' })).toHaveLength(2)
// Autonomy has returned to humans: the next bid waits for approval.
const run = await h.rt.mastra.getWorkflow('dayAheadBid').createRun()
const res = await run.start({ inputData: { market_date: '2026-03-16' } })
expect(res.status === 'success' && res.result.lifecycle_status).toBe('suspended')
await h.rt.close()
})
it('a compliant streak produces a WIDEN request that only a human can apply (envelope-review)', async () => {
const h = await harness({ llm: null, config: { widenAfterCompliantDays: 2 } })
const { t, review } = await throughExecution(h, 1.0)
expect(review.envelope_requests).toHaveLength(0)
const digest = h.rt.ctx.caseDesk.proposals.list().find((p) => p.value.type === 'DISPATCH_PLAN')!.value.digest
t.onExecutionReports(h.rt.ctx.simulationGateway.execute(digest, {}, '2026-03-16T01:00:00Z'))
const second = (await t.onMetering({ id: 'meter-2', market_date: MARKET_DATE, metered_mw: flat('12.000'), actual_price_yuan_per_mwh: flat('421.00'), received_at: '2026-03-16T02:00:00Z' })).result!
expect(second.envelope_requests.map((r: { envelope_id: string }) => r.envelope_id).sort()).toEqual(['env-bid-001', 'env-dispatch-001'])
expect(second.envelope_review_run_ids).toHaveLength(2)
const inbox = h.rt.ctx.caseDesk.inbox().filter((p) => p.run_id !== '')
expect(inbox.every((p) => p.workflow === 'envelopeReview')).toBe(true)
expect(h.rt.ctx.envelopes.get('env-bid-001')!.bounds['price_deviation_pct']).toBeUndefined() // bid envelope had only max_energy
const bidReq = second.envelope_requests.find((r: { envelope_id: string }) => r.envelope_id === 'env-dispatch-001')!
const runId = inbox.find((p) => p.proposal_id === bidReq.id)!.run_id
const refused = await t.decide(runId, { decision: 'approve', approver: { id: 'resource-agent', role: 'ops-lead' } })
expect(refused.status).toBe('suspended')
const ok = await t.decide(runId, { decision: 'approve', approver: APPROVER, comment: 'shadow data supports widening' })
expect(ok.status).toBe('success')
expect(h.rt.ctx.envelopes.get('env-dispatch-001')!.bounds['max_total_mw']).toBe('36.00')
expect(h.rt.ctx.envelopes.get('env-dispatch-001')!.approval.approved_by).toEqual(['user-ops-lead'])
expect(h.rt.ctx.events.list({ event_type: 'EnvelopeChanged' })).toHaveLength(1)
await h.rt.close()
})
it('insight cards project situation, review and resource state with lineage refs', async () => {
const h = await harness({ llm: null })
const sit = await h.rt.mastra.getWorkflow('dayAheadSituation').createRun()
await sit.start({ inputData: { market_date: MARKET_DATE } })
await throughExecution(h, 0.8)
const trading = insightCards(h.rt.ctx, 'TRADING_DESK')
expect(trading.map((c) => c.title)).toEqual(['Day-ahead situation', 'D+1 review'])
expect(trading[0]!.metrics[0]!.ref?.path).toMatch(/^quantiles\.p50\.values\.\d+$/)
const pool = insightCards(h.rt.ctx, 'RESOURCE_POOL')
expect(pool[0]!.headline).toMatch(/Reliability 0\.900 → 0\.880/)
const ops = insightCards(h.rt.ctx, 'OPERATIONS_DASHBOARD')
expect(ops.find((c) => c.id === 'card-ops-cases')).toBeDefined()
await h.rt.close()
})
})