From 381b6d35217e91de03b3a27260ae893f4e3aa7b9 Mon Sep 17 00:00:00 2001 From: Thomas Bayes Date: Wed, 2 Sep 2026 22:54:14 -0400 Subject: [PATCH] =?UTF-8?q?M5:=20shadow=20run,=20kill-switch=20hierarchy?= =?UTF-8?q?=20L0=E2=80=93L4,=20KPI=20dashboard,=20shadow=20replay=20CLI?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Shadow run (ROADMAP M5, phase-1 acceptance form): SHADOW runtime mode records bids without submitting them, simulates the market answer from the actual day-ahead clearing price, dispatches to the simulation gateway, runs the D+1 review, and scores every day shadow-vs-human-vs-hindsight (ShadowDayRecord) with a lineage-completeness audit. KpiReport regenerated after every day per docs/12 §4 definitions (C1/C2 placeholders as named config). Kill switches (docs/13 §8): BreakerService with L0 permit revocation, L1 envelope suspension, L2 loss breaker (mark-to-market, reduce-only bids), L3 channel breaker (bids fall back to the file channel, dispatch BLOCKED), L4 AI-off (templates run, no Proposal created); abnormal-day protocol on EXTREME situations; per-level authority (B8 placeholder); auditable drill. Runtime: shadow-close workflow, shadow schedule entries, live-data ingestion through the quality gate, human-bid ingestion, breaker/KPI/shadow endpoints and insight cards, replay CLI (npm run shadow). Ledger, time series and streak counters are file-backed so a multi-week run survives restarts. Domain: HumanBidRecord, ShadowDayRecord, KpiReport, BreakerRecord, SHADOW receipt channel; contracts, fixtures and pydantic models regenerated. Services: L2 metrics moved from evals so the shadow run and the harness share one implementation. Fixes: envelope/permit validity compared ISO timestamps as strings ('…00Z' vs '…00.000Z'); L2 baseline was stale since M4 (skill_versions only, metrics unchanged) — rewritten from the live service. Docs: docs/14 shadow-run runbook (timeline, breaker trigger/authority/ recovery, KPI definitions as implemented); README and CLAUDE.md status. Tests: 21 consecutive shadow days with complete lineage, KPI report, WIDEN recommendation produced but not acted on; restart durability; drill; L2/L3/L4 and abnormal-day paths; API. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_017wrZgPL9LoKaD69BpEQU4v --- CLAUDE.md | 33 +- README.md | 25 +- .../breaker_record/unknown-level.json | 13 + .../execution_receipt/unknown-channel.json | 9 + .../fixtures/breaker_record/l0-armed.json | 13 + .../fixtures/breaker_record/l2-tripped.json | 16 + .../fixtures/execution_receipt/shadow.json | 9 + .../human_bid_record/platform-export.json | 210 ++++++++ contracts/fixtures/kpi_report/window-30d.json | 51 ++ .../shadow_day_record/released-day.json | 71 +++ contracts/schema/breaker_record.json | 139 ++++++ contracts/schema/execution_receipt.json | 3 +- contracts/schema/human_bid_record.json | 94 ++++ contracts/schema/insight_card.json | 5 +- contracts/schema/kpi_report.json | 285 +++++++++++ contracts/schema/shadow_day_record.json | 453 ++++++++++++++++++ docs/14-shadow-run-runbook.md | 74 +++ packages/domain/scripts/make-fixtures.ts | 97 ++++ packages/domain/src/chain.ts | 3 +- packages/domain/src/index.ts | 6 + packages/domain/src/review.ts | 2 +- packages/domain/src/shadow.ts | 155 ++++++ .../baselines/l2-synthetic-hubei-v0.json | 6 +- packages/evals/src/metrics.ts | 106 +--- packages/runtime/package.json | 3 +- packages/runtime/src/api.ts | 49 +- packages/runtime/src/cards.ts | 45 ++ packages/runtime/src/context.ts | 45 +- packages/runtime/src/index.ts | 2 + packages/runtime/src/main.ts | 6 +- packages/runtime/src/runtime.ts | 72 ++- packages/runtime/src/shadow-cli.ts | 155 ++++++ packages/runtime/src/shadow.ts | 170 +++++++ packages/runtime/src/trigger.ts | 127 ++++- .../src/workflows/award-decomposition.ts | 17 +- .../runtime/src/workflows/day-ahead-bid.ts | 19 +- .../src/workflows/proposal-lifecycle.ts | 33 +- packages/runtime/src/workflows/review.ts | 2 + .../runtime/src/workflows/shadow-close.ts | 330 +++++++++++++ packages/runtime/test/api.test.ts | 2 +- packages/runtime/test/lifecycle.test.ts | 42 +- packages/runtime/test/m4.test.ts | 14 +- packages/runtime/test/m5.test.ts | 274 +++++++++++ packages/services/src/authority.ts | 7 +- packages/services/src/breaker.ts | 240 ++++++++++ packages/services/src/counters.ts | 26 + packages/services/src/envelope.ts | 32 +- packages/services/src/gateway.ts | 8 +- packages/services/src/index.ts | 5 + packages/services/src/kpi.ts | 141 ++++++ packages/services/src/ledger.ts | 20 +- packages/services/src/metrics.ts | 103 ++++ packages/services/src/review.ts | 10 +- packages/services/src/shadow.ts | 113 +++++ packages/services/src/timeseries.ts | 29 ++ skills-py/vpp_contracts/breaker_record.py | 54 +++ skills-py/vpp_contracts/execution_receipt.py | 1 + skills-py/vpp_contracts/human_bid_record.py | 49 ++ skills-py/vpp_contracts/insight_card.py | 3 + skills-py/vpp_contracts/kpi_report.py | 93 ++++ skills-py/vpp_contracts/shadow_day_record.py | 148 ++++++ 61 files changed, 4146 insertions(+), 221 deletions(-) create mode 100644 contracts/fixtures-invalid/breaker_record/unknown-level.json create mode 100644 contracts/fixtures-invalid/execution_receipt/unknown-channel.json create mode 100644 contracts/fixtures/breaker_record/l0-armed.json create mode 100644 contracts/fixtures/breaker_record/l2-tripped.json create mode 100644 contracts/fixtures/execution_receipt/shadow.json create mode 100644 contracts/fixtures/human_bid_record/platform-export.json create mode 100644 contracts/fixtures/kpi_report/window-30d.json create mode 100644 contracts/fixtures/shadow_day_record/released-day.json create mode 100644 contracts/schema/breaker_record.json create mode 100644 contracts/schema/human_bid_record.json create mode 100644 contracts/schema/kpi_report.json create mode 100644 contracts/schema/shadow_day_record.json create mode 100644 docs/14-shadow-run-runbook.md create mode 100644 packages/domain/src/shadow.ts create mode 100644 packages/runtime/src/shadow-cli.ts create mode 100644 packages/runtime/src/shadow.ts create mode 100644 packages/runtime/src/workflows/shadow-close.ts create mode 100644 packages/runtime/test/m5.test.ts create mode 100644 packages/services/src/breaker.ts create mode 100644 packages/services/src/counters.ts create mode 100644 packages/services/src/kpi.ts create mode 100644 packages/services/src/metrics.ts create mode 100644 packages/services/src/shadow.ts create mode 100644 skills-py/vpp_contracts/breaker_record.py create mode 100644 skills-py/vpp_contracts/human_bid_record.py create mode 100644 skills-py/vpp_contracts/kpi_report.py create mode 100644 skills-py/vpp_contracts/shadow_day_record.py diff --git a/CLAUDE.md b/CLAUDE.md index 68a1b38..bf34ec1 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -6,23 +6,32 @@ docs win — or write an ADR changing the doc first. ## Current phase -Docs complete; M1–M4 implemented: schemas + contracts pipeline; deterministic +Docs complete; M1–M5 implemented: schemas + contracts pipeline; deterministic services (ledger, stores, policy engine incl. dispatch rules, envelope with -deviation-streak suspension, authority/permits, file-export + simulation +deviation-streak suspension, authority/permits, file-export + shadow + simulation gateways, events, lineage assemblers, revenue + power-balance simulation, -review service, Case Desk); Python skill service (forecasts, bid MILP, report, -potential assessment, dispatch optimization); L2 eval harness with committed -baseline (synthetic data); Mastra runtime with the proposal-lifecycle safety -chain (durable suspend/resume), situation / bid / award-decomposition / review / -envelope-review workflows, trigger service, LLM port with LLM-down mode, Case -Desk API with insight cards. Storage is file-backed reference semantics — -Postgres/Timescale adapters later. Next is M5 (shadow run). Build order is ROADMAP.md (M1→M5). A skill change that moves an L2 metric +review service, Case Desk, kill-switch hierarchy L0–L4 with loss breaker and +abnormal-day protocol, shadow clearing/scoring, KPI computation); Python skill +service (forecasts, bid MILP, report, potential assessment, dispatch +optimization); L2 eval harness with committed baseline (synthetic data); Mastra +runtime with the proposal-lifecycle safety chain (durable suspend/resume), +situation / bid / award-decomposition / review / envelope-review / shadow-close +workflows, trigger service (incl. shadow schedule and live-data ingestion), LLM +port with LLM-down mode, Case Desk API with insight cards, KPI dashboard and +breaker endpoints, shadow replay CLI (`npm run shadow -w @vpp/runtime`). +Storage is file-backed reference semantics (ledger, time series, counters and +repositories all persist under the data dir) — Postgres/Timescale adapters +later. Phase 1 is feature-complete; what remains is running the shadow period on +real Hubei data (`ROADMAP.md` M5 acceptance: 20+ consecutive days) and the +governance steps that need business answers (`docs/open-questions.md`). Build +order is ROADMAP.md (M1→M5). A skill change that moves an L2 metric must update the baseline in the same change (`npm run eval -w @vpp/evals -- --write-baseline`). Workflows are per-runtime factories (`createProposalLifecycle(ctx)` etc.) — never module singletons, or a second Mastra instance steals their context. -Do not start a milestone's work before its predecessor's acceptance criteria are -testable, and do not build phase-2 items (edge control links, federation, -interaction/load-control agents) unless explicitly asked. +Do not build phase-2 items (edge control links, federation, +interaction/load-control agents, programmatic bid submission, acting on +envelope-widening recommendations without the docs/03 envelope-review approval) +unless explicitly asked. ## Hard rules (from docs/01 invariants — treat as review criteria) diff --git a/README.md b/README.md index 27d8504..4e201be 100644 --- a/README.md +++ b/README.md @@ -6,8 +6,8 @@ a deterministic safety chain (rule check → simulation → envelope/human appro execution permit) governs everything before any external effect. **The LLM never computes numbers and never touches the second-level control loop.** -**Status: M1–M4 implemented.** Design docs 00–13 are complete; implementation -follows [ROADMAP.md](ROADMAP.md). Present today: +**Status: M1–M5 implemented (phase 1 feature-complete).** Design docs 00–14 are complete; +implementation follows [ROADMAP.md](ROADMAP.md). Present today: - `packages/domain` — zod schemas for every business and safety-chain object, exported to `contracts/` and regenerated as pydantic models (`skills-py/vpp_contracts`). @@ -27,13 +27,24 @@ follows [ROADMAP.md](ROADMAP.md). Present today: human-approved); envelope deviation-streak suspension; trigger service (scheduled / event / manual via router agent); provider-abstracted LLM port with an LLM-down mode; Case Desk HTTP API incl. AI insight cards. Bid release is a file export for manual upload (degraded - channel by design). + channel by design) or, in shadow mode, a recorded-never-submitted receipt. +- **Shadow run (M5, the phase-1 acceptance form)**: `VPP_MODE=SHADOW` runs the full loop on + live data with every external effect simulated — bids recorded, never submitted; the + market's answer simulated from the actual clearing price; dispatch to the simulation + gateway; D+1 review — and scores every day shadow-vs-human-vs-hindsight (`ShadowDayRecord`) + with lineage-completeness audit, plus an auto-generated docs/12 §4 KPI report. Kill-switch + hierarchy L0–L4 (`BreakerService`: permit revocation, envelope suspension, loss breaker, + channel breaker, AI-off) with the abnormal-day protocol and an auditable drill + (`POST /breakers/drill`). Replay driver: `npm run shadow -w @vpp/runtime -- --days 21`. All eight docs/01 invariants have automated tests (`packages/services/test/chain.test.ts`, `packages/runtime/test/lifecycle.test.ts`); the docs/07 D-1 16:00 and D+1 sections run in -`packages/runtime/test/m4.test.ts`. Storage is file-backed reference semantics; +`packages/runtime/test/m4.test.ts`; the ROADMAP M5 acceptance (21 consecutive shadow days with +complete lineage, KPI report, one widening recommendation not acted on) and the kill-switch +drill run in `packages/runtime/test/m5.test.ts`. Storage is file-backed reference semantics; Postgres/Timescale adapters and the programmatic trading-platform channel are later work. -M5 (shadow run) is next. +Next: run the shadow period on real Hubei data and close the business parameters in +`docs/open-questions.md`. ## Development @@ -44,7 +55,8 @@ pydantic models use `StrEnum` and PEP 604 unions). npm ci npm run check # typecheck, re-export contracts, run TS tests npm run eval -w @vpp/evals -- --check # L2 harness vs baseline (needs the skill service below) -npm run start -w @vpp/runtime # runtime + Case Desk API on :4100 (VPP_LLM_MODEL unset = LLM-down mode) +npm run start -w @vpp/runtime # runtime + Case Desk API on :4100 (VPP_MODE=SHADOW default; VPP_LLM_MODEL unset = LLM-down mode) +npm run shadow -w @vpp/runtime -- --days 21 # replay 21 shadow days from the dataset through the live skill service cd skills-py uv venv --python 3.11 .venv && uv pip install -r requirements.txt # or python3.11 -m venv @@ -78,6 +90,7 @@ CI (`.github/workflows/ci.yml`) runs both sides and fails if `contracts/` or | [04-control-plane](docs/04-control-plane.md) | Execution engine, edge autonomy, time/space cascades | | [05-skills-and-data](docs/05-skills-and-data.md) | Skill contracts, five-store data layer, policy packs | | [06-integration](docs/06-integration.md) | External system boundaries and degraded channels | +| [14-shadow-run-runbook](docs/14-shadow-run-runbook.md) | Shadow-run operations, kill-switch levels (trigger / authority / recovery), KPI definitions as implemented | | [07-scenario-walkthrough](docs/07-scenario-walkthrough.md) | Day-ahead spot bidding, D-1 → D → D+1 | | [08-implementation](docs/08-implementation.md) | Stack, LLM abstraction, deployment, milestones | | [09-runtime-implementation](docs/09-runtime-implementation.md) | Runtime on Mastra: workflows, suspend/resume, lineage | diff --git a/contracts/fixtures-invalid/breaker_record/unknown-level.json b/contracts/fixtures-invalid/breaker_record/unknown-level.json new file mode 100644 index 0000000..611380c --- /dev/null +++ b/contracts/fixtures-invalid/breaker_record/unknown-level.json @@ -0,0 +1,13 @@ +{ + "level": "L5", + "status": "ARMED", + "scope": null, + "reason": null, + "tripped_at": null, + "tripped_by": null, + "reset_at": null, + "reset_by": null, + "reset_basis": null, + "drill": false, + "trip_count": 0 +} diff --git a/contracts/fixtures-invalid/execution_receipt/unknown-channel.json b/contracts/fixtures-invalid/execution_receipt/unknown-channel.json new file mode 100644 index 0000000..4a5c0cd --- /dev/null +++ b/contracts/fixtures-invalid/execution_receipt/unknown-channel.json @@ -0,0 +1,9 @@ +{ + "receipt_id": "rcpt-001", + "proposal_digest": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", + "permit_id": "permit-001", + "channel": "EMAIL", + "idempotency_key": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb:permit-001", + "artifact_ref": "exports/bid-2026-03-15-prop-001.json", + "accepted_at": "2026-03-14T08:30:00Z" +} diff --git a/contracts/fixtures/breaker_record/l0-armed.json b/contracts/fixtures/breaker_record/l0-armed.json new file mode 100644 index 0000000..16a99b3 --- /dev/null +++ b/contracts/fixtures/breaker_record/l0-armed.json @@ -0,0 +1,13 @@ +{ + "level": "L0", + "status": "ARMED", + "scope": null, + "reason": null, + "tripped_at": null, + "tripped_by": null, + "reset_at": null, + "reset_by": null, + "reset_basis": null, + "drill": false, + "trip_count": 0 +} diff --git a/contracts/fixtures/breaker_record/l2-tripped.json b/contracts/fixtures/breaker_record/l2-tripped.json new file mode 100644 index 0000000..dcede2b --- /dev/null +++ b/contracts/fixtures/breaker_record/l2-tripped.json @@ -0,0 +1,16 @@ +{ + "level": "L2", + "status": "TRIPPED", + "scope": "*", + "reason": "daily expected loss 120000.00 yuan exceeds budget 100000", + "tripped_at": "2026-03-15T06:30:00Z", + "tripped_by": { + "id": "runtime", + "role": "system" + }, + "reset_at": null, + "reset_by": null, + "reset_basis": null, + "drill": false, + "trip_count": 1 +} diff --git a/contracts/fixtures/execution_receipt/shadow.json b/contracts/fixtures/execution_receipt/shadow.json new file mode 100644 index 0000000..531ad4f --- /dev/null +++ b/contracts/fixtures/execution_receipt/shadow.json @@ -0,0 +1,9 @@ +{ + "receipt_id": "rcpt-shadow-001", + "proposal_digest": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", + "permit_id": "permit-001", + "channel": "SHADOW", + "idempotency_key": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb:permit-001", + "artifact_ref": "shadow://bid/bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", + "accepted_at": "2026-03-14T08:30:00Z" +} diff --git a/contracts/fixtures/human_bid_record/platform-export.json b/contracts/fixtures/human_bid_record/platform-export.json new file mode 100644 index 0000000..b9376cc --- /dev/null +++ b/contracts/fixtures/human_bid_record/platform-export.json @@ -0,0 +1,210 @@ +{ + "id": "hb-2026-03-15", + "market_date": "2026-03-15", + "prices_yuan_per_mwh": { + "interval_minutes": 15, + "date": "2026-03-15", + "values": [ + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00", + "380.00" + ] + }, + "quantities_mwh": { + "interval_minutes": 15, + "date": "2026-03-15", + "values": [ + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0", + "11.0" + ] + }, + "source": "TRADING_PLATFORM_EXPORT", + "received_at": "2026-03-14T08:30:00Z" +} diff --git a/contracts/fixtures/kpi_report/window-30d.json b/contracts/fixtures/kpi_report/window-30d.json new file mode 100644 index 0000000..a34a1d3 --- /dev/null +++ b/contracts/fixtures/kpi_report/window-30d.json @@ -0,0 +1,51 @@ +{ + "id": "kpi-2026-03-15", + "window": { + "from": "2026-02-14", + "to": "2026-03-15", + "days": 30 + }, + "kpis": [ + { + "id": "FORECAST_LOAD_MAPE", + "value": "0.052", + "unit": "1", + "target": "0.08", + "comparator": "LTE", + "samples": 30, + "status": "MEET", + "definition": "aggregate day-ahead 96-interval load MAPE, rolling window mean" + }, + { + "id": "CROSS_REGION_MATCH", + "value": null, + "unit": "1", + "target": "0.85", + "comparator": "GTE", + "samples": 0, + "status": "NOT_APPLICABLE", + "definition": "federation commitments vs delivery confirmations (phase 2)" + } + ], + "comparison": { + "shadow_yuan": "15156000.00", + "human_yuan": "13337280.00", + "hindsight_yuan": "16200000.00", + "naive_yuan": "15913800.00", + "capture_ratio": "0.935556", + "uplift_vs_naive": "0.952383", + "days_with_human_baseline": 30 + }, + "shadow": { + "days": 30, + "complete_days": 30, + "consecutive_complete_days": 30, + "first_date": "2026-02-14", + "last_date": "2026-03-15", + "released_days": 27, + "pending_days": 3, + "widen_recommendations": 1, + "breaker_trips": 0 + }, + "generated_at": "2026-03-16T03:00:00Z" +} diff --git a/contracts/fixtures/shadow_day_record/released-day.json b/contracts/fixtures/shadow_day_record/released-day.json new file mode 100644 index 0000000..49ec9ce --- /dev/null +++ b/contracts/fixtures/shadow_day_record/released-day.json @@ -0,0 +1,71 @@ +{ + "id": "shadow-2026-03-15", + "market_date": "2026-03-15", + "shadow": { + "proposal_id": "prop-001", + "digest": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", + "outcome": "RELEASED", + "expected_revenue_yuan": "510600.00", + "line": { + "energy_mwh": "1200.000", + "cleared_energy_mwh": "1200.000", + "realised_revenue_yuan": "505200.00" + }, + "llm_used": false + }, + "human": { + "record_id": "hb-2026-03-15", + "source": "TRADING_PLATFORM_EXPORT", + "line": { + "energy_mwh": "1056.000", + "cleared_energy_mwh": "1056.000", + "realised_revenue_yuan": "444576.00" + } + }, + "hindsight": { + "energy_mwh": "1260.000", + "cleared_energy_mwh": "1260.000", + "realised_revenue_yuan": "540000.00" + }, + "naive": { + "energy_mwh": "1260.000", + "cleared_energy_mwh": "1260.000", + "realised_revenue_yuan": "530460.00" + }, + "award": { + "id": "award-2026-03-15", + "energy_mwh": "1200.000" + }, + "dispatch": { + "proposal_id": "prop-002", + "outcome": "RELEASED", + "shortfall_mwh": "0.000" + }, + "execution": { + "planned_mwh": "1200.000", + "delivered_mwh": "1164.000", + "deviation_mwh": "36.000", + "fulfillment_ratio": "0.970", + "within_band": true, + "simulated": true + }, + "forecast": { + "load_mape": "0.041", + "pv_nrmse": "0.062", + "price_mape": "0.017", + "price_coverage_p10_p90": "0.812" + }, + "decision_latency_ms": 1840, + "review_finding_id": "rf-2026-03-15-response", + "envelope_recommendations": [ + { + "envelope_id": "env-bid-001", + "action": "KEEP" + } + ], + "breakers_tripped": [], + "abnormal_day": false, + "lineage_complete": true, + "lineage_gaps": [], + "generated_at": "2026-03-16T03:00:00Z" +} diff --git a/contracts/schema/breaker_record.json b/contracts/schema/breaker_record.json new file mode 100644 index 0000000..3e0a320 --- /dev/null +++ b/contracts/schema/breaker_record.json @@ -0,0 +1,139 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "level": { + "type": "string", + "enum": [ + "L0", + "L1", + "L2", + "L3", + "L4" + ] + }, + "status": { + "type": "string", + "enum": [ + "ARMED", + "TRIPPED" + ] + }, + "scope": { + "type": [ + "string", + "null" + ] + }, + "reason": { + "type": [ + "string", + "null" + ] + }, + "tripped_at": { + "anyOf": [ + { + "type": "string", + "format": "date-time", + "pattern": "^(?:(?:\\d\\d[2468][048]|\\d\\d[13579][26]|\\d\\d0[48]|[02468][048]00|[13579][26]00)-02-29|\\d{4}-(?:(?:0[13578]|1[02])-(?:0[1-9]|[12]\\d|3[01])|(?:0[469]|11)-(?:0[1-9]|[12]\\d|30)|(?:02)-(?:0[1-9]|1\\d|2[0-8])))T(?:(?:[01]\\d|2[0-3]):[0-5]\\d:[0-5]\\d(?:\\.\\d+)?(?:Z))$" + }, + { + "type": "null" + } + ] + }, + "tripped_by": { + "anyOf": [ + { + "type": "object", + "properties": { + "id": { + "type": "string", + "minLength": 1 + }, + "role": { + "type": "string", + "minLength": 1 + } + }, + "required": [ + "id", + "role" + ], + "additionalProperties": false + }, + { + "type": "null" + } + ] + }, + "reset_at": { + "anyOf": [ + { + "type": "string", + "format": "date-time", + "pattern": "^(?:(?:\\d\\d[2468][048]|\\d\\d[13579][26]|\\d\\d0[48]|[02468][048]00|[13579][26]00)-02-29|\\d{4}-(?:(?:0[13578]|1[02])-(?:0[1-9]|[12]\\d|3[01])|(?:0[469]|11)-(?:0[1-9]|[12]\\d|30)|(?:02)-(?:0[1-9]|1\\d|2[0-8])))T(?:(?:[01]\\d|2[0-3]):[0-5]\\d:[0-5]\\d(?:\\.\\d+)?(?:Z))$" + }, + { + "type": "null" + } + ] + }, + "reset_by": { + "anyOf": [ + { + "type": "object", + "properties": { + "id": { + "type": "string", + "minLength": 1 + }, + "role": { + "type": "string", + "minLength": 1 + } + }, + "required": [ + "id", + "role" + ], + "additionalProperties": false + }, + { + "type": "null" + } + ] + }, + "reset_basis": { + "type": [ + "string", + "null" + ] + }, + "drill": { + "type": "boolean" + }, + "trip_count": { + "type": "integer", + "minimum": 0, + "maximum": 9007199254740991 + } + }, + "required": [ + "level", + "status", + "scope", + "reason", + "tripped_at", + "tripped_by", + "reset_at", + "reset_by", + "reset_basis", + "drill", + "trip_count" + ], + "additionalProperties": false, + "$id": "https://vpp-ai-platform/contracts/breaker_record.json", + "title": "BreakerRecord" +} diff --git a/contracts/schema/execution_receipt.json b/contracts/schema/execution_receipt.json index 9cdf27b..6ba8ee2 100644 --- a/contracts/schema/execution_receipt.json +++ b/contracts/schema/execution_receipt.json @@ -19,7 +19,8 @@ "enum": [ "FILE_EXPORT", "TRADING_PLATFORM_API", - "SIMULATION_GATEWAY" + "SIMULATION_GATEWAY", + "SHADOW" ] }, "idempotency_key": { diff --git a/contracts/schema/human_bid_record.json b/contracts/schema/human_bid_record.json new file mode 100644 index 0000000..7e252d1 --- /dev/null +++ b/contracts/schema/human_bid_record.json @@ -0,0 +1,94 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "id": { + "type": "string", + "minLength": 1 + }, + "market_date": { + "type": "string", + "pattern": "^\\d{4}-\\d{2}-\\d{2}$" + }, + "prices_yuan_per_mwh": { + "type": "object", + "properties": { + "interval_minutes": { + "type": "number", + "const": 15 + }, + "date": { + "type": "string", + "pattern": "^\\d{4}-\\d{2}-\\d{2}$" + }, + "values": { + "minItems": 96, + "maxItems": 96, + "type": "array", + "items": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + } + } + }, + "required": [ + "interval_minutes", + "date", + "values" + ], + "additionalProperties": false + }, + "quantities_mwh": { + "type": "object", + "properties": { + "interval_minutes": { + "type": "number", + "const": 15 + }, + "date": { + "type": "string", + "pattern": "^\\d{4}-\\d{2}-\\d{2}$" + }, + "values": { + "minItems": 96, + "maxItems": 96, + "type": "array", + "items": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + } + } + }, + "required": [ + "interval_minutes", + "date", + "values" + ], + "additionalProperties": false + }, + "source": { + "type": "string", + "enum": [ + "TRADING_PLATFORM_EXPORT", + "MANUAL_ENTRY", + "SYNTHETIC_NAIVE" + ] + }, + "received_at": { + "type": "string", + "format": "date-time", + "pattern": "^(?:(?:\\d\\d[2468][048]|\\d\\d[13579][26]|\\d\\d0[48]|[02468][048]00|[13579][26]00)-02-29|\\d{4}-(?:(?:0[13578]|1[02])-(?:0[1-9]|[12]\\d|3[01])|(?:0[469]|11)-(?:0[1-9]|[12]\\d|30)|(?:02)-(?:0[1-9]|1\\d|2[0-8])))T(?:(?:[01]\\d|2[0-3]):[0-5]\\d:[0-5]\\d(?:\\.\\d+)?(?:Z))$" + } + }, + "required": [ + "id", + "market_date", + "prices_yuan_per_mwh", + "quantities_mwh", + "source", + "received_at" + ], + "additionalProperties": false, + "$id": "https://vpp-ai-platform/contracts/human_bid_record.json", + "title": "HumanBidRecord" +} diff --git a/contracts/schema/insight_card.json b/contracts/schema/insight_card.json index db4be85..b31cec0 100644 --- a/contracts/schema/insight_card.json +++ b/contracts/schema/insight_card.json @@ -87,7 +87,10 @@ "SITUATION_REPORT", "REVIEW_FINDING", "DECISION_CASE", - "POTENTIAL_ASSESSMENT" + "POTENTIAL_ASSESSMENT", + "KPI_REPORT", + "SHADOW_DAY", + "BREAKER" ] }, "id": { diff --git a/contracts/schema/kpi_report.json b/contracts/schema/kpi_report.json new file mode 100644 index 0000000..05aebc9 --- /dev/null +++ b/contracts/schema/kpi_report.json @@ -0,0 +1,285 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "id": { + "type": "string", + "minLength": 1 + }, + "window": { + "type": "object", + "properties": { + "from": { + "anyOf": [ + { + "type": "string", + "pattern": "^\\d{4}-\\d{2}-\\d{2}$" + }, + { + "type": "null" + } + ] + }, + "to": { + "anyOf": [ + { + "type": "string", + "pattern": "^\\d{4}-\\d{2}-\\d{2}$" + }, + { + "type": "null" + } + ] + }, + "days": { + "type": "integer", + "minimum": 0, + "maximum": 9007199254740991 + } + }, + "required": [ + "from", + "to", + "days" + ], + "additionalProperties": false + }, + "kpis": { + "type": "array", + "items": { + "type": "object", + "properties": { + "id": { + "type": "string", + "enum": [ + "FORECAST_LOAD_MAPE", + "FORECAST_PV_NRMSE", + "POTENTIAL_ACCURACY", + "DECISION_LATENCY_P95_MS", + "DISPATCH_SUCCESS_RATE", + "REVENUE_UPLIFT_VS_HUMAN", + "CROSS_REGION_MATCH" + ] + }, + "value": { + "anyOf": [ + { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + { + "type": "null" + } + ] + }, + "unit": { + "type": "string", + "minLength": 1 + }, + "target": { + "anyOf": [ + { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + { + "type": "null" + } + ] + }, + "comparator": { + "type": "string", + "enum": [ + "LTE", + "GTE" + ] + }, + "samples": { + "type": "integer", + "minimum": 0, + "maximum": 9007199254740991 + }, + "status": { + "type": "string", + "enum": [ + "MEET", + "MISS", + "NO_DATA", + "NOT_APPLICABLE" + ] + }, + "definition": { + "type": "string", + "minLength": 1 + } + }, + "required": [ + "id", + "value", + "unit", + "target", + "comparator", + "samples", + "status", + "definition" + ], + "additionalProperties": false + } + }, + "comparison": { + "type": "object", + "properties": { + "shadow_yuan": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "human_yuan": { + "anyOf": [ + { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + { + "type": "null" + } + ] + }, + "hindsight_yuan": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "naive_yuan": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "capture_ratio": { + "anyOf": [ + { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + { + "type": "null" + } + ] + }, + "uplift_vs_naive": { + "anyOf": [ + { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + { + "type": "null" + } + ] + }, + "days_with_human_baseline": { + "type": "integer", + "minimum": 0, + "maximum": 9007199254740991 + } + }, + "required": [ + "shadow_yuan", + "human_yuan", + "hindsight_yuan", + "naive_yuan", + "capture_ratio", + "uplift_vs_naive", + "days_with_human_baseline" + ], + "additionalProperties": false + }, + "shadow": { + "type": "object", + "properties": { + "days": { + "type": "integer", + "minimum": 0, + "maximum": 9007199254740991 + }, + "complete_days": { + "type": "integer", + "minimum": 0, + "maximum": 9007199254740991 + }, + "consecutive_complete_days": { + "type": "integer", + "minimum": 0, + "maximum": 9007199254740991 + }, + "first_date": { + "anyOf": [ + { + "type": "string", + "pattern": "^\\d{4}-\\d{2}-\\d{2}$" + }, + { + "type": "null" + } + ] + }, + "last_date": { + "anyOf": [ + { + "type": "string", + "pattern": "^\\d{4}-\\d{2}-\\d{2}$" + }, + { + "type": "null" + } + ] + }, + "released_days": { + "type": "integer", + "minimum": 0, + "maximum": 9007199254740991 + }, + "pending_days": { + "type": "integer", + "minimum": 0, + "maximum": 9007199254740991 + }, + "widen_recommendations": { + "type": "integer", + "minimum": 0, + "maximum": 9007199254740991 + }, + "breaker_trips": { + "type": "integer", + "minimum": 0, + "maximum": 9007199254740991 + } + }, + "required": [ + "days", + "complete_days", + "consecutive_complete_days", + "first_date", + "last_date", + "released_days", + "pending_days", + "widen_recommendations", + "breaker_trips" + ], + "additionalProperties": false + }, + "generated_at": { + "type": "string", + "format": "date-time", + "pattern": "^(?:(?:\\d\\d[2468][048]|\\d\\d[13579][26]|\\d\\d0[48]|[02468][048]00|[13579][26]00)-02-29|\\d{4}-(?:(?:0[13578]|1[02])-(?:0[1-9]|[12]\\d|3[01])|(?:0[469]|11)-(?:0[1-9]|[12]\\d|30)|(?:02)-(?:0[1-9]|1\\d|2[0-8])))T(?:(?:[01]\\d|2[0-3]):[0-5]\\d:[0-5]\\d(?:\\.\\d+)?(?:Z))$" + } + }, + "required": [ + "id", + "window", + "kpis", + "comparison", + "shadow", + "generated_at" + ], + "additionalProperties": false, + "$id": "https://vpp-ai-platform/contracts/kpi_report.json", + "title": "KpiReport" +} diff --git a/contracts/schema/shadow_day_record.json b/contracts/schema/shadow_day_record.json new file mode 100644 index 0000000..18e9c84 --- /dev/null +++ b/contracts/schema/shadow_day_record.json @@ -0,0 +1,453 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "id": { + "type": "string", + "minLength": 1 + }, + "market_date": { + "type": "string", + "pattern": "^\\d{4}-\\d{2}-\\d{2}$" + }, + "shadow": { + "type": "object", + "properties": { + "proposal_id": { + "anyOf": [ + { + "type": "string", + "minLength": 1 + }, + { + "type": "null" + } + ] + }, + "digest": { + "anyOf": [ + { + "type": "string", + "pattern": "^[0-9a-f]{64}$" + }, + { + "type": "null" + } + ] + }, + "outcome": { + "type": "string", + "minLength": 1 + }, + "expected_revenue_yuan": { + "anyOf": [ + { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + { + "type": "null" + } + ] + }, + "line": { + "anyOf": [ + { + "type": "object", + "properties": { + "energy_mwh": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "cleared_energy_mwh": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "realised_revenue_yuan": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + } + }, + "required": [ + "energy_mwh", + "cleared_energy_mwh", + "realised_revenue_yuan" + ], + "additionalProperties": false + }, + { + "type": "null" + } + ] + }, + "llm_used": { + "type": "boolean" + } + }, + "required": [ + "proposal_id", + "digest", + "outcome", + "expected_revenue_yuan", + "line", + "llm_used" + ], + "additionalProperties": false + }, + "human": { + "anyOf": [ + { + "type": "object", + "properties": { + "record_id": { + "type": "string", + "minLength": 1 + }, + "source": { + "type": "string", + "enum": [ + "TRADING_PLATFORM_EXPORT", + "MANUAL_ENTRY", + "SYNTHETIC_NAIVE" + ] + }, + "line": { + "type": "object", + "properties": { + "energy_mwh": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "cleared_energy_mwh": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "realised_revenue_yuan": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + } + }, + "required": [ + "energy_mwh", + "cleared_energy_mwh", + "realised_revenue_yuan" + ], + "additionalProperties": false + } + }, + "required": [ + "record_id", + "source", + "line" + ], + "additionalProperties": false + }, + { + "type": "null" + } + ] + }, + "hindsight": { + "type": "object", + "properties": { + "energy_mwh": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "cleared_energy_mwh": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "realised_revenue_yuan": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + } + }, + "required": [ + "energy_mwh", + "cleared_energy_mwh", + "realised_revenue_yuan" + ], + "additionalProperties": false + }, + "naive": { + "type": "object", + "properties": { + "energy_mwh": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "cleared_energy_mwh": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "realised_revenue_yuan": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + } + }, + "required": [ + "energy_mwh", + "cleared_energy_mwh", + "realised_revenue_yuan" + ], + "additionalProperties": false + }, + "award": { + "anyOf": [ + { + "type": "object", + "properties": { + "id": { + "type": "string", + "minLength": 1 + }, + "energy_mwh": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + } + }, + "required": [ + "id", + "energy_mwh" + ], + "additionalProperties": false + }, + { + "type": "null" + } + ] + }, + "dispatch": { + "anyOf": [ + { + "type": "object", + "properties": { + "proposal_id": { + "type": "string", + "minLength": 1 + }, + "outcome": { + "type": "string", + "minLength": 1 + }, + "shortfall_mwh": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + } + }, + "required": [ + "proposal_id", + "outcome", + "shortfall_mwh" + ], + "additionalProperties": false + }, + { + "type": "null" + } + ] + }, + "execution": { + "anyOf": [ + { + "type": "object", + "properties": { + "planned_mwh": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "delivered_mwh": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "deviation_mwh": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "fulfillment_ratio": { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + "within_band": { + "type": "boolean" + }, + "simulated": { + "type": "boolean" + } + }, + "required": [ + "planned_mwh", + "delivered_mwh", + "deviation_mwh", + "fulfillment_ratio", + "within_band", + "simulated" + ], + "additionalProperties": false + }, + { + "type": "null" + } + ] + }, + "forecast": { + "type": "object", + "properties": { + "load_mape": { + "anyOf": [ + { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + { + "type": "null" + } + ] + }, + "pv_nrmse": { + "anyOf": [ + { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + { + "type": "null" + } + ] + }, + "price_mape": { + "anyOf": [ + { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + { + "type": "null" + } + ] + }, + "price_coverage_p10_p90": { + "anyOf": [ + { + "type": "string", + "pattern": "^-?\\d+(\\.\\d+)?$" + }, + { + "type": "null" + } + ] + } + }, + "required": [ + "load_mape", + "pv_nrmse", + "price_mape", + "price_coverage_p10_p90" + ], + "additionalProperties": false + }, + "decision_latency_ms": { + "anyOf": [ + { + "type": "integer", + "minimum": 0, + "maximum": 9007199254740991 + }, + { + "type": "null" + } + ] + }, + "review_finding_id": { + "anyOf": [ + { + "type": "string", + "minLength": 1 + }, + { + "type": "null" + } + ] + }, + "envelope_recommendations": { + "type": "array", + "items": { + "type": "object", + "properties": { + "envelope_id": { + "type": "string", + "minLength": 1 + }, + "action": { + "type": "string", + "enum": [ + "WIDEN", + "NARROW", + "SUSPEND", + "KEEP" + ] + } + }, + "required": [ + "envelope_id", + "action" + ], + "additionalProperties": false + } + }, + "breakers_tripped": { + "type": "array", + "items": { + "type": "string", + "enum": [ + "L0", + "L1", + "L2", + "L3", + "L4" + ] + } + }, + "abnormal_day": { + "type": "boolean" + }, + "lineage_complete": { + "type": "boolean" + }, + "lineage_gaps": { + "type": "array", + "items": { + "type": "string" + } + }, + "generated_at": { + "type": "string", + "format": "date-time", + "pattern": "^(?:(?:\\d\\d[2468][048]|\\d\\d[13579][26]|\\d\\d0[48]|[02468][048]00|[13579][26]00)-02-29|\\d{4}-(?:(?:0[13578]|1[02])-(?:0[1-9]|[12]\\d|3[01])|(?:0[469]|11)-(?:0[1-9]|[12]\\d|30)|(?:02)-(?:0[1-9]|1\\d|2[0-8])))T(?:(?:[01]\\d|2[0-3]):[0-5]\\d:[0-5]\\d(?:\\.\\d+)?(?:Z))$" + } + }, + "required": [ + "id", + "market_date", + "shadow", + "human", + "hindsight", + "naive", + "award", + "dispatch", + "execution", + "forecast", + "decision_latency_ms", + "review_finding_id", + "envelope_recommendations", + "breakers_tripped", + "abnormal_day", + "lineage_complete", + "lineage_gaps", + "generated_at" + ], + "additionalProperties": false, + "$id": "https://vpp-ai-platform/contracts/shadow_day_record.json", + "title": "ShadowDayRecord" +} diff --git a/docs/14-shadow-run-runbook.md b/docs/14-shadow-run-runbook.md new file mode 100644 index 0000000..ddaf7fd --- /dev/null +++ b/docs/14-shadow-run-runbook.md @@ -0,0 +1,74 @@ +# 14 · 影子运行手册(M5 一期验收形态) + +> 08 篇 §4 与 ROADMAP M5 的运行手册:影子运行怎么跑、熔断层级的「三要素」、 +> KPI 口径在代码中的落点。本篇是运行侧文档,架构决策仍以 03/12/13 篇为准。 + +## 1. 影子运行是什么 + +全链路接实时数据,**一切外部效果仿真**:申报只生成不提交(`SHADOW` 通道回执), +出清结果由当日实际日前出清价对影子申报仿真撮合(统一出清价规则),调度方案投递仿真网关, +D+1 复盘照常。每个市场日产出一条 `ShadowDayRecord`(影子 vs 人工 vs 后见之明三线对比 + +血缘完整性审计),并自动重算 `KpiReport`。运行 20+ 个连续完整血缘日即满足 M5 验收条件。 + +| 时点(Asia/Shanghai) | 触发项 | 流程模板 | +|---|---|---| +| D-1 06:00 | `situation-0600` | `day-ahead-situation` | +| D-1 08:00 | `bid-0800` | `day-ahead-bid` → `proposal-lifecycle`(影子网关) | +| D-1 16:00 | `shadow-clearing-1600` | 影子撮合 → `award-decomposition` → `proposal-lifecycle`(仿真网关) | +| D+1 02:00 | `shadow-close-0200` | `shadow-close`:仿真执行 → 计量 → `review` → 三线对比 → KPI | + +数据前提:D 日实际负荷/光伏/出清价须在 D-1 16:00 前经 `POST /market-data` 接入 +(走质量门禁,不合格曲线隔离并留痕);人工实际申报经 `POST /human-bids` 接入 +(无人工申报时对比记录 `human = null`,收益提升 KPI 记 `NO_DATA`)。 +触发失败(如出清价未到)记 `ScheduledTriggerFailed` 事件并在下一轮巡检重试。 + +- 实时模式:`VPP_MODE=SHADOW npm run start -w @vpp/runtime` +- 历史重放:`npm run shadow -w @vpp/runtime -- --days 21`(人工基线可用 `--human-baseline naive` + 生成占位,标记 `SYNTHETIC_NAIVE`,验收时必须替换为真实人工申报) + +**影子运行的两条建模假设**(不是业务参数):仿真执行的履约率取单元内资源可靠性评分均值 +(或固定比例,`shadow.fulfillment`);影子 P&L = 实际价撮合收益 − 边际成本 × 成交电量。 + +## 2. 熔断层级三要素(13 篇 §8 的落地) + +| 级别 | 触发条件 | 授权岗位(`breaker.authority`,B8 待定) | 效果(代码落点) | 恢复条件 | +|---|---|---|---|---| +| L0 单笔 | 人工判定某许可需撤 | senior-trader / ops-lead / risk-officer | `authority.revoke` → 网关拒收该 (Proposal, Permit) | 新许可须重走现势复核;L0 记录由人工复位 | +| L1 单包络 | 自动:连续 N 次偏差超阈(B4);人工 | ops-lead / risk-officer | `envelopes.suspend` → 该类动作回归人工 | 仅经 `envelope-review` 人工再批准(REACTIVATE) | +| L2 资金 | 自动:单日预期损失 > `breaker.daily_loss_budget`(B5) | risk-officer / ops-lead | 包络门全部改人工;规则校核只放行减仓类申报(`breaker-l2-reduce-only`) | 人工复位并填写依据(复盘结论编号 + 量化条件) | +| L3 通道 | 人工:申报/控制通道故障 | ops-lead / platform-admin | 调度方案 AUTHORIZED 但 `BLOCKED`(边缘进入断连策略);申报转人工文件通道 | 通道恢复后人工复位 | +| L4 全平台 | 人工:AI 建议停用 | platform-admin / ops-lead | 周期模板照跑出数据,不生成 Proposal(`ProposalSuppressed`) | 人工复位 | + +- 自动触发只允许 L1/L2(`breaker.automatic`),执行者记为 `runtime/system`;**复位永远是人**。 +- 智能体身份(`*-agent`)对任何级别无操作权(I1/I2)。 +- 每次触发/复位写 `BreakerTripped` / `BreakerReset` 事件;`GET /breakers` 给出五级状态。 +- **演练**:`POST /breakers/drill`(或 `runBreakerDrill`)对五级逐一「触发→在链路上验证效果→复位」, + 全部使用演练对象,不触碰在途 Proposal;事件 `BreakerDrillStep` / `BreakerDrillCompleted` 留痕。 + ROADMAP M5 要求演练一次,13 篇要求每季度至少一级。 + +**异常日协议**(13 篇 §1):态势报告 `EXTREME`(B7 触发条件待定,当前以价格区间比占位)→ +该市场日自动登记为异常日,所有 Proposal 走人工(`abnormal-day protocol` 原因进收件箱), +同时开专题工单;人工可 `POST /abnormal-days/:date` 登记或 `/clear` 解除。 + +## 3. KPI 口径(12 篇 §4 → `computeKpiReport`) + +| KPI | 实现口径 | 目标(占位,C1/C2 待冻结) | +|---|---|---| +| FORECAST_LOAD_MAPE | 聚合负荷 P50 vs 计量 96 点 MAPE,窗口均值 | ≤ 0.08 | +| FORECAST_PV_NRMSE | 光伏 P50 RMSE / 装机(登记的 PV 资源额定功率之和,缺省 `shadow.pvCapacityMw`) | ≤ 0.10(占位) | +| POTENTIAL_ACCURACY | 已调度日中 \|计划 − 履约\| / 计划 ≤ 容差 的占比 | ≥ 0.90,容差 0.10(占位) | +| DECISION_LATENCY_P95_MS | 工单开启 → Proposal 到达 AUTO_APPROVED/PENDING_HUMAN 的 P95(不含人工等待) | ≤ 180000 | +| DISPATCH_SUCCESS_RATE | 持有效许可的调度方案中,回执确认且偏差 ≤ 带宽 的占比 | ≥ 0.98,带宽 10%(占位) | +| REVENUE_UPLIFT_VS_HUMAN | 同日、同实际价撮合下 (Σ影子收益 − Σ人工收益) / Σ人工收益 | ≥ 0.15(基线定义须冻结) | +| CROSS_REGION_MATCH | 二期联邦工件,`NOT_APPLICABLE` | ≥ 0.85 | + +窗口 = 最近 `kpi.windowDays`(默认 30)个影子日;每条 KPI 附口径文字,验收时以报告中的文字为准签认。 +另附三线合计(影子/人工/后见之明/朴素)、`capture_ratio`、连续完整血缘天数、 +包络放宽建议数(**只记录不执行**——放宽须经 `envelope-review` 人工批准,属二期治理动作)。 + +## 4. 相关配置键 + +`mode`、`breaker.dailyLossBudgetYuan`(B5)、`breaker.authority`(B8)、`breaker.automatic`、 +`kpi.*`(C1/C2)、`shadow.fulfillment`、`shadow.pvCapacityMw`(C2)、`extremeDayPriceRatio`(B7)、 +`widenAfterCompliantDays`(B4/B9)。全部在 `packages/runtime/src/runtime.ts` 的 +`DEFAULT_RUNTIME_CONFIG` 以占位值出现,并带 `OPEN-QUESTION` 注释。 diff --git a/packages/domain/scripts/make-fixtures.ts b/packages/domain/scripts/make-fixtures.ts index d340776..e78c1f3 100644 --- a/packages/domain/scripts/make-fixtures.ts +++ b/packages/domain/scripts/make-fixtures.ts @@ -329,6 +329,15 @@ const valid: Record> = { }, }, execution_receipt: { + shadow: { + receipt_id: 'rcpt-shadow-001', + proposal_digest: REF_B, + permit_id: 'permit-001', + channel: 'SHADOW', + idempotency_key: `${REF_B}:permit-001`, + artifact_ref: `shadow://bid/${REF_B}`, + accepted_at: T1, + }, 'file-export': { receipt_id: 'rcpt-001', proposal_digest: REF_B, @@ -581,6 +590,86 @@ const valid: Record> = { payload: { situation_report_id: 'sit-001' }, }, }, + human_bid_record: { + 'platform-export': { + id: 'hb-2026-03-15', + market_date: DATE, + prices_yuan_per_mwh: curve('380.00'), + quantities_mwh: curve('11.0'), + source: 'TRADING_PLATFORM_EXPORT', + received_at: T1, + }, + }, + breaker_record: { + 'l2-tripped': { + level: 'L2', + status: 'TRIPPED', + scope: '*', + reason: 'daily expected loss 120000.00 yuan exceeds budget 100000', + tripped_at: '2026-03-15T06:30:00Z', + tripped_by: { id: 'runtime', role: 'system' }, + reset_at: null, + reset_by: null, + reset_basis: null, + drill: false, + trip_count: 1, + }, + 'l0-armed': { + level: 'L0', + status: 'ARMED', + scope: null, + reason: null, + tripped_at: null, + tripped_by: null, + reset_at: null, + reset_by: null, + reset_basis: null, + drill: false, + trip_count: 0, + }, + }, + shadow_day_record: { + 'released-day': { + id: 'shadow-2026-03-15', + market_date: DATE, + shadow: { + proposal_id: 'prop-001', + digest: REF_B, + outcome: 'RELEASED', + expected_revenue_yuan: '510600.00', + line: { energy_mwh: '1200.000', cleared_energy_mwh: '1200.000', realised_revenue_yuan: '505200.00' }, + llm_used: false, + }, + human: { record_id: 'hb-2026-03-15', source: 'TRADING_PLATFORM_EXPORT', line: { energy_mwh: '1056.000', cleared_energy_mwh: '1056.000', realised_revenue_yuan: '444576.00' } }, + hindsight: { energy_mwh: '1260.000', cleared_energy_mwh: '1260.000', realised_revenue_yuan: '540000.00' }, + naive: { energy_mwh: '1260.000', cleared_energy_mwh: '1260.000', realised_revenue_yuan: '530460.00' }, + award: { id: 'award-2026-03-15', energy_mwh: '1200.000' }, + dispatch: { proposal_id: 'prop-002', outcome: 'RELEASED', shortfall_mwh: '0.000' }, + execution: { planned_mwh: '1200.000', delivered_mwh: '1164.000', deviation_mwh: '36.000', fulfillment_ratio: '0.970', within_band: true, simulated: true }, + forecast: { load_mape: '0.041', pv_nrmse: '0.062', price_mape: '0.017', price_coverage_p10_p90: '0.812' }, + decision_latency_ms: 1840, + review_finding_id: 'rf-2026-03-15-response', + envelope_recommendations: [{ envelope_id: 'env-bid-001', action: 'KEEP' }], + breakers_tripped: [], + abnormal_day: false, + lineage_complete: true, + lineage_gaps: [], + generated_at: '2026-03-16T03:00:00Z', + }, + }, + kpi_report: { + 'window-30d': { + id: 'kpi-2026-03-15', + window: { from: '2026-02-14', to: DATE, days: 30 }, + kpis: [ + { id: 'FORECAST_LOAD_MAPE', value: '0.052', unit: '1', target: '0.08', comparator: 'LTE', samples: 30, status: 'MEET', definition: 'aggregate day-ahead 96-interval load MAPE, rolling window mean' }, + { id: 'CROSS_REGION_MATCH', value: null, unit: '1', target: '0.85', comparator: 'GTE', samples: 0, status: 'NOT_APPLICABLE', definition: 'federation commitments vs delivery confirmations (phase 2)' }, + ], + comparison: { shadow_yuan: '15156000.00', human_yuan: '13337280.00', hindsight_yuan: '16200000.00', naive_yuan: '15913800.00', capture_ratio: '0.935556', uplift_vs_naive: '0.952383', days_with_human_baseline: 30 }, + shadow: { days: 30, complete_days: 30, consecutive_complete_days: 30, first_date: '2026-02-14', last_date: DATE, released_days: 27, pending_days: 3, widen_recommendations: 1, breaker_trips: 0 }, + generated_at: '2026-03-16T03:00:00Z', + }, + }, } // Each invalid case breaks exactly one rule and must fail on both sides. @@ -639,6 +728,14 @@ const invalid: Record> = { return o })(), }, + breaker_record: { + // a level outside the L0–L4 hierarchy (docs/13 §8) + 'unknown-level': { ...(valid['breaker_record']!['l0-armed'] as object), level: 'L5' }, + }, + execution_receipt: { + // channel outside the enum: a receipt must say which channel took the effect (I5) + 'unknown-channel': { ...(valid['execution_receipt']!['file-export'] as object), channel: 'EMAIL' }, + }, } ;(invalid['proposal']!['bid-float-money'] as any).payload.expected_revenue_yuan = 510600.0 diff --git a/packages/domain/src/chain.ts b/packages/domain/src/chain.ts index 88de7a5..759b86e 100644 --- a/packages/domain/src/chain.ts +++ b/packages/domain/src/chain.ts @@ -67,7 +67,8 @@ export const ExecutionReceipt = z.object({ receipt_id: Id, proposal_digest: SnapshotRef, permit_id: Id, - channel: z.enum(['FILE_EXPORT', 'TRADING_PLATFORM_API', 'SIMULATION_GATEWAY']), + /** SHADOW: bid generated and recorded, never submitted (docs/08 §4 M5). */ + channel: z.enum(['FILE_EXPORT', 'TRADING_PLATFORM_API', 'SIMULATION_GATEWAY', 'SHADOW']), idempotency_key: z.string().min(1), artifact_ref: z.string().min(1), accepted_at: IsoUtc, diff --git a/packages/domain/src/index.ts b/packages/domain/src/index.ts index 2634964..b55653f 100644 --- a/packages/domain/src/index.ts +++ b/packages/domain/src/index.ts @@ -36,6 +36,7 @@ import { PotentialAssessmentResult, } from './dispatch.js' import { EnvelopeChangeRequest, InsightCard, ReviewFinding, SemanticMemoryEntry } from './review.js' +import { BreakerRecord, HumanBidRecord, KpiReport, ShadowDayRecord } from './shadow.js' export * from './common.js' export * from './proposal.js' @@ -51,6 +52,7 @@ export * from './skill.js' export * from './chain.js' export * from './dispatch.js' export * from './review.js' +export * from './shadow.js' /** * Registry driving the contracts pipeline: keys become schema/fixture/module @@ -94,4 +96,8 @@ export const schemaRegistry: Record = { semantic_memory_entry: SemanticMemoryEntry, envelope_change_request: EnvelopeChangeRequest, insight_card: InsightCard, + human_bid_record: HumanBidRecord, + shadow_day_record: ShadowDayRecord, + kpi_report: KpiReport, + breaker_record: BreakerRecord, } diff --git a/packages/domain/src/review.ts b/packages/domain/src/review.ts index c182359..506d9aa 100644 --- a/packages/domain/src/review.ts +++ b/packages/domain/src/review.ts @@ -94,7 +94,7 @@ export const InsightCard = z.object({ title: z.string().min(1), headline: z.string().min(1), metrics: z.array(z.object({ name: z.string().min(1), value: DecimalString, unit: z.string().min(1), ref: LineageRef.nullable() })), - source: z.object({ kind: z.enum(['SITUATION_REPORT', 'REVIEW_FINDING', 'DECISION_CASE', 'POTENTIAL_ASSESSMENT']), id: Id, ref: SnapshotRef.nullable() }), + source: z.object({ kind: z.enum(['SITUATION_REPORT', 'REVIEW_FINDING', 'DECISION_CASE', 'POTENTIAL_ASSESSMENT', 'KPI_REPORT', 'SHADOW_DAY', 'BREAKER']), id: Id, ref: SnapshotRef.nullable() }), generated_at: IsoUtc, }) export type InsightCard = z.infer diff --git a/packages/domain/src/shadow.ts b/packages/domain/src/shadow.ts new file mode 100644 index 0000000..323428b --- /dev/null +++ b/packages/domain/src/shadow.ts @@ -0,0 +1,155 @@ +import { z } from 'zod' +import { Curve96, DecimalString, Id, IsoUtc, MarketDate, SnapshotRef } from './common.js' + +/** + * Shadow-run objects (docs/08 §4 M5, docs/12 §1 L4, docs/13 §8). The shadow + * run is the phase-1 acceptance form: the full loop runs on live data with + * every external effect simulated, and each day is scored against the human + * decision and perfect hindsight. Every number here is computed by the + * deterministic shadow services from recorded objects — never by an LLM. + */ + +/** The human trader's actual submission for a market date (ingested from the legacy channel). */ +export const HumanBidSource = z.enum(['TRADING_PLATFORM_EXPORT', 'MANUAL_ENTRY', 'SYNTHETIC_NAIVE']) + +export const HumanBidRecord = z.object({ + id: Id, + market_date: MarketDate, + prices_yuan_per_mwh: Curve96, + quantities_mwh: Curve96, + source: HumanBidSource, + received_at: IsoUtc, +}) +export type HumanBidRecord = z.infer + +/** One line of the shadow-vs-human-vs-hindsight comparison, all cleared at the day's actual prices. */ +export const BidLine = z.object({ + energy_mwh: DecimalString, + cleared_energy_mwh: DecimalString, + realised_revenue_yuan: DecimalString, +}) +export type BidLine = z.infer + +/** Kill-switch hierarchy (docs/13 §8): L0 permit → L1 envelope → L2 loss → L3 channel → L4 AI off. */ +export const BreakerLevel = z.enum(['L0', 'L1', 'L2', 'L3', 'L4']) +export type BreakerLevel = z.infer + +const Actor = z.object({ id: Id, role: z.string().min(1) }) + +export const BreakerRecord = z.object({ + level: BreakerLevel, + status: z.enum(['ARMED', 'TRIPPED']), + /** What the trip applied to: permit id (L0), envelope id (L1), '*' for global levels. */ + scope: z.string().nullable(), + reason: z.string().nullable(), + tripped_at: IsoUtc.nullable(), + tripped_by: Actor.nullable(), + reset_at: IsoUtc.nullable(), + reset_by: Actor.nullable(), + /** Quantified recovery condition + review reference (docs/13 §8 三要素). */ + reset_basis: z.string().nullable(), + /** True when the last trip was a drill (docs/13 §8: 每季度演练至少一级). */ + drill: z.boolean(), + trip_count: z.int().nonnegative(), +}) +export type BreakerRecord = z.infer + +/** Daily shadow record: what the platform would have done, what the human did, what hindsight allows. */ +export const ShadowDayRecord = z.object({ + id: Id, + market_date: MarketDate, + shadow: z.object({ + proposal_id: Id.nullable(), + digest: SnapshotRef.nullable(), + /** Terminal proposal status that day (RELEASED / PENDING_HUMAN / REJECTED / STALE / NONE / SUPPRESSED). */ + outcome: z.string().min(1), + expected_revenue_yuan: DecimalString.nullable(), + line: BidLine.nullable(), + llm_used: z.boolean(), + }), + human: z.object({ record_id: Id, source: HumanBidSource, line: BidLine }).nullable(), + hindsight: BidLine, + naive: BidLine, + award: z.object({ id: Id, energy_mwh: DecimalString }).nullable(), + dispatch: z.object({ proposal_id: Id, outcome: z.string().min(1), shortfall_mwh: DecimalString }).nullable(), + execution: z + .object({ + planned_mwh: DecimalString, + delivered_mwh: DecimalString, + deviation_mwh: DecimalString, + fulfillment_ratio: DecimalString, + within_band: z.boolean(), + /** True when execution was simulated by the simulation gateway rather than reported from the field. */ + simulated: z.boolean(), + }) + .nullable(), + forecast: z.object({ + load_mape: DecimalString.nullable(), + pv_nrmse: DecimalString.nullable(), + price_mape: DecimalString.nullable(), + price_coverage_p10_p90: DecimalString.nullable(), + }), + /** Trigger → proposal reaches AUTO_APPROVED/PENDING_HUMAN (docs/12 §4: human waiting time excluded). */ + decision_latency_ms: z.int().nonnegative().nullable(), + review_finding_id: Id.nullable(), + envelope_recommendations: z.array(z.object({ envelope_id: Id, action: z.enum(['WIDEN', 'NARROW', 'SUSPEND', 'KEEP']) })), + breakers_tripped: z.array(BreakerLevel), + abnormal_day: z.boolean(), + lineage_complete: z.boolean(), + lineage_gaps: z.array(z.string()), + generated_at: IsoUtc, +}) +export type ShadowDayRecord = z.infer + +/** docs/12 §4 KPI table, one row per proposal indicator. Definitions are OPEN-QUESTION C1/C2 until frozen. */ +export const KpiId = z.enum([ + 'FORECAST_LOAD_MAPE', + 'FORECAST_PV_NRMSE', + 'POTENTIAL_ACCURACY', + 'DECISION_LATENCY_P95_MS', + 'DISPATCH_SUCCESS_RATE', + 'REVENUE_UPLIFT_VS_HUMAN', + 'CROSS_REGION_MATCH', +]) +export type KpiId = z.infer + +export const KpiEntry = z.object({ + id: KpiId, + value: DecimalString.nullable(), + unit: z.string().min(1), + target: DecimalString.nullable(), + comparator: z.enum(['LTE', 'GTE']), + samples: z.int().nonnegative(), + status: z.enum(['MEET', 'MISS', 'NO_DATA', 'NOT_APPLICABLE']), + definition: z.string().min(1), +}) +export type KpiEntry = z.infer + +export const KpiReport = z.object({ + id: Id, + window: z.object({ from: MarketDate.nullable(), to: MarketDate.nullable(), days: z.int().nonnegative() }), + kpis: z.array(KpiEntry), + /** Three-line comparison totals over the window (docs/12 §1 L4 影子运行). */ + comparison: z.object({ + shadow_yuan: DecimalString, + human_yuan: DecimalString.nullable(), + hindsight_yuan: DecimalString, + naive_yuan: DecimalString, + capture_ratio: DecimalString.nullable(), + uplift_vs_naive: DecimalString.nullable(), + days_with_human_baseline: z.int().nonnegative(), + }), + shadow: z.object({ + days: z.int().nonnegative(), + complete_days: z.int().nonnegative(), + consecutive_complete_days: z.int().nonnegative(), + first_date: MarketDate.nullable(), + last_date: MarketDate.nullable(), + released_days: z.int().nonnegative(), + pending_days: z.int().nonnegative(), + widen_recommendations: z.int().nonnegative(), + breaker_trips: z.int().nonnegative(), + }), + generated_at: IsoUtc, +}) +export type KpiReport = z.infer diff --git a/packages/evals/reports/baselines/l2-synthetic-hubei-v0.json b/packages/evals/reports/baselines/l2-synthetic-hubei-v0.json index 35ebb40..9f5f0b8 100644 --- a/packages/evals/reports/baselines/l2-synthetic-hubei-v0.json +++ b/packages/evals/reports/baselines/l2-synthetic-hubei-v0.json @@ -10,7 +10,9 @@ "pv-forecast": "1.0.0", "price-forecast": "1.0.0", "bid-optimization-milp": "1.0.0", - "report-generator": "1.0.0" + "report-generator": "1.0.0", + "potential-assessment": "1.0.0", + "dispatch-optimization": "1.0.0" }, "config": { "window": 28, @@ -58,5 +60,5 @@ "uplift_vs_naive": 1.188643 } }, - "digest": "4018e32ab695422c613e211aba1392164f3bb9c30076986fdfda220fc9bba047" + "digest": "0a6cafb6209114f7c6e064499f63918e8cebe04fa7f65dac252b0d0216a77764" } diff --git a/packages/evals/src/metrics.ts b/packages/evals/src/metrics.ts index 69aa940..0441982 100644 --- a/packages/evals/src/metrics.ts +++ b/packages/evals/src/metrics.ts @@ -1,103 +1,7 @@ /** - * L2 skill metrics (docs/12 §1 L2, §4 口径). Pure functions over plain numbers; - * the harness converts decimal strings at the boundary and quantizes results - * when it writes a report. KPI thresholds (≤ 8% etc.) are judged on real data - * and are OPEN-QUESTION C2 in their exact statistical level — nothing here - * hardcodes a pass/fail number. + * L2 skill metrics now live in @vpp/services (the shadow run scores days with + * the same functions the harness uses — one definition, docs/12 §4). Re-exported + * here so the harness and its tests keep their import path. */ - -export const mean = (xs: number[]): number => - xs.length === 0 ? Number.NaN : xs.reduce((a, b) => a + b, 0) / xs.length - -/** Mean absolute percentage error over intervals where actual ≠ 0. */ -export function mape(actual: number[], pred: number[]): number { - const terms: number[] = [] - for (let i = 0; i < actual.length; i++) { - const a = actual[i]! - if (a !== 0) terms.push(Math.abs((pred[i]! - a) / a)) - } - return mean(terms) -} - -/** RMSE normalised by installed capacity — the PV metric docs/12 §4 suggests. */ -export function nrmse(actual: number[], pred: number[], capacity: number): number { - const se = actual.map((a, i) => (pred[i]! - a) ** 2) - return Math.sqrt(mean(se)) / capacity -} - -/** Share of actuals inside [lower, upper]; nominal for a P10–P90 band is 0.80. */ -export function coverage(actual: number[], lower: number[], upper: number[]): number { - let hits = 0 - for (let i = 0; i < actual.length; i++) { - if (actual[i]! >= lower[i]! && actual[i]! <= upper[i]!) hits++ - } - return hits / actual.length -} - -/** - * Direction accuracy (docs/12 L2 电价 方向准确率): share of interval pairs whose - * high/low ordering the forecast gets right. Bid optimisation depends on the - * ranking of intervals far more than on absolute price level. - */ -export function directionAccuracy(actual: number[], pred: number[]): number { - let agree = 0 - let pairs = 0 - for (let i = 0; i < actual.length; i++) { - for (let j = i + 1; j < actual.length; j++) { - const da = actual[j]! - actual[i]! - const dp = pred[j]! - pred[i]! - if (da === 0 && dp === 0) continue - pairs++ - if (Math.sign(da) === Math.sign(dp)) agree++ - } - } - return pairs === 0 ? Number.NaN : agree / pairs -} - -export interface Bid { - offers: number[] - quantities: number[] -} - -/** Uniform-price clearing: an offer clears where it does not exceed the realised price. */ -export function realisedRevenue(bid: Bid, actualPrice: number[]): number { - let total = 0 - for (let t = 0; t < actualPrice.length; t++) { - if (bid.offers[t]! <= actualPrice[t]!) total += actualPrice[t]! * bid.quantities[t]! - } - return total -} - -/** - * Lower-bound baseline: a price-taker offering a flat profile that meets the - * upper energy bound (or as much as capacity allows), pro-rata to capacity. - */ -export function naiveBid(capMwh: number[], energyMax: number): Bid { - const sellable = capMwh.reduce((a, b) => a + b, 0) - const scale = sellable === 0 ? 0 : Math.min(1, energyMax / sellable) - return { offers: capMwh.map(() => 0), quantities: capMwh.map((c) => c * scale) } -} - -/** - * Upper-bound baseline: perfect hindsight — fill the highest-priced intervals - * first up to the energy bound, honouring per-interval capacity and block - * size. Offers at zero so everything clears. - */ -export function hindsightBid( - actualPrice: number[], - capMwh: number[], - energyMax: number, - minBlock: number, -): Bid { - const order = actualPrice.map((_, i) => i).sort((a, b) => actualPrice[b]! - actualPrice[a]!) - const quantities = capMwh.map(() => 0) - let room = energyMax - for (const t of order) { - const q = Math.min(capMwh[t]!, room) - if (q < minBlock) continue - quantities[t] = q - room -= q - if (room <= 0) break - } - return { offers: capMwh.map(() => 0), quantities } -} +export { coverage, directionAccuracy, hindsightBid, mape, mean, naiveBid, nrmse, realisedRevenue } from '@vpp/services' +export type { Bid } from '@vpp/services' diff --git a/packages/runtime/package.json b/packages/runtime/package.json index 9c8ce62..e15ed47 100644 --- a/packages/runtime/package.json +++ b/packages/runtime/package.json @@ -9,7 +9,8 @@ "scripts": { "typecheck": "tsc -p tsconfig.json", "test": "vitest run", - "start": "tsx src/main.ts" + "start": "tsx src/main.ts", + "shadow": "tsx src/shadow-cli.ts" }, "dependencies": { "@mastra/core": "^1.63.2", diff --git a/packages/runtime/src/api.ts b/packages/runtime/src/api.ts index 3caf231..5d596fc 100644 --- a/packages/runtime/src/api.ts +++ b/packages/runtime/src/api.ts @@ -1,10 +1,12 @@ import { createServer } from 'node:http' import type { IncomingMessage, Server, ServerResponse } from 'node:http' import { z } from 'zod' -import { InsightCard } from '@vpp/domain' +import { BreakerLevel, InsightCard } from '@vpp/domain' +import { BreakerAuthorityError } from '@vpp/services' import { insightCards } from './cards.js' import { LlmUnavailable } from './llm.js' import type { Runtime } from './runtime.js' +import { runBreakerDrill } from './shadow.js' import type { TriggerService } from './trigger.js' import { ResumeDecision } from './workflows/proposal-lifecycle.js' @@ -17,6 +19,9 @@ import { ResumeDecision } from './workflows/proposal-lifecycle.js' */ const DecideBody = z.object({ decision: z.enum(['approve', 'reject']), comment: z.string().optional() }) const TaskBody = z.object({ message: z.string().min(1) }) +const TripBody = z.object({ reason: z.string().min(1), scope: z.string().min(1).optional() }) +const ResetBody = z.object({ basis: z.string().min(1) }) +const AbnormalBody = z.object({ reason: z.string().min(1) }) const json = (res: ServerResponse, status: number, body: unknown) => { res.writeHead(status, { 'content-type': 'application/json' }) @@ -42,7 +47,7 @@ export function createApi(rt: Runtime, triggers: TriggerService): Server { const parts = url.pathname.split('/').filter(Boolean) const user = { id: req.headers['x-user-id'], role: req.headers['x-user-role'] } try { - if (req.method === 'GET' && url.pathname === '/health') return json(res, 200, { status: 'ok', llm: rt.ctx.llm.backend }) + if (req.method === 'GET' && url.pathname === '/health') return json(res, 200, { status: 'ok', llm: rt.ctx.llm.backend, mode: rt.ctx.config.mode, breakers_tripped: rt.ctx.breakers.state().filter((b) => b.status === 'TRIPPED').map((b) => b.level) }) if (req.method === 'GET' && url.pathname === '/inbox') return json(res, 200, rt.ctx.caseDesk.inbox().filter((p) => p.run_id !== '')) if (req.method === 'GET' && url.pathname === '/cases') return json(res, 200, rt.ctx.caseDesk.cases.list().map((r) => r.value)) if (req.method === 'GET' && parts[0] === 'cases' && parts[1] && parts.length === 2) return json(res, 200, rt.ctx.caseDesk.read(parts[1])) @@ -71,10 +76,50 @@ export function createApi(rt: Runtime, triggers: TriggerService): Server { const body = TaskBody.parse(await readBody(req)) return json(res, 200, await triggers.manual(body.message)) } + + // ---- M5: live data, shadow comparison, KPI dashboard, kill switches + if (req.method === 'POST' && url.pathname === '/market-data') return json(res, 200, triggers.onMarketData((await readBody(req)) as never)) + if (req.method === 'POST' && url.pathname === '/human-bids') return json(res, 200, triggers.onHumanBid((await readBody(req)) as never)) + if (req.method === 'GET' && url.pathname === '/shadow/days') return json(res, 200, rt.ctx.shadowDays.list().map((r) => r.value)) + if (req.method === 'GET' && parts[0] === 'shadow' && parts[1] === 'days' && parts[2]) { + const row = rt.ctx.shadowDays.get(`shadow-${parts[2]}`) + return row ? json(res, 200, row.value) : json(res, 404, { error: `no shadow record for ${parts[2]}` }) + } + if (req.method === 'GET' && url.pathname === '/kpi') { + const latest = rt.ctx.kpiReports.list().map((r) => r.value).sort((a, b) => (a.generated_at < b.generated_at ? -1 : 1)).at(-1) + return latest ? json(res, 200, latest) : json(res, 404, { error: 'no KPI report yet' }) + } + if (req.method === 'GET' && url.pathname === '/breakers') return json(res, 200, { levels: rt.ctx.breakers.state() }) + if (req.method === 'POST' && parts[0] === 'breakers') { + if (typeof user.id !== 'string' || typeof user.role !== 'string') return json(res, 401, { error: 'x-user-id and x-user-role required' }) + const actor = { id: user.id, role: user.role } + if (parts[1] === 'drill' && parts.length === 2) return json(res, 200, await runBreakerDrill(rt.ctx, actor)) + const level = BreakerLevel.parse(parts[1]) + if (parts[2] === 'trip') { + const body = TripBody.parse(await readBody(req)) + return json(res, 200, rt.ctx.breakers.trip(level, actor, body.reason, body.scope ?? null, rt.ctx.clock())) + } + if (parts[2] === 'reset') { + const body = ResetBody.parse(await readBody(req)) + return json(res, 200, rt.ctx.breakers.reset(level, actor, body.basis, rt.ctx.clock())) + } + } + if (req.method === 'POST' && parts[0] === 'abnormal-days' && parts[1]) { + if (typeof user.id !== 'string' || typeof user.role !== 'string') return json(res, 401, { error: 'x-user-id and x-user-role required' }) + const actor = { id: user.id, role: user.role } + if (parts[2] === 'clear') { + rt.ctx.breakers.clearAbnormalDay(parts[1], actor, rt.ctx.clock()) + return json(res, 200, { market_date: parts[1], abnormal: false }) + } + const body = AbnormalBody.parse(await readBody(req)) + rt.ctx.breakers.declareAbnormalDay(parts[1], actor, body.reason, rt.ctx.clock()) + return json(res, 200, { market_date: parts[1], abnormal: true }) + } return json(res, 404, { error: 'not found' }) } catch (e) { if (e instanceof z.ZodError) return json(res, 400, { error: e.issues }) if (e instanceof LlmUnavailable) return json(res, 503, { error: e.message }) + if (e instanceof BreakerAuthorityError) return json(res, 403, { error: e.message }) return json(res, 500, { error: (e as Error).message }) } }) diff --git a/packages/runtime/src/cards.ts b/packages/runtime/src/cards.ts index e872d5e..b406bb4 100644 --- a/packages/runtime/src/cards.ts +++ b/packages/runtime/src/cards.ts @@ -69,7 +69,52 @@ export function insightCards(ctx: RuntimeContext, page: InsightCard['page']): In }) } } + const shadowDays = ctx.shadowDays.list().map((r) => r.value).sort((a, b) => (a.market_date < b.market_date ? -1 : 1)) + const latestDay = shadowDays.at(-1) + if (page === 'TRADING_DESK' && latestDay) { + // Three-line comparison for the latest shadow day (docs/12 §1 L4): the only legitimate basis for widening envelopes. + cards.push({ + id: `card-shadow-${latestDay.market_date}`, + page, + market_date: latestDay.market_date, + title: 'Shadow vs human vs hindsight', + headline: `Shadow bid ${latestDay.shadow.outcome}; ${latestDay.human ? `human baseline ${latestDay.human.source}` : 'no human baseline ingested'}; lineage ${latestDay.lineage_complete ? 'complete' : `gaps: ${latestDay.lineage_gaps.length}`}`, + metrics: [ + ...(latestDay.shadow.line ? [{ name: 'shadow_realised_revenue', value: latestDay.shadow.line.realised_revenue_yuan, unit: 'yuan', ref: null }] : []), + ...(latestDay.human ? [{ name: 'human_realised_revenue', value: latestDay.human.line.realised_revenue_yuan, unit: 'yuan', ref: null }] : []), + { name: 'hindsight_realised_revenue', value: latestDay.hindsight.realised_revenue_yuan, unit: 'yuan', ref: null }, + { name: 'naive_realised_revenue', value: latestDay.naive.realised_revenue_yuan, unit: 'yuan', ref: null }, + ], + source: { kind: 'SHADOW_DAY', id: latestDay.id, ref: null }, + generated_at: now, + }) + } if (page === 'OPERATIONS_DASHBOARD') { + const kpi = ctx.kpiReports.list().map((r) => r.value).sort((a, b) => (a.generated_at < b.generated_at ? -1 : 1)).at(-1) + if (kpi) { + const scored = kpi.kpis.filter((k) => k.status === 'MEET' || k.status === 'MISS') + cards.push({ + id: `card-kpi-${kpi.id}`, + page, + market_date: kpi.window.to ?? now.slice(0, 10), + title: 'KPI dashboard', + headline: `${scored.filter((k) => k.status === 'MEET').length}/${scored.length} scored KPIs met over ${kpi.window.days} shadow days; ${kpi.shadow.consecutive_complete_days} consecutive days with complete lineage`, + metrics: kpi.kpis.filter((k) => k.value !== null).map((k) => ({ name: k.id.toLowerCase(), value: k.value!, unit: k.unit, ref: null })), + source: { kind: 'KPI_REPORT', id: kpi.id, ref: null }, + generated_at: now, + }) + } + const tripped = ctx.breakers.state().filter((b) => b.status === 'TRIPPED') + cards.push({ + id: 'card-breakers', + page, + market_date: now.slice(0, 10), + title: 'Kill switches', + headline: tripped.length === 0 ? 'All levels armed (L0–L4)' : `Tripped: ${tripped.map((b) => `${b.level} (${b.reason ?? ''})`).join('; ')}`, + metrics: [{ name: 'levels_tripped', value: String(tripped.length), unit: '1', ref: null }], + source: { kind: 'BREAKER', id: 'breakers', ref: null }, + generated_at: now, + }) const open = ctx.caseDesk.cases.list().map((c) => c.value).filter((c) => !c.status.startsWith('CLOSED')) cards.push({ id: 'card-ops-cases', diff --git a/packages/runtime/src/context.ts b/packages/runtime/src/context.ts index 461700c..03a7786 100644 --- a/packages/runtime/src/context.ts +++ b/packages/runtime/src/context.ts @@ -1,10 +1,14 @@ import type { AuthorityService, + BreakerConfig, + BreakerService, CaseDeskService, EnvelopeService, EventBus, FileExportGateway, GatewayPort, + IngestionPipeline, + KpiConfig, LedgerService, LineageRecorder, PolicyEngine, @@ -17,7 +21,7 @@ import type { SnapshotStore, TimeSeriesStore, } from '@vpp/services' -import type { AwardNotice, EnvelopeChangeRequest, ExecutionReport, MeteringRecord, Proposal, ReviewFinding } from '@vpp/domain' +import type { AwardNotice, EnvelopeChangeRequest, ExecutionReceipt, ExecutionReport, HumanBidRecord, KpiReport, MeteringRecord, Proposal, ReviewFinding, ShadowDayRecord } from '@vpp/domain' import type { LlmPort } from './llm.js' /** @@ -27,7 +31,26 @@ import type { LlmPort } from './llm.js' * `llm`. Workflows are created per runtime and close over their context, so * two runtimes can coexist in one process — which is what the restart test does. */ +/** + * SHADOW: bids are recorded, never submitted; dispatch goes to the simulation + * gateway (docs/08 §4 M5 — the phase-1 acceptance form). FILE_EXPORT: the M3 + * degraded channel (bid file for manual upload). + */ +export type RuntimeMode = 'SHADOW' | 'FILE_EXPORT' + +export interface ShadowConfig { + /** + * How simulated execution fills a released dispatch plan when no field + * reports arrive: each unit delivers its resources' mean reliability score, + * or a fixed ratio. A shadow-run modelling assumption, not a market parameter. + */ + fulfillment: { kind: 'RELIABILITY_SCORE' } | { kind: 'FIXED'; ratio: string } + /** Installed PV capacity for nRMSE when no PV resource is registered. OPEN-QUESTION C2. */ + pvCapacityMw: string +} + export interface RuntimeConfig { + mode: RuntimeMode policyPackId: string /** OPEN-QUESTION B3: which roles may approve; placeholder list. */ approverRoles: string[] @@ -49,8 +72,16 @@ export interface RuntimeConfig { reliabilityAlpha: string widenAfterCompliantDays: number widenStepPct: string + /** Kill-switch hierarchy parameters — OPEN-QUESTION B5 (loss budget), B8 (authority). */ + breaker: BreakerConfig + /** docs/12 §4 KPI definitions — OPEN-QUESTION C1/C2 placeholders. */ + kpi: KpiConfig + shadow: ShadowConfig } +/** A bid release channel: any GatewayPort whose receipts are inspectable (file export, shadow, later the trading-platform API). */ +export type BidGateway = GatewayPort & { receipts: Repository } + export interface RuntimeContext { config: RuntimeConfig clock: () => string @@ -63,8 +94,13 @@ export interface RuntimeContext { policy: PolicyEngine envelopes: EnvelopeService authority: AuthorityService - gateway: FileExportGateway - /** Simulation and gateway are chosen by proposal type (ADR-0005: type-specific behaviour is data, not workflow structure). */ + /** Bid channel for this runtime's mode. */ + gateway: BidGateway + /** The manual channel, always present: L3 falls bids back to it (docs/06 §2.1 降级策略). */ + fileExportGateway: FileExportGateway + breakers: BreakerService + ingestion: IngestionPipeline + /** Simulation and gateway are chosen by proposal type (ADR-0005: type-specific behaviour is data, not workflow structure); the L3 breaker is consulted here. */ simulationFor: (type: Proposal['type']) => SimulationPort gatewayFor: (type: Proposal['type']) => GatewayPort simulationGateway: SimulationGateway @@ -74,6 +110,9 @@ export interface RuntimeContext { metering: Repository envelopeRequests: Repository findings: Repository + humanBids: Repository + shadowDays: Repository + kpiReports: Repository caseDesk: CaseDeskService skills: SkillClient lineage: LineageRecorder diff --git a/packages/runtime/src/index.ts b/packages/runtime/src/index.ts index d4ccc9c..510540c 100644 --- a/packages/runtime/src/index.ts +++ b/packages/runtime/src/index.ts @@ -11,3 +11,5 @@ export * from './cards.js' export * from './workflows/award-decomposition.js' export * from './workflows/review.js' export * from './workflows/envelope-review.js' +export * from './workflows/shadow-close.js' +export * from './shadow.js' diff --git a/packages/runtime/src/main.ts b/packages/runtime/src/main.ts index 5087918..bd5426d 100644 --- a/packages/runtime/src/main.ts +++ b/packages/runtime/src/main.ts @@ -4,7 +4,8 @@ * VPP_SKILLS_URL Python skill service (default http://127.0.0.1:8000) * VPP_LLM_MODEL Mastra router model id, e.g. openai/gpt-... ; unset = LLM down (I6 mode) * VPP_API_PORT Case Desk API port (default 4100) - * Scheduled triggers (D-1 06:00 situation, 08:00 bid, Asia/Shanghai) run in-process. + * VPP_MODE SHADOW (default; bids recorded, never submitted — ROADMAP M5) or FILE_EXPORT (M3 manual channel) + * Scheduled triggers (D-1 06:00 situation, 08:00 bid; in shadow mode also 16:00 clearing and D+1 02:00 close; Asia/Shanghai) run in-process. */ import { createApi } from './api.js' import { createRuntime } from './runtime.js' @@ -14,11 +15,12 @@ const rt = await createRuntime({ dataDir: process.env['VPP_DATA_DIR'] ?? './data', skills: process.env['VPP_SKILLS_URL'] ?? 'http://127.0.0.1:8000', llm: process.env['VPP_LLM_MODEL'] ? (process.env['VPP_LLM_MODEL'] as `${string}/${string}`) : null, + config: { mode: process.env['VPP_MODE'] === 'FILE_EXPORT' ? 'FILE_EXPORT' : 'SHADOW' }, }) const triggers = new TriggerService(rt) triggers.startEventConsumers() triggers.startScheduler() const port = Number(process.env['VPP_API_PORT'] ?? 4100) createApi(rt, triggers).listen(port, () => { - console.log(`vpp runtime: api on :${port}, llm=${rt.ctx.llm.backend}, data=${process.env['VPP_DATA_DIR'] ?? './data'}`) + console.log(`vpp runtime: api on :${port}, mode=${rt.ctx.config.mode}, llm=${rt.ctx.llm.backend}, data=${process.env['VPP_DATA_DIR'] ?? './data'}`) }) diff --git a/packages/runtime/src/runtime.ts b/packages/runtime/src/runtime.ts index ff0eefd..d7b8cd8 100644 --- a/packages/runtime/src/runtime.ts +++ b/packages/runtime/src/runtime.ts @@ -2,29 +2,36 @@ import { join } from 'node:path' import { Mastra } from '@mastra/core' import type { MastraModelConfig } from '@mastra/core/llm' import { LibSQLStore } from '@mastra/libsql' -import { Approval, AwardNotice, DecisionCase, Envelope, EnvelopeChangeRequest, ExecutionPermit, ExecutionReceipt, ExecutionReport, MeteringRecord, Proposal, ResourceProfile, ReviewFinding, SemanticMemoryEntry } from '@vpp/domain' +import { Approval, AwardNotice, BreakerRecord, DecisionCase, Envelope, EnvelopeChangeRequest, ExecutionPermit, ExecutionReceipt, ExecutionReport, HumanBidRecord, KpiReport, MeteringRecord, Proposal, ResourceProfile, ReviewFinding, SemanticMemoryEntry, ShadowDayRecord } from '@vpp/domain' import { AuthorityService, + BREAKER_CONFIG_PLACEHOLDER, + BreakerService, CaseDeskService, EnvelopeService, FileEventBus, FileExportGateway, FsRepository, FsSnapshotStore, + FsTimeSeriesStore, + GatewayRejected, HUBEI_BID_PACK_PLACEHOLDER, HttpSkillClient, + IngestionPipeline, + KPI_CONFIG_PLACEHOLDER, + KeyValueRow, LedgerService, LineageRecorder, - MemoryTimeSeriesStore, PendingApprovalSchema, PolicyEngine, PowerBalanceSimulator, RevenueScenarioSimulator, ReviewService, + ShadowBidGateway, SimulationGateway, hubeiSpotBiddingPack, } from '@vpp/services' -import type { SkillClient } from '@vpp/services' +import type { BreakerAction, GatewayPort, SkillClient } from '@vpp/services' import type { RuntimeConfig, RuntimeContext } from './context.js' import { AGENT_SPECS, MastraLlm, NullLlm } from './llm.js' import type { LlmPort } from './llm.js' @@ -34,6 +41,7 @@ import { createEnvelopeReview } from './workflows/envelope-review.js' import { createReview } from './workflows/review.js' import { createDayAheadSituation } from './workflows/day-ahead-situation.js' import { createProposalLifecycle } from './workflows/proposal-lifecycle.js' +import { createShadowClose } from './workflows/shadow-close.js' /** * Runtime = thin shell over Mastra + the self-built services (docs/09 §0). @@ -53,6 +61,7 @@ export interface RuntimeOptions { } export const DEFAULT_RUNTIME_CONFIG: RuntimeConfig = { + mode: 'FILE_EXPORT', policyPackId: 'hubei-spot-bidding', approverRoles: ['senior-trader', 'ops-lead', 'risk-officer'], // OPEN-QUESTION B3 bidPermitTtlMs: 6 * 3_600_000, // OPEN-QUESTION A1 @@ -65,6 +74,9 @@ export const DEFAULT_RUNTIME_CONFIG: RuntimeConfig = { reliabilityAlpha: '0.2', // OPEN-QUESTION B10 widenAfterCompliantDays: 20, // OPEN-QUESTION B4/B9 widenStepPct: '20', // OPEN-QUESTION B1 + breaker: BREAKER_CONFIG_PLACEHOLDER, // OPEN-QUESTION B5 / B8 + kpi: KPI_CONFIG_PLACEHOLDER, // OPEN-QUESTION C1 / C2 + shadow: { fulfillment: { kind: 'RELIABILITY_SCORE' }, pvCapacityMw: '30' }, // pvCapacityMw: OPEN-QUESTION C2 } export interface Runtime { @@ -84,10 +96,11 @@ export async function createRuntime(opts: RuntimeOptions): Promise { const snapshots = new FsSnapshotStore(dir('snapshots')) const events = new FileEventBus(dir('events.jsonl'), clock) - const ledger = new LedgerService({ daMonthlyDeviationBand: '0.05', clock }) // OPEN-QUESTION A5 + const ledger = new LedgerService({ daMonthlyDeviationBand: '0.05', clock, path: dir('ledger.jsonl') }) // OPEN-QUESTION A5 const authority = new AuthorityService(new FsRepository(dir('permits.json'), ExecutionPermit, (p) => p.id, clock), () => newId('permit')) const policy = new PolicyEngine([hubeiSpotBiddingPack('2026.03', HUBEI_BID_PACK_PLACEHOLDER)]) - const envelopes = new EnvelopeService(new FsRepository(dir('envelopes.json'), Envelope, (e) => e.id, clock)) + const kv = (name: string) => new FsRepository(dir(`counters-${name}.json`), KeyValueRow, (r) => r.key, clock) + const envelopes = new EnvelopeService(new FsRepository(dir('envelopes.json'), Envelope, (e) => e.id, clock), kv('envelope')) const caseDesk = new CaseDeskService({ cases: new FsRepository(dir('cases.json'), DecisionCase, (c) => c.id, clock), proposals: new FsRepository(dir('proposals.json'), Proposal, (p) => p.id, clock), @@ -98,7 +111,9 @@ export async function createRuntime(opts: RuntimeOptions): Promise { events, clock, }) - const gateway = new FileExportGateway(authority, dir('exports'), new FsRepository(dir('receipts.json'), ExecutionReceipt, (r) => r.idempotency_key, clock)) + const fileExportGateway = new FileExportGateway(authority, dir('exports'), new FsRepository(dir('receipts.json'), ExecutionReceipt, (r) => r.idempotency_key, clock)) + const shadowGateway = new ShadowBidGateway(authority, new FsRepository(dir('shadow-receipts.json'), ExecutionReceipt, (r) => r.idempotency_key, clock)) + const gateway = config.mode === 'SHADOW' ? shadowGateway : fileExportGateway const simulationGateway = new SimulationGateway(authority, new FsRepository(dir('sim-receipts.json'), ExecutionReceipt, (r) => r.idempotency_key, clock)) const resources = new FsRepository(dir('resources.json'), ResourceProfile, (r) => r.resource_id, clock) const revenueSim = new RevenueScenarioSimulator() @@ -112,10 +127,24 @@ export async function createRuntime(opts: RuntimeOptions): Promise { } return limits }) - const review = new ReviewService(resources, envelopes, new FsRepository(dir('memory.json'), SemanticMemoryEntry, (m) => m.id, clock), { - reliabilityAlpha: config.reliabilityAlpha, - widenAfterCompliantDays: config.widenAfterCompliantDays, - widenStepPct: config.widenStepPct, + const review = new ReviewService( + resources, + envelopes, + new FsRepository(dir('memory.json'), SemanticMemoryEntry, (m) => m.id, clock), + { reliabilityAlpha: config.reliabilityAlpha, widenAfterCompliantDays: config.widenAfterCompliantDays, widenStepPct: config.widenStepPct }, + kv('review'), + ) + const timeseries = opts.timeseries ?? new FsTimeSeriesStore(dir('timeseries.jsonl'), clock) + const BREAKER_EVENT: Record = { TRIP: 'BreakerTripped', RESET: 'BreakerReset', ABNORMAL_DAY_DECLARED: 'AbnormalDayDeclared', ABNORMAL_DAY_CLEARED: 'AbnormalDayCleared' } + const breakers = new BreakerService({ + records: new FsRepository(dir('breakers.json'), BreakerRecord, (r) => r.level, clock), + kv: kv('breaker'), + authority, + envelopes, + cfg: config.breaker, + onChange: (action, detail) => { + events.append({ event_type: BREAKER_EVENT[action], payload: detail, correlation_id: `breaker-${String(detail['level'] ?? detail['market_date'] ?? '*')}` }) + }, }) let llm: LlmPort @@ -133,7 +162,7 @@ export async function createRuntime(opts: RuntimeOptions): Promise { clock, newId, snapshots, - timeseries: opts.timeseries ?? new MemoryTimeSeriesStore(clock), + timeseries, resources, ledger, events, @@ -141,8 +170,20 @@ export async function createRuntime(opts: RuntimeOptions): Promise { envelopes, authority, gateway, + fileExportGateway, + breakers, + ingestion: new IngestionPipeline({ timeseries, snapshots, clock }), simulationFor: (type) => (type === 'DISPATCH_PLAN' ? powerSim : revenueSim), - gatewayFor: (type) => (type === 'DISPATCH_PLAN' ? simulationGateway : gateway), + gatewayFor: (type): GatewayPort => { + // L3 channel breaker (docs/13 §8): control effects stop (edge disconnection strategy); + // bids fall back to the manual file channel (docs/06 §2.1 降级策略). + const blocked = breakers.channelBlocked(type) + if (blocked) { + if (type === 'BID') return fileExportGateway + return { dispatch: () => { throw new GatewayRejected([blocked]) } } + } + return type === 'DISPATCH_PLAN' ? simulationGateway : gateway + }, simulationGateway, review, awards: new FsRepository(dir('awards.json'), AwardNotice, (a) => a.id, clock), @@ -150,6 +191,9 @@ export async function createRuntime(opts: RuntimeOptions): Promise { metering: new FsRepository(dir('metering.json'), MeteringRecord, (m) => m.id, clock), envelopeRequests: new FsRepository(dir('envelope-requests.json'), EnvelopeChangeRequest, (r) => r.id, clock), findings: new FsRepository(dir('findings.json'), ReviewFinding, (f) => f.id, clock), + humanBids: new FsRepository(dir('human-bids.json'), HumanBidRecord, (h) => h.id, clock), + shadowDays: new FsRepository(dir('shadow-days.json'), ShadowDayRecord, (d) => d.id, clock), + kpiReports: new FsRepository(dir('kpi-reports.json'), KpiReport, (k) => k.id, clock), caseDesk, skills: typeof opts.skills === 'string' ? new HttpSkillClient(opts.skills) : opts.skills, lineage: new LineageRecorder(), @@ -161,6 +205,7 @@ export async function createRuntime(opts: RuntimeOptions): Promise { // module-level singletons would bind to whichever Mastra registered last. const proposalLifecycle = createProposalLifecycle(ctx) const envelopeReview = createEnvelopeReview(ctx) + const reviewWorkflow = createReview(ctx, envelopeReview) const mastra = new Mastra({ storage, workflows: { @@ -169,7 +214,8 @@ export async function createRuntime(opts: RuntimeOptions): Promise { dayAheadSituation: createDayAheadSituation(ctx), dayAheadBid: createDayAheadBid(ctx, proposalLifecycle), awardDecomposition: createAwardDecomposition(ctx, proposalLifecycle), - review: createReview(ctx, envelopeReview), + review: reviewWorkflow, + shadowClose: createShadowClose(ctx, reviewWorkflow), }, agents, }) diff --git a/packages/runtime/src/shadow-cli.ts b/packages/runtime/src/shadow-cli.ts new file mode 100644 index 0000000..f52964b --- /dev/null +++ b/packages/runtime/src/shadow-cli.ts @@ -0,0 +1,155 @@ +/** + * Shadow-run replay (ROADMAP M5 acceptance driver): + * npm run shadow -w @vpp/runtime -- [--dataset PATH] [--from INDEX|DATE] [--days N] + * [--data-dir DIR] [--skills-url URL] [--human-baseline naive|none] + * [--contract-share 0.6] + * + * Replays consecutive market days from a replay dataset through the real + * runtime in SHADOW mode: live-data ingestion → D-1 situation → bid → safety + * chain → shadow clearing → award decomposition → simulated execution → D+1 + * review → shadow-vs-human-vs-hindsight scoring → KPI report. Uses the Python + * skill service (VPP_SKILLS_URL) and the LLM if VPP_LLM_MODEL is set — the + * numbers are identical either way (I6). Writes the daily comparison table and + * the KPI report under /shadow/. + */ +import { mkdirSync, readFileSync, writeFileSync } from 'node:fs' +import { join } from 'node:path' +import { fileURLToPath } from 'node:url' +import type { Envelope, KpiReport, ShadowDayRecord } from '@vpp/domain' +import { Decimal, naiveBid } from '@vpp/services' +import { createRuntime } from './runtime.js' +import { ShadowRunner, seedMonthlyContracts } from './shadow.js' +import { TriggerService } from './trigger.js' + +interface Day { date: string; load_mw: string[]; pv_mw: string[]; price_yuan_per_mwh: string[]; adjustable_capacity_mw: string[] } +interface Dataset { meta: { name: string; synthetic: boolean; adjustable_capacity_mw: string; pv_capacity_mw: string }; days: Day[] } + +const here = fileURLToPath(new URL('.', import.meta.url)) +const args = process.argv.slice(2) +const opt = (name: string, dflt: string) => { + const i = args.indexOf(name) + return i >= 0 && args[i + 1] ? args[i + 1]! : dflt +} + +const datasetPath = opt('--dataset', join(here, '../../evals/datasets/synthetic-hubei-v0.json')) +const dataset = JSON.parse(readFileSync(datasetPath, 'utf8')) as Dataset +const fromArg = opt('--from', '60') +const fromIdx = /^\d{4}-\d{2}-\d{2}$/.test(fromArg) ? dataset.days.findIndex((d) => d.date === fromArg) : Number(fromArg) +if (fromIdx < 0 || fromIdx >= dataset.days.length) throw new Error(`--from ${fromArg} not in dataset`) +const days = Number(opt('--days', '21')) +const dataDir = opt('--data-dir', './data-shadow') +const skillsUrl = opt('--skills-url', process.env['VPP_SKILLS_URL'] ?? 'http://127.0.0.1:8000') +const humanBaseline = opt('--human-baseline', 'naive') +const contractShare = opt('--contract-share', '0.6') + +let now = new Date().toISOString() +const rt = await createRuntime({ + dataDir, + skills: skillsUrl, + llm: process.env['VPP_LLM_MODEL'] ? (process.env['VPP_LLM_MODEL'] as `${string}/${string}`) : null, + clock: () => now, + config: { mode: 'SHADOW', shadow: { fulfillment: { kind: 'RELIABILITY_SCORE' }, pvCapacityMw: dataset.meta.pv_capacity_mw } }, +}) +const triggers = new TriggerService(rt) +triggers.startEventConsumers() +const runner = new ShadowRunner(triggers, { setClock: (iso) => (now = iso) }) + +// ---- seed the shadow world: one aggregated flexible resource sized to the dataset, monthly contracts, placeholder envelopes +const replayDays = dataset.days.slice(fromIdx, fromIdx + days) +if (replayDays.length === 0) throw new Error('nothing to replay') +now = `${dataset.days[0]!.date}T00:00:00Z` +const capMw = dataset.meta.adjustable_capacity_mw +if (!rt.ctx.resources.get('res-shadow-pool')) { + rt.ctx.resources.put({ + resource_id: 'res-shadow-pool', + name: `Shadow pool (${dataset.meta.name})`, + type: 'STORAGE', + rated_power_mw: capMw, + certified_adjustable_mw: capMw, + confidence: '0.9', + reliability_score: '0.95', + constraints: { min_duration_min: 60, recovery_rate_mw_per_min: '0.5' }, + evidence_refs: [], + updated_at: now, + }) +} +const k = rt.ctx.config.risk.commitment_buffer_k +const sellablePerDay = new Decimal(capMw).mul(k).mul('0.25').mul(96).toFixed(3) +seedMonthlyContracts(rt.ctx, [...new Set(replayDays.map((d) => d.date.slice(0, 7)))], sellablePerDay, contractShare) +const envelope = (id: string, scope: Envelope['scope'], bounds: Record): Envelope => ({ + id, + scope, + bounds, + validity: { from: `${dataset.days[0]!.date}T00:00:00Z`, to: `${dataset.days.at(-1)!.date}T23:59:59Z` }, + approval: { level: 'L1', approved_by: ['shadow-seed'] }, // placeholder initial envelopes — OPEN-QUESTION B1/B2 + escalation: { max_consecutive_deviations: 3, deviation_threshold_pct: '10.0' }, // OPEN-QUESTION B4 + status: 'ACTIVE', +}) +if (!rt.ctx.envelopes.get('env-shadow-bid')) rt.ctx.envelopes.register(envelope('env-shadow-bid', { proposal_type: 'BID', timescales: ['DAY_AHEAD'], resource_set: 'shadow-pool' }, { max_energy_mwh: sellablePerDay })) +if (!rt.ctx.envelopes.get('env-shadow-dispatch')) rt.ctx.envelopes.register(envelope('env-shadow-dispatch', { proposal_type: 'DISPATCH_PLAN', timescales: ['DAY_AHEAD'], resource_set: 'shadow-pool' }, { max_total_mw: capMw })) + +// ---- history before the first replay day +for (const d of dataset.days.slice(0, fromIdx)) triggers.onMarketData({ market_date: d.date, load_mw: d.load_mw, pv_mw: d.pv_mw, price_yuan_per_mwh: d.price_yuan_per_mwh, source: dataset.meta.name }) + +const fmt = (x: string | null | undefined) => (x === null || x === undefined ? '—' : x) +console.log(`shadow replay: ${replayDays.length} days from ${replayDays[0]!.date}, dataset ${dataset.meta.name}${dataset.meta.synthetic ? ' (SYNTHETIC)' : ''}, skills ${skillsUrl}, llm=${rt.ctx.llm.backend}`) +console.log('date outcome shadow¥ human¥ hindsight¥ naive¥ loadMAPE latency lineage') +const records: ShadowDayRecord[] = [] +let kpi: KpiReport | null = null +for (const d of replayDays) { + // The day's actuals arrive as live data; forecasts only see history strictly before the market date. + triggers.onMarketData({ market_date: d.date, load_mw: d.load_mw, pv_mw: d.pv_mw, price_yuan_per_mwh: d.price_yuan_per_mwh, source: dataset.meta.name }) + if (humanBaseline === 'naive') { + // Placeholder human decision: a price-taker flat bid at the position's upper bound, labelled SYNTHETIC_NAIVE. + const capMwh = d.adjustable_capacity_mw.map((c) => Number(c) * Number(k) * 0.25) + const eMax = Number(rt.ctx.ledger.dayAheadBounds(d.date).daily_energy_max_mwh) + const b = naiveBid(capMwh, eMax) + triggers.onHumanBid({ + id: `hb-${d.date}`, + market_date: d.date, + prices_yuan_per_mwh: { interval_minutes: 15, date: d.date, values: b.offers.map(() => '0.00') }, + quantities_mwh: { interval_minutes: 15, date: d.date, values: b.quantities.map((q) => q.toFixed(3)) }, + source: 'SYNTHETIC_NAIVE', + received_at: `${d.date}T00:00:00Z`, + }) + } + const out = await runner.replayDay(d.date) + records.push(out.record) + kpi = out.kpi + const r = out.record + console.log( + `${r.market_date} ${r.shadow.outcome.padEnd(13)} ${fmt(r.shadow.line?.realised_revenue_yuan).padStart(12)} ${fmt(r.human?.line.realised_revenue_yuan).padStart(12)} ${r.hindsight.realised_revenue_yuan.padStart(12)} ${r.naive.realised_revenue_yuan.padStart(12)} ${fmt(r.forecast.load_mape).padEnd(8)} ${String(r.decision_latency_ms ?? '—').padStart(6)}ms ${r.lineage_complete ? 'complete' : `gaps ${r.lineage_gaps.length}`}`, + ) +} + +if (kpi) { + console.log('\nKPI report', kpi.id) + for (const e of kpi.kpis) console.log(` ${e.id.padEnd(26)} ${fmt(e.value).padStart(12)} ${e.unit.padEnd(4)} target ${e.comparator} ${fmt(e.target).padStart(8)} ${e.status.padEnd(14)} n=${e.samples}`) + console.log(` comparison: shadow ${kpi.comparison.shadow_yuan} / human ${fmt(kpi.comparison.human_yuan)} / hindsight ${kpi.comparison.hindsight_yuan} / naive ${kpi.comparison.naive_yuan}; capture ${fmt(kpi.comparison.capture_ratio)}`) + console.log(` shadow days ${kpi.shadow.days}, consecutive complete ${kpi.shadow.consecutive_complete_days}, released ${kpi.shadow.released_days}, pending ${kpi.shadow.pending_days}, widen recommendations ${kpi.shadow.widen_recommendations}, breaker trips ${kpi.shadow.breaker_trips}`) + const outDir = join(dataDir, 'shadow') + mkdirSync(outDir, { recursive: true }) + const last = records.at(-1)!.market_date + writeFileSync(join(outDir, `kpi-${last}.json`), JSON.stringify(kpi, null, 2) + '\n') + const md = [ + `# Shadow run report — ${records[0]!.market_date} → ${last}`, + '', + `Dataset: ${dataset.meta.name}${dataset.meta.synthetic ? ' (SYNTHETIC — measures the loop, not the business KPI)' : ''}. Human baseline: ${humanBaseline}.`, + '', + '| date | outcome | shadow ¥ | human ¥ | hindsight ¥ | naive ¥ | load MAPE | price MAPE | latency ms | lineage |', + '|---|---|---|---|---|---|---|---|---|---|', + ...records.map((r) => `| ${r.market_date} | ${r.shadow.outcome} | ${fmt(r.shadow.line?.realised_revenue_yuan)} | ${fmt(r.human?.line.realised_revenue_yuan)} | ${r.hindsight.realised_revenue_yuan} | ${r.naive.realised_revenue_yuan} | ${fmt(r.forecast.load_mape)} | ${fmt(r.forecast.price_mape)} | ${r.decision_latency_ms ?? '—'} | ${r.lineage_complete ? 'complete' : r.lineage_gaps.join('; ')} |`), + '', + '## KPI (docs/12 §4 definitions; targets are OPEN-QUESTION C1/C2 placeholders)', + '', + '| KPI | value | target | status | samples | definition |', + '|---|---|---|---|---|---|', + ...kpi.kpis.map((e) => `| ${e.id} | ${fmt(e.value)} | ${e.comparator} ${fmt(e.target)} | ${e.status} | ${e.samples} | ${e.definition} |`), + '', + `Consecutive shadow days with complete lineage: **${kpi.shadow.consecutive_complete_days}**. Envelope-widening recommendations produced (not acted on): **${kpi.shadow.widen_recommendations}**.`, + '', + ].join('\n') + writeFileSync(join(outDir, `report-${last}.md`), md) + console.log(`\nwritten: ${outDir}/report-${last}.md, kpi-${last}.json`) +} +await rt.close() diff --git a/packages/runtime/src/shadow.ts b/packages/runtime/src/shadow.ts new file mode 100644 index 0000000..3fff8a4 --- /dev/null +++ b/packages/runtime/src/shadow.ts @@ -0,0 +1,170 @@ +import type { BreakerLevel, Envelope, ExecutionPermit, KpiReport, Proposal, ShadowDayRecord } from '@vpp/domain' +import { Decimal, GatewayRejected, canonicalJson, contentRef } from '@vpp/services' +import type { Actor } from '@vpp/services' +import type { RuntimeContext } from './context.js' +import type { TriggerService } from './trigger.js' + +/** + * Shadow-run orchestration (docs/08 §4 M5). In live mode the trigger service + * fires each stage at its local time; `ShadowRunner.replayDay` drives the same + * handlers over historical data, moving the injected clock through the + * docs/07 timeline (D-1 06:00 situation → 08:00 bid → 16:00 clearing → D+1 + * 02:00 close). The runner never touches the safety chain directly. + */ +export const localToUtc = (date: string, hhmm: string, dayOffset = 0): string => + new Date(new Date(`${date}T${hhmm}:00Z`).getTime() + dayOffset * 86_400_000 - 8 * 3_600_000).toISOString() + +export class ShadowRunner { + constructor( + private readonly triggers: TriggerService, + private readonly opts: { setClock?: (iso: string) => void } = {}, + ) {} + + private at(iso: string): void { + this.opts.setClock?.(iso) + } + + async replayDay(date: string): Promise<{ record: ShadowDayRecord; kpi: KpiReport }> { + this.at(localToUtc(date, '06:00', -1)) + await this.triggers.startTemplate('dayAheadSituation', { market_date: date }, 'SCHEDULED', 'shadow-replay') + this.at(localToUtc(date, '08:00', -1)) + await this.triggers.startTemplate('dayAheadBid', { market_date: date }, 'SCHEDULED', 'shadow-replay') + this.at(localToUtc(date, '16:00', -1)) + await this.triggers.shadowClearing(date) + this.at(localToUtc(date, '02:00', 1)) + const out = await this.triggers.shadowClose(date) + return { record: out.record, kpi: out.kpi } + } + + async replay(dates: string[]): Promise { + const out: ShadowDayRecord[] = [] + for (const d of dates) out.push((await this.replayDay(d)).record) + return out + } +} + +export interface DrillStep { + level: BreakerLevel + verified: boolean + detail: string +} + +export interface DrillResult { + drill_id: string + all_verified: boolean + steps: DrillStep[] +} + +/** + * Kill-switch drill (docs/13 §8: 每季度演练至少一级; ROADMAP M5: drilled once). + * Every level is tripped on throwaway drill objects, its effect verified where + * the real chain would feel it, then reset — all recorded in the event log so + * the drill itself is auditable. Nothing here touches a live proposal. + */ +export async function runBreakerDrill(ctx: RuntimeContext, by: Actor): Promise { + const now = ctx.clock() + const drillId = ctx.newId('drill') + const today = now.slice(0, 10) + const flat = (v: string) => ({ interval_minutes: 15 as const, date: today, values: Array.from({ length: 96 }, () => v) }) + const digest = contentRef(canonicalJson({ drill: drillId })) + const lineage = { tool_calls: [], data_refs: [], ledger_version: ctx.ledger.read().version, policy_pack_version: ctx.policy.current(ctx.config.policyPackId).version } + const drillBid: Proposal = { + id: `${drillId}-bid`, + type: 'BID', + timescale: 'DAY_AHEAD', + originator: { agent: 'trading-agent', trigger: 'MANUAL', task_id: drillId }, + payload: { kind: 'BID', market_date: today, prices_yuan_per_mwh: flat('0.00'), quantities_mwh: flat('1.000'), expected_revenue_yuan: '0.00' }, + lineage, + envelope_ref: null, + digest, + status: 'DRAFT', + created_at: now, + } + const drillDispatch: Proposal = { + id: `${drillId}-dispatch`, + type: 'DISPATCH_PLAN', + timescale: 'DAY_AHEAD', + originator: { agent: 'resource-agent', trigger: 'MANUAL', task_id: drillId }, + payload: { kind: 'DISPATCH_PLAN', market_date: today, award_ref: `${drillId}-award`, total_target_mw: flat('1.000'), allocations: [{ unit_id: 'drill-unit', target_mw: flat('1.000') }] }, + lineage, + envelope_ref: null, + digest: contentRef(canonicalJson({ drill: drillId, dispatch: true })), + status: 'DRAFT', + created_at: now, + } + const steps: DrillStep[] = [] + const record = (level: BreakerLevel, verified: boolean, detail: string) => { + steps.push({ level, verified, detail }) + ctx.events.append({ event_type: 'BreakerDrillStep', payload: { drill_id: drillId, level, verified, detail, by }, correlation_id: drillId }) + } + + // L0 — permit revocation: the gateway-side validation must refuse the revoked permit. + const permit: ExecutionPermit = { id: `${drillId}-permit`, proposal_digest: drillBid.digest, issued_at: now, expires_at: new Date(new Date(now).getTime() + 3_600_000).toISOString(), effect_limits: {}, revoked_at: null, issuer: 'authority-service' } + ctx.authority.permits.put(permit) + ctx.breakers.trip('L0', by, `drill ${drillId}`, permit.id, now, true) + const l0 = ctx.authority.validate(permit, drillBid, now) + record('L0', l0.some((r) => /revoked/.test(r)), `gateway validation: ${l0.join('; ') || 'accepted'}`) + ctx.breakers.reset('L0', by, `drill ${drillId}: revocation verified at gateway validation`, now) + + // L1 — envelope suspension: the drill envelope (validity in the past, so it never matches live work) must be SUSPENDED. + const env: Envelope = { + id: `${drillId}-env`, + scope: { proposal_type: 'BID', timescales: ['DAY_AHEAD'], resource_set: 'drill' }, + bounds: { max_energy_mwh: '1.0' }, + validity: { from: new Date(new Date(now).getTime() - 7_200_000).toISOString(), to: new Date(new Date(now).getTime() - 3_600_000).toISOString() }, + approval: { level: 'L1', approved_by: [by.id] }, + escalation: { max_consecutive_deviations: 3, deviation_threshold_pct: '10.0' }, + status: 'ACTIVE', + } + ctx.envelopes.register(env) + ctx.breakers.trip('L1', by, `drill ${drillId}`, env.id, now, true) + record('L1', ctx.envelopes.get(env.id)?.status === 'SUSPENDED', `envelope ${env.id} status ${ctx.envelopes.get(env.id)?.status}`) + ctx.breakers.reset('L1', by, `drill ${drillId}: suspension verified; drill envelope left inert`, now) + + // L2 — loss breaker: auto-approval frozen and an exposure-increasing bid rejected by rule check. + ctx.breakers.trip('L2', by, `drill ${drillId}`, '*', now, true) + const freeze = ctx.breakers.autoApprovalFreeze(drillBid) + const l2 = ctx.breakers.guard(drillBid, ctx.ledger.read()) + record('L2', freeze !== null && l2.some((v) => v.rule_id === 'breaker-l2-reduce-only'), `freeze: ${freeze ?? 'none'}; guard: ${l2.map((v) => v.rule_id).join(',') || 'none'}`) + ctx.breakers.reset('L2', by, `drill ${drillId}: freeze and reduce-only guard verified`, now) + + // L3 — channel breaker: control effects refused, bids routed to the manual file channel. + ctx.breakers.trip('L3', by, `drill ${drillId}`, '*', now, true) + let dispatchRefused = false + try { + ctx.gatewayFor('DISPATCH_PLAN').dispatch(drillDispatch, permit, now) + } catch (e) { + dispatchRefused = e instanceof GatewayRejected && e.reasons.some((r) => /L3/.test(r)) + } + const bidFallsBack = ctx.gatewayFor('BID') === ctx.fileExportGateway + record('L3', dispatchRefused && bidFallsBack, `dispatch refused: ${dispatchRefused}; bid → file export: ${bidFallsBack}`) + ctx.breakers.reset('L3', by, `drill ${drillId}: channel closure verified`, now) + + // L4 — AI off: rule check rejects any AI proposal; business workflows suppress creation (covered by the L4 test). + ctx.breakers.trip('L4', by, `drill ${drillId}`, '*', now, true) + const l4 = ctx.breakers.guard(drillBid, ctx.ledger.read()) + record('L4', ctx.breakers.isTripped('L4') && l4.some((v) => v.rule_id === 'breaker-l4'), `guard: ${l4.map((v) => v.rule_id).join(',') || 'none'}`) + ctx.breakers.reset('L4', by, `drill ${drillId}: AI-off guard verified`, now) + + const all = steps.every((s) => s.verified) + ctx.events.append({ event_type: 'BreakerDrillCompleted', payload: { drill_id: drillId, all_verified: all, levels: steps.map((s) => s.level), by, at: now }, correlation_id: drillId }) + return { drill_id: drillId, all_verified: all, steps } +} + +/** Monthly contract positions from a daily sellable estimate — the same seeding rule the L2 harness uses. */ +export function seedMonthlyContracts(ctx: RuntimeContext, months: string[], sellableMwhPerDay: string, contractShare: string): void { + for (const month of months) { + if (ctx.ledger.read().entries.some((e) => e.timescale === 'MONTHLY' && e.kind === 'CONTRACT' && e.period === month)) continue + const daysInMonth = new Date(Date.UTC(Number(month.slice(0, 4)), Number(month.slice(5, 7)), 0)).getUTCDate() + ctx.ledger.append({ + id: `contract-${month}`, + timescale: 'MONTHLY', + period: month, + kind: 'CONTRACT', + energy_mwh: new Decimal(sellableMwhPerDay).mul(contractShare).mul(daysInMonth).toFixed(3), + curve: null, + source_ref: `shadow-seed-contract-${month}`, + expected_version: ctx.ledger.read().version, + }) + } +} diff --git a/packages/runtime/src/trigger.ts b/packages/runtime/src/trigger.ts index 54a6632..61d6d93 100644 --- a/packages/runtime/src/trigger.ts +++ b/packages/runtime/src/trigger.ts @@ -1,5 +1,7 @@ -import { AwardNotice, ExecutionReport, MeteringRecord, RouterDecision } from '@vpp/domain' -import type { EventEnvelope } from '@vpp/domain' +import { z } from 'zod' +import { AwardNotice, Curve96, ExecutionReport, HumanBidRecord, MarketDate, MeteringRecord, RouterDecision } from '@vpp/domain' +import type { EventEnvelope, KpiReport, ShadowDayRecord } from '@vpp/domain' +import { shadowClearing } from '@vpp/services' import { ENVELOPE_REVIEW_STEP_ID } from './workflows/envelope-review.js' import { LlmUnavailable } from './llm.js' import type { Runtime } from './runtime.js' @@ -16,11 +18,15 @@ import type { ResumeDecision } from './workflows/proposal-lifecycle.js' * operator picks the template explicitly. * Approval is a resume on the lifecycle run (docs/09 §2). */ +export type ScheduledTemplate = 'dayAheadSituation' | 'dayAheadBid' | 'shadowClearing' | 'shadowClose' + export interface ScheduleEntry { id: string - /** 'HH:MM' Asia/Shanghai on D-1 */ + /** 'HH:MM' Asia/Shanghai */ localTime: string - workflow: 'dayAheadSituation' | 'dayAheadBid' + workflow: ScheduledTemplate + /** Market date relative to the local day the entry fires on: +1 = tomorrow (D-1 stages), -1 = yesterday (D+1 close). */ + marketDateOffsetDays?: 1 | -1 } export const DEFAULT_SCHEDULE: ScheduleEntry[] = [ @@ -28,12 +34,31 @@ export const DEFAULT_SCHEDULE: ScheduleEntry[] = [ { id: 'bid-0800', localTime: '08:00', workflow: 'dayAheadBid' }, // docs/07 D-1 08:00; window is OPEN-QUESTION A1 ] -export const nextMarketDate = (nowIso: string): string => { +/** Shadow mode adds the simulated market answer (docs/07 D-1 16:00) and the D+1 close (execution → metering → review → scoring). */ +export const SHADOW_SCHEDULE: ScheduleEntry[] = [ + ...DEFAULT_SCHEDULE, + { id: 'shadow-clearing-1600', localTime: '16:00', workflow: 'shadowClearing' }, // docs/07 D-1 16:00; publication time is OPEN-QUESTION A1 + { id: 'shadow-close-0200', localTime: '02:00', workflow: 'shadowClose', marketDateOffsetDays: -1 }, // D+1 02:00 for D +] + +export const marketDateFor = (nowIso: string, offsetDays: number): string => { const shanghai = new Date(new Date(nowIso).getTime() + 8 * 3_600_000) - shanghai.setUTCDate(shanghai.getUTCDate() + 1) + shanghai.setUTCDate(shanghai.getUTCDate() + offsetDays) return shanghai.toISOString().slice(0, 10) } +export const nextMarketDate = (nowIso: string): string => marketDateFor(nowIso, 1) + +/** Live-data feed (docs/06 §3): actual curves for a market date; each goes through the quality gate. */ +export const MarketDataInput = z.object({ + market_date: MarketDate, + load_mw: z.array(z.string()).length(96).optional(), + pv_mw: z.array(z.string()).length(96).optional(), + price_yuan_per_mwh: z.array(z.string()).length(96).optional(), + source: z.string().min(1).optional(), +}) +export type MarketDataInput = z.infer + export class TriggerService { private timer: NodeJS.Timeout | null = null private readonly firedToday = new Set() @@ -41,7 +66,7 @@ export class TriggerService { constructor( private readonly rt: Runtime, - private readonly schedule: ScheduleEntry[] = DEFAULT_SCHEDULE, + private readonly schedule: ScheduleEntry[] = rt.ctx.config.mode === 'SHADOW' ? SHADOW_SCHEDULE : DEFAULT_SCHEDULE, ) {} // ---- scheduled --------------------------------------------------------- @@ -67,22 +92,39 @@ export class TriggerService { for (const entry of this.schedule) { const key = `${day}:${entry.id}` if (hhmm >= entry.localTime && !this.firedToday.has(key)) { - this.firedToday.add(key) - await this.startTemplate(entry.workflow, { market_date: nextMarketDate(nowIso) }, 'SCHEDULED', entry.id) - fired.push(entry.id) + const marketDate = marketDateFor(nowIso, entry.marketDateOffsetDays ?? 1) + try { + await this.fire(entry, marketDate) + this.firedToday.add(key) + fired.push(entry.id) + } catch (e) { + // Late data (e.g. no clearing price yet) is normal in a live shadow run: record it and retry next tick. + this.rt.ctx.events.append({ event_type: 'ScheduledTriggerFailed', payload: { entry: entry.id, workflow: entry.workflow, market_date: marketDate, error: (e as Error).message }, correlation_id: `schedule-${entry.id}` }) + } } } return fired } + private async fire(entry: ScheduleEntry, marketDate: string): Promise { + if (entry.workflow === 'shadowClearing') await this.shadowClearing(marketDate) + else if (entry.workflow === 'shadowClose') await this.shadowClose(marketDate) + else await this.startTemplate(entry.workflow, { market_date: marketDate }, 'SCHEDULED', entry.id) + } + // ---- event ------------------------------------------------------------- - /** Deterministic event rules. M3: a SituationPublished with EXTREME risk opens an abnormal-day case (docs/13 §1). */ + /** + * Deterministic event rules. A SituationPublished with EXTREME risk starts the + * abnormal-day protocol (docs/13 §1): envelopes are off for that market date — + * every proposal goes to a human — and a case is opened for the review. + */ startEventConsumers(): void { this.unsubscribe.push( this.rt.ctx.events.subscribe('SituationPublished', (evt: EventEnvelope) => { const payload = evt.payload as { report: { market_date: string; risk_level: string; id: string } } if (payload.report.risk_level === 'EXTREME') { + this.rt.ctx.breakers.declareAbnormalDay(payload.report.market_date, { id: 'runtime', role: 'system' }, `situation ${payload.report.id}: risk EXTREME (OPEN-QUESTION B7 triggers)`, this.rt.ctx.clock()) this.rt.ctx.caseDesk.open({ id: this.rt.ctx.newId('case'), kind: 'ADHOC_ANALYSIS', @@ -115,6 +157,69 @@ export class TriggerService { } } + /** Live-data feed: actual curves for a market date through the ingestion pipeline (quality gate + evidence snapshot). */ + onMarketData(input: MarketDataInput) { + const m = MarketDataInput.parse(input) + const source = m.source ?? 'market-data-feed' + const series: Array<[string, string[] | undefined]> = [['load:aggregate', m.load_mw], ['pv:aggregate', m.pv_mw], ['price:da', m.price_yuan_per_mwh]] + const accepted: string[] = [] + const quarantined: Array<{ series: string; issues: string[] }> = [] + for (const [id, values] of series) { + if (!values) continue + const curve = Curve96.parse({ interval_minutes: 15, date: m.market_date, values }) + const r = this.rt.ctx.ingestion.ingestCurve(id, curve, source) + if (r.accepted) accepted.push(id) + else quarantined.push({ series: id, issues: r.issues }) + } + this.rt.ctx.events.append({ event_type: 'MarketDataIngested', payload: { market_date: m.market_date, source, accepted, quarantined }, correlation_id: `market-data-${m.market_date}` }) + return { market_date: m.market_date, accepted, quarantined } + } + + /** The human trader's actual submission for a market date (shadow comparison baseline). */ + onHumanBid(record: HumanBidRecord) { + const h = HumanBidRecord.parse(record) + this.rt.ctx.humanBids.put(h) + this.rt.ctx.events.append({ event_type: 'HumanBidRecorded', payload: { id: h.id, market_date: h.market_date, source: h.source }, correlation_id: `shadow-${h.market_date}` }) + return h + } + + /** + * Shadow market answer (docs/07 D-1 16:00 in shadow mode): the released + * shadow bid is cleared against the actual day-ahead price and the resulting + * AwardNotice goes down the normal award path. Idempotent per bid digest. + */ + async shadowClearing(marketDate: string) { + const ctx = this.rt.ctx + const bid = ctx.caseDesk.proposals + .list() + .map((r) => r.value) + .filter((p) => p.type === 'BID' && p.payload.market_date === marketDate && p.status === 'RELEASED') + .sort((a, b) => (a.created_at < b.created_at ? -1 : 1)) + .at(-1) + if (!bid) { + ctx.events.append({ event_type: 'ShadowClearingSkipped', payload: { market_date: marketDate, reason: 'no RELEASED bid for the day' }, correlation_id: `shadow-${marketDate}` }) + return null + } + const existing = ctx.awards.list().find((a) => a.value.bid_proposal_digest === bid.digest) + if (existing) return { award: existing.value, decomposition: null } + const price = ctx.timeseries.latest('price:da', marketDate)?.curve + if (!price) throw new Error(`no clearing price for ${marketDate} yet — ingest market data first`) + const award = shadowClearing(bid, price, { id: `award-shadow-${marketDate}`, now: ctx.clock() }) + ctx.events.append({ event_type: 'ShadowCleared', payload: { market_date: marketDate, award_id: award.id, bid_proposal_digest: bid.digest, awarded_mwh: award.awarded_mwh.values.reduce((s, v) => s + Number(v), 0).toFixed(3) }, correlation_id: `shadow-${marketDate}` }) + const decomposition = await this.onAward(award) + return { award, decomposition } + } + + /** D+1 close for a shadow day: simulated execution → metering → review → comparison → KPI (shadow-close workflow). */ + async shadowClose(marketDate: string): Promise<{ run_id: string; record: ShadowDayRecord; kpi: KpiReport; review_ran: boolean; execution_simulated: boolean }> { + const wf = this.rt.mastra.getWorkflow('shadowClose') + const run = await wf.createRun() + const result = await run.start({ inputData: { market_date: marketDate } }) + if (result.status === 'failed') throw result.error + if (result.status !== 'success') throw new Error(`shadow close for ${marketDate} ended ${result.status}`) + return { run_id: run.runId, ...result.result } + } + /** Metering adapter stand-in: D+1 metering arrives → review workflow (docs/07 D+1). */ async onMetering(record: MeteringRecord) { const m = MeteringRecord.parse(record) diff --git a/packages/runtime/src/workflows/award-decomposition.ts b/packages/runtime/src/workflows/award-decomposition.ts index e365ce1..aa68960 100644 --- a/packages/runtime/src/workflows/award-decomposition.ts +++ b/packages/runtime/src/workflows/award-decomposition.ts @@ -19,11 +19,11 @@ export const AwardInput = z.object({ award: AwardNotice }) export const AwardOutput = z.object({ case_id: z.string(), - proposal_id: z.string(), - proposal: ProposalSchema, - lifecycle_run_id: z.string(), + proposal_id: z.string().nullable(), + proposal: ProposalSchema.nullable(), + lifecycle_run_id: z.string().nullable(), lifecycle: LifecycleOutput.nullable(), - lifecycle_status: z.enum(['success', 'suspended', 'failed']), + lifecycle_status: z.enum(['success', 'suspended', 'failed', 'SUPPRESSED']), llm_used: z.boolean(), shortfall_mwh: z.string(), }) @@ -146,6 +146,7 @@ export function createAwardDecomposition(ctx: RuntimeContext, lifecycle: ReturnT allocations_ref: { tool_call_id: tc, path: 'allocations' }, rationale: `Dispatch ${tc}: ${out.allocations.length} unit(s), shortfall ${out.shortfall_mwh} MWh (template — LLM unavailable)`, } + if (ctx.breakers.isTripped('L4')) return { ...inputData, draft: template, llm_used: false } try { const prompt = `Award ${inputData.award_id} for ${inputData.market_date}. Dispatch tool call ${tc} output has fields total_target_mw and allocations (units: ${out.allocations.map((a) => a.unit_id).join(', ')}; shortfall_mwh = ${out.shortfall_mwh}). Return a DispatchProposalDraft referencing those two fields plus a short rationale. No numbers beyond those quoted.` const draft = await ctx.llm.structured('resource-agent', prompt, DispatchProposalDraftSchema) @@ -162,6 +163,12 @@ export function createAwardDecomposition(ctx: RuntimeContext, lifecycle: ReturnT inputSchema: resourceStrategy.outputSchema, outputSchema: AwardOutput, execute: async ({ inputData }) => { + const shortfall = (inputData.dispatch_output as { shortfall_mwh: string }).shortfall_mwh + if (ctx.breakers.isTripped('L4')) { + ctx.events.append({ event_type: 'ProposalSuppressed', payload: { case_id: inputData.case_id, market_date: inputData.market_date, type: 'DISPATCH_PLAN', by: 'L4', tool_calls: inputData.tool_calls.map((t) => t.tool_call_id) }, correlation_id: inputData.case_id }) + ctx.caseDesk.close(inputData.case_id, null, false) + return { case_id: inputData.case_id, proposal_id: null, proposal: null, lifecycle_run_id: null, lifecycle: null, lifecycle_status: 'SUPPRESSED' as const, llm_used: false, shortfall_mwh: shortfall } + } const proposal: Proposal = assembleDispatchProposal(inputData.draft, { id: ctx.newId('prop'), award_ref: inputData.award_id, @@ -187,7 +194,7 @@ export function createAwardDecomposition(ctx: RuntimeContext, lifecycle: ReturnT lifecycle: result.status === 'success' ? result.result : null, lifecycle_status: (result.status === 'success' ? 'success' : result.status === 'suspended' ? 'suspended' : 'failed') as 'success' | 'suspended' | 'failed', llm_used: inputData.llm_used, - shortfall_mwh: (inputData.dispatch_output as { shortfall_mwh: string }).shortfall_mwh, + shortfall_mwh: shortfall, } }, }) diff --git a/packages/runtime/src/workflows/day-ahead-bid.ts b/packages/runtime/src/workflows/day-ahead-bid.ts index 6ce65d1..66e3ae0 100644 --- a/packages/runtime/src/workflows/day-ahead-bid.ts +++ b/packages/runtime/src/workflows/day-ahead-bid.ts @@ -19,13 +19,14 @@ import type { createProposalLifecycle } from './proposal-lifecycle.js' */ export const BidInput = z.object({ market_date: MarketDate, case_id: z.string().min(1).optional() }) +/** SUPPRESSED: L4 kill switch — the template ran and produced data, but no Proposal was created (docs/13 §8). */ export const BidOutput = z.object({ case_id: z.string(), - proposal_id: z.string(), - proposal: ProposalSchema, - lifecycle_run_id: z.string(), + proposal_id: z.string().nullable(), + proposal: ProposalSchema.nullable(), + lifecycle_run_id: z.string().nullable(), lifecycle: LifecycleOutput.nullable(), - lifecycle_status: z.enum(['success', 'suspended', 'failed']), + lifecycle_status: z.enum(['success', 'suspended', 'failed', 'SUPPRESSED']), llm_used: z.boolean(), }) @@ -105,6 +106,7 @@ export function createDayAheadBid(ctx: RuntimeContext, lifecycle: ReturnType { + if (ctx.breakers.isTripped('L4')) { + // L4: back to pure human operation — the numbers exist in lineage snapshots, no Proposal is created. + ctx.events.append({ event_type: 'ProposalSuppressed', payload: { case_id: inputData.case_id, market_date: inputData.market_date, type: 'BID', by: 'L4', tool_calls: inputData.tool_calls.map((t) => t.tool_call_id) }, correlation_id: inputData.case_id }) + ctx.caseDesk.close(inputData.case_id, null, false) + return { case_id: inputData.case_id, proposal_id: null, proposal: null, lifecycle_run_id: null, lifecycle: null, lifecycle_status: 'SUPPRESSED' as const, llm_used: false } + } const proposal: Proposal = assembleBidProposal(inputData.draft, { id: ctx.newId('prop'), originator: { agent: 'trading-agent', trigger: 'SCHEDULED', task_id: inputData.case_id }, @@ -132,8 +140,9 @@ export function createDayAheadBid(ctx: RuntimeContext, lifecycle: ReturnType { const pack = ctx.policy.pack(ctx.config.policyPackId, inputData.proposal.lineage.policy_pack_version) let p = transition(ctx, inputData.case_id, inputData.proposal, 'RULE_CHECK') - const result = ctx.policy.check(p, { id: pack.id, version: pack.version }, { ledger: ctx.ledger.read(), ledgerService: ctx.ledger, snapshots: ctx.snapshots, now: ctx.clock() }) + const ledgerView = ctx.ledger.read() + const result = ctx.policy.check(p, { id: pack.id, version: pack.version }, { ledger: ledgerView, ledgerService: ctx.ledger, snapshots: ctx.snapshots, now: ctx.clock() }) + // Kill-switch state is checked alongside the policy pack (L4: no AI proposals; L2: reduce-only bids) — docs/13 §8. + const breakerViolations = ctx.breakers.guard(p, ledgerView) + const violations = [...result.violations, ...breakerViolations] + const ok = result.ok && breakerViolations.length === 0 const ref = ctx.snapshots.put(result) - ctx.events.append({ event_type: 'RuleCheckCompleted', payload: { proposal_id: p.id, ok: result.ok, result_ref: ref, violations: result.violations }, correlation_id: inputData.case_id, causation_id: p.id }) + ctx.events.append({ event_type: 'RuleCheckCompleted', payload: { proposal_id: p.id, ok, result_ref: ref, violations, breaker_violations: breakerViolations }, correlation_id: inputData.case_id, causation_id: p.id }) const carry: Carry = { proposal: p, case_id: inputData.case_id, simulation_alert: false, simulation_alerts: [], approvals: [], envelope_match: null, done: null } - if (!result.ok) { - p = transition(ctx, inputData.case_id, p, 'REJECTED', { by: 'rule-check', violations: result.violations }) - return finish({ ...carry, proposal: p }, 'REJECTED', result.violations.map((v) => `${v.rule_id}: ${v.message}`)) + if (!ok) { + p = transition(ctx, inputData.case_id, p, 'REJECTED', { by: 'rule-check', violations }) + return finish({ ...carry, proposal: p }, 'REJECTED', violations.map((v) => `${v.rule_id}: ${v.message}`)) } return carry }, @@ -159,6 +165,13 @@ export function createProposalLifecycle(ctx: RuntimeContext) { ctx.caseDesk.requestApproval({ proposal_id: p.id, proposal_digest: p.digest, case_id: inputData.case_id, run_id: '', required_level: 'L2', reasons: inputData.simulation_alerts, since: now }) return await suspend({ reason: 'simulation alert', proposal_id: p.id, required_level: 'L2', reasons: inputData.simulation_alerts }) } + // Loss breaker / abnormal-day protocol: envelopes are off, every proposal waits for a human (docs/13 §1, §8 L2). + const freeze = ctx.breakers.autoApprovalFreeze(p) + if (freeze) { + p = transition(ctx, inputData.case_id, p, 'PENDING_HUMAN', { because: 'breaker', reasons: [freeze] }) + ctx.caseDesk.requestApproval({ proposal_id: p.id, proposal_digest: p.digest, case_id: inputData.case_id, run_id: '', required_level: 'L2', reasons: [freeze], since: now }) + return await suspend({ reason: 'breaker', proposal_id: p.id, required_level: 'L2', reasons: [freeze] }) + } const match = ctx.envelopes.match(p, { now, priceBaseline: priceForecastFromLineage(ctx, p)?.quantiles.p50 ?? null }) ctx.snapshots.put(match) if (match.within) { @@ -204,9 +217,15 @@ export function createProposalLifecycle(ctx: RuntimeContext) { const permitId = (inputData.permit as { id: string } | undefined)?.id const permit = permitId ? ctx.authority.permits.get(permitId)?.value : undefined if (!permit) throw new Error('release reached without a permit — invariant I4 violated in workflow wiring') + const blocked = ctx.breakers.channelBlocked(inputData.proposal.type) + if (blocked && inputData.proposal.type !== 'BID') { + // L3: the control channel is closed; the plan stays AUTHORIZED and the case open for the operator. + ctx.events.append({ event_type: 'ChannelBlocked', payload: { proposal_id: inputData.proposal.id, permit_id: permit.id, reason: blocked }, correlation_id: inputData.case_id, causation_id: inputData.proposal.id }) + return { outcome: 'BLOCKED' as const, proposal: inputData.proposal, permit_id: permit.id, receipt_id: null, reasons: [blocked] } + } try { const receipt = ctx.gatewayFor(inputData.proposal.type).dispatch(inputData.proposal, permit, now) - const p = transition(ctx, inputData.case_id, inputData.proposal, 'RELEASED', { receipt_id: receipt.receipt_id, artifact_ref: receipt.artifact_ref }) + const p = transition(ctx, inputData.case_id, inputData.proposal, 'RELEASED', { receipt_id: receipt.receipt_id, artifact_ref: receipt.artifact_ref, channel: receipt.channel }) ctx.events.append({ event_type: 'ExecutionReceipt', payload: receipt, correlation_id: inputData.case_id, causation_id: p.id }) if (p.type === 'BID') { ctx.ledger.append({ diff --git a/packages/runtime/src/workflows/review.ts b/packages/runtime/src/workflows/review.ts index a396bae..942bb3d 100644 --- a/packages/runtime/src/workflows/review.ts +++ b/packages/runtime/src/workflows/review.ts @@ -86,6 +86,8 @@ export function createReview(ctx: RuntimeContext, envelopeReview: ReturnType new Decimal(v) +const energyOf = (mw: string[]) => mw.reduce((s, v) => s.add(v), D(0)).mul('0.25') + +function latestProposal(ctx: RuntimeContext, date: string, type: Proposal['type']): Proposal | null { + const all = ctx.caseDesk.proposals + .list() + .map((r) => r.value) + .filter((p) => p.type === type && p.payload.market_date === date) + .sort((a, b) => (a.created_at < b.created_at ? -1 : a.created_at > b.created_at ? 1 : 0)) + return all.at(-1) ?? null +} + +function forecastFromLineage(ctx: RuntimeContext, p: Proposal | null, tool: string): ForecastBundle | null { + if (!p) return null + for (const tc of p.lineage.tool_calls) { + if (tc.tool !== tool) continue + const parsed = ForecastBundleSchema.safeParse(ctx.snapshots.get(tc.outputs_ref)) + if (parsed.success) return parsed.data + } + return null +} + +function pvForecastFromSituation(ctx: RuntimeContext, date: string): ForecastBundle | null { + const sits = ctx.events.list({ event_type: 'SituationPublished' }).filter((e) => (e.payload as { report: { market_date: string } }).report.market_date === date) + const last = sits.at(-1)?.payload as { forecast_tool_calls: Array<{ tool: string; outputs_ref: string }> } | undefined + const pv = last?.forecast_tool_calls.find((t) => t.tool === 'pv-forecast') + if (!pv) return null + const parsed = ForecastBundleSchema.safeParse(ctx.snapshots.get(pv.outputs_ref)) + return parsed.success ? parsed.data : null +} + +/** Per-unit fulfilment ratio for simulated execution (shadow modelling assumption, see ShadowConfig). */ +export function shadowFulfillment(ctx: RuntimeContext): Record { + const cfg = ctx.config.shadow.fulfillment + const units = unitResources(ctx) + const out: Record = {} + for (const [unit, rids] of Object.entries(units)) { + if (cfg.kind === 'FIXED') { + out[unit] = Number(cfg.ratio) + continue + } + const scores = rids.map((r) => ctx.resources.get(r)?.value.reliability_score).filter((s): s is string => s !== undefined).map(Number) + out[unit] = scores.length === 0 ? 1 : scores.reduce((a, b) => a + b, 0) / scores.length + } + return out +} + +const fmt = (x: number, dp: number) => x.toFixed(dp) + +export function createShadowClose(ctx: RuntimeContext, review: ReturnType) { + const execute = createStep({ + id: 'execute', + inputSchema: ShadowCloseInput, + outputSchema: Stage, + execute: async ({ inputData }) => { + const date = inputData.market_date + const now = ctx.clock() + const dispatch = latestProposal(ctx, date, 'DISPATCH_PLAN') + let simulated = false + if (dispatch && dispatch.status === 'RELEASED') { + const existing = ctx.executionReports.list().filter((r) => r.value.market_date === date && r.value.dispatch_proposal_digest === dispatch.digest) + if (existing.length === 0) { + // No field feedback for this plan: the simulation gateway executes it with the configured fulfilment model. + const reports = ctx.simulationGateway.executePlan(dispatch, shadowFulfillment(ctx), now) + for (const r of reports) { + ctx.executionReports.put(r) + ctx.events.append({ event_type: 'ExecutionReport', payload: { ...r, simulated: true }, correlation_id: dispatch.digest }) + } + simulated = true + } + } + return { market_date: date, execution_simulated: simulated, review_ran: false } + }, + }) + + const meter = createStep({ + id: 'meter', + inputSchema: Stage, + outputSchema: Stage, + execute: async ({ inputData }) => { + const date = inputData.market_date + const now = ctx.clock() + if (ctx.metering.list().some((m) => m.value.market_date === date)) return inputData + const price = ctx.timeseries.latest('price:da', date)?.curve + if (!price) throw new Error(`no actual price for ${date} — ingest market data before closing the shadow day`) + const reports = ctx.executionReports.list().map((r) => r.value).filter((r) => r.market_date === date) + const metered: Curve96 = { + interval_minutes: 15, + date, + values: Array.from({ length: 96 }, (_, t) => reports.reduce((s, r) => s.add(r.actual_mw.values[t]!), D(0)).toFixed(3)), + } + const record: MeteringRecord = { id: `meter-${date}`, market_date: date, metered_mw: metered, actual_price_yuan_per_mwh: price, received_at: now } + ctx.metering.put(record) + ctx.events.append({ event_type: 'MeteringArrived', payload: { ...record, simulated: true }, correlation_id: record.id }) + return inputData + }, + }) + + const runReview = createStep({ + id: 'review', + inputSchema: Stage, + outputSchema: Stage, + execute: async ({ inputData }) => { + const date = inputData.market_date + // Real metering may already have triggered the review (trigger service); never review a day twice. + if (ctx.events.list({ event_type: 'ReviewCompleted', correlation_id: `review-${date}` }).length > 0) return inputData + const run = await review.createRun() + const res = await run.start({ inputData: { market_date: date } }) + if (res.status === 'failed') throw res.error + return { ...inputData, review_ran: true } + }, + }) + + const compare = createStep({ + id: 'compare', + inputSchema: Stage, + outputSchema: ShadowCloseOutput, + execute: async ({ inputData }) => { + const date = inputData.market_date + const now = ctx.clock() + const actualPrice = ctx.timeseries.latest('price:da', date)?.curve + if (!actualPrice) throw new Error(`no actual price for ${date}`) + const gaps: string[] = [] + + // ---- shadow bid + const bid = latestProposal(ctx, date, 'BID') + const suppressed = ctx.events.list({ event_type: 'ProposalSuppressed' }).some((e) => { + const p = e.payload as { market_date: string; type: string } + return p.market_date === date && p.type === 'BID' + }) + let shadowLine: BidLine | null = null + let expected: string | null = null + let llmUsed = false + if (bid?.type === 'BID') { + shadowLine = bidLine(bid.payload.prices_yuan_per_mwh.values, bid.payload.quantities_mwh.values, actualPrice.values) + expected = bid.payload.expected_revenue_yuan + const drafted = ctx.events.list({ event_type: 'BidDrafted' }).find((e) => (e.payload as { proposal_id: string }).proposal_id === bid.id) + llmUsed = !!(drafted?.payload as { llm_used?: boolean } | undefined)?.llm_used + gaps.push(...auditProposalLineage(bid, ctx.snapshots)) + if (bid.status === 'RELEASED') { + if (!ctx.gateway.receipts.list().some((r) => r.value.proposal_digest === bid.digest) && !ctx.fileExportGateway.receipts.list().some((r) => r.value.proposal_digest === bid.digest)) gaps.push(`${bid.id}: RELEASED without a gateway receipt`) + if (!ctx.authority.permits.list().some((r) => r.value.proposal_digest === bid.digest)) gaps.push(`${bid.id}: RELEASED without a permit on record`) + if (!ctx.ledger.read().entries.some((e) => e.kind === 'BID_SUBMITTED' && e.source_ref === bid.digest)) gaps.push(`${bid.id}: no BID_SUBMITTED ledger entry`) + } + const trail = ctx.events.list({ event_type: 'ProposalStatusChanged', correlation_id: bid.originator.task_id }).map((e) => (e.payload as { status: string }).status) + if (!trail.includes('RULE_CHECK')) gaps.push(`${bid.id}: no RULE_CHECK transition in the event log`) + } else if (!suppressed) { + gaps.push('no BID proposal for the day') + } + + // ---- baselines + const k = D(ctx.config.risk.commitment_buffer_k) + const capMw = ctx.resources.list().reduce((s, r) => s.add(r.value.certified_adjustable_mw), D(0)) + const capMwh = Array.from({ length: 96 }, () => capMw.mul(k).mul('0.25').toFixed(6)) + let eMax: string + try { + eMax = ctx.ledger.dayAheadBounds(date).daily_energy_max_mwh + } catch { + eMax = capMwh.reduce((s, v) => s.add(v), D(0)).toString() + } + const naive = naiveLine(capMwh, eMax, actualPrice.values) + const hindsight = hindsightLine(capMwh, eMax, ctx.config.risk.min_block_mwh, actualPrice.values) + const humanRec = ctx.humanBids + .list() + .map((r) => r.value) + .filter((h) => h.market_date === date) + .sort((a, b) => (a.received_at < b.received_at ? -1 : 1)) + .at(-1) + const human = humanRec ? { record_id: humanRec.id, source: humanRec.source, line: bidLine(humanRec.prices_yuan_per_mwh.values, humanRec.quantities_mwh.values, actualPrice.values) } : null + + // ---- award, dispatch, execution + const awardRec = ctx.awards.list().map((r) => r.value).filter((a) => a.market_date === date).at(-1) + const award = awardRec ? { id: awardRec.id, energy_mwh: awardRec.awarded_mwh.values.reduce((s, v) => s.add(v), D(0)).toFixed(3) } : null + const dispatchP = latestProposal(ctx, date, 'DISPATCH_PLAN') + let dispatch: ShadowDayRecord['dispatch'] = null + if (dispatchP) { + const opt = dispatchP.lineage.tool_calls.find((t) => t.tool === 'dispatch-optimization') + const out = opt ? (ctx.snapshots.get(opt.outputs_ref) as { shortfall_mwh?: string } | undefined) : undefined + dispatch = { proposal_id: dispatchP.id, outcome: dispatchP.status, shortfall_mwh: out?.shortfall_mwh ?? '0.000' } + gaps.push(...auditProposalLineage(dispatchP, ctx.snapshots)) + if (dispatchP.status === 'RELEASED' && !ctx.simulationGateway.receipts.list().some((r) => r.value.proposal_digest === dispatchP.digest)) gaps.push(`${dispatchP.id}: RELEASED without a simulation-gateway receipt`) + } else if (award && bid?.status === 'RELEASED') { + gaps.push('award recorded but no DISPATCH_PLAN proposal') + } + const reports: ExecutionReport[] = ctx.executionReports.list().map((r) => r.value).filter((r) => r.market_date === date) + let execution: ShadowDayRecord['execution'] = null + if (reports.length > 0) { + const planned = reports.reduce((s, r) => s.add(energyOf(r.planned_mw.values)), D(0)) + const delivered = reports.reduce((s, r) => s.add(energyOf(r.actual_mw.values)), D(0)) + const deviation = reports.reduce((s, r) => s.add(r.deviation_mwh), D(0)) + const band = D(ctx.config.kpi.executionDeviationBandPct).div(100) + const simulatedFlag = inputData.execution_simulated || ctx.events.list({ event_type: 'ExecutionReport' }).some((e) => (e.payload as { market_date: string; simulated?: boolean }).market_date === date && (e.payload as { simulated?: boolean }).simulated === true) + execution = { + planned_mwh: planned.toFixed(3), + delivered_mwh: delivered.toFixed(3), + deviation_mwh: deviation.toFixed(3), + fulfillment_ratio: planned.gt(0) ? delivered.div(planned).toFixed(3) : '1.000', + within_band: planned.gt(0) ? deviation.div(planned).lte(band) : true, + simulated: simulatedFlag, + } + } else if (dispatchP?.status === 'RELEASED') { + gaps.push(`${dispatchP.id}: RELEASED dispatch plan has no execution report`) + } + + // ---- forecast quality (same metric functions as the L2 harness, docs/12 §4) + const actualLoad = ctx.timeseries.latest('load:aggregate', date)?.curve ?? null + const actualPv = ctx.timeseries.latest('pv:aggregate', date)?.curve ?? null + const loadFc = forecastFromLineage(ctx, bid, 'load-forecast') + const priceFc = forecastFromLineage(ctx, bid, 'price-forecast') + const pvFc = pvForecastFromSituation(ctx, date) + const pvCap = ctx.resources.list().filter((r) => r.value.type === 'PV').reduce((s, r) => s.add(r.value.rated_power_mw), D(0)) + const pvCapacity = pvCap.gt(0) ? Number(pvCap.toString()) : Number(ctx.config.shadow.pvCapacityMw) + const num = (c: Curve96) => c.values.map(Number) + const forecast = { + load_mape: loadFc && actualLoad ? fmt(mape(num(actualLoad), num(loadFc.quantiles.p50)), 6) : null, + pv_nrmse: pvFc && actualPv && pvCapacity > 0 ? fmt(nrmse(num(actualPv), num(pvFc.quantiles.p50), pvCapacity), 6) : null, + price_mape: priceFc ? fmt(mape(num(actualPrice), num(priceFc.quantiles.p50)), 6) : null, + price_coverage_p10_p90: priceFc ? fmt(coverage(num(actualPrice), num(priceFc.quantiles.p10), num(priceFc.quantiles.p90)), 6) : null, + } + + // ---- decision latency: case opened → proposal reached AUTO_APPROVED / PENDING_HUMAN (human waiting excluded) + let latency: number | null = null + if (bid) { + const caseId = bid.originator.task_id + const opened = ctx.events.list({ event_type: 'CaseOpened', correlation_id: caseId })[0] + const decided = [...ctx.events.list({ event_type: 'Lifecycle.AUTO_APPROVED', correlation_id: caseId }), ...ctx.events.list({ event_type: 'Lifecycle.PENDING_HUMAN', correlation_id: caseId })] + .filter((e) => (e.payload as { proposal_id: string }).proposal_id === bid.id) + .sort((a, b) => (a.occurred_at < b.occurred_at ? -1 : 1))[0] + if (opened && decided) latency = Math.max(0, new Date(decided.occurred_at).getTime() - new Date(opened.occurred_at).getTime()) + } + + // ---- review outcome + const finding: ReviewFinding | undefined = ctx.findings + .list() + .map((r) => r.value) + .filter((f) => f.market_date === date) + .sort((a, b) => (a.generated_at < b.generated_at ? -1 : 1)) + .at(-1) + if (!ctx.metering.list().some((m) => m.value.market_date === date)) gaps.push('no metering record for the day') + if (!finding) gaps.push('no ReviewFinding for the day') + const recommendations = (finding?.writebacks ?? []) + .filter((w): w is Extract => w.target === 'ENVELOPE_RECOMMENDATION') + .map((w) => ({ envelope_id: w.envelope_id, action: w.action })) + + // ---- mark-to-market for the L2 loss breaker (docs/13 §1 单日亏损失控) + if (shadowLine) { + const cost = D(ctx.config.risk.marginal_cost_yuan_per_mwh).mul(shadowLine.cleared_energy_mwh) + const pnl = D(shadowLine.realised_revenue_yuan).sub(cost) + const mtm = ctx.breakers.recordDailyPnl(date, pnl.toFixed(2), now) + ctx.events.append({ event_type: 'ShadowMarkToMarket', payload: { market_date: date, pnl_yuan: pnl.toFixed(2), cumulative_yuan: mtm.cumulative_yuan, l2_tripped: mtm.tripped }, correlation_id: `shadow-${date}` }) + } + + const record: ShadowDayRecord = ShadowDayRecordSchema.parse({ + id: `shadow-${date}`, + market_date: date, + shadow: { + proposal_id: bid?.id ?? null, + digest: bid?.digest ?? null, + outcome: bid ? bid.status : suppressed ? 'SUPPRESSED' : 'NONE', + expected_revenue_yuan: expected, + line: shadowLine, + llm_used: llmUsed, + }, + human, + hindsight, + naive, + award, + dispatch, + execution, + forecast, + decision_latency_ms: latency, + review_finding_id: finding?.id ?? null, + envelope_recommendations: recommendations, + breakers_tripped: ctx.breakers.state().filter((b) => b.status === 'TRIPPED').map((b) => b.level), + abnormal_day: ctx.breakers.isAbnormalDay(date), + lineage_complete: gaps.length === 0, + lineage_gaps: gaps, + generated_at: now, + }) + ctx.shadowDays.put(record) + const ref = ctx.snapshots.put(record) + ctx.events.append({ event_type: 'ShadowDayRecorded', payload: { market_date: date, record_ref: ref, lineage_complete: record.lineage_complete, gaps, outcome: record.shadow.outcome }, correlation_id: `shadow-${date}` }) + + // ---- KPI dashboard regenerated after every day (docs/12 §4, §5: the run itself is evidence) + const kpi = computeKpiReport(ctx.shadowDays.list().map((r) => r.value), ctx.config.kpi, { id: `kpi-${date}`, now }) + ctx.kpiReports.put(kpi) + const kpiRef = ctx.snapshots.put(kpi) + ctx.events.append({ event_type: 'KpiReportGenerated', payload: { id: kpi.id, report_ref: kpiRef, consecutive_complete_days: kpi.shadow.consecutive_complete_days, kpis: kpi.kpis.map((e) => ({ id: e.id, value: e.value, status: e.status })) }, correlation_id: `shadow-${date}` }) + + return { market_date: date, record, kpi, review_ran: inputData.review_ran, execution_simulated: inputData.execution_simulated } + }, + }) + + return createWorkflow({ id: 'shadow-close', inputSchema: ShadowCloseInput, outputSchema: ShadowCloseOutput }) + .then(execute) + .then(meter) + .then(runReview) + .then(compare) + .commit() +} diff --git a/packages/runtime/test/api.test.ts b/packages/runtime/test/api.test.ts index 6c54525..5fc6484 100644 --- a/packages/runtime/test/api.test.ts +++ b/packages/runtime/test/api.test.ts @@ -22,7 +22,7 @@ describe('Case Desk HTTP API', () => { it('inbox → decide (approve) → case view + lineage expansion', async () => { const h = await harness({ llm: null }) const { call, close } = await serve(h) - expect((await call('GET', '/health')).body).toEqual({ status: 'ok', llm: 'null' }) + expect((await call('GET', '/health')).body).toMatchObject({ status: 'ok', llm: 'null', mode: 'FILE_EXPORT' }) const run = await h.rt.mastra.getWorkflow('dayAheadBid').createRun() const out = await run.start({ inputData: { market_date: MARKET_DATE } }) diff --git a/packages/runtime/test/lifecycle.test.ts b/packages/runtime/test/lifecycle.test.ts index 701af29..1df8b1d 100644 --- a/packages/runtime/test/lifecycle.test.ts +++ b/packages/runtime/test/lifecycle.test.ts @@ -42,9 +42,9 @@ describe('docs/07 timeline D-1 06:00 → 08:30 (LLM down: I6)', () => { expect(statuses(h, out.case_id)).toEqual(['DRAFT', 'RULE_CHECK', 'SIMULATION', 'ENVELOPE_CHECK', 'AUTO_APPROVED', 'FRESH_CHECK', 'AUTHORIZED', 'RELEASED']) // P2: every payload number traces to the MILP tool output; P7: the ledger recorded the bid. - const lin = h.rt.ctx.caseDesk.expandLineage(out.proposal_id) + const lin = h.rt.ctx.caseDesk.expandLineage(out.proposal_id!) expect(lin.tool_calls.map((t) => t.tool)).toEqual(['load-forecast', 'pv-forecast', 'price-forecast', 'bid-optimization-milp']) - expect((lin.tool_calls[3]!.outputs as { expected_revenue_yuan: string }).expected_revenue_yuan).toBe(out.proposal.payload.expected_revenue_yuan) + expect((lin.tool_calls[3]!.outputs as { expected_revenue_yuan: string }).expected_revenue_yuan).toBe(out.proposal!.payload.expected_revenue_yuan) expect(h.rt.ctx.ledger.viewByTimescale('DAY_AHEAD')).toHaveLength(1) expect(out.proposal.lineage.ledger_version).toBe(1) @@ -72,7 +72,7 @@ describe('docs/07 timeline D-1 06:00 → 08:30 (LLM down: I6)', () => { const b = await submitBid(without) expect(a.llm_used).toBe(true) expect(b.llm_used).toBe(false) - expect(a.proposal.payload).toEqual(b.proposal.payload) + expect(a.proposal!.payload).toEqual(b.proposal!.payload) expect(llm.prompts.map((p) => p.agentId)).toContain('trading-agent') await withLlm.rt.close() await without.rt.close() @@ -100,17 +100,17 @@ describe('human approval path (PENDING_HUMAN → resume)', () => { expect(statuses(h, out.case_id).at(-1)).toBe('PENDING_HUMAN') const inbox = h.rt.ctx.caseDesk.inbox() expect(inbox).toHaveLength(1) - expect(inbox[0]!.run_id).toBe(out.lifecycle_run_id) + expect(inbox[0]!.run_id).toBe(out.lifecycle_run_id!) expect(inbox[0]!.reasons[0]).toMatch(/no ACTIVE envelope/) const triggers = new TriggerService(h.rt) - const res = await triggers.decide(out.lifecycle_run_id, { decision: 'approve', approver: APPROVER, comment: 'ok' }) + const res = await triggers.decide(out.lifecycle_run_id!, { decision: 'approve', approver: APPROVER, comment: 'ok' }) expect(res.status).toBe('success') if (res.status === 'success') expect(res.result.outcome).toBe('RELEASED') expect(statuses(h, out.case_id).slice(-4)).toEqual(['APPROVED', 'FRESH_CHECK', 'AUTHORIZED', 'RELEASED']) const view = h.rt.ctx.caseDesk.read(out.case_id) expect(view.approvals[0]!.approver).toEqual(APPROVER) - expect(view.approvals[0]!.proposal_digest).toBe(out.proposal.digest) + expect(view.approvals[0]!.proposal_digest).toBe(out.proposal!.digest) await h.rt.close() }) @@ -119,12 +119,12 @@ describe('human approval path (PENDING_HUMAN → resume)', () => { const out = await submitBid(h) const triggers = new TriggerService(h.rt) for (const approver of [{ id: 'trading-agent', role: 'senior-trader' }, { id: 'user-x', role: 'agent' }, { id: 'bid-optimization-milp', role: 'ops-lead' }]) { - const res = await triggers.decide(out.lifecycle_run_id, { decision: 'approve', approver }) + const res = await triggers.decide(out.lifecycle_run_id!, { decision: 'approve', approver }) expect(res.status).toBe('suspended') } expect(h.rt.ctx.events.list({ event_type: 'ApprovalRefused' })).toHaveLength(3) expect(h.rt.ctx.caseDesk.read(out.case_id).approvals).toHaveLength(0) - const still = h.rt.ctx.caseDesk.proposals.get(out.proposal_id)!.value.status + const still = h.rt.ctx.caseDesk.proposals.get(out.proposal_id!)!.value.status expect(still).toBe('PENDING_HUMAN') await h.rt.close() }) @@ -132,10 +132,10 @@ describe('human approval path (PENDING_HUMAN → resume)', () => { it('a human reject ends the run as REJECTED', async () => { const h = await harness({ llm: null }) const out = await submitBid(h) - const res = await new TriggerService(h.rt).decide(out.lifecycle_run_id, { decision: 'reject', approver: APPROVER, comment: 'too aggressive' }) + const res = await new TriggerService(h.rt).decide(out.lifecycle_run_id!, { decision: 'reject', approver: APPROVER, comment: 'too aggressive' }) expect(res.status).toBe('success') if (res.status === 'success') expect(res.result.outcome).toBe('REJECTED') - expect(h.rt.ctx.caseDesk.proposals.get(out.proposal_id)!.value.status).toBe('REJECTED') + expect(h.rt.ctx.caseDesk.proposals.get(out.proposal_id!)!.value.status).toBe('REJECTED') await h.rt.close() }) @@ -157,14 +157,14 @@ describe('durability: a suspended approval survives process restart', () => { await h1.rt.close() const h2 = await harness({ llm: null, dataDir: h1.dataDir, seed: false }) - expect(h2.rt.ctx.caseDesk.inbox()[0]!.run_id).toBe(out.lifecycle_run_id) - expect(h2.rt.ctx.caseDesk.proposals.get(out.proposal_id)!.value.status).toBe('PENDING_HUMAN') - // The ledger is in-memory in phase 1: replay its monthly anchor so the fresh check sees the same version. - h2.rt.ctx.ledger.append({ id: 'contract-2026-03', timescale: 'MONTHLY', period: '2026-03', kind: 'CONTRACT', energy_mwh: (311.04 * 31).toFixed(3), curve: null, source_ref: 'contract-2026-03-001', expected_version: 0 }) - const state = await h2.rt.mastra.getWorkflow('proposalLifecycle').getWorkflowRunById(out.lifecycle_run_id) + expect(h2.rt.ctx.caseDesk.inbox()[0]!.run_id).toBe(out.lifecycle_run_id!) + expect(h2.rt.ctx.caseDesk.proposals.get(out.proposal_id!)!.value.status).toBe('PENDING_HUMAN') + // The ledger is file-backed: the monthly anchor and version survive the restart (M5 shadow run needs this). + expect(h2.rt.ctx.ledger.read().version).toBe(1) + const state = await h2.rt.mastra.getWorkflow('proposalLifecycle').getWorkflowRunById(out.lifecycle_run_id!) expect(state?.status).toBe('suspended') - const res = await new TriggerService(h2.rt).decide(out.lifecycle_run_id, { decision: 'approve', approver: APPROVER }) + const res = await new TriggerService(h2.rt).decide(out.lifecycle_run_id!, { decision: 'approve', approver: APPROVER }) expect(res.status).toBe('success') if (res.status === 'success') expect(res.result.outcome).toBe('RELEASED') expect(h2.rt.ctx.caseDesk.read(out.case_id).case.status).toBe('CLOSED_DONE') @@ -178,7 +178,7 @@ describe('staleness and permits (I3/I4)', () => { const out = await submitBid(h) h.rt.ctx.ledger.append({ id: 'late', timescale: 'MONTHLY', period: '2026-04', kind: 'CONTRACT', energy_mwh: '1.0', curve: null, source_ref: 'c', expected_version: 1 }) // Approval evidence is captured at resume time; the auto path captured version 1 in lineage. - const res = await new TriggerService(h.rt).decide(out.lifecycle_run_id, { decision: 'approve', approver: APPROVER }) + const res = await new TriggerService(h.rt).decide(out.lifecycle_run_id!, { decision: 'approve', approver: APPROVER }) expect(res.status).toBe('success') if (res.status === 'success') { // The approval itself now references ledger v2 — consistent — so this passes fresh check; @@ -207,7 +207,7 @@ describe('staleness and permits (I3/I4)', () => { // the permit at `now`, then the gateway is asked at now + TTL + 1s. const triggers = new TriggerService(h.rt) h.rt.ctx.events.subscribe('Lifecycle.AUTHORIZED', () => h.setNow('2026-03-14T06:00:01Z')) // TTL is 6h - const res = await triggers.decide(out.lifecycle_run_id, { decision: 'approve', approver: APPROVER }) + const res = await triggers.decide(out.lifecycle_run_id!, { decision: 'approve', approver: APPROVER }) expect(res.status).toBe('success') if (res.status === 'success') { expect(res.result.outcome).toBe('STALE') @@ -223,7 +223,7 @@ describe('staleness and permits (I3/I4)', () => { h.rt.ctx.events.subscribe('Lifecycle.AUTHORIZED', (evt) => { h.rt.ctx.authority.revoke((evt.payload as { permit_id: string }).permit_id, 'L1 kill switch', h.rt.ctx.clock()) }) - const res = await new TriggerService(h.rt).decide(out.lifecycle_run_id, { decision: 'approve', approver: APPROVER }) + const res = await new TriggerService(h.rt).decide(out.lifecycle_run_id!, { decision: 'approve', approver: APPROVER }) if (res.status === 'success') expect(res.result.reasons[0]).toMatch(/revoked/) else throw new Error(res.status) await h.rt.close() @@ -235,8 +235,8 @@ describe('replay (I7) and authoritative store (I8)', () => { const h = await harness({ llm: null }) h.rt.ctx.envelopes.register(envelope({ max_energy_mwh: '1500.0' })) const out = await submitBid(h) - const stored: Proposal = h.rt.ctx.caseDesk.proposals.get(out.proposal_id)!.value - expect(proposalDigest(stored)).toBe(out.proposal.digest) + const stored: Proposal = h.rt.ctx.caseDesk.proposals.get(out.proposal_id!)!.value + expect(proposalDigest(stored)).toBe(out.proposal!.digest) const originalCheck = h.rt.ctx.events.list({ event_type: 'RuleCheckCompleted' })[0]!.payload as { result_ref: string } const recorded = h.rt.ctx.snapshots.get(originalCheck.result_ref) as { ok: boolean; rules_evaluated: string[] } // Replay against a ledger reconstructed to the lineage version (the bid append moved it to 2). diff --git a/packages/runtime/test/m4.test.ts b/packages/runtime/test/m4.test.ts index 785fcea..a89b150 100644 --- a/packages/runtime/test/m4.test.ts +++ b/packages/runtime/test/m4.test.ts @@ -34,19 +34,19 @@ describe('docs/07 D-1 16:00: award → resource agent → dispatch plan through const bid = await bidDay(h) h.rt.ctx.envelopes.register(dispatchEnvelope({ max_total_mw: '30.0' })) const t = new TriggerService(h.rt) - const out = await t.onAward(awardFor(bid.proposal.digest, '3.0')) // 12 MW target, resource offers 24 MW + const out = await t.onAward(awardFor(bid.proposal!.digest, '3.0')) // 12 MW target, resource offers 24 MW expect(out.status).toBe('success') const r = out.result! expect(r.lifecycle?.outcome).toBe('RELEASED') - expect(r.proposal.type).toBe('DISPATCH_PLAN') + expect(r.proposal!.type).toBe('DISPATCH_PLAN') expect(r.shortfall_mwh).toBe('0.000') expect(h.rt.ctx.ledger.read().entries.map((e) => e.kind)).toEqual(['CONTRACT', 'BID_SUBMITTED', 'AWARD']) expect(h.skills.calls.filter((c) => c === 'potential' || c === 'dispatch')).toEqual(['potential', 'dispatch']) expect(h.rt.ctx.simulationGateway.receipts.list()).toHaveLength(1) // P2 for dispatch numbers: allocations are byte-identical to the tool output. - const lin = h.rt.ctx.caseDesk.expandLineage(r.proposal_id) + const lin = h.rt.ctx.caseDesk.expandLineage(r.proposal_id!) expect(lin.tool_calls.map((c) => c.tool)).toEqual(['potential-assessment', 'dispatch-optimization']) - expect((lin.tool_calls[1]!.outputs as { allocations: unknown }).allocations).toEqual(r.proposal.type === 'DISPATCH_PLAN' ? r.proposal.payload.allocations : null) + expect((lin.tool_calls[1]!.outputs as { allocations: unknown }).allocations).toEqual(r.proposal!.type === 'DISPATCH_PLAN' ? r.proposal!.payload.allocations : null) await h.rt.close() }) @@ -54,7 +54,7 @@ describe('docs/07 D-1 16:00: award → resource agent → dispatch plan through const h = await harness({ llm: null }) const bid = await bidDay(h) h.rt.ctx.envelopes.register(dispatchEnvelope({ max_total_mw: '100.0' })) - const out = await new TriggerService(h.rt).onAward(awardFor(bid.proposal.digest, '10.0')) // 40 MW target > 24 MW + const out = await new TriggerService(h.rt).onAward(awardFor(bid.proposal!.digest, '10.0')) // 40 MW target > 24 MW expect(out.status).toBe('success') expect(out.result!.lifecycle_status).toBe('suspended') expect(Number(out.result!.shortfall_mwh)).toBeGreaterThan(0) @@ -68,8 +68,8 @@ describe('docs/07 D+1: execution feedback → review → writebacks', () => { const bid = await bidDay(h) h.rt.ctx.envelopes.register(dispatchEnvelope({ max_total_mw: '30.0' })) const t = new TriggerService(h.rt) - const out = await t.onAward(awardFor(bid.proposal.digest, '3.0')) - const digest = out.result!.proposal.digest + const out = await t.onAward(awardFor(bid.proposal!.digest, '3.0')) + const digest = out.result!.proposal!.digest t.onExecutionReports(h.rt.ctx.simulationGateway.execute(digest, { 'agg-unit-01': fulfilment }, '2026-03-16T01:00:00Z')) const review = await t.onMetering({ id: `meter-${MARKET_DATE}`, market_date: MARKET_DATE, metered_mw: flat((12 * fulfilment).toFixed(3)), actual_price_yuan_per_mwh: flat('421.00'), received_at: '2026-03-16T02:00:00Z' }) return { t, review: review.result!, bid } diff --git a/packages/runtime/test/m5.test.ts b/packages/runtime/test/m5.test.ts new file mode 100644 index 0000000..34f992e --- /dev/null +++ b/packages/runtime/test/m5.test.ts @@ -0,0 +1,274 @@ +import { existsSync, readdirSync } from 'node:fs' +import type { AddressInfo } from 'node:net' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import type { Envelope, ShadowDayRecord } from '@vpp/domain' +import { BreakerAuthorityError } from '@vpp/services' +import { createApi } from '../src/api.js' +import { insightCards } from '../src/cards.js' +import type { RuntimeOptions } from '../src/runtime.js' +import { ShadowRunner, localToUtc, runBreakerDrill } from '../src/shadow.js' +import { TriggerService } from '../src/trigger.js' +import { envelope, harness } from './helpers.js' + +const OPS = { id: 'user-ops-lead', role: 'ops-lead' } +const RISK = { id: 'user-risk-officer', role: 'risk-officer' } +const ADMIN = { id: 'user-platform-admin', role: 'platform-admin' } + +const dispatchEnvelope = (bounds: Record): Envelope => + envelope(bounds, { id: 'env-dispatch-001', scope: { proposal_type: 'DISPATCH_PLAN', timescales: ['DAY_AHEAD'], resource_set: 'pool-hubei-01' } }) + +/** n consecutive market dates from `from` (all inside March 2026, the month the harness has a contract for). */ +const dates = (from: string, n: number): string[] => + Array.from({ length: n }, (_, i) => new Date(new Date(`${from}T00:00:00Z`).getTime() + i * 86_400_000).toISOString().slice(0, 10)) + +const flat = (date: string, v: string) => ({ interval_minutes: 15 as const, date, values: Array(96).fill(v) as string[] }) + +/** Shadow-mode harness: docs/07 world + both envelopes + trigger consumers + a runner that drives the injected clock. */ +async function shadowHarness(opts: { dataDir?: string; seed?: boolean; config?: RuntimeOptions['config'] } = {}) { + const h = await harness({ + llm: null, + ...(opts.dataDir ? { dataDir: opts.dataDir } : {}), + ...(opts.seed === false ? { seed: false } : {}), + config: { mode: 'SHADOW', shadow: { fulfillment: { kind: 'FIXED', ratio: '0.97' }, pvCapacityMw: '30' }, ...opts.config }, + }) + if (opts.seed !== false) { + h.rt.ctx.envelopes.register(envelope({ max_energy_mwh: '1500.0' })) + h.rt.ctx.envelopes.register(dispatchEnvelope({ max_total_mw: '30.0' })) + } + const t = new TriggerService(h.rt) + t.startEventConsumers() + const runner = new ShadowRunner(t, { setClock: h.setNow }) + /** The human trader's actual bid for the day: a flat price-taker block, labelled synthetic. */ + const humanBid = (date: string) => t.onHumanBid({ id: `hb-${date}`, market_date: date, prices_yuan_per_mwh: flat(date, '0.00'), quantities_mwh: flat(date, '3.000'), source: 'SYNTHETIC_NAIVE', received_at: localToUtc(date, '09:00', -1) }) + return { h, t, runner, humanBid } +} + +describe('ROADMAP M5 acceptance: shadow run on the docs/07 loop', () => { + it('21 consecutive shadow days with complete lineage, KPI report auto-generated, one envelope-widening recommendation produced but not acted on', async () => { + const { h, t, runner, humanBid } = await shadowHarness() + const days = dates('2026-03-02', 21) + const records: ShadowDayRecord[] = [] + for (const d of days) { + humanBid(d) + records.push((await runner.replayDay(d)).record) + } + // --- every day closed the full loop with complete lineage + expect(records.map((r) => r.lineage_gaps).flat()).toEqual([]) + expect(records.every((r) => r.lineage_complete)).toBe(true) + expect(records.map((r) => r.shadow.outcome)).toEqual(Array(21).fill('RELEASED')) + expect(records.every((r) => r.award !== null && r.dispatch?.outcome === 'RELEASED' && r.execution?.simulated === true && r.execution.fulfillment_ratio === '0.970' && r.execution.within_band)).toBe(true) + expect(records.every((r) => r.human?.source === 'SYNTHETIC_NAIVE' && r.review_finding_id !== null && r.forecast.load_mape !== null && r.forecast.pv_nrmse !== null && r.decision_latency_ms !== null)).toBe(true) + // --- external effects were simulated: bids recorded, never submitted; dispatch to the simulation gateway + expect(h.rt.ctx.gateway.receipts.list().map((r) => r.value.channel)).toEqual(Array(21).fill('SHADOW')) + expect(h.rt.ctx.simulationGateway.receipts.list()).toHaveLength(21) + expect(readdirSync(join(h.dataDir, 'exports'))).toEqual([]) + expect(h.rt.ctx.ledger.read().entries.filter((e) => e.kind === 'AWARD')).toHaveLength(21) + // --- KPI report auto-generated after every day, per docs/12 §4 definitions + const kpi = h.rt.ctx.kpiReports.list().map((r) => r.value).sort((a, b) => (a.generated_at < b.generated_at ? -1 : 1)).at(-1)! + expect(h.rt.ctx.events.list({ event_type: 'KpiReportGenerated' })).toHaveLength(21) + expect(kpi.shadow).toMatchObject({ days: 21, complete_days: 21, consecutive_complete_days: 21, first_date: '2026-03-02', last_date: '2026-03-22', released_days: 21, pending_days: 0, breaker_trips: 0 }) + expect(kpi.kpis.map((k) => k.id)).toEqual(['FORECAST_LOAD_MAPE', 'FORECAST_PV_NRMSE', 'POTENTIAL_ACCURACY', 'DECISION_LATENCY_P95_MS', 'DISPATCH_SUCCESS_RATE', 'REVENUE_UPLIFT_VS_HUMAN', 'CROSS_REGION_MATCH']) + const by = Object.fromEntries(kpi.kpis.map((k) => [k.id, k])) + expect(by['FORECAST_LOAD_MAPE']!.samples).toBe(21) + expect(by['POTENTIAL_ACCURACY']).toMatchObject({ value: '1.000000', status: 'MEET', samples: 21 }) + expect(by['DISPATCH_SUCCESS_RATE']).toMatchObject({ value: '1.000000', status: 'MEET', samples: 21 }) + expect(by['DECISION_LATENCY_P95_MS']).toMatchObject({ value: '0.000000', status: 'MEET', samples: 21 }) // frozen test clock; real clock in live mode + expect(by['REVENUE_UPLIFT_VS_HUMAN']!.samples).toBe(21) + expect(by['REVENUE_UPLIFT_VS_HUMAN']!.value).not.toBeNull() + expect(by['CROSS_REGION_MATCH']!.status).toBe('NOT_APPLICABLE') + expect(kpi.comparison.days_with_human_baseline).toBe(21) + expect(Number(kpi.comparison.hindsight_yuan)).toBeGreaterThanOrEqual(Number(kpi.comparison.shadow_yuan)) + // --- widening recommendation from shadow data (20 compliant days), surfaced for humans, envelopes untouched + expect(kpi.shadow.widen_recommendations).toBeGreaterThanOrEqual(1) + const widen = h.rt.ctx.envelopeRequests.list().map((r) => r.value).filter((r) => r.action === 'WIDEN') + expect(widen.map((r) => r.envelope_id).sort()).toEqual(expect.arrayContaining(['env-bid-001', 'env-dispatch-001'])) + expect(records[19]!.envelope_recommendations.some((e) => e.action === 'WIDEN')).toBe(true) + expect(h.rt.ctx.envelopes.get('env-bid-001')!.bounds).toEqual({ max_energy_mwh: '1500.0' }) + expect(h.rt.ctx.envelopes.get('env-dispatch-001')!.bounds).toEqual({ max_total_mw: '30.0' }) + expect(h.rt.ctx.events.list({ event_type: 'EnvelopeChanged' })).toHaveLength(0) + expect(h.rt.ctx.caseDesk.inbox().filter((p) => p.run_id !== '' && p.workflow === 'envelopeReview').length).toBeGreaterThanOrEqual(2) + // --- projections + expect(insightCards(h.rt.ctx, 'OPERATIONS_DASHBOARD').map((c) => c.title)).toEqual(['D+1 review', 'KPI dashboard', 'Kill switches', 'Open cases']) + expect(insightCards(h.rt.ctx, 'TRADING_DESK').find((c) => c.title === 'Shadow vs human vs hindsight')!.metrics.map((m) => m.name)).toContain('human_realised_revenue') + await h.rt.close() + }, 120_000) + + it('the shadow run survives a process restart: ledger, streaks and shadow records are file-backed', async () => { + const a = await shadowHarness({ config: { widenAfterCompliantDays: 4 } }) + for (const d of dates('2026-03-02', 3)) { + a.humanBid(d) + await a.runner.replayDay(d) + } + expect(a.h.rt.ctx.envelopeRequests.list()).toHaveLength(0) + await a.h.rt.close() + + const b = await shadowHarness({ dataDir: a.h.dataDir, seed: false, config: { widenAfterCompliantDays: 4 } }) + expect(b.h.rt.ctx.ledger.read().entries.map((e) => e.kind)).toEqual(['CONTRACT', ...Array(3).fill(['BID_SUBMITTED', 'AWARD']).flat()]) + expect(b.h.rt.ctx.shadowDays.list()).toHaveLength(3) + for (const d of dates('2026-03-05', 2)) { + b.humanBid(d) + await b.runner.replayDay(d) + } + const kpi = b.h.rt.ctx.kpiReports.list().map((r) => r.value).sort((x, y) => (x.generated_at < y.generated_at ? -1 : 1)).at(-1)! + expect(kpi.shadow.consecutive_complete_days).toBe(5) + // The compliant-day streak (3 before, 4th after the restart) crossed the widening threshold only because it was persisted. + expect(b.h.rt.ctx.envelopeRequests.list().map((r) => r.value.action)).toContain('WIDEN') + await b.h.rt.close() + }, 60_000) + + it('live data enters through the quality gate; quarantined curves never reach the forecast history', async () => { + const { h, t } = await shadowHarness() + const out = t.onMarketData({ market_date: '2026-05-01', load_mw: Array(96).fill('0'), price_yuan_per_mwh: Array(96).fill('400.00'), source: 'test-feed' }) + expect(out.accepted).toEqual(['price:da']) + expect(out.quarantined[0]).toMatchObject({ series: 'load:aggregate' }) + expect(h.rt.ctx.timeseries.latest('load:aggregate', '2026-05-01')).toBeUndefined() + expect(h.rt.ctx.ingestion.quarantined()).toHaveLength(1) + expect(h.rt.ctx.events.list({ event_type: 'MarketDataIngested' })).toHaveLength(1) + await h.rt.close() + }) +}) + +describe('kill-switch hierarchy L0–L4 (docs/13 §8)', () => { + it('drill: every level trips on drill objects, is verified where the chain feels it, and is reset — all in the event log', async () => { + const { h } = await shadowHarness() + const drill = await runBreakerDrill(h.rt.ctx, OPS) + expect(drill.steps.map((s) => [s.level, s.verified])).toEqual([['L0', true], ['L1', true], ['L2', true], ['L3', true], ['L4', true]]) + expect(drill.all_verified).toBe(true) + expect(h.rt.ctx.breakers.state().map((b) => [b.level, b.status, b.trip_count, b.drill])).toEqual([['L0', 'ARMED', 1, true], ['L1', 'ARMED', 1, true], ['L2', 'ARMED', 1, true], ['L3', 'ARMED', 1, true], ['L4', 'ARMED', 1, true]]) + expect(h.rt.ctx.events.list({ event_type: 'BreakerDrillStep' })).toHaveLength(5) + expect(h.rt.ctx.events.list({ event_type: 'BreakerTripped' })).toHaveLength(5) + expect(h.rt.ctx.events.list({ event_type: 'BreakerReset' })).toHaveLength(5) + expect(h.rt.ctx.events.list({ event_type: 'BreakerDrillCompleted' })[0]!.payload).toMatchObject({ all_verified: true }) + // The drill left no live effect: the next bid still auto-approves and releases. + const run = await h.rt.mastra.getWorkflow('dayAheadBid').createRun() + const res = await run.start({ inputData: { market_date: '2026-03-15' } }) + expect(res.status === 'success' && res.result.lifecycle?.outcome).toBe('RELEASED') + // Agents and unauthorized roles cannot pull a switch (I1/I2, OPEN-QUESTION B8 mapping). + await expect(runBreakerDrill(h.rt.ctx, { id: 'trading-agent', role: 'ops-lead' })).rejects.toThrow(BreakerAuthorityError) + expect(() => h.rt.ctx.breakers.trip('L4', { id: 'user-t', role: 'senior-trader' }, 'x', null, h.rt.ctx.clock())).toThrow(BreakerAuthorityError) + expect(() => h.rt.ctx.breakers.reset('L2', { id: 'runtime', role: 'system' }, 'x', h.rt.ctx.clock())).toThrow(BreakerAuthorityError) + await h.rt.close() + }) + + it('L2 loss breaker: a day marked to market beyond the loss budget freezes auto-approval and allows only position-reducing bids until a human resets it', async () => { + // Marginal cost far above price → every cleared MWh loses money; simulation budget lifted so the chain releases the bid. + const { h, t, runner, humanBid } = await shadowHarness({ + config: { risk: { risk_aversion: '0.3', commitment_buffer_k: '0.9', min_block_mwh: '0.5', marginal_cost_yuan_per_mwh: '100000' }, worstCaseLossBudgetYuan: '10000000000000' }, + }) + humanBid('2026-03-02') + const day1 = (await runner.replayDay('2026-03-02')).record + expect(day1.shadow.outcome).toBe('RELEASED') + expect(day1.breakers_tripped).toEqual(['L2']) + expect(h.rt.ctx.breakers.get('L2')).toMatchObject({ status: 'TRIPPED', tripped_by: { id: 'runtime', role: 'system' } }) + expect(h.rt.ctx.breakers.get('L2').reason).toMatch(/expected loss .* exceeds daily budget 100000/) + expect(h.rt.ctx.events.list({ event_type: 'BreakerTripped' }).map((e) => (e.payload as { level: string }).level)).toEqual(['L2']) + + // Next day: an exposure-increasing bid is rejected at rule check (reduce-only), so nothing reaches the envelope gate. + humanBid('2026-03-03') + const day2 = (await runner.replayDay('2026-03-03')).record + expect(day2.shadow.outcome).toBe('REJECTED') + const rejected = h.rt.ctx.events.list({ event_type: 'Lifecycle.REJECTED' }).at(-1)!.payload as { violations: Array<{ rule_id: string }> } + expect(rejected.violations.map((v) => v.rule_id)).toContain('breaker-l2-reduce-only') + expect(day2.award).toBeNull() + + // Reset needs an authorized human and a basis; then autonomy returns. + expect(() => h.rt.ctx.breakers.reset('L2', OPS, '', h.rt.ctx.clock())).toThrow(/basis/) + h.rt.ctx.breakers.reset('L2', RISK, `loss reviewed in ${day1.review_finding_id}; budget restored`, h.rt.ctx.clock()) + expect(h.rt.ctx.breakers.autoApprovalFreeze(h.rt.ctx.caseDesk.proposals.list()[0]!.value)).toBeNull() + expect(h.rt.ctx.events.list({ event_type: 'BreakerReset' })[0]!.payload).toMatchObject({ level: 'L2', by: RISK }) + void t + await h.rt.close() + }) + + it('abnormal-day protocol: an EXTREME situation turns envelopes off for that date — the bid waits for a human', async () => { + const { h, t, runner, humanBid } = await shadowHarness() + h.skills.spread = 1.2 // p90/p50 = 2.2 ≥ extreme ratio 2.0 (OPEN-QUESTION B7) + humanBid('2026-03-02') + const day = (await runner.replayDay('2026-03-02')).record + expect(h.rt.ctx.events.list({ event_type: 'AbnormalDayDeclared' })[0]!.payload).toMatchObject({ market_date: '2026-03-02' }) + expect(day.abnormal_day).toBe(true) + expect(day.shadow.outcome).toBe('PENDING_HUMAN') + expect(day.award).toBeNull() + const pending = h.rt.ctx.caseDesk.inbox().find((p) => p.run_id !== '' && p.workflow !== 'envelopeReview')! + expect(pending.reasons[0]).toMatch(/abnormal-day protocol for 2026-03-02/) + // A human can still approve the (unchanged) proposal — the protocol removes autonomy, not the chain. + const res = await t.decide(pending.run_id, { decision: 'approve', approver: OPS, comment: 'reviewed extreme-day forecast' }) + expect(res.status === 'success' && res.result.outcome).toBe('RELEASED') + h.rt.ctx.breakers.clearAbnormalDay('2026-03-02', OPS, h.rt.ctx.clock()) + expect(h.rt.ctx.breakers.isAbnormalDay('2026-03-02')).toBe(false) + await h.rt.close() + }) + + it('L3 channel breaker: bids fall back to the manual file channel, dispatch plans are authorized but BLOCKED', async () => { + const { h, t, humanBid } = await shadowHarness() + h.rt.ctx.breakers.trip('L3', OPS, 'trading-platform link down', '*', h.rt.ctx.clock()) + humanBid('2026-03-02') + h.setNow(localToUtc('2026-03-02', '08:00', -1)) + const run = await h.rt.mastra.getWorkflow('dayAheadBid').createRun() + const res = await run.start({ inputData: { market_date: '2026-03-02' } }) + expect(res.status === 'success' && res.result.lifecycle?.outcome).toBe('RELEASED') + const receipt = h.rt.ctx.fileExportGateway.receipts.list()[0]!.value + expect(receipt.channel).toBe('FILE_EXPORT') + expect(existsSync(receipt.artifact_ref)).toBe(true) + expect(h.rt.ctx.gateway.receipts.list()).toHaveLength(0) + h.setNow(localToUtc('2026-03-02', '16:00', -1)) + const cleared = await t.shadowClearing('2026-03-02') + expect(cleared?.decomposition?.result?.lifecycle?.outcome).toBe('BLOCKED') + expect(h.rt.ctx.events.list({ event_type: 'ChannelBlocked' })).toHaveLength(1) + expect(h.rt.ctx.simulationGateway.receipts.list()).toHaveLength(0) + await h.rt.close() + }) + + it('L4: AI suggestions off — the periodic templates still run the tools and produce data, but no Proposal is created', async () => { + const { h } = await shadowHarness() + h.rt.ctx.breakers.trip('L4', ADMIN, 'quarterly no-AI operation day', '*', h.rt.ctx.clock()) + const run = await h.rt.mastra.getWorkflow('dayAheadBid').createRun() + const res = await run.start({ inputData: { market_date: '2026-03-15' } }) + expect(res.status).toBe('success') + if (res.status !== 'success') return + expect(res.result.lifecycle_status).toBe('SUPPRESSED') + expect(res.result.proposal).toBeNull() + expect(h.skills.calls).toContain('milp') // the numbers were still computed and snapshotted + expect(h.rt.ctx.caseDesk.proposals.list()).toHaveLength(0) + expect(h.rt.ctx.events.list({ event_type: 'ProposalSuppressed' })).toHaveLength(1) + expect(h.rt.ctx.caseDesk.read(res.result.case_id).case.status).toBe('CLOSED_ABORTED') + h.rt.ctx.breakers.reset('L4', ADMIN, 'no-AI day complete', h.rt.ctx.clock()) + const again = await (await h.rt.mastra.getWorkflow('dayAheadBid').createRun()).start({ inputData: { market_date: '2026-03-15' } }) + expect(again.status === 'success' && again.result.lifecycle?.outcome).toBe('RELEASED') + await h.rt.close() + }) +}) + +describe('Case Desk API: shadow, KPI and breaker endpoints', () => { + it('serves breakers/KPI/shadow days and enforces breaker authority', async () => { + const { h, t, runner, humanBid } = await shadowHarness() + const server = createApi(h.rt, t) + await new Promise((r) => server.listen(0, '127.0.0.1', r)) + const base = `http://127.0.0.1:${(server.address() as AddressInfo).port}` + // eslint-disable-next-line @typescript-eslint/no-explicit-any + const call = async (method: string, path: string, body?: unknown, headers: Record = {}): Promise<{ status: number; body: any }> => { + const init: RequestInit = { method, headers: { 'content-type': 'application/json', ...headers } } + if (body !== undefined) init.body = JSON.stringify(body) + const res = await fetch(base + path, init) + return { status: res.status, body: await res.json() } + } + expect((await call('GET', '/health')).body).toMatchObject({ mode: 'SHADOW', breakers_tripped: [] }) + expect((await call('GET', '/kpi')).status).toBe(404) + expect((await call('GET', '/breakers')).body.levels).toHaveLength(5) + expect((await call('POST', '/breakers/L2/trip', { reason: 'x' })).status).toBe(401) + expect((await call('POST', '/breakers/L2/trip', { reason: 'x' }, { 'x-user-id': 'user-t', 'x-user-role': 'senior-trader' })).status).toBe(403) + expect((await call('POST', '/breakers/L2/trip', { reason: 'manual risk hold' }, { 'x-user-id': RISK.id, 'x-user-role': RISK.role })).body).toMatchObject({ level: 'L2', status: 'TRIPPED' }) + expect((await call('POST', '/breakers/L2/reset', { basis: 'hold lifted after review' }, { 'x-user-id': RISK.id, 'x-user-role': RISK.role })).body).toMatchObject({ level: 'L2', status: 'ARMED' }) + expect((await call('POST', '/breakers/drill', undefined, { 'x-user-id': OPS.id, 'x-user-role': OPS.role })).body.all_verified).toBe(true) + expect((await call('POST', '/market-data', { market_date: '2026-05-02', price_yuan_per_mwh: Array(96).fill('410.00') })).body.accepted).toEqual(['price:da']) + humanBid('2026-03-02') + await runner.replayDay('2026-03-02') + expect((await call('GET', '/shadow/days')).body).toHaveLength(1) + expect((await call('GET', '/shadow/days/2026-03-02')).body.lineage_complete).toBe(true) + expect((await call('GET', '/kpi')).body.shadow.days).toBe(1) + expect((await call('GET', '/cards/OPERATIONS_DASHBOARD')).body.map((c: { title: string }) => c.title)).toContain('KPI dashboard') + await new Promise((r) => server.close(() => r())) + await h.rt.close() + }) +}) diff --git a/packages/services/src/authority.ts b/packages/services/src/authority.ts index 10a8f83..a3c1903 100644 --- a/packages/services/src/authority.ts +++ b/packages/services/src/authority.ts @@ -5,6 +5,9 @@ import type { LedgerService } from './ledger.js' import { MemoryRepository } from './relational.js' import type { Repository } from './relational.js' +/** ISO timestamps must be compared as instants, not strings: '…00Z' and '…00.000Z' are the same moment. */ +const ts = (iso: string) => Date.parse(iso) + /** * Authority service (docs/03 §2.2, I3/I4): fresh-state check + permit issue + * revocation. Approval ≠ execution: hours may pass, the world changes. The @@ -62,7 +65,7 @@ export class AuthorityService { for (const a of basis.approvals) { if (a.decision !== 'APPROVE') reasons.push(`approval ${a.id} is ${a.decision}`) if (a.proposal_digest !== proposal.digest) reasons.push(`approval ${a.id} bound to digest ${a.proposal_digest.slice(0, 8)}…, proposal is ${proposal.digest.slice(0, 8)}…`) - if (ctx.now < a.validity.from || ctx.now > a.validity.to) reasons.push(`approval ${a.id} validity [${a.validity.from}, ${a.validity.to}] does not cover ${ctx.now}`) + if (ts(ctx.now) < ts(a.validity.from) || ts(ctx.now) > ts(a.validity.to)) reasons.push(`approval ${a.id} validity [${a.validity.from}, ${a.validity.to}] does not cover ${ctx.now}`) if (a.evidence_versions.ledger_version !== ledgerVersion) reasons.push(`ledger version ${a.evidence_versions.ledger_version} at approval, now ${ledgerVersion}`) if (a.evidence_versions.policy_pack_version !== ctx.currentPolicyPackVersion) reasons.push(`policy pack ${a.evidence_versions.policy_pack_version} at approval, now ${ctx.currentPolicyPackVersion}`) } @@ -115,7 +118,7 @@ export class AuthorityService { if (!stored) reasons.push(`permit ${permit.id} not on record`) else if (stored.revoked_at) reasons.push(`permit ${permit.id} revoked at ${stored.revoked_at}`) if (permit.proposal_digest !== proposal.digest) reasons.push('permit digest does not match proposal') - if (now < permit.issued_at || now >= permit.expires_at) reasons.push(`permit expired at ${permit.expires_at} (now ${now})`) + if (ts(now) < ts(permit.issued_at) || ts(now) >= ts(permit.expires_at)) reasons.push(`permit expired at ${permit.expires_at} (now ${now})`) if (proposal.type === 'BID' && permit.effect_limits['max_energy_mwh'] !== undefined) { const total = proposal.payload.quantities_mwh.values.reduce((s, v) => s.add(new Decimal(v)), new Decimal(0)) if (total.gt(permit.effect_limits['max_energy_mwh'])) reasons.push(`energy ${total.toString()} exceeds permit limit ${permit.effect_limits['max_energy_mwh']}`) diff --git a/packages/services/src/breaker.ts b/packages/services/src/breaker.ts new file mode 100644 index 0000000..9016a90 --- /dev/null +++ b/packages/services/src/breaker.ts @@ -0,0 +1,240 @@ +import type { BreakerLevel, BreakerRecord, LedgerView, Proposal } from '@vpp/domain' +import { BreakerRecord as BreakerRecordSchema } from '@vpp/domain' +import type { AuthorityService } from './authority.js' +import { createKeyValueStore, kvGet, kvSet } from './counters.js' +import type { KeyValueStore } from './counters.js' +import { Decimal } from './decimal.js' +import type { EnvelopeService } from './envelope.js' +import type { Violation } from './policy.js' +import { MemoryRepository } from './relational.js' +import type { Repository } from './relational.js' + +/** + * Kill-switch hierarchy (docs/13 §8), from precise to total: + * L0 permit revocation → gateway refuses that one effect + * L1 envelope suspension → that class of action returns to humans + * L2 loss breaker → every auto-approval frozen; only position-reducing bids pass rule check + * L3 channel breaker → no external channel accepts effects (bids fall back to the manual file channel) + * L4 AI suggestions off → periodic workflows still produce data, no Proposal is created + * Each level records who tripped it, why, and the quantified basis for reset + * (三要素). The loss breaker and abnormal-day protocol (docs/13 §1) are the two + * automatic trips; everything else needs a human with an authorized role. + */ +export const BREAKER_LEVELS: BreakerLevel[] = ['L0', 'L1', 'L2', 'L3', 'L4'] + +export interface Actor { + id: string + role: string +} + +export interface BreakerConfig { + /** L2 threshold: cumulative expected loss for one market date. OPEN-QUESTION B5. */ + dailyLossBudgetYuan: string + /** Roles allowed to trip and reset each level. OPEN-QUESTION B8. */ + authority: Record + /** Levels the runtime may trip on its own (actor role 'system'): loss breaker, streak suspension. */ + automatic: BreakerLevel[] +} + +export const BREAKER_CONFIG_PLACEHOLDER: BreakerConfig = { + dailyLossBudgetYuan: '100000', // OPEN-QUESTION B5 + authority: { + // OPEN-QUESTION B8: which posts may pull each level — placeholder mapping. + L0: ['senior-trader', 'ops-lead', 'risk-officer'], + L1: ['ops-lead', 'risk-officer'], + L2: ['risk-officer', 'ops-lead'], + L3: ['ops-lead', 'platform-admin'], + L4: ['platform-admin', 'ops-lead'], + }, + automatic: ['L1', 'L2'], +} + +export class BreakerAuthorityError extends Error { + constructor(message: string) { + super(message) + this.name = 'BreakerAuthorityError' + } +} + +export type BreakerAction = 'TRIP' | 'RESET' | 'ABNORMAL_DAY_DECLARED' | 'ABNORMAL_DAY_CLEARED' + +export interface BreakerDeps { + records?: Repository + kv?: KeyValueStore + authority: AuthorityService + envelopes: EnvelopeService + cfg?: BreakerConfig + /** Event hook: the runtime appends a BreakerTripped/BreakerReset/AbnormalDay* event. */ + onChange?: (action: BreakerAction, detail: Record) => void +} + +const armed = (level: BreakerLevel): BreakerRecord => ({ + level, + status: 'ARMED', + scope: null, + reason: null, + tripped_at: null, + tripped_by: null, + reset_at: null, + reset_by: null, + reset_basis: null, + drill: false, + trip_count: 0, +}) + +export class BreakerService { + readonly records: Repository + private readonly kv: KeyValueStore + readonly cfg: BreakerConfig + + constructor(private readonly deps: BreakerDeps) { + this.records = deps.records ?? new MemoryRepository(BreakerRecordSchema, (r) => r.level) + this.kv = deps.kv ?? createKeyValueStore() + this.cfg = deps.cfg ?? BREAKER_CONFIG_PLACEHOLDER + } + + get(level: BreakerLevel): BreakerRecord { + return this.records.get(level)?.value ?? armed(level) + } + + state(): BreakerRecord[] { + return BREAKER_LEVELS.map((l) => this.get(l)) + } + + isTripped(level: BreakerLevel): boolean { + return this.get(level).status === 'TRIPPED' + } + + private assertMayOperate(level: BreakerLevel, by: Actor, op: 'trip' | 'reset'): void { + if (by.role === 'system') { + if (op === 'trip' && this.cfg.automatic.includes(level)) return + throw new BreakerAuthorityError(`the runtime may not ${op} ${level} on its own (docs/13 §8: recovery is a human decision)`) + } + if (by.id.endsWith('-agent')) throw new BreakerAuthorityError(`agents may not ${op} breakers (I1/I2)`) + if (!this.cfg.authority[level].includes(by.role)) throw new BreakerAuthorityError(`role '${by.role}' may not ${op} ${level} (OPEN-QUESTION B8 mapping)`) + } + + /** + * Trip a level. L0/L1 act on a specific permit/envelope (scope = its id); + * L2–L4 are global. The effect is applied here so "tripped" and "effective" + * cannot drift apart. + */ + trip(level: BreakerLevel, by: Actor, reason: string, scope: string | null, now: string, drill = false): BreakerRecord { + this.assertMayOperate(level, by, 'trip') + if (level === 'L0') { + if (!scope) throw new Error('L0 needs a permit id as scope') + this.deps.authority.revoke(scope, reason, now) + } else if (level === 'L1') { + if (!scope) throw new Error('L1 needs an envelope id as scope') + this.deps.envelopes.suspend(scope, now) + } + const prev = this.get(level) + const rec: BreakerRecord = { + ...prev, + status: 'TRIPPED', + scope: scope ?? '*', + reason, + tripped_at: now, + tripped_by: by, + reset_at: null, + reset_by: null, + reset_basis: null, + drill, + trip_count: prev.trip_count + 1, + } + this.records.put(rec) + this.deps.onChange?.('TRIP', { level, scope: rec.scope, reason, by, drill, at: now }) + return rec + } + + /** + * Re-arm a level. Requires an authorized human and a basis (quantified + * recovery condition + review reference). Reset never un-revokes a permit or + * re-activates an envelope: those go back through the safety chain and the + * envelope-review workflow respectively. + */ + reset(level: BreakerLevel, by: Actor, basis: string, now: string): BreakerRecord { + this.assertMayOperate(level, by, 'reset') + if (!basis.trim()) throw new Error('reset needs a basis (recovery condition + review reference)') + const prev = this.get(level) + if (prev.status === 'ARMED') return prev + const rec: BreakerRecord = { ...prev, status: 'ARMED', reset_at: now, reset_by: by, reset_basis: basis } + this.records.put(rec) + this.deps.onChange?.('RESET', { level, basis, by, at: now, was_drill: prev.drill }) + return rec + } + + // ---- L2: mark-to-market ---------------------------------------------------- + + /** Accumulate a market date's realised/expected P&L; trips L2 when the day's loss exceeds the budget. */ + recordDailyPnl(marketDate: string, pnlYuan: string, now: string): { cumulative_yuan: string; tripped: boolean } { + const key = `pnl:${marketDate}` + const cumulative = new Decimal(kvGet(this.kv, key) || '0').add(pnlYuan) + kvSet(this.kv, key, cumulative.toFixed(2)) + const budget = new Decimal(this.cfg.dailyLossBudgetYuan) + if (cumulative.lt(budget.neg()) && !this.isTripped('L2')) { + this.trip('L2', { id: 'runtime', role: 'system' }, `expected loss ${cumulative.neg().toFixed(2)} yuan on ${marketDate} exceeds daily budget ${budget.toString()}`, '*', now) + return { cumulative_yuan: cumulative.toFixed(2), tripped: true } + } + return { cumulative_yuan: cumulative.toFixed(2), tripped: false } + } + + dailyPnl(marketDate: string): string { + return kvGet(this.kv, `pnl:${marketDate}`) || '0.00' + } + + // ---- abnormal-day protocol (docs/13 §1) ---------------------------------- + + /** Extreme forecast / new price regime → every proposal for that date goes to a human (envelopes temporarily off). */ + declareAbnormalDay(marketDate: string, by: Actor, reason: string, now: string): void { + if (by.role !== 'system') this.assertMayOperate('L1', by, 'trip') + if (this.isAbnormalDay(marketDate)) return + kvSet(this.kv, `abnormal:${marketDate}`, reason) + this.deps.onChange?.('ABNORMAL_DAY_DECLARED', { market_date: marketDate, reason, by, at: now }) + } + + clearAbnormalDay(marketDate: string, by: Actor, now: string): void { + this.assertMayOperate('L1', by, 'reset') + if (!this.isAbnormalDay(marketDate)) return + kvSet(this.kv, `abnormal:${marketDate}`, '') + this.deps.onChange?.('ABNORMAL_DAY_CLEARED', { market_date: marketDate, by, at: now }) + } + + isAbnormalDay(marketDate: string): boolean { + return !!kvGet(this.kv, `abnormal:${marketDate}`) + } + + abnormalDayReason(marketDate: string): string | null { + return kvGet(this.kv, `abnormal:${marketDate}`) || null + } + + // ---- guards consulted by the safety chain -------------------------------- + + /** Reason the envelope gate must send this proposal to a human regardless of envelope match; null = normal. */ + autoApprovalFreeze(p: Proposal): string | null { + if (this.isTripped('L2')) return `L2 loss breaker tripped: auto-approval frozen (${this.get('L2').reason ?? ''})` + if (this.isAbnormalDay(p.payload.market_date)) return `abnormal-day protocol for ${p.payload.market_date}: ${this.abnormalDayReason(p.payload.market_date) ?? ''} — all proposals manual` + return null + } + + /** Rule-check violations from breaker state: L4 blocks every AI proposal; L2 allows only position-reducing bids. */ + guard(p: Proposal, ledger: LedgerView): Violation[] { + const out: Violation[] = [] + if (this.isTripped('L4')) out.push({ rule_id: 'breaker-l4', severity: 'REJECT', message: `L4 tripped: AI proposals disabled (${this.get('L4').reason ?? ''})` }) + if (this.isTripped('L2') && p.type === 'BID') { + const existing = ledger.entries.filter((e) => e.kind === 'BID_SUBMITTED' && e.timescale === 'DAY_AHEAD' && e.period === p.payload.market_date).at(-1) + const current = new Decimal(existing?.energy_mwh ?? '0') + const proposed = p.payload.quantities_mwh.values.reduce((s, v) => s.add(v), new Decimal(0)) + if (proposed.gt(current)) { + out.push({ rule_id: 'breaker-l2-reduce-only', severity: 'REJECT', message: `L2 tripped: only position-reducing bids allowed; ${proposed.toFixed(3)} MWh > current ${current.toFixed(3)} MWh for ${p.payload.market_date}` }) + } + } + return out + } + + /** L3: reason the external channel for this proposal type is closed; null = open. */ + channelBlocked(type: Proposal['type']): string | null { + if (!this.isTripped('L3')) return null + return `L3 channel breaker tripped for ${type}: ${this.get('L3').reason ?? ''}` + } +} diff --git a/packages/services/src/counters.ts b/packages/services/src/counters.ts new file mode 100644 index 0000000..bed1b28 --- /dev/null +++ b/packages/services/src/counters.ts @@ -0,0 +1,26 @@ +import { z } from 'zod' +import { MemoryRepository } from './relational.js' +import type { Repository } from './relational.js' + +/** + * Tiny persisted key→value store for operational counters that must survive a + * process restart during a multi-week shadow run: envelope deviation streaks, + * review compliance streaks, daily mark-to-market, abnormal-day flags. Values + * are strings (decimal strings for money) so the same row schema fits every + * counter and the FsRepository adapter needs nothing new. + */ +export const KeyValueRow = z.object({ key: z.string().min(1), value: z.string() }) +export type KeyValueRow = z.infer + +export type KeyValueStore = Repository + +export const createKeyValueStore = (clock?: () => string): KeyValueStore => new MemoryRepository(KeyValueRow, (r) => r.key, clock) + +export const kvGet = (kv: KeyValueStore, key: string): string | undefined => kv.get(key)?.value.value +export const kvSet = (kv: KeyValueStore, key: string, value: string): void => { + kv.put({ key, value }) +} +export const kvDelete = (kv: KeyValueStore, key: string): void => { + // Repositories have no delete; an empty value is "unset" for every counter that uses this store. + if (kv.get(key)) kv.put({ key, value: '' }) +} diff --git a/packages/services/src/envelope.ts b/packages/services/src/envelope.ts index 336b9c8..9c5f7a8 100644 --- a/packages/services/src/envelope.ts +++ b/packages/services/src/envelope.ts @@ -1,5 +1,7 @@ import type { ApprovalLevel, Curve96, Envelope, EnvelopeMatch, Proposal } from '@vpp/domain' import { Envelope as EnvelopeSchema } from '@vpp/domain' +import { createKeyValueStore, kvGet, kvSet } from './counters.js' +import type { KeyValueStore } from './counters.js' import { Decimal } from './decimal.js' import { MemoryRepository } from './relational.js' import type { Repository } from './relational.js' @@ -21,7 +23,11 @@ export interface EnvelopeContext { } export class EnvelopeService { - constructor(readonly repo: Repository = new MemoryRepository(EnvelopeSchema, (e) => e.id)) {} + constructor( + readonly repo: Repository = new MemoryRepository(EnvelopeSchema, (e) => e.id), + /** Deviation streak per envelope; persisted so a suspension condition survives restarts. */ + private readonly streaks: KeyValueStore = createKeyValueStore(), + ) {} register(envelope: Envelope): void { this.repo.put(envelope) @@ -31,8 +37,6 @@ export class EnvelopeService { return this.repo.get(id)?.value } - private readonly streaks = new Map() - /** * Escalation rule (docs/03 §3): N consecutive within-envelope executions * whose deviation exceeds the threshold suspend the envelope — autonomy @@ -44,16 +48,24 @@ export class EnvelopeService { if (!row) throw new Error(`envelope ${envelopeId} not found`) const env = row.value const exceeded = new Decimal(deviationPct).gt(env.escalation.deviation_threshold_pct) - const streak = exceeded ? (this.streaks.get(envelopeId) ?? 0) + 1 : 0 - this.streaks.set(envelopeId, streak) + const streak = exceeded ? Number(kvGet(this.streaks, `streak:${envelopeId}`) || 0) + 1 : 0 + kvSet(this.streaks, `streak:${envelopeId}`, String(streak)) if (exceeded && streak >= env.escalation.max_consecutive_deviations && env.status === 'ACTIVE') { - this.repo.put({ ...env, status: 'SUSPENDED' }, row.version) - void now + this.suspend(envelopeId, now) return { streak, suspended: true } } return { streak, suspended: false } } + /** L1 kill switch (docs/13 §8): the envelope stops auto-approving until a human re-approves it via envelope-review. */ + suspend(envelopeId: string, now: string): Envelope { + const row = this.repo.get(envelopeId) + if (!row) throw new Error(`envelope ${envelopeId} not found`) + if (row.value.status === 'SUSPENDED') return row.value + void now + return this.repo.put({ ...row.value, status: 'SUSPENDED' }, row.version).value + } + /** Applied only by the envelope-review workflow after a human approval. */ apply(change: { envelope_id: string; action: 'WIDEN' | 'NARROW' | 'SUSPEND' | 'REACTIVATE'; bounds: Record }, approvedBy: string[]): Envelope { const row = this.repo.get(change.envelope_id) @@ -65,7 +77,7 @@ export class EnvelopeService { status: change.action === 'SUSPEND' ? 'SUSPENDED' : 'ACTIVE', approval: { ...env.approval, approved_by: approvedBy }, } - this.streaks.set(env.id, 0) + kvSet(this.streaks, `streak:${env.id}`, '0') return this.repo.put(next, row.version).value } @@ -79,8 +91,8 @@ export class EnvelopeService { e.status === 'ACTIVE' && e.scope.proposal_type === proposal.type && e.scope.timescales.includes(proposal.timescale) && - e.validity.from <= ctx.now && - ctx.now <= e.validity.to, + Date.parse(e.validity.from) <= Date.parse(ctx.now) && + Date.parse(ctx.now) <= Date.parse(e.validity.to), ) if (candidates.length === 0) { return { diff --git a/packages/services/src/gateway.ts b/packages/services/src/gateway.ts index 48784fe..8e538e1 100644 --- a/packages/services/src/gateway.ts +++ b/packages/services/src/gateway.ts @@ -113,7 +113,13 @@ export class SimulationGateway implements GatewayPort { execute(proposalDigest: string, fulfillment: Record, now: string): ExecutionReport[] { const order = this.orders.get(proposalDigest) if (!order || order.proposal.type !== 'DISPATCH_PLAN') throw new Error(`no dispatch order for ${proposalDigest.slice(0, 8)}…`) - const p = order.proposal + return this.executePlan(order.proposal, fulfillment, now) + } + + /** Same, from a persisted proposal (after a restart the in-memory order book is empty; the receipt store is not). */ + executePlan(p: Proposal, fulfillment: Record, now: string): ExecutionReport[] { + if (p.type !== 'DISPATCH_PLAN') throw new Error(`simulation gateway executes DISPATCH_PLAN only, got ${p.type}`) + if (!this.receipts.list().some((r) => r.value.proposal_digest === p.digest)) throw new Error(`no receipt for ${p.digest.slice(0, 8)}…: plan was never released to the simulation gateway`) return p.payload.allocations.map((a) => { const f = fulfillment[a.unit_id] ?? 1 const actual = a.target_mw.values.map((v) => (Number(v) * f).toFixed(3)) diff --git a/packages/services/src/index.ts b/packages/services/src/index.ts index 639ff46..3f1b9de 100644 --- a/packages/services/src/index.ts +++ b/packages/services/src/index.ts @@ -16,3 +16,8 @@ export * from './simulation.js' export * from './casedesk.js' export * from './skillclient.js' export * from './review.js' +export * from './metrics.js' +export * from './counters.js' +export * from './shadow.js' +export * from './breaker.js' +export * from './kpi.js' diff --git a/packages/services/src/kpi.ts b/packages/services/src/kpi.ts new file mode 100644 index 0000000..52175f2 --- /dev/null +++ b/packages/services/src/kpi.ts @@ -0,0 +1,141 @@ +import type { KpiEntry, KpiId, KpiReport, ShadowDayRecord } from '@vpp/domain' +import { Decimal } from './decimal.js' + +/** + * KPI dashboard (docs/12 §4): each proposal indicator as a measurable + * definition over shadow-day records. Targets and statistical levels are + * OPEN-QUESTION C1/C2 — placeholders below are named config, and every entry + * carries its definition text so acceptance can freeze it in writing. + */ +export interface KpiConfig { + /** Rolling window in market days (docs/12 §4: 月滚动均值). */ + windowDays: number + loadMapeTarget: string // ≤ 0.08 + pvNrmseTarget: string // OPEN-QUESTION C2: PV metric and level + potentialAccuracyTarget: string // ≥ 0.90 + /** |planned − delivered| / planned tolerance for a unit-day to count as accurate. OPEN-QUESTION C1. */ + potentialToleranceRatio: string + decisionLatencyTargetMs: number // ≤ 3 min + dispatchSuccessTarget: string // ≥ 0.98 + /** Execution deviation band (% of planned) for "executed successfully". OPEN-QUESTION C1. */ + executionDeviationBandPct: string + revenueUpliftTarget: string // ≥ 0.15 vs frozen human baseline (OPEN-QUESTION C1) + crossRegionTarget: string // ≥ 0.85, phase 2 +} + +export const KPI_CONFIG_PLACEHOLDER: KpiConfig = { + windowDays: 30, + loadMapeTarget: '0.08', + pvNrmseTarget: '0.10', // OPEN-QUESTION C2 + potentialAccuracyTarget: '0.90', + potentialToleranceRatio: '0.10', // OPEN-QUESTION C1 + decisionLatencyTargetMs: 180_000, + dispatchSuccessTarget: '0.98', + executionDeviationBandPct: '10', // OPEN-QUESTION C1 + revenueUpliftTarget: '0.15', // OPEN-QUESTION C1: baseline definition must be frozen + crossRegionTarget: '0.85', +} + +const D = (v: string | number) => new Decimal(v) +const ratio6 = (x: Decimal) => x.toFixed(6) + +function entry(id: KpiId, value: Decimal | null, unit: string, target: string | null, comparator: 'LTE' | 'GTE', samples: number, definition: string, notApplicable = false): KpiEntry { + let status: KpiEntry['status'] = 'NO_DATA' + if (notApplicable) status = 'NOT_APPLICABLE' + else if (value !== null && target !== null) status = (comparator === 'LTE' ? value.lte(target) : value.gte(target)) ? 'MEET' : 'MISS' + return { id, value: value === null ? null : ratio6(value), unit, target, comparator, samples, status, definition } +} + +const meanOf = (xs: Decimal[]): Decimal | null => (xs.length === 0 ? null : xs.reduce((s, v) => s.add(v), D(0)).div(xs.length)) + +/** Nearest-rank percentile on integers. */ +const percentile = (xs: number[], p: number): number | null => { + if (xs.length === 0) return null + const sorted = [...xs].sort((a, b) => a - b) + return sorted[Math.min(sorted.length - 1, Math.max(0, Math.ceil(p * sorted.length) - 1))]! +} + +const nextDate = (d: string) => new Date(new Date(`${d}T00:00:00Z`).getTime() + 86_400_000).toISOString().slice(0, 10) + +export function computeKpiReport(records: ShadowDayRecord[], cfg: KpiConfig, opts: { id: string; now: string }): KpiReport { + const all = [...records].sort((a, b) => (a.market_date < b.market_date ? -1 : 1)) + const win = all.slice(-cfg.windowDays) + + // ---- forecast + const loadMape = meanOf(win.map((r) => r.forecast.load_mape).filter((v): v is string => v !== null).map(D)) + const pvNrmse = meanOf(win.map((r) => r.forecast.pv_nrmse).filter((v): v is string => v !== null).map(D)) + + // ---- potential accuracy: per dispatched day, |planned − delivered| / planned ≤ tolerance + const executed = win.filter((r) => r.execution !== null && D(r.execution.planned_mwh).gt(0)) + const accurate = executed.filter((r) => D(r.execution!.planned_mwh).sub(r.execution!.delivered_mwh).abs().div(r.execution!.planned_mwh).lte(cfg.potentialToleranceRatio)) + const potential = executed.length === 0 ? null : D(accurate.length).div(executed.length) + + // ---- decision latency P95 (human waiting time excluded by construction: the sample ends at PENDING_HUMAN/AUTO_APPROVED) + const latencies = win.map((r) => r.decision_latency_ms).filter((v): v is number => v !== null) + const p95 = percentile(latencies, 0.95) + + // ---- dispatch execution success: permitted dispatch orders executed within the deviation band + const dispatched = win.filter((r) => r.dispatch !== null && r.dispatch.outcome === 'RELEASED') + const succeeded = dispatched.filter((r) => r.execution !== null && r.execution.within_band) + const dispatchSuccess = dispatched.length === 0 ? null : D(succeeded.length).div(dispatched.length) + + // ---- revenue uplift vs human baseline, over days where both lines exist + const paired = win.filter((r) => r.shadow.line !== null && r.human !== null) + const shadowPaired = paired.reduce((s, r) => s.add(r.shadow.line!.realised_revenue_yuan), D(0)) + const humanPaired = paired.reduce((s, r) => s.add(r.human!.line.realised_revenue_yuan), D(0)) + const uplift = paired.length === 0 || humanPaired.eq(0) ? null : shadowPaired.sub(humanPaired).div(humanPaired.abs()) + + const kpis: KpiEntry[] = [ + entry('FORECAST_LOAD_MAPE', loadMape, '1', cfg.loadMapeTarget, 'LTE', win.filter((r) => r.forecast.load_mape !== null).length, `aggregate day-ahead 96-interval load MAPE (P50 vs metered), mean over the last ${cfg.windowDays} shadow days (docs/12 §4; level OPEN-QUESTION C2)`), + entry('FORECAST_PV_NRMSE', pvNrmse, '1', cfg.pvNrmseTarget, 'LTE', win.filter((r) => r.forecast.pv_nrmse !== null).length, 'PV P50 RMSE normalised by installed capacity, window mean (OPEN-QUESTION C2: metric and level)'), + entry('POTENTIAL_ACCURACY', potential, '1', cfg.potentialAccuracyTarget, 'GTE', executed.length, `share of dispatched days with |planned − delivered| / planned ≤ ${cfg.potentialToleranceRatio} (docs/12 §4; denominator choice OPEN-QUESTION C1)`), + entry('DECISION_LATENCY_P95_MS', p95 === null ? null : D(p95), 'ms', String(cfg.decisionLatencyTargetMs), 'LTE', latencies.length, 'P95 of trigger → proposal reaches AUTO_APPROVED or PENDING_HUMAN; human approval waiting time excluded (docs/12 §4)'), + entry('DISPATCH_SUCCESS_RATE', dispatchSuccess, '1', cfg.dispatchSuccessTarget, 'GTE', dispatched.length, `share of permitted dispatch orders confirmed by execution report with deviation ≤ ${cfg.executionDeviationBandPct}% of plan (docs/12 §4)`), + entry('REVENUE_UPLIFT_VS_HUMAN', uplift, '1', cfg.revenueUpliftTarget, 'GTE', paired.length, '(Σ shadow realised revenue − Σ human realised revenue) / Σ human, both cleared at actual prices on the same days (docs/12 §4; baseline to be frozen, OPEN-QUESTION C1)'), + entry('CROSS_REGION_MATCH', null, '1', cfg.crossRegionTarget, 'GTE', 0, 'federation commitments vs delivery confirmations (docs/10 §3) — phase 2', true), + ] + + // ---- three-line totals over the whole shadow history + const sum = (pick: (r: ShadowDayRecord) => string | null) => all.reduce((s, r) => { const v = pick(r); return v === null ? s : s.add(v) }, D(0)) + const shadow = sum((r) => r.shadow.line?.realised_revenue_yuan ?? null) + const human = sum((r) => r.human?.line.realised_revenue_yuan ?? null) + const hindsight = sum((r) => r.hindsight.realised_revenue_yuan) + const naive = sum((r) => r.naive.realised_revenue_yuan) + const humanDays = all.filter((r) => r.human !== null).length + + // ---- shadow-run status: consecutive complete days counted from the latest day backwards + let consecutive = 0 + for (let i = all.length - 1; i >= 0; i--) { + const r = all[i]! + if (!r.lineage_complete) break + if (i < all.length - 1 && nextDate(r.market_date) !== all[i + 1]!.market_date) break + consecutive++ + } + + return { + id: opts.id, + window: { from: win[0]?.market_date ?? null, to: win.at(-1)?.market_date ?? null, days: win.length }, + kpis, + comparison: { + shadow_yuan: shadow.toFixed(2), + human_yuan: humanDays === 0 ? null : human.toFixed(2), + hindsight_yuan: hindsight.toFixed(2), + naive_yuan: naive.toFixed(2), + capture_ratio: hindsight.eq(0) ? null : ratio6(shadow.div(hindsight)), + uplift_vs_naive: naive.eq(0) ? null : ratio6(shadow.div(naive)), + days_with_human_baseline: humanDays, + }, + shadow: { + days: all.length, + complete_days: all.filter((r) => r.lineage_complete).length, + consecutive_complete_days: consecutive, + first_date: all[0]?.market_date ?? null, + last_date: all.at(-1)?.market_date ?? null, + released_days: all.filter((r) => r.shadow.outcome === 'RELEASED').length, + pending_days: all.filter((r) => r.shadow.outcome === 'PENDING_HUMAN').length, + widen_recommendations: all.filter((r) => r.envelope_recommendations.some((e) => e.action === 'WIDEN')).length, + breaker_trips: all.filter((r) => r.breakers_tripped.length > 0).length, + }, + generated_at: opts.now, + } +} diff --git a/packages/services/src/ledger.ts b/packages/services/src/ledger.ts index 70f1d70..e4c373f 100644 --- a/packages/services/src/ledger.ts +++ b/packages/services/src/ledger.ts @@ -1,3 +1,5 @@ +import { appendFileSync, existsSync, mkdirSync, readFileSync } from 'node:fs' +import { dirname } from 'node:path' import { Decimal } from './decimal.js' import type { LedgerView, PositionBounds, PositionEntry, PositionUpdate, Timescale } from '@vpp/domain' import { PositionUpdate as PositionUpdateSchema } from '@vpp/domain' @@ -25,6 +27,12 @@ export interface LedgerConfig { */ daMonthlyDeviationBand: string // e.g. "0.05" = ±5% clock?: () => string + /** + * JSONL file holding every accepted entry in order. When given, the ledger + * restores itself on construction so a shadow run's positions survive a + * process restart (the Postgres adapter replaces this later). + */ + path?: string } /** @@ -36,7 +44,16 @@ export class LedgerService { private version = 0 private entries: PositionEntry[] = [] - constructor(private readonly cfg: LedgerConfig) {} + constructor(private readonly cfg: LedgerConfig) { + if (cfg.path) { + mkdirSync(dirname(cfg.path), { recursive: true }) + if (existsSync(cfg.path)) { + const lines = readFileSync(cfg.path, 'utf8').split('\n').filter(Boolean) + this.entries = lines.map((l) => JSON.parse(l) as PositionEntry) + this.version = this.entries.length + } + } + } read(): LedgerView { return { version: this.version, entries: [...this.entries] } @@ -74,6 +91,7 @@ export class LedgerService { ...rest, recorded_at: this.cfg.clock?.() ?? new Date().toISOString(), } + if (this.cfg.path) appendFileSync(this.cfg.path, JSON.stringify(entry) + '\n') this.entries.push(entry) this.version += 1 return this.read() diff --git a/packages/services/src/metrics.ts b/packages/services/src/metrics.ts new file mode 100644 index 0000000..69aa940 --- /dev/null +++ b/packages/services/src/metrics.ts @@ -0,0 +1,103 @@ +/** + * L2 skill metrics (docs/12 §1 L2, §4 口径). Pure functions over plain numbers; + * the harness converts decimal strings at the boundary and quantizes results + * when it writes a report. KPI thresholds (≤ 8% etc.) are judged on real data + * and are OPEN-QUESTION C2 in their exact statistical level — nothing here + * hardcodes a pass/fail number. + */ + +export const mean = (xs: number[]): number => + xs.length === 0 ? Number.NaN : xs.reduce((a, b) => a + b, 0) / xs.length + +/** Mean absolute percentage error over intervals where actual ≠ 0. */ +export function mape(actual: number[], pred: number[]): number { + const terms: number[] = [] + for (let i = 0; i < actual.length; i++) { + const a = actual[i]! + if (a !== 0) terms.push(Math.abs((pred[i]! - a) / a)) + } + return mean(terms) +} + +/** RMSE normalised by installed capacity — the PV metric docs/12 §4 suggests. */ +export function nrmse(actual: number[], pred: number[], capacity: number): number { + const se = actual.map((a, i) => (pred[i]! - a) ** 2) + return Math.sqrt(mean(se)) / capacity +} + +/** Share of actuals inside [lower, upper]; nominal for a P10–P90 band is 0.80. */ +export function coverage(actual: number[], lower: number[], upper: number[]): number { + let hits = 0 + for (let i = 0; i < actual.length; i++) { + if (actual[i]! >= lower[i]! && actual[i]! <= upper[i]!) hits++ + } + return hits / actual.length +} + +/** + * Direction accuracy (docs/12 L2 电价 方向准确率): share of interval pairs whose + * high/low ordering the forecast gets right. Bid optimisation depends on the + * ranking of intervals far more than on absolute price level. + */ +export function directionAccuracy(actual: number[], pred: number[]): number { + let agree = 0 + let pairs = 0 + for (let i = 0; i < actual.length; i++) { + for (let j = i + 1; j < actual.length; j++) { + const da = actual[j]! - actual[i]! + const dp = pred[j]! - pred[i]! + if (da === 0 && dp === 0) continue + pairs++ + if (Math.sign(da) === Math.sign(dp)) agree++ + } + } + return pairs === 0 ? Number.NaN : agree / pairs +} + +export interface Bid { + offers: number[] + quantities: number[] +} + +/** Uniform-price clearing: an offer clears where it does not exceed the realised price. */ +export function realisedRevenue(bid: Bid, actualPrice: number[]): number { + let total = 0 + for (let t = 0; t < actualPrice.length; t++) { + if (bid.offers[t]! <= actualPrice[t]!) total += actualPrice[t]! * bid.quantities[t]! + } + return total +} + +/** + * Lower-bound baseline: a price-taker offering a flat profile that meets the + * upper energy bound (or as much as capacity allows), pro-rata to capacity. + */ +export function naiveBid(capMwh: number[], energyMax: number): Bid { + const sellable = capMwh.reduce((a, b) => a + b, 0) + const scale = sellable === 0 ? 0 : Math.min(1, energyMax / sellable) + return { offers: capMwh.map(() => 0), quantities: capMwh.map((c) => c * scale) } +} + +/** + * Upper-bound baseline: perfect hindsight — fill the highest-priced intervals + * first up to the energy bound, honouring per-interval capacity and block + * size. Offers at zero so everything clears. + */ +export function hindsightBid( + actualPrice: number[], + capMwh: number[], + energyMax: number, + minBlock: number, +): Bid { + const order = actualPrice.map((_, i) => i).sort((a, b) => actualPrice[b]! - actualPrice[a]!) + const quantities = capMwh.map(() => 0) + let room = energyMax + for (const t of order) { + const q = Math.min(capMwh[t]!, room) + if (q < minBlock) continue + quantities[t] = q + room -= q + if (room <= 0) break + } + return { offers: capMwh.map(() => 0), quantities } +} diff --git a/packages/services/src/review.ts b/packages/services/src/review.ts index 71c3072..a61754d 100644 --- a/packages/services/src/review.ts +++ b/packages/services/src/review.ts @@ -13,6 +13,8 @@ import type { Writeback, } from '@vpp/domain' import { SemanticMemoryEntry as SemanticMemoryEntrySchema } from '@vpp/domain' +import { createKeyValueStore, kvGet, kvSet } from './counters.js' +import type { KeyValueStore } from './counters.js' import { Decimal } from './decimal.js' import type { EnvelopeService } from './envelope.js' import { MemoryRepository } from './relational.js' @@ -72,13 +74,13 @@ const mape = (actual: string[], pred: string[]): string | null => { } export class ReviewService { - private readonly compliantDays = new Map() - constructor( private readonly resources: ResourceRegistry, private readonly envelopes: EnvelopeService, readonly memory: Repository = new MemoryRepository(SemanticMemoryEntrySchema, (m) => m.id), private readonly cfg: ReviewConfig = DEFAULT_REVIEW_CONFIG, + /** Compliant-day streak per envelope; persisted so a 20-day widening streak survives restarts. */ + private readonly compliantDays: KeyValueStore = createKeyValueStore(), ) {} /** Plan-vs-actual per timescale → attribution → finding with writebacks (not yet applied). */ @@ -127,8 +129,8 @@ export class ReviewService { const env = this.envelopes.get(u.envelope_id) if (!env) continue const compliant = D(u.deviation_pct).lte(env.escalation.deviation_threshold_pct) - const days = compliant ? (this.compliantDays.get(env.id) ?? 0) + 1 : 0 - this.compliantDays.set(env.id, days) + const days = compliant ? Number(kvGet(this.compliantDays, `compliant:${env.id}`) || 0) + 1 : 0 + kvSet(this.compliantDays, `compliant:${env.id}`, String(days)) writebacks.push(this.envelopeRecommendation(env, days, compliant)) } // ---- FORECAST lesson diff --git a/packages/services/src/shadow.ts b/packages/services/src/shadow.ts new file mode 100644 index 0000000..082f8e1 --- /dev/null +++ b/packages/services/src/shadow.ts @@ -0,0 +1,113 @@ +import type { AwardNotice, BidLine, Curve96, ExecutionPermit, ExecutionReceipt, Proposal } from '@vpp/domain' +import { ExecutionReceipt as ReceiptSchema } from '@vpp/domain' +import type { AuthorityService } from './authority.js' +import { Decimal } from './decimal.js' +import { digestMatches } from './digest.js' +import { GatewayRejected } from './gateway.js' +import type { GatewayPort } from './gateway.js' +import { hindsightBid, naiveBid, realisedRevenue } from './metrics.js' +import { MemoryRepository } from './relational.js' +import type { Repository } from './relational.js' +import type { SnapshotStore } from './snapshot.js' + +/** + * Shadow run (docs/08 §4 M5, docs/12 §1 L4): the full loop runs, but the bid + * is generated and recorded — never submitted — and the market's answer is + * simulated from the actual day-ahead clearing price. Everything here is + * arithmetic on recorded objects; no LLM, no live effect. + */ + +/** Bid gateway for shadow mode: accepts (Proposal, Permit) like any channel, records a receipt, submits nothing. */ +export class ShadowBidGateway implements GatewayPort { + constructor( + private readonly authority: AuthorityService, + readonly receipts: Repository = new MemoryRepository(ReceiptSchema, (r) => r.idempotency_key), + ) {} + + dispatch(proposal: Proposal, permit: ExecutionPermit, now: string): ExecutionReceipt { + const key = `${proposal.digest}:${permit.id}` + const existing = this.receipts.get(key) + if (existing) return existing.value + const reasons = this.authority.validate(permit, proposal, now) + if (reasons.length > 0) throw new GatewayRejected(reasons) + if (proposal.type !== 'BID') throw new GatewayRejected([`shadow bid gateway accepts BID only, got ${proposal.type}`]) + const receipt: ExecutionReceipt = { + receipt_id: `rcpt-${proposal.id}-${permit.id}`, + proposal_digest: proposal.digest, + permit_id: permit.id, + channel: 'SHADOW', + idempotency_key: key, + artifact_ref: `shadow://bid/${proposal.digest}`, + accepted_at: now, + } + this.receipts.put(receipt) + return receipt + } +} + +/** + * Simulated clearing of a shadow bid against the day's actual day-ahead price + * (uniform-price rule, same as the L2 harness): an interval clears when the + * offer does not exceed the clearing price. The award answers the bid digest so + * the award-decomposition flow treats it exactly like a real notice. + */ +export function shadowClearing(bid: Proposal, clearingPrice: Curve96, opts: { id: string; now: string }): AwardNotice { + if (bid.type !== 'BID') throw new Error(`shadow clearing needs a BID, got ${bid.type}`) + if (clearingPrice.date !== bid.payload.market_date) throw new Error(`clearing price dated ${clearingPrice.date}, bid is for ${bid.payload.market_date}`) + const awarded = bid.payload.quantities_mwh.values.map((q, t) => + new Decimal(bid.payload.prices_yuan_per_mwh.values[t]!).lte(clearingPrice.values[t]!) ? q : '0', + ) + return { + id: opts.id, + market_date: bid.payload.market_date, + bid_proposal_digest: bid.digest, + awarded_mwh: { interval_minutes: 15, date: bid.payload.market_date, values: awarded }, + clearing_price_yuan_per_mwh: clearingPrice, + received_at: opts.now, + } +} + +const nums = (v: string[]) => v.map(Number) +const money = (x: number) => x.toFixed(2) +const mwh = (x: number) => x.toFixed(3) + +/** Score a bid (offers/quantities) at the actual price: energy offered, energy cleared, revenue. */ +export function bidLine(offers: string[], quantities: string[], actualPrice: string[]): BidLine { + const o = nums(offers) + const q = nums(quantities) + const p = nums(actualPrice) + const cleared = q.reduce((s, v, t) => s + (o[t]! <= p[t]! ? v : 0), 0) + return { + energy_mwh: mwh(q.reduce((s, v) => s + v, 0)), + cleared_energy_mwh: mwh(cleared), + realised_revenue_yuan: money(realisedRevenue({ offers: o, quantities: q }, p)), + } +} + +/** Lower bound: price-taker flat profile pro-rata to capacity (docs/12 §1 L2 朴素策略). */ +export function naiveLine(capMwh: string[], energyMax: string, actualPrice: string[]): BidLine { + const b = naiveBid(nums(capMwh), Number(energyMax)) + return bidLine(b.offers.map(String), b.quantities.map((x) => x.toFixed(6)), actualPrice) +} + +/** Upper bound: perfect hindsight (docs/12 §1 L2 完美后见之明). */ +export function hindsightLine(capMwh: string[], energyMax: string, minBlock: string, actualPrice: string[]): BidLine { + const b = hindsightBid(nums(actualPrice), nums(capMwh), Number(energyMax), Number(minBlock)) + return bidLine(b.offers.map(String), b.quantities.map((x) => x.toFixed(6)), actualPrice) +} + +/** + * Lineage completeness for one proposal (I7): digest matches, every tool call's + * input/output snapshot is present, every data ref resolves. Returns gaps. + */ +export function auditProposalLineage(p: Proposal, snapshots: SnapshotStore): string[] { + const gaps: string[] = [] + if (!digestMatches(p)) gaps.push(`${p.id}: digest does not match payload + lineage`) + if (p.lineage.tool_calls.length === 0) gaps.push(`${p.id}: no tool calls in lineage`) + for (const tc of p.lineage.tool_calls) { + if (!snapshots.has(tc.inputs_ref)) gaps.push(`${p.id}: ${tc.tool_call_id} input snapshot missing`) + if (!snapshots.has(tc.outputs_ref)) gaps.push(`${p.id}: ${tc.tool_call_id} output snapshot missing`) + } + for (const ref of p.lineage.data_refs) if (!snapshots.has(ref)) gaps.push(`${p.id}: data ref ${ref.slice(0, 8)}… missing`) + return gaps +} diff --git a/packages/services/src/timeseries.ts b/packages/services/src/timeseries.ts index 6d21062..bff9034 100644 --- a/packages/services/src/timeseries.ts +++ b/packages/services/src/timeseries.ts @@ -1,3 +1,5 @@ +import { appendFileSync, existsSync, mkdirSync, readFileSync } from 'node:fs' +import { dirname } from 'node:path' import { Curve96 as Curve96Schema, MarketDate } from '@vpp/domain' import type { Curve96 } from '@vpp/domain' import { canonicalJson } from './snapshot.js' @@ -60,9 +62,18 @@ export class MemoryTimeSeriesStore implements TimeSeriesStore { recorded_at: this.clock(), } this.byKey.set(key, [...history, record]) + this.persist(record) return record } + protected persist(_record: CurveRecord): void {} + + /** Load a persisted revision verbatim (versions preserved, in file order). */ + protected restore(record: CurveRecord): void { + const key = keyOf(record.series_id, record.date) + this.byKey.set(key, [...(this.byKey.get(key) ?? []), record]) + } + latest(series_id: string, date: string): CurveRecord | undefined { const history = this.byKey.get(keyOf(series_id, date)) return history?.[history.length - 1] @@ -94,3 +105,21 @@ export class MemoryTimeSeriesStore implements TimeSeriesStore { return [...ids].sort() } } + +/** JSONL-backed time-series store: every accepted revision is appended before it is served. */ +export class FsTimeSeriesStore extends MemoryTimeSeriesStore { + constructor( + private readonly path: string, + clock?: () => string, + ) { + super(clock) + mkdirSync(dirname(path), { recursive: true }) + if (existsSync(path)) { + for (const line of readFileSync(path, 'utf8').split('\n').filter(Boolean)) this.restore(JSON.parse(line) as CurveRecord) + } + } + + protected override persist(record: CurveRecord): void { + appendFileSync(this.path, JSON.stringify(record) + '\n') + } +} diff --git a/skills-py/vpp_contracts/breaker_record.py b/skills-py/vpp_contracts/breaker_record.py new file mode 100644 index 0000000..33ca98a --- /dev/null +++ b/skills-py/vpp_contracts/breaker_record.py @@ -0,0 +1,54 @@ +# generated by datamodel-codegen: +# filename: breaker_record.json + +from __future__ import annotations + +from enum import StrEnum + +from pydantic import AwareDatetime, BaseModel, ConfigDict, conint, constr + + +class Level(StrEnum): + L0 = 'L0' + L1 = 'L1' + L2 = 'L2' + L3 = 'L3' + L4 = 'L4' + + +class Status(StrEnum): + ARMED = 'ARMED' + TRIPPED = 'TRIPPED' + + +class TrippedBy(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + id: constr(min_length=1) + role: constr(min_length=1) + + +class ResetBy(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + id: constr(min_length=1) + role: constr(min_length=1) + + +class BreakerRecord(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + level: Level + status: Status + scope: str | None + reason: str | None + tripped_at: AwareDatetime | None + tripped_by: TrippedBy | None + reset_at: AwareDatetime | None + reset_by: ResetBy | None + reset_basis: str | None + drill: bool + trip_count: conint(ge=0, le=9007199254740991) diff --git a/skills-py/vpp_contracts/execution_receipt.py b/skills-py/vpp_contracts/execution_receipt.py index f91b9fc..6c79d61 100644 --- a/skills-py/vpp_contracts/execution_receipt.py +++ b/skills-py/vpp_contracts/execution_receipt.py @@ -12,6 +12,7 @@ class Channel(StrEnum): FILE_EXPORT = 'FILE_EXPORT' TRADING_PLATFORM_API = 'TRADING_PLATFORM_API' SIMULATION_GATEWAY = 'SIMULATION_GATEWAY' + SHADOW = 'SHADOW' class ExecutionReceipt(BaseModel): diff --git a/skills-py/vpp_contracts/human_bid_record.py b/skills-py/vpp_contracts/human_bid_record.py new file mode 100644 index 0000000..f25e432 --- /dev/null +++ b/skills-py/vpp_contracts/human_bid_record.py @@ -0,0 +1,49 @@ +# generated by datamodel-codegen: +# filename: human_bid_record.json + +from __future__ import annotations + +from enum import StrEnum +from typing import Literal + +from pydantic import AwareDatetime, BaseModel, ConfigDict, Field, RootModel, constr + + +class Value(RootModel[constr(pattern=r'^-?\d+(\.\d+)?$')]): + root: constr(pattern=r'^-?\d+(\.\d+)?$') + + +class PricesYuanPerMwh(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + interval_minutes: Literal[15] + date: constr(pattern=r'^\d{4}-\d{2}-\d{2}$') + values: list[Value] = Field(..., max_length=96, min_length=96) + + +class QuantitiesMwh(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + interval_minutes: Literal[15] + date: constr(pattern=r'^\d{4}-\d{2}-\d{2}$') + values: list[Value] = Field(..., max_length=96, min_length=96) + + +class Source(StrEnum): + TRADING_PLATFORM_EXPORT = 'TRADING_PLATFORM_EXPORT' + MANUAL_ENTRY = 'MANUAL_ENTRY' + SYNTHETIC_NAIVE = 'SYNTHETIC_NAIVE' + + +class HumanBidRecord(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + id: constr(min_length=1) + market_date: constr(pattern=r'^\d{4}-\d{2}-\d{2}$') + prices_yuan_per_mwh: PricesYuanPerMwh + quantities_mwh: QuantitiesMwh + source: Source + received_at: AwareDatetime diff --git a/skills-py/vpp_contracts/insight_card.py b/skills-py/vpp_contracts/insight_card.py index bd6c929..6e55d4b 100644 --- a/skills-py/vpp_contracts/insight_card.py +++ b/skills-py/vpp_contracts/insight_card.py @@ -37,6 +37,9 @@ class Kind(StrEnum): REVIEW_FINDING = 'REVIEW_FINDING' DECISION_CASE = 'DECISION_CASE' POTENTIAL_ASSESSMENT = 'POTENTIAL_ASSESSMENT' + KPI_REPORT = 'KPI_REPORT' + SHADOW_DAY = 'SHADOW_DAY' + BREAKER = 'BREAKER' class Source(BaseModel): diff --git a/skills-py/vpp_contracts/kpi_report.py b/skills-py/vpp_contracts/kpi_report.py new file mode 100644 index 0000000..e3bfa25 --- /dev/null +++ b/skills-py/vpp_contracts/kpi_report.py @@ -0,0 +1,93 @@ +# generated by datamodel-codegen: +# filename: kpi_report.json + +from __future__ import annotations + +from enum import StrEnum + +from pydantic import AwareDatetime, BaseModel, ConfigDict, Field, conint, constr + + +class Window(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + from_: constr(pattern=r'^\d{4}-\d{2}-\d{2}$') | None = Field(..., alias='from') + to: constr(pattern=r'^\d{4}-\d{2}-\d{2}$') | None + days: conint(ge=0, le=9007199254740991) + + +class Id(StrEnum): + FORECAST_LOAD_MAPE = 'FORECAST_LOAD_MAPE' + FORECAST_PV_NRMSE = 'FORECAST_PV_NRMSE' + POTENTIAL_ACCURACY = 'POTENTIAL_ACCURACY' + DECISION_LATENCY_P95_MS = 'DECISION_LATENCY_P95_MS' + DISPATCH_SUCCESS_RATE = 'DISPATCH_SUCCESS_RATE' + REVENUE_UPLIFT_VS_HUMAN = 'REVENUE_UPLIFT_VS_HUMAN' + CROSS_REGION_MATCH = 'CROSS_REGION_MATCH' + + +class Comparator(StrEnum): + LTE = 'LTE' + GTE = 'GTE' + + +class Status(StrEnum): + MEET = 'MEET' + MISS = 'MISS' + NO_DATA = 'NO_DATA' + NOT_APPLICABLE = 'NOT_APPLICABLE' + + +class Kpi(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + id: Id + value: constr(pattern=r'^-?\d+(\.\d+)?$') | None + unit: constr(min_length=1) + target: constr(pattern=r'^-?\d+(\.\d+)?$') | None + comparator: Comparator + samples: conint(ge=0, le=9007199254740991) + status: Status + definition: constr(min_length=1) + + +class Comparison(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + shadow_yuan: constr(pattern=r'^-?\d+(\.\d+)?$') + human_yuan: constr(pattern=r'^-?\d+(\.\d+)?$') | None + hindsight_yuan: constr(pattern=r'^-?\d+(\.\d+)?$') + naive_yuan: constr(pattern=r'^-?\d+(\.\d+)?$') + capture_ratio: constr(pattern=r'^-?\d+(\.\d+)?$') | None + uplift_vs_naive: constr(pattern=r'^-?\d+(\.\d+)?$') | None + days_with_human_baseline: conint(ge=0, le=9007199254740991) + + +class Shadow(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + days: conint(ge=0, le=9007199254740991) + complete_days: conint(ge=0, le=9007199254740991) + consecutive_complete_days: conint(ge=0, le=9007199254740991) + first_date: constr(pattern=r'^\d{4}-\d{2}-\d{2}$') | None + last_date: constr(pattern=r'^\d{4}-\d{2}-\d{2}$') | None + released_days: conint(ge=0, le=9007199254740991) + pending_days: conint(ge=0, le=9007199254740991) + widen_recommendations: conint(ge=0, le=9007199254740991) + breaker_trips: conint(ge=0, le=9007199254740991) + + +class KpiReport(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + id: constr(min_length=1) + window: Window + kpis: list[Kpi] + comparison: Comparison + shadow: Shadow + generated_at: AwareDatetime diff --git a/skills-py/vpp_contracts/shadow_day_record.py b/skills-py/vpp_contracts/shadow_day_record.py new file mode 100644 index 0000000..562f1bb --- /dev/null +++ b/skills-py/vpp_contracts/shadow_day_record.py @@ -0,0 +1,148 @@ +# generated by datamodel-codegen: +# filename: shadow_day_record.json + +from __future__ import annotations + +from enum import StrEnum + +from pydantic import AwareDatetime, BaseModel, ConfigDict, conint, constr + + +class Line(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + energy_mwh: constr(pattern=r'^-?\d+(\.\d+)?$') + cleared_energy_mwh: constr(pattern=r'^-?\d+(\.\d+)?$') + realised_revenue_yuan: constr(pattern=r'^-?\d+(\.\d+)?$') + + +class Shadow(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + proposal_id: constr(min_length=1) | None + digest: constr(pattern=r'^[0-9a-f]{64}$') | None + outcome: constr(min_length=1) + expected_revenue_yuan: constr(pattern=r'^-?\d+(\.\d+)?$') | None + line: Line | None + llm_used: bool + + +class Source(StrEnum): + TRADING_PLATFORM_EXPORT = 'TRADING_PLATFORM_EXPORT' + MANUAL_ENTRY = 'MANUAL_ENTRY' + SYNTHETIC_NAIVE = 'SYNTHETIC_NAIVE' + + +class Human(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + record_id: constr(min_length=1) + source: Source + line: Line + + +class Hindsight(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + energy_mwh: constr(pattern=r'^-?\d+(\.\d+)?$') + cleared_energy_mwh: constr(pattern=r'^-?\d+(\.\d+)?$') + realised_revenue_yuan: constr(pattern=r'^-?\d+(\.\d+)?$') + + +class Naive(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + energy_mwh: constr(pattern=r'^-?\d+(\.\d+)?$') + cleared_energy_mwh: constr(pattern=r'^-?\d+(\.\d+)?$') + realised_revenue_yuan: constr(pattern=r'^-?\d+(\.\d+)?$') + + +class Award(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + id: constr(min_length=1) + energy_mwh: constr(pattern=r'^-?\d+(\.\d+)?$') + + +class Dispatch(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + proposal_id: constr(min_length=1) + outcome: constr(min_length=1) + shortfall_mwh: constr(pattern=r'^-?\d+(\.\d+)?$') + + +class Execution(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + planned_mwh: constr(pattern=r'^-?\d+(\.\d+)?$') + delivered_mwh: constr(pattern=r'^-?\d+(\.\d+)?$') + deviation_mwh: constr(pattern=r'^-?\d+(\.\d+)?$') + fulfillment_ratio: constr(pattern=r'^-?\d+(\.\d+)?$') + within_band: bool + simulated: bool + + +class Forecast(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + load_mape: constr(pattern=r'^-?\d+(\.\d+)?$') | None + pv_nrmse: constr(pattern=r'^-?\d+(\.\d+)?$') | None + price_mape: constr(pattern=r'^-?\d+(\.\d+)?$') | None + price_coverage_p10_p90: constr(pattern=r'^-?\d+(\.\d+)?$') | None + + +class Action(StrEnum): + WIDEN = 'WIDEN' + NARROW = 'NARROW' + SUSPEND = 'SUSPEND' + KEEP = 'KEEP' + + +class EnvelopeRecommendation(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + envelope_id: constr(min_length=1) + action: Action + + +class BreakersTrippedEnum(StrEnum): + L0 = 'L0' + L1 = 'L1' + L2 = 'L2' + L3 = 'L3' + L4 = 'L4' + + +class ShadowDayRecord(BaseModel): + model_config = ConfigDict( + extra='forbid', + ) + id: constr(min_length=1) + market_date: constr(pattern=r'^\d{4}-\d{2}-\d{2}$') + shadow: Shadow + human: Human | None + hindsight: Hindsight + naive: Naive + award: Award | None + dispatch: Dispatch | None + execution: Execution | None + forecast: Forecast + decision_latency_ms: conint(ge=0, le=9007199254740991) | None + review_finding_id: constr(min_length=1) | None + envelope_recommendations: list[EnvelopeRecommendation] + breakers_tripped: list[BreakersTrippedEnum] + abnormal_day: bool + lineage_complete: bool + lineage_gaps: list[str] + generated_at: AwareDatetime