vpp-ai-platform/packages/domain/scripts/make-fixtures.ts

754 lines
24 KiB
TypeScript
Raw Normal View History

/**
* Generates golden fixtures into contracts/fixtures/<name>/*.json and invalid
* fixtures into contracts/fixtures-invalid/<name>/*.json. Run once and commit;
* fixtures are reviewed via diff. Invalid cases are crafted to fail on BOTH the
* zod and pydantic sides (dual-side contract test, docs/11 §3.2).
*/
import { mkdirSync, writeFileSync } from 'node:fs'
import { fileURLToPath } from 'node:url'
import { schemaRegistry } from '../src/index.js'
const root = fileURLToPath(new URL('../../../contracts/', import.meta.url))
const REF_A = 'a'.repeat(64)
const REF_B = 'b'.repeat(64)
const REF_C = 'c'.repeat(64)
const DATE = '2026-03-15'
const T0 = '2026-03-14T06:00:00Z'
const T1 = '2026-03-14T08:30:00Z'
const T2 = '2026-03-15T23:59:59Z'
const curve = (v: string, date = DATE) => ({
interval_minutes: 15,
date,
values: Array.from({ length: 96 }, () => v),
})
const valid: Record<string, Record<string, unknown>> = {
proposal: {
'bid-basic': {
id: 'prop-001',
type: 'BID',
timescale: 'DAY_AHEAD',
originator: { agent: 'trading-agent', trigger: 'SCHEDULED', task_id: 'task-001' },
payload: {
kind: 'BID',
market_date: DATE,
prices_yuan_per_mwh: curve('425.50'),
quantities_mwh: curve('12.5'),
expected_revenue_yuan: '510600.00',
},
lineage: {
tool_calls: [
{
tool_call_id: 'tc-001',
tool: 'bid-optimization-milp',
version: '1.0.0',
inputs_ref: REF_A,
outputs_ref: REF_B,
},
],
data_refs: [REF_C],
ledger_version: 42,
policy_pack_version: '2026.03',
},
envelope_ref: 'env-bid-001',
digest: REF_B,
status: 'DRAFT',
created_at: T1,
},
M4: resource agent, envelopes live, review loop, insight cards - packages/domain: AwardNotice, DISPATCH_PLAN proposal payload, ExecutionReport, MeteringRecord, potential-assessment and dispatch-optimization contracts, ReviewFinding (+ typed writebacks), SemanticMemoryEntry, EnvelopeChangeRequest, InsightCard — exported with fixtures on both sides. - skills-py: potential-assessment (certified × rolling fulfilment, evidence days) and dispatch-optimization (per-interval LP on HiGHS, shortfall reported) skills + routes + tests. - packages/services: dispatch rules in the policy pack (over-allocation, award anchor, lineage integrity for allocations); PowerBalanceSimulator; SimulationGateway (permit-only, idempotent, seeded execute → ExecutionReports); envelope deviation-streak suspension + apply(); ReviewService (attribution, reliability EWMA writeback, semantic memory, envelope recommendations as change requests); dispatch assembler; skill client methods. - packages/runtime: resource agent; award-decomposition, review and envelope-review workflows; lifecycle selects simulator/gateway by proposal type; trigger hooks for awards, execution reports, metering; decide() resumes either lifecycle or envelope-review runs; insight cards API. - Tests: docs/07 D-1 16:00 and D+1 end to end; reliability score 0.9 → 0.880 and the next assessment de-rates capacity; envelope suspension on a seeded 3-day streak; WIDEN request applied only by a human. 184 TS + 80 Python. - docs/open-questions: B10 (reliability/potential parameters). README and CLAUDE.md status → M4 done, M5 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 19:31:12 -04:00
'dispatch-plan': {
id: 'prop-002',
type: 'DISPATCH_PLAN',
timescale: 'DAY_AHEAD',
originator: { agent: 'resource-agent', trigger: 'EVENT', task_id: 'task-002' },
payload: {
kind: 'DISPATCH_PLAN',
market_date: DATE,
award_ref: 'award-2026-03-15',
total_target_mw: curve('48.0'),
allocations: [{ unit_id: 'agg-unit-wuhan-01', target_mw: curve('48.0') }],
},
lineage: {
tool_calls: [
{ tool_call_id: 'tc-002', tool: 'dispatch-optimization', version: '1.0.0', inputs_ref: REF_A, outputs_ref: REF_C },
],
data_refs: [REF_C],
ledger_version: 43,
policy_pack_version: '2026.03',
},
envelope_ref: null,
digest: REF_C,
status: 'DRAFT',
created_at: T1,
},
},
approval: {
approve: {
id: 'appr-001',
proposal_digest: REF_B,
decision: 'APPROVE',
approver: { id: 'user-trader-01', role: 'senior-trader' },
scope: { effect_type: 'BID', limits: { max_energy_mwh: '1500.0' } },
validity: { from: T1, to: T2 },
evidence_versions: {
policy_pack_version: '2026.03',
ledger_version: 42,
data_snapshot_refs: [REF_C],
},
comment: 'within monthly position band',
decided_at: T1,
},
},
execution_permit: {
active: {
id: 'permit-001',
proposal_digest: REF_B,
issued_at: T1,
expires_at: T2,
effect_limits: { max_energy_mwh: '1500.0' },
revoked_at: null,
issuer: 'authority-service',
},
},
envelope: {
'bid-envelope': {
id: 'env-bid-001',
scope: { proposal_type: 'BID', timescales: ['DAY_AHEAD'], resource_set: 'pool-hubei-01' },
bounds: { price_deviation_pct: '5.0', max_energy_mwh: '1500.0' },
validity: { from: T0, to: T2 },
approval: { level: 'L2', approved_by: ['user-ops-lead', 'user-risk-officer'] },
escalation: { max_consecutive_deviations: 3, deviation_threshold_pct: '10.0' },
status: 'ACTIVE',
},
},
flexibility_envelope: {
'da-aggregation-unit': {
id: 'flex-001',
provider: { unit_id: 'agg-unit-wuhan-01', level: 'AGGREGATION_UNIT' },
market_date: DATE,
up_mw: curve('8.5'),
down_mw: curve('6.0'),
valid_until: T2,
schema_version: '1.0.0',
signature: null,
},
},
position_update: {
'award-da': {
id: 'pos-001',
timescale: 'DAY_AHEAD',
period: DATE,
kind: 'AWARD',
energy_mwh: '1180.0',
curve: curve('12.3'),
source_ref: REF_B,
expected_version: 42,
},
},
ledger_view: {
basic: {
version: 43,
entries: [
{
id: 'pos-000',
timescale: 'MONTHLY',
period: '2026-03',
kind: 'CONTRACT',
energy_mwh: '36000.0',
curve: null,
source_ref: 'contract-2026-03-001',
recorded_at: T0,
},
],
},
},
forecast_bundle: {
'load-da': {
id: 'fc-001',
kind: 'LOAD',
market_date: DATE,
unit: 'mw',
quantiles: { p10: curve('40.1'), p50: curve('45.7'), p90: curve('52.3') },
model: { name: 'load-forecast', version: '1.2.0' },
features_snapshot_ref: REF_A,
generated_at: T0,
},
},
M2: skill contracts, Python skill service, L2 eval harness with baseline - packages/domain: ForecastRequest, BidOptimizationRequest/Result, ReportRequest, SkillReport (+ golden and invalid fixtures, exported to contracts/ and regenerated as pydantic models). - skills-py/vpp_skills: FastAPI service with versioned registry; load/PV/ price forecasts (same-day-type EWM point forecast, conformal residual quantiles — coverage test as acceptance gate); bid-optimization MILP on HiGHS (binary block participation, hard ledger energy bounds, exact Decimal fit of the rounded curve inside the bounds, revenue distribution over quantile paths); report generator whose every figure is a {tool_call_id, path} reference, with a verifier. 48 tests incl. hypothesis property test that bids respect ledger constraints. - packages/services: LedgerService.dayAheadBounds (the P7 cascade band handed to the optimizer); Decimal resolved once for CJS/ESM interop. - packages/evals: L2 metrics (MAPE, nRMSE, coverage, direction accuracy, naive/hindsight revenue baselines), HTTP skill client, rolling-origin harness that pushes each bid through the real ledger, CLI with --check/--write-baseline; committed baseline on the SYNTHETIC dataset (no historical Hubei data yet — baselines measure the harness, not KPI). - CI: evals job boots the skill service and fails on baseline digest drift. - docs/open-questions: A6 (flexibility marginal cost = offer floor); A4/B6 wired as placeholders. README/CLAUDE.md status → M2 done, M3 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 06:29:08 -04:00
forecast_request: {
'price-da': {
kind: 'PRICE',
market_date: DATE,
unit: 'yuan_per_mwh',
history: [curve('398.20', '2026-03-13'), curve('410.75', '2026-03-14')],
exogenous: {},
features_snapshot_ref: REF_A,
},
},
bid_optimization_request: {
'da-basic': {
market_date: DATE,
price_forecast: {
id: 'fc-price-001',
kind: 'PRICE',
market_date: DATE,
unit: 'yuan_per_mwh',
quantiles: { p10: curve('360.00'), p50: curve('425.50'), p90: curve('490.00') },
model: { name: 'price-forecast', version: '1.0.0' },
features_snapshot_ref: REF_A,
generated_at: T0,
},
adjustable_capacity_mw: curve('60.0'),
position_bounds: {
ledger_version: 42,
daily_energy_min_mwh: '1140.0',
daily_energy_max_mwh: '1260.0',
},
risk: {
risk_aversion: '0.3',
commitment_buffer_k: '0.9',
min_block_mwh: '1.0',
marginal_cost_yuan_per_mwh: '0',
},
},
},
bid_optimization_result: {
'da-basic': {
market_date: DATE,
prices_yuan_per_mwh: curve('0.00'),
quantities_mwh: curve('12.5'),
daily_energy_mwh: '1200.0',
expected_revenue_yuan: '510600.00',
revenue_distribution_yuan: { p10: '432000.00', p50: '510600.00', p90: '588000.00' },
position_bounds: {
ledger_version: 42,
daily_energy_min_mwh: '1140.0',
daily_energy_max_mwh: '1260.0',
},
solver: {
name: 'highs',
version: '1.7.0',
status: 'OPTIMAL',
objective_value: '487020.00',
wall_time_ms: 12,
},
binding_constraints: ['daily_energy_max'],
skill_version: '1.0.0',
},
},
report_request: {
'da-bid-summary': {
kind: 'DAY_AHEAD_BID_SUMMARY',
market_date: DATE,
sources: [
{
tool_call_id: 'tc-001',
tool: 'bid-optimization-milp',
version: '1.0.0',
output: { expected_revenue_yuan: '510600.00', daily_energy_mwh: '1200.0' },
},
],
},
},
skill_report: {
'da-bid-summary': {
id: 'rep-001',
kind: 'DAY_AHEAD_BID_SUMMARY',
market_date: DATE,
sections: [
{
title: 'Bid summary',
metrics: [
{
name: 'expected_revenue',
value: '510600.00',
unit: 'yuan',
ref: { tool_call_id: 'tc-001', path: 'expected_revenue_yuan' },
},
],
notes: ['all figures reference solver output tc-001'],
},
],
skill_version: '1.0.0',
generated_at: T1,
},
},
M3: Mastra runtime, safety chain, two agents, Case Desk v1 - packages/domain: safety-chain objects (ValidationResult, SimulationResult, EnvelopeMatch, StaleDenial, ExecutionReceipt, LineageRef/BidProposalDraft, RouterDecision, BidExportFile) + fixtures on both sides. - packages/services: proposal digest; PolicyEngine + hubei-spot-bidding pack (digest-valid, bid-format, price-limits, quantity-non-negative, ledger-consistency, lineage-integrity, originator-permission — each with pass/fail tests); EnvelopeService; AuthorityService (fresh check, permits, revoke, gateway validate); FileExportGateway (idempotent receipts); Memory/File EventBus; LineageRecorder + P2 assembler; RevenueScenario simulator; CaseDeskService; FsRepository; skill HTTP client moved here. - packages/runtime: createRuntime (LibSQL storage, per-runtime workflow factories), proposal-lifecycle (rule check → simulation → envelope gate with suspend/resume → fresh check + permit → release), day-ahead-situation, day-ahead-bid, TriggerService (scheduled/event/manual), LlmPort (Mastra/Scripted/Null), Case Desk HTTP API, dev entry point. - Tests: all eight docs/01 invariants, docs/07 06:00→08:30 end to end with LLM down, restart survival of a suspended approval, permit expiry and revocation, replay of a released proposal, trigger scheduling. 141 TS + 60 Python tests. - Known gaps: ledger not yet persisted (replayed on restart); STALE ends the run instead of looping to rule check; synthetic data stands in for historical replay. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 06:55:55 -04:00
validation_result: {
pass: {
proposal_digest: REF_B,
policy_pack: { id: 'hubei-spot-bidding', version: '2026.03' },
ok: true,
violations: [],
rules_evaluated: ['bid-format', 'price-limits', 'ledger-consistency', 'lineage-integrity'],
checked_at: T1,
},
'reject-price': {
proposal_digest: REF_B,
policy_pack: { id: 'hubei-spot-bidding', version: '2026.03' },
ok: false,
violations: [{ rule_id: 'price-limits', severity: 'REJECT', message: 'price 1500.00 above cap' }],
rules_evaluated: ['bid-format', 'price-limits'],
checked_at: T1,
},
},
simulation_result: {
'revenue-ok': {
proposal_digest: REF_B,
kind: 'REVENUE_SCENARIOS',
alert: false,
alerts: [],
metrics: { revenue_p05_yuan: '410200.00', revenue_p50_yuan: '510600.00', worst_loss_yuan: '0.00' },
scenario_count: 1000,
simulator: { name: 'revenue-scenarios', version: '1.0.0', seed: 20260315 },
simulated_at: T1,
},
},
envelope_match: {
within: {
proposal_digest: REF_B,
within: true,
envelope_id: 'env-bid-001',
required_level: 'L1',
reasons: [],
checked_at: T1,
},
outside: {
proposal_digest: REF_B,
within: false,
envelope_id: null,
required_level: 'L2',
reasons: ['no ACTIVE envelope covers BID/DAY_AHEAD for pool-hubei-01'],
checked_at: T1,
},
},
stale_denial: {
'ledger-moved': {
proposal_digest: REF_B,
reasons: ['ledger version 42 at approval, now 43'],
checked_at: T2,
},
},
execution_receipt: {
M5: shadow run, kill-switch hierarchy L0–L4, KPI dashboard, shadow replay CLI Shadow run (ROADMAP M5, phase-1 acceptance form): SHADOW runtime mode records bids without submitting them, simulates the market answer from the actual day-ahead clearing price, dispatches to the simulation gateway, runs the D+1 review, and scores every day shadow-vs-human-vs-hindsight (ShadowDayRecord) with a lineage-completeness audit. KpiReport regenerated after every day per docs/12 §4 definitions (C1/C2 placeholders as named config). Kill switches (docs/13 §8): BreakerService with L0 permit revocation, L1 envelope suspension, L2 loss breaker (mark-to-market, reduce-only bids), L3 channel breaker (bids fall back to the file channel, dispatch BLOCKED), L4 AI-off (templates run, no Proposal created); abnormal-day protocol on EXTREME situations; per-level authority (B8 placeholder); auditable drill. Runtime: shadow-close workflow, shadow schedule entries, live-data ingestion through the quality gate, human-bid ingestion, breaker/KPI/shadow endpoints and insight cards, replay CLI (npm run shadow). Ledger, time series and streak counters are file-backed so a multi-week run survives restarts. Domain: HumanBidRecord, ShadowDayRecord, KpiReport, BreakerRecord, SHADOW receipt channel; contracts, fixtures and pydantic models regenerated. Services: L2 metrics moved from evals so the shadow run and the harness share one implementation. Fixes: envelope/permit validity compared ISO timestamps as strings ('…00Z' vs '…00.000Z'); L2 baseline was stale since M4 (skill_versions only, metrics unchanged) — rewritten from the live service. Docs: docs/14 shadow-run runbook (timeline, breaker trigger/authority/ recovery, KPI definitions as implemented); README and CLAUDE.md status. Tests: 21 consecutive shadow days with complete lineage, KPI report, WIDEN recommendation produced but not acted on; restart durability; drill; L2/L3/L4 and abnormal-day paths; API. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017wrZgPL9LoKaD69BpEQU4v
2026-09-02 22:54:14 -04:00
shadow: {
receipt_id: 'rcpt-shadow-001',
proposal_digest: REF_B,
permit_id: 'permit-001',
channel: 'SHADOW',
idempotency_key: `${REF_B}:permit-001`,
artifact_ref: `shadow://bid/${REF_B}`,
accepted_at: T1,
},
M3: Mastra runtime, safety chain, two agents, Case Desk v1 - packages/domain: safety-chain objects (ValidationResult, SimulationResult, EnvelopeMatch, StaleDenial, ExecutionReceipt, LineageRef/BidProposalDraft, RouterDecision, BidExportFile) + fixtures on both sides. - packages/services: proposal digest; PolicyEngine + hubei-spot-bidding pack (digest-valid, bid-format, price-limits, quantity-non-negative, ledger-consistency, lineage-integrity, originator-permission — each with pass/fail tests); EnvelopeService; AuthorityService (fresh check, permits, revoke, gateway validate); FileExportGateway (idempotent receipts); Memory/File EventBus; LineageRecorder + P2 assembler; RevenueScenario simulator; CaseDeskService; FsRepository; skill HTTP client moved here. - packages/runtime: createRuntime (LibSQL storage, per-runtime workflow factories), proposal-lifecycle (rule check → simulation → envelope gate with suspend/resume → fresh check + permit → release), day-ahead-situation, day-ahead-bid, TriggerService (scheduled/event/manual), LlmPort (Mastra/Scripted/Null), Case Desk HTTP API, dev entry point. - Tests: all eight docs/01 invariants, docs/07 06:00→08:30 end to end with LLM down, restart survival of a suspended approval, permit expiry and revocation, replay of a released proposal, trigger scheduling. 141 TS + 60 Python tests. - Known gaps: ledger not yet persisted (replayed on restart); STALE ends the run instead of looping to rule check; synthetic data stands in for historical replay. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 06:55:55 -04:00
'file-export': {
receipt_id: 'rcpt-001',
proposal_digest: REF_B,
permit_id: 'permit-001',
channel: 'FILE_EXPORT',
idempotency_key: `${REF_B}:permit-001`,
artifact_ref: 'exports/bid-2026-03-15-prop-001.json',
accepted_at: T1,
},
},
bid_proposal_draft: {
basic: {
market_date: DATE,
prices_ref: { tool_call_id: 'tc-001', path: 'prices_yuan_per_mwh' },
quantities_ref: { tool_call_id: 'tc-001', path: 'quantities_mwh' },
expected_revenue_ref: { tool_call_id: 'tc-001', path: 'expected_revenue_yuan' },
rationale: 'Evening peak priced above P50; allocate bounded energy to 18:00–20:00.',
},
},
router_decision: {
'day-ahead-bid': {
workflow_id: 'day-ahead-bid',
params: { market_date: DATE },
confidence: 'HIGH',
},
},
bid_export_file: {
basic: {
format_version: '1.0.0',
proposal_id: 'prop-001',
proposal_digest: REF_B,
permit_id: 'permit-001',
proposal_type: 'BID',
market_date: DATE,
prices_yuan_per_mwh: curve('0.00'),
quantities_mwh: curve('12.5'),
exported_at: T1,
},
},
M4: resource agent, envelopes live, review loop, insight cards - packages/domain: AwardNotice, DISPATCH_PLAN proposal payload, ExecutionReport, MeteringRecord, potential-assessment and dispatch-optimization contracts, ReviewFinding (+ typed writebacks), SemanticMemoryEntry, EnvelopeChangeRequest, InsightCard — exported with fixtures on both sides. - skills-py: potential-assessment (certified × rolling fulfilment, evidence days) and dispatch-optimization (per-interval LP on HiGHS, shortfall reported) skills + routes + tests. - packages/services: dispatch rules in the policy pack (over-allocation, award anchor, lineage integrity for allocations); PowerBalanceSimulator; SimulationGateway (permit-only, idempotent, seeded execute → ExecutionReports); envelope deviation-streak suspension + apply(); ReviewService (attribution, reliability EWMA writeback, semantic memory, envelope recommendations as change requests); dispatch assembler; skill client methods. - packages/runtime: resource agent; award-decomposition, review and envelope-review workflows; lifecycle selects simulator/gateway by proposal type; trigger hooks for awards, execution reports, metering; decide() resumes either lifecycle or envelope-review runs; insight cards API. - Tests: docs/07 D-1 16:00 and D+1 end to end; reliability score 0.9 → 0.880 and the next assessment de-rates capacity; envelope suspension on a seeded 3-day streak; WIDEN request applied only by a human. 184 TS + 80 Python. - docs/open-questions: B10 (reliability/potential parameters). README and CLAUDE.md status → M4 done, M5 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 19:31:12 -04:00
award_notice: {
'da-award': {
id: 'award-2026-03-15',
market_date: DATE,
bid_proposal_digest: REF_B,
awarded_mwh: curve('12.0'),
clearing_price_yuan_per_mwh: curve('418.30'),
received_at: '2026-03-14T08:00:00Z',
},
},
dispatch_proposal_draft: {
basic: {
market_date: DATE,
total_target_ref: { tool_call_id: 'tc-002', path: 'total_target_mw' },
allocations_ref: { tool_call_id: 'tc-002', path: 'allocations' },
rationale: 'Storage covers the evening block; industrial load takes the midday share.',
},
},
execution_report: {
'unit-01': {
id: 'exec-2026-03-15-agg-unit-wuhan-01',
market_date: DATE,
unit_id: 'agg-unit-wuhan-01',
dispatch_proposal_digest: REF_B,
planned_mw: curve('48.0'),
actual_mw: curve('45.6'),
deviation_mwh: '57.600',
fulfillment_ratio: '0.950',
reported_at: '2026-03-16T01:00:00Z',
},
},
metering_record: {
'da-metering': {
id: 'meter-2026-03-15',
market_date: DATE,
metered_mw: curve('45.6'),
actual_price_yuan_per_mwh: curve('421.00'),
received_at: '2026-03-16T02:00:00Z',
},
},
potential_assessment_request: {
basic: {
market_date: DATE,
resources: [
{
profile: {
resource_id: 'res-storage-01',
name: 'Wuhan industrial park storage #1',
type: 'STORAGE',
rated_power_mw: '10.0',
certified_adjustable_mw: '8.0',
confidence: '0.92',
reliability_score: '0.88',
constraints: { min_duration_min: 60, recovery_rate_mw_per_min: '0.5' },
evidence_refs: [REF_C],
updated_at: T0,
},
fulfillment: [
{ market_date: '2026-03-13', planned_mwh: '40.0', delivered_mwh: '38.0' },
{ market_date: '2026-03-14', planned_mwh: '40.0', delivered_mwh: '39.0' },
],
},
],
features_snapshot_ref: REF_A,
},
},
potential_assessment_result: {
basic: {
market_date: DATE,
assessments: [
{ resource_id: 'res-storage-01', adjustable_mw: curve('7.700'), confidence: '0.90', fulfillment_rate: '0.9625', evidence_days: 2 },
],
total_adjustable_mw: curve('7.700'),
skill_version: '1.0.0',
},
},
dispatch_optimization_request: {
basic: {
market_date: DATE,
target_mw: curve('48.0'),
units: [
{ unit_id: 'agg-unit-wuhan-01', available_mw: curve('30.0'), cost_weight: '1.0' },
{ unit_id: 'agg-unit-wuhan-02', available_mw: curve('30.0'), cost_weight: '1.2' },
],
},
},
dispatch_optimization_result: {
basic: {
market_date: DATE,
total_target_mw: curve('48.0'),
allocations: [
{ unit_id: 'agg-unit-wuhan-01', target_mw: curve('30.000') },
{ unit_id: 'agg-unit-wuhan-02', target_mw: curve('18.000') },
],
shortfall_mwh: '0.000',
solver: { name: 'highs', version: 'scipy-1.17.1', status: 'OPTIMAL', wall_time_ms: 4 },
skill_version: '1.0.0',
},
},
review_finding: {
'd-plus-1': {
id: 'rf-2026-03-15-response',
market_date: DATE,
kind: 'RESPONSE',
summary: 'Unit agg-unit-wuhan-01 delivered 95.0% of plan; reliability score adjusted.',
attribution: {
trading_mwh: '48.000',
control_mwh: '0.000',
response_mwh: '57.600',
forecast_load_mape: '0.041',
forecast_price_mape: '0.017',
net_deviation_mwh: '57.600',
},
writebacks: [
{ target: 'SEMANTIC_MEMORY', topic: 'response:agg-unit-wuhan-01', lesson: 'Evening block under-delivery on 2026-03-15.' },
{ target: 'RESOURCE_PROFILE', resource_id: 'res-storage-01', previous_reliability_score: '0.88', new_reliability_score: '0.894', fulfillment_ratio: '0.950' },
{ target: 'ENVELOPE_RECOMMENDATION', envelope_id: 'env-bid-001', action: 'WIDEN', bound: 'price_deviation_pct', current_value: '5.0', recommended_value: '6.0', basis: '26/26 within-envelope days under threshold' },
],
evidence_refs: [REF_A, REF_C],
lineage_refs: [{ tool_call_id: 'tc-review-001', path: 'attribution' }],
generated_at: '2026-03-16T03:00:00Z',
},
},
semantic_memory_entry: {
basic: {
id: 'mem-001',
topic: 'forecast:load:hot-day',
lesson: 'Hot-day HVAC load systematically under-forecast; feature-engineering candidate.',
source_finding_id: 'rf-2026-03-15-response',
evidence_refs: [REF_A],
created_at: '2026-03-16T03:00:00Z',
},
},
envelope_change_request: {
widen: {
id: 'ecr-001',
envelope_id: 'env-bid-001',
action: 'WIDEN',
bounds: { price_deviation_pct: '6.0', max_energy_mwh: '1500.0' },
basis: 'ReviewFinding rf-2026-03-15-response: 26/26 within-envelope days under threshold',
source_finding_id: 'rf-2026-03-15-response',
requested_at: '2026-03-16T03:00:00Z',
},
},
insight_card: {
'trading-desk': {
id: 'card-001',
page: 'TRADING_DESK',
market_date: DATE,
title: 'Day-ahead situation',
headline: 'Risk LOW; P50 peak price 425.50 yuan/MWh',
metrics: [{ name: 'peak_price_p50', value: '425.50', unit: 'yuan_per_mwh', ref: { tool_call_id: 'tc-001', path: 'quantiles.p50.values.76' } }],
source: { kind: 'SITUATION_REPORT', id: 'sit-001', ref: REF_A },
generated_at: T0,
},
},
situation_report: {
'normal-day': {
id: 'sit-001',
market_date: DATE,
risk_level: 'LOW',
findings: [
{ kind: 'TREND', summary: 'mild temperatures, load near seasonal norm', refs: [REF_A] },
],
forecast_refs: ['fc-001'],
generated_at: T0,
},
},
resource_profile: {
storage: {
resource_id: 'res-storage-01',
name: 'Wuhan industrial park storage #1',
type: 'STORAGE',
rated_power_mw: '10.0',
certified_adjustable_mw: '8.0',
confidence: '0.92',
reliability_score: '0.88',
constraints: { min_duration_min: 60, recovery_rate_mw_per_min: '0.5' },
evidence_refs: [REF_C],
updated_at: T0,
},
},
decision_case: {
'da-bid': {
id: 'case-001',
kind: 'DAY_AHEAD_BID',
objective: 'Complete day-ahead bid for 2026-03-15',
owner: 'user-trader-01',
deadline: '2026-03-14T10:00:00Z',
status: 'AWAITING_APPROVAL',
refs: {
evidence: ['sit-001', 'fc-001'],
proposals: ['prop-001'],
approvals: [],
permits: [],
scenarios: [],
outcome: null,
},
opened_at: T0,
closed_at: null,
},
},
event_envelope: {
'situation-published': {
event_id: 'evt-001',
event_type: 'SituationReportPublished',
schema_version: '1.0.0',
occurred_at: T0,
causation_id: 'task-001',
correlation_id: 'case-001',
payload: { situation_report_id: 'sit-001' },
},
},
M5: shadow run, kill-switch hierarchy L0–L4, KPI dashboard, shadow replay CLI Shadow run (ROADMAP M5, phase-1 acceptance form): SHADOW runtime mode records bids without submitting them, simulates the market answer from the actual day-ahead clearing price, dispatches to the simulation gateway, runs the D+1 review, and scores every day shadow-vs-human-vs-hindsight (ShadowDayRecord) with a lineage-completeness audit. KpiReport regenerated after every day per docs/12 §4 definitions (C1/C2 placeholders as named config). Kill switches (docs/13 §8): BreakerService with L0 permit revocation, L1 envelope suspension, L2 loss breaker (mark-to-market, reduce-only bids), L3 channel breaker (bids fall back to the file channel, dispatch BLOCKED), L4 AI-off (templates run, no Proposal created); abnormal-day protocol on EXTREME situations; per-level authority (B8 placeholder); auditable drill. Runtime: shadow-close workflow, shadow schedule entries, live-data ingestion through the quality gate, human-bid ingestion, breaker/KPI/shadow endpoints and insight cards, replay CLI (npm run shadow). Ledger, time series and streak counters are file-backed so a multi-week run survives restarts. Domain: HumanBidRecord, ShadowDayRecord, KpiReport, BreakerRecord, SHADOW receipt channel; contracts, fixtures and pydantic models regenerated. Services: L2 metrics moved from evals so the shadow run and the harness share one implementation. Fixes: envelope/permit validity compared ISO timestamps as strings ('…00Z' vs '…00.000Z'); L2 baseline was stale since M4 (skill_versions only, metrics unchanged) — rewritten from the live service. Docs: docs/14 shadow-run runbook (timeline, breaker trigger/authority/ recovery, KPI definitions as implemented); README and CLAUDE.md status. Tests: 21 consecutive shadow days with complete lineage, KPI report, WIDEN recommendation produced but not acted on; restart durability; drill; L2/L3/L4 and abnormal-day paths; API. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017wrZgPL9LoKaD69BpEQU4v
2026-09-02 22:54:14 -04:00
human_bid_record: {
'platform-export': {
id: 'hb-2026-03-15',
market_date: DATE,
prices_yuan_per_mwh: curve('380.00'),
quantities_mwh: curve('11.0'),
source: 'TRADING_PLATFORM_EXPORT',
received_at: T1,
},
},
breaker_record: {
'l2-tripped': {
level: 'L2',
status: 'TRIPPED',
scope: '*',
reason: 'daily expected loss 120000.00 yuan exceeds budget 100000',
tripped_at: '2026-03-15T06:30:00Z',
tripped_by: { id: 'runtime', role: 'system' },
reset_at: null,
reset_by: null,
reset_basis: null,
drill: false,
trip_count: 1,
},
'l0-armed': {
level: 'L0',
status: 'ARMED',
scope: null,
reason: null,
tripped_at: null,
tripped_by: null,
reset_at: null,
reset_by: null,
reset_basis: null,
drill: false,
trip_count: 0,
},
},
shadow_day_record: {
'released-day': {
id: 'shadow-2026-03-15',
market_date: DATE,
shadow: {
proposal_id: 'prop-001',
digest: REF_B,
outcome: 'RELEASED',
expected_revenue_yuan: '510600.00',
line: { energy_mwh: '1200.000', cleared_energy_mwh: '1200.000', realised_revenue_yuan: '505200.00' },
llm_used: false,
},
human: { record_id: 'hb-2026-03-15', source: 'TRADING_PLATFORM_EXPORT', line: { energy_mwh: '1056.000', cleared_energy_mwh: '1056.000', realised_revenue_yuan: '444576.00' } },
hindsight: { energy_mwh: '1260.000', cleared_energy_mwh: '1260.000', realised_revenue_yuan: '540000.00' },
naive: { energy_mwh: '1260.000', cleared_energy_mwh: '1260.000', realised_revenue_yuan: '530460.00' },
award: { id: 'award-2026-03-15', energy_mwh: '1200.000' },
dispatch: { proposal_id: 'prop-002', outcome: 'RELEASED', shortfall_mwh: '0.000' },
execution: { planned_mwh: '1200.000', delivered_mwh: '1164.000', deviation_mwh: '36.000', fulfillment_ratio: '0.970', within_band: true, simulated: true },
forecast: { load_mape: '0.041', pv_nrmse: '0.062', price_mape: '0.017', price_coverage_p10_p90: '0.812' },
decision_latency_ms: 1840,
review_finding_id: 'rf-2026-03-15-response',
envelope_recommendations: [{ envelope_id: 'env-bid-001', action: 'KEEP' }],
breakers_tripped: [],
abnormal_day: false,
lineage_complete: true,
lineage_gaps: [],
generated_at: '2026-03-16T03:00:00Z',
},
},
kpi_report: {
'window-30d': {
id: 'kpi-2026-03-15',
window: { from: '2026-02-14', to: DATE, days: 30 },
kpis: [
{ id: 'FORECAST_LOAD_MAPE', value: '0.052', unit: '1', target: '0.08', comparator: 'LTE', samples: 30, status: 'MEET', definition: 'aggregate day-ahead 96-interval load MAPE, rolling window mean' },
{ id: 'CROSS_REGION_MATCH', value: null, unit: '1', target: '0.85', comparator: 'GTE', samples: 0, status: 'NOT_APPLICABLE', definition: 'federation commitments vs delivery confirmations (phase 2)' },
],
comparison: { shadow_yuan: '15156000.00', human_yuan: '13337280.00', hindsight_yuan: '16200000.00', naive_yuan: '15913800.00', capture_ratio: '0.935556', uplift_vs_naive: '0.952383', days_with_human_baseline: 30 },
shadow: { days: 30, complete_days: 30, consecutive_complete_days: 30, first_date: '2026-02-14', last_date: DATE, released_days: 27, pending_days: 3, widen_recommendations: 1, breaker_trips: 0 },
generated_at: '2026-03-16T03:00:00Z',
},
},
}
// Each invalid case breaks exactly one rule and must fail on both sides.
const invalid: Record<string, Record<string, unknown>> = {
proposal: {
// money as JSON float — the forbidden representation (docs/11 §3.3)
'bid-float-money': structuredClone(valid['proposal']!['bid-basic']!),
},
approval: {
// digest not sha256 hex
'bad-digest': { ...(valid['approval']!['approve'] as object), proposal_digest: 'not-a-digest' },
},
M2: skill contracts, Python skill service, L2 eval harness with baseline - packages/domain: ForecastRequest, BidOptimizationRequest/Result, ReportRequest, SkillReport (+ golden and invalid fixtures, exported to contracts/ and regenerated as pydantic models). - skills-py/vpp_skills: FastAPI service with versioned registry; load/PV/ price forecasts (same-day-type EWM point forecast, conformal residual quantiles — coverage test as acceptance gate); bid-optimization MILP on HiGHS (binary block participation, hard ledger energy bounds, exact Decimal fit of the rounded curve inside the bounds, revenue distribution over quantile paths); report generator whose every figure is a {tool_call_id, path} reference, with a verifier. 48 tests incl. hypothesis property test that bids respect ledger constraints. - packages/services: LedgerService.dayAheadBounds (the P7 cascade band handed to the optimizer); Decimal resolved once for CJS/ESM interop. - packages/evals: L2 metrics (MAPE, nRMSE, coverage, direction accuracy, naive/hindsight revenue baselines), HTTP skill client, rolling-origin harness that pushes each bid through the real ledger, CLI with --check/--write-baseline; committed baseline on the SYNTHETIC dataset (no historical Hubei data yet — baselines measure the harness, not KPI). - CI: evals job boots the skill service and fails on baseline digest drift. - docs/open-questions: A6 (flexibility marginal cost = offer floor); A4/B6 wired as placeholders. README/CLAUDE.md status → M2 done, M3 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 06:29:08 -04:00
skill_report: {
// a metric without a lineage reference — the one thing a report may never contain (I1)
'metric-without-ref': (() => {
const o = structuredClone(valid['skill_report']!['da-bid-summary']) as any
delete o.sections[0].metrics[0].ref
return o
})(),
},
bid_optimization_result: {
// solver status outside the enum
'bad-solver-status': (() => {
const o = structuredClone(valid['bid_optimization_result']!['da-basic']) as any
o.solver.status = 'SOLVED'
return o
})(),
},
M3: Mastra runtime, safety chain, two agents, Case Desk v1 - packages/domain: safety-chain objects (ValidationResult, SimulationResult, EnvelopeMatch, StaleDenial, ExecutionReceipt, LineageRef/BidProposalDraft, RouterDecision, BidExportFile) + fixtures on both sides. - packages/services: proposal digest; PolicyEngine + hubei-spot-bidding pack (digest-valid, bid-format, price-limits, quantity-non-negative, ledger-consistency, lineage-integrity, originator-permission — each with pass/fail tests); EnvelopeService; AuthorityService (fresh check, permits, revoke, gateway validate); FileExportGateway (idempotent receipts); Memory/File EventBus; LineageRecorder + P2 assembler; RevenueScenario simulator; CaseDeskService; FsRepository; skill HTTP client moved here. - packages/runtime: createRuntime (LibSQL storage, per-runtime workflow factories), proposal-lifecycle (rule check → simulation → envelope gate with suspend/resume → fresh check + permit → release), day-ahead-situation, day-ahead-bid, TriggerService (scheduled/event/manual), LlmPort (Mastra/Scripted/Null), Case Desk HTTP API, dev entry point. - Tests: all eight docs/01 invariants, docs/07 06:00→08:30 end to end with LLM down, restart survival of a suspended approval, permit expiry and revocation, replay of a released proposal, trigger scheduling. 141 TS + 60 Python tests. - Known gaps: ledger not yet persisted (replayed on restart); STALE ends the run instead of looping to rule check; synthetic data stands in for historical replay. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 06:55:55 -04:00
bid_proposal_draft: {
// a literal number where only a lineage reference is allowed (I1/P2)
'literal-number': {
market_date: DATE,
prices_ref: { tool_call_id: 'tc-001', path: 'prices_yuan_per_mwh' },
quantities_ref: { tool_call_id: 'tc-001', path: 'quantities_mwh' },
expected_revenue_ref: '510600.00',
rationale: 'x',
},
},
router_decision: {
// a workflow id that is not a registered template
'unknown-template': { workflow_id: 'transfer-funds', params: { market_date: DATE }, confidence: 'HIGH' },
},
M4: resource agent, envelopes live, review loop, insight cards - packages/domain: AwardNotice, DISPATCH_PLAN proposal payload, ExecutionReport, MeteringRecord, potential-assessment and dispatch-optimization contracts, ReviewFinding (+ typed writebacks), SemanticMemoryEntry, EnvelopeChangeRequest, InsightCard — exported with fixtures on both sides. - skills-py: potential-assessment (certified × rolling fulfilment, evidence days) and dispatch-optimization (per-interval LP on HiGHS, shortfall reported) skills + routes + tests. - packages/services: dispatch rules in the policy pack (over-allocation, award anchor, lineage integrity for allocations); PowerBalanceSimulator; SimulationGateway (permit-only, idempotent, seeded execute → ExecutionReports); envelope deviation-streak suspension + apply(); ReviewService (attribution, reliability EWMA writeback, semantic memory, envelope recommendations as change requests); dispatch assembler; skill client methods. - packages/runtime: resource agent; award-decomposition, review and envelope-review workflows; lifecycle selects simulator/gateway by proposal type; trigger hooks for awards, execution reports, metering; decide() resumes either lifecycle or envelope-review runs; insight cards API. - Tests: docs/07 D-1 16:00 and D+1 end to end; reliability score 0.9 → 0.880 and the next assessment de-rates capacity; envelope suspension on a seeded 3-day streak; WIDEN request applied only by a human. 184 TS + 80 Python. - docs/open-questions: B10 (reliability/potential parameters). README and CLAUDE.md status → M4 done, M5 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 19:31:12 -04:00
review_finding: {
// a writeback with an unknown target — findings may only write to the three sanctioned stores
'unknown-writeback': (() => {
const o = structuredClone(valid['review_finding']!['d-plus-1']) as any
o.writebacks = [{ target: 'LEDGER', note: 'nope' }]
return o
})(),
},
position_update: {
// missing optimistic-concurrency field
'missing-version': (() => {
const o = structuredClone(valid['position_update']!['award-da']) as Record<string, unknown>
delete o['expected_version']
return o
})(),
},
M5: shadow run, kill-switch hierarchy L0–L4, KPI dashboard, shadow replay CLI Shadow run (ROADMAP M5, phase-1 acceptance form): SHADOW runtime mode records bids without submitting them, simulates the market answer from the actual day-ahead clearing price, dispatches to the simulation gateway, runs the D+1 review, and scores every day shadow-vs-human-vs-hindsight (ShadowDayRecord) with a lineage-completeness audit. KpiReport regenerated after every day per docs/12 §4 definitions (C1/C2 placeholders as named config). Kill switches (docs/13 §8): BreakerService with L0 permit revocation, L1 envelope suspension, L2 loss breaker (mark-to-market, reduce-only bids), L3 channel breaker (bids fall back to the file channel, dispatch BLOCKED), L4 AI-off (templates run, no Proposal created); abnormal-day protocol on EXTREME situations; per-level authority (B8 placeholder); auditable drill. Runtime: shadow-close workflow, shadow schedule entries, live-data ingestion through the quality gate, human-bid ingestion, breaker/KPI/shadow endpoints and insight cards, replay CLI (npm run shadow). Ledger, time series and streak counters are file-backed so a multi-week run survives restarts. Domain: HumanBidRecord, ShadowDayRecord, KpiReport, BreakerRecord, SHADOW receipt channel; contracts, fixtures and pydantic models regenerated. Services: L2 metrics moved from evals so the shadow run and the harness share one implementation. Fixes: envelope/permit validity compared ISO timestamps as strings ('…00Z' vs '…00.000Z'); L2 baseline was stale since M4 (skill_versions only, metrics unchanged) — rewritten from the live service. Docs: docs/14 shadow-run runbook (timeline, breaker trigger/authority/ recovery, KPI definitions as implemented); README and CLAUDE.md status. Tests: 21 consecutive shadow days with complete lineage, KPI report, WIDEN recommendation produced but not acted on; restart durability; drill; L2/L3/L4 and abnormal-day paths; API. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017wrZgPL9LoKaD69BpEQU4v
2026-09-02 22:54:14 -04:00
breaker_record: {
// a level outside the L0–L4 hierarchy (docs/13 §8)
'unknown-level': { ...(valid['breaker_record']!['l0-armed'] as object), level: 'L5' },
},
execution_receipt: {
// channel outside the enum: a receipt must say which channel took the effect (I5)
'unknown-channel': { ...(valid['execution_receipt']!['file-export'] as object), channel: 'EMAIL' },
},
}
;(invalid['proposal']!['bid-float-money'] as any).payload.expected_revenue_yuan = 510600.0
const write = (dir: string, sets: Record<string, Record<string, unknown>>) => {
for (const [name, cases] of Object.entries(sets)) {
mkdirSync(`${root}${dir}/${name}`, { recursive: true })
for (const [caseName, obj] of Object.entries(cases)) {
writeFileSync(`${root}${dir}/${name}/${caseName}.json`, JSON.stringify(obj, null, 2) + '\n')
}
}
}
write('fixtures', valid)
write('fixtures-invalid', invalid)
console.log('fixtures written')