vpp-ai-platform/packages/runtime/test/helpers.ts

134 lines
8.1 KiB
TypeScript
Raw Normal View History

M3: Mastra runtime, safety chain, two agents, Case Desk v1 - packages/domain: safety-chain objects (ValidationResult, SimulationResult, EnvelopeMatch, StaleDenial, ExecutionReceipt, LineageRef/BidProposalDraft, RouterDecision, BidExportFile) + fixtures on both sides. - packages/services: proposal digest; PolicyEngine + hubei-spot-bidding pack (digest-valid, bid-format, price-limits, quantity-non-negative, ledger-consistency, lineage-integrity, originator-permission — each with pass/fail tests); EnvelopeService; AuthorityService (fresh check, permits, revoke, gateway validate); FileExportGateway (idempotent receipts); Memory/File EventBus; LineageRecorder + P2 assembler; RevenueScenario simulator; CaseDeskService; FsRepository; skill HTTP client moved here. - packages/runtime: createRuntime (LibSQL storage, per-runtime workflow factories), proposal-lifecycle (rule check → simulation → envelope gate with suspend/resume → fresh check + permit → release), day-ahead-situation, day-ahead-bid, TriggerService (scheduled/event/manual), LlmPort (Mastra/Scripted/Null), Case Desk HTTP API, dev entry point. - Tests: all eight docs/01 invariants, docs/07 06:00→08:30 end to end with LLM down, restart survival of a suspended approval, permit expiry and revocation, replay of a released proposal, trigger scheduling. 141 TS + 60 Python tests. - Known gaps: ledger not yet persisted (replayed on restart); STALE ends the run instead of looping to rule check; synthetic data stands in for historical replay. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 06:55:55 -04:00
import { mkdtempSync, readFileSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { fileURLToPath } from 'node:url'
M4: resource agent, envelopes live, review loop, insight cards - packages/domain: AwardNotice, DISPATCH_PLAN proposal payload, ExecutionReport, MeteringRecord, potential-assessment and dispatch-optimization contracts, ReviewFinding (+ typed writebacks), SemanticMemoryEntry, EnvelopeChangeRequest, InsightCard — exported with fixtures on both sides. - skills-py: potential-assessment (certified × rolling fulfilment, evidence days) and dispatch-optimization (per-interval LP on HiGHS, shortfall reported) skills + routes + tests. - packages/services: dispatch rules in the policy pack (over-allocation, award anchor, lineage integrity for allocations); PowerBalanceSimulator; SimulationGateway (permit-only, idempotent, seeded execute → ExecutionReports); envelope deviation-streak suspension + apply(); ReviewService (attribution, reliability EWMA writeback, semantic memory, envelope recommendations as change requests); dispatch assembler; skill client methods. - packages/runtime: resource agent; award-decomposition, review and envelope-review workflows; lifecycle selects simulator/gateway by proposal type; trigger hooks for awards, execution reports, metering; decide() resumes either lifecycle or envelope-review runs; insight cards API. - Tests: docs/07 D-1 16:00 and D+1 end to end; reliability score 0.9 → 0.880 and the next assessment de-rates capacity; envelope suspension on a seeded 3-day streak; WIDEN request applied only by a human. 184 TS + 80 Python. - docs/open-questions: B10 (reliability/potential parameters). README and CLAUDE.md status → M4 done, M5 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 19:31:12 -04:00
import type { BidOptimizationRequest, BidOptimizationResult, DispatchOptimizationRequest, DispatchOptimizationResult, Envelope, ForecastBundle, ForecastKind, ForecastRequest, PotentialAssessmentRequest, PotentialAssessmentResult, ReportRequest, SkillReport } from '@vpp/domain'
M3: Mastra runtime, safety chain, two agents, Case Desk v1 - packages/domain: safety-chain objects (ValidationResult, SimulationResult, EnvelopeMatch, StaleDenial, ExecutionReceipt, LineageRef/BidProposalDraft, RouterDecision, BidExportFile) + fixtures on both sides. - packages/services: proposal digest; PolicyEngine + hubei-spot-bidding pack (digest-valid, bid-format, price-limits, quantity-non-negative, ledger-consistency, lineage-integrity, originator-permission — each with pass/fail tests); EnvelopeService; AuthorityService (fresh check, permits, revoke, gateway validate); FileExportGateway (idempotent receipts); Memory/File EventBus; LineageRecorder + P2 assembler; RevenueScenario simulator; CaseDeskService; FsRepository; skill HTTP client moved here. - packages/runtime: createRuntime (LibSQL storage, per-runtime workflow factories), proposal-lifecycle (rule check → simulation → envelope gate with suspend/resume → fresh check + permit → release), day-ahead-situation, day-ahead-bid, TriggerService (scheduled/event/manual), LlmPort (Mastra/Scripted/Null), Case Desk HTTP API, dev entry point. - Tests: all eight docs/01 invariants, docs/07 06:00→08:30 end to end with LLM down, restart survival of a suspended approval, permit expiry and revocation, replay of a released proposal, trigger scheduling. 141 TS + 60 Python tests. - Known gaps: ledger not yet persisted (replayed on restart); STALE ends the run instead of looping to rule check; synthetic data stands in for historical replay. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 06:55:55 -04:00
import { MemoryTimeSeriesStore } from '@vpp/services'
import type { SkillClient } from '@vpp/services'
import { createRuntime } from '../src/runtime.js'
import type { RuntimeOptions } from '../src/runtime.js'
import type { LlmPort } from '../src/llm.js'
const datasetPath = fileURLToPath(new URL('../../evals/datasets/synthetic-hubei-v0.json', import.meta.url))
interface Day { date: string; load_mw: string[]; pv_mw: string[]; price_yuan_per_mwh: string[]; adjustable_capacity_mw: string[] }
export const MARKET_DATE = '2026-03-15'
export const NOW = '2026-03-14T00:00:00Z' // D-1 08:00 Asia/Shanghai
/**
* Deterministic stand-in for the Python skill service: forecast = last
* history day with a ±spread band; bid = flat energy at the position's
* lower bound offered at `offer` — enough to steer the lifecycle down every
* branch (envelope in/out, simulation alert) from tests.
*/
export class StubSkills implements SkillClient {
spread = 0.1
offer = '0.00'
energyFraction = 0 // 0 → E_min, 1 → E_max
calls: string[] = []
async skills() {
return ['load-forecast', 'pv-forecast', 'price-forecast', 'bid-optimization-milp'].map((id) => ({ id, version: '0.0.1', endpoint: '/stub' }))
}
async forecast(kind: ForecastKind, req: ForecastRequest): Promise<ForecastBundle> {
this.calls.push(`forecast:${kind}`)
const last = req.history[req.history.length - 1]!
const scaled = (f: number) => ({ interval_minutes: 15 as const, date: req.market_date, values: last.values.map((v) => (Number(v) * f).toFixed(kind === 'PRICE' ? 2 : 3)) })
return {
id: `fc-${kind.toLowerCase()}-${req.market_date}`, kind, market_date: req.market_date, unit: req.unit,
quantiles: { p10: scaled(1 - this.spread), p50: scaled(1), p90: scaled(1 + this.spread) },
model: { name: `${kind.toLowerCase()}-forecast`, version: '0.0.1' }, features_snapshot_ref: req.features_snapshot_ref, generated_at: '2026-03-14T06:00:00Z',
}
}
async optimizeBid(req: BidOptimizationRequest): Promise<BidOptimizationResult> {
this.calls.push('milp')
const lo = Number(req.position_bounds.daily_energy_min_mwh)
const hi = Number(req.position_bounds.daily_energy_max_mwh)
const per = ((lo + (hi - lo) * this.energyFraction) / 96).toFixed(3)
const q = { interval_minutes: 15 as const, date: req.market_date, values: Array(96).fill(per) as string[] }
const energy = (Number(per) * 96).toFixed(3)
const p50 = req.price_forecast.quantiles.p50.values
const revenue = p50.reduce((s, p) => s + (Number(this.offer) <= Number(p) ? Number(p) * Number(per) : 0), 0).toFixed(2)
return {
market_date: req.market_date, prices_yuan_per_mwh: { ...q, values: Array(96).fill(this.offer) }, quantities_mwh: q, daily_energy_mwh: energy,
expected_revenue_yuan: revenue, revenue_distribution_yuan: { p10: revenue, p50: revenue, p90: revenue }, position_bounds: req.position_bounds,
solver: { name: 'stub', version: '0', status: 'OPTIMAL', objective_value: revenue, wall_time_ms: 1 }, binding_constraints: ['daily_energy_min'], skill_version: '0.0.1',
}
}
M4: resource agent, envelopes live, review loop, insight cards - packages/domain: AwardNotice, DISPATCH_PLAN proposal payload, ExecutionReport, MeteringRecord, potential-assessment and dispatch-optimization contracts, ReviewFinding (+ typed writebacks), SemanticMemoryEntry, EnvelopeChangeRequest, InsightCard — exported with fixtures on both sides. - skills-py: potential-assessment (certified × rolling fulfilment, evidence days) and dispatch-optimization (per-interval LP on HiGHS, shortfall reported) skills + routes + tests. - packages/services: dispatch rules in the policy pack (over-allocation, award anchor, lineage integrity for allocations); PowerBalanceSimulator; SimulationGateway (permit-only, idempotent, seeded execute → ExecutionReports); envelope deviation-streak suspension + apply(); ReviewService (attribution, reliability EWMA writeback, semantic memory, envelope recommendations as change requests); dispatch assembler; skill client methods. - packages/runtime: resource agent; award-decomposition, review and envelope-review workflows; lifecycle selects simulator/gateway by proposal type; trigger hooks for awards, execution reports, metering; decide() resumes either lifecycle or envelope-review runs; insight cards API. - Tests: docs/07 D-1 16:00 and D+1 end to end; reliability score 0.9 → 0.880 and the next assessment de-rates capacity; envelope suspension on a seeded 3-day streak; WIDEN request applied only by a human. 184 TS + 80 Python. - docs/open-questions: B10 (reliability/potential parameters). README and CLAUDE.md status → M4 done, M5 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 19:31:12 -04:00
async report(r: ReportRequest): Promise<SkillReport> {
this.calls.push('report')
return { id: `rep-${r.kind.toLowerCase()}-${r.market_date}`, kind: r.kind, market_date: r.market_date, sections: [], skill_version: '0.0.1', generated_at: '2026-03-16T03:00:00Z' }
}
async assessPotential(req: PotentialAssessmentRequest): Promise<PotentialAssessmentResult> {
this.calls.push('potential')
const curveOf = (v: string) => ({ interval_minutes: 15 as const, date: req.market_date, values: Array(96).fill(v) as string[] })
const assessments = req.resources.map((r) => {
const planned = r.fulfillment.reduce((s, f) => s + Number(f.planned_mwh), 0)
const delivered = r.fulfillment.reduce((s, f) => s + Number(f.delivered_mwh), 0)
const rate = planned > 0 ? Math.min(1, delivered / planned) : null
const mw = (Number(r.profile.certified_adjustable_mw) * (rate ?? 1)).toFixed(3)
return { resource_id: r.profile.resource_id, adjustable_mw: curveOf(mw), confidence: '0.900', fulfillment_rate: rate === null ? null : rate.toFixed(4), evidence_days: r.fulfillment.length }
})
const total = assessments.reduce((s, a) => s + Number(a.adjustable_mw.values[0]), 0).toFixed(3)
return { market_date: req.market_date, assessments, total_adjustable_mw: curveOf(total), skill_version: '0.0.1' }
}
async optimizeDispatch(req: DispatchOptimizationRequest): Promise<DispatchOptimizationResult> {
this.calls.push('dispatch')
// Greedy fill in unit order, per interval.
const remaining = req.target_mw.values.map(Number)
const allocations = req.units.map((u) => {
const values = u.available_mw.values.map((cap, t) => {
const x = Math.min(Number(cap), remaining[t]!)
remaining[t]! -= x
return x.toFixed(3)
})
return { unit_id: u.unit_id, target_mw: { interval_minutes: 15 as const, date: req.market_date, values } }
})
const shortfall = (remaining.reduce((s, v) => s + Math.max(0, v), 0) * 0.25).toFixed(3)
return { market_date: req.market_date, total_target_mw: req.target_mw, allocations, shortfall_mwh: shortfall, solver: { name: 'stub', version: '0', status: 'OPTIMAL', wall_time_ms: 0 }, skill_version: '0.0.1' }
}
M3: Mastra runtime, safety chain, two agents, Case Desk v1 - packages/domain: safety-chain objects (ValidationResult, SimulationResult, EnvelopeMatch, StaleDenial, ExecutionReceipt, LineageRef/BidProposalDraft, RouterDecision, BidExportFile) + fixtures on both sides. - packages/services: proposal digest; PolicyEngine + hubei-spot-bidding pack (digest-valid, bid-format, price-limits, quantity-non-negative, ledger-consistency, lineage-integrity, originator-permission — each with pass/fail tests); EnvelopeService; AuthorityService (fresh check, permits, revoke, gateway validate); FileExportGateway (idempotent receipts); Memory/File EventBus; LineageRecorder + P2 assembler; RevenueScenario simulator; CaseDeskService; FsRepository; skill HTTP client moved here. - packages/runtime: createRuntime (LibSQL storage, per-runtime workflow factories), proposal-lifecycle (rule check → simulation → envelope gate with suspend/resume → fresh check + permit → release), day-ahead-situation, day-ahead-bid, TriggerService (scheduled/event/manual), LlmPort (Mastra/Scripted/Null), Case Desk HTTP API, dev entry point. - Tests: all eight docs/01 invariants, docs/07 06:00→08:30 end to end with LLM down, restart survival of a suspended approval, permit expiry and revocation, replay of a released proposal, trigger scheduling. 141 TS + 60 Python tests. - Known gaps: ledger not yet persisted (replayed on restart); STALE ends the run instead of looping to rule check; synthetic data stands in for historical replay. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 06:55:55 -04:00
}
export function seededTimeseries(clock: () => string) {
const ts = new MemoryTimeSeriesStore(clock)
const days = (JSON.parse(readFileSync(datasetPath, 'utf8')) as { days: Day[] }).days
for (const d of days) {
ts.write('load:aggregate', { interval_minutes: 15, date: d.date, values: d.load_mw }, 'synthetic')
ts.write('pv:aggregate', { interval_minutes: 15, date: d.date, values: d.pv_mw }, 'synthetic')
ts.write('price:da', { interval_minutes: 15, date: d.date, values: d.price_yuan_per_mwh }, 'synthetic')
}
return ts
}
export const envelope = (bounds: Record<string, string>, over: Partial<Envelope> = {}): Envelope => ({
id: 'env-bid-001',
scope: { proposal_type: 'BID', timescales: ['DAY_AHEAD'], resource_set: 'pool-hubei-01' },
bounds,
validity: { from: '2026-03-01T00:00:00Z', to: '2026-03-31T23:59:59Z' },
approval: { level: 'L1', approved_by: ['user-ops-lead'] },
escalation: { max_consecutive_deviations: 3, deviation_threshold_pct: '10.0' },
status: 'ACTIVE',
...over,
})
export interface Harness { rt: Awaited<ReturnType<typeof createRuntime>>; skills: StubSkills; dataDir: string; setNow: (iso: string) => void }
/** Fresh runtime on a temp dir, seeded like docs/07 D-1: history, one storage resource, a monthly contract. */
export async function harness(opts: { llm?: LlmPort | null; dataDir?: string; seed?: boolean; config?: RuntimeOptions['config'] } = {}): Promise<Harness> {
let now = NOW
const clock = () => now
const skills = new StubSkills()
const dataDir = opts.dataDir ?? mkdtempSync(join(tmpdir(), 'vpp-rt-'))
const rt = await createRuntime({ dataDir, llm: opts.llm ?? null, skills, clock, timeseries: seededTimeseries(clock), ...(opts.config ? { config: opts.config } : {}) })
if (opts.seed !== false) {
rt.ctx.resources.put({
resource_id: 'res-storage-01', name: 'Wuhan storage #1', type: 'STORAGE', rated_power_mw: '30.0', certified_adjustable_mw: '24.0', confidence: '0.9', reliability_score: '0.9',
constraints: { min_duration_min: 60, recovery_rate_mw_per_min: '0.5' }, evidence_refs: [], updated_at: NOW,
})
// 24 MW × 0.9 × 0.25 h × 96 = 518.4 MWh sellable/day; contract 60% of it → daily share 311.04, band ±5%
rt.ctx.ledger.append({ id: 'contract-2026-03', timescale: 'MONTHLY', period: '2026-03', kind: 'CONTRACT', energy_mwh: (311.04 * 31).toFixed(3), curve: null, source_ref: 'contract-2026-03-001', expected_version: 0 })
}
return { rt, skills, dataDir, setNow: (iso) => (now = iso) }
}