vpp-ai-platform/packages/evals/test/harness.test.ts

105 lines
4.3 KiB
TypeScript
Raw Normal View History

M2: skill contracts, Python skill service, L2 eval harness with baseline - packages/domain: ForecastRequest, BidOptimizationRequest/Result, ReportRequest, SkillReport (+ golden and invalid fixtures, exported to contracts/ and regenerated as pydantic models). - skills-py/vpp_skills: FastAPI service with versioned registry; load/PV/ price forecasts (same-day-type EWM point forecast, conformal residual quantiles — coverage test as acceptance gate); bid-optimization MILP on HiGHS (binary block participation, hard ledger energy bounds, exact Decimal fit of the rounded curve inside the bounds, revenue distribution over quantile paths); report generator whose every figure is a {tool_call_id, path} reference, with a verifier. 48 tests incl. hypothesis property test that bids respect ledger constraints. - packages/services: LedgerService.dayAheadBounds (the P7 cascade band handed to the optimizer); Decimal resolved once for CJS/ESM interop. - packages/evals: L2 metrics (MAPE, nRMSE, coverage, direction accuracy, naive/hindsight revenue baselines), HTTP skill client, rolling-origin harness that pushes each bid through the real ledger, CLI with --check/--write-baseline; committed baseline on the SYNTHETIC dataset (no historical Hubei data yet — baselines measure the harness, not KPI). - CI: evals job boots the skill service and fails on baseline digest drift. - docs/open-questions: A6 (flexibility marginal cost = offer floor); A4/B6 wired as placeholders. README/CLAUDE.md status → M2 done, M3 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 06:29:08 -04:00
import { fileURLToPath } from 'node:url'
import { describe, expect, it } from 'vitest'
import type {
BidOptimizationRequest,
BidOptimizationResult,
ForecastBundle,
ForecastKind,
ForecastRequest,
ReportRequest,
SkillReport,
} from '@vpp/domain'
M3: Mastra runtime, safety chain, two agents, Case Desk v1 - packages/domain: safety-chain objects (ValidationResult, SimulationResult, EnvelopeMatch, StaleDenial, ExecutionReceipt, LineageRef/BidProposalDraft, RouterDecision, BidExportFile) + fixtures on both sides. - packages/services: proposal digest; PolicyEngine + hubei-spot-bidding pack (digest-valid, bid-format, price-limits, quantity-non-negative, ledger-consistency, lineage-integrity, originator-permission — each with pass/fail tests); EnvelopeService; AuthorityService (fresh check, permits, revoke, gateway validate); FileExportGateway (idempotent receipts); Memory/File EventBus; LineageRecorder + P2 assembler; RevenueScenario simulator; CaseDeskService; FsRepository; skill HTTP client moved here. - packages/runtime: createRuntime (LibSQL storage, per-runtime workflow factories), proposal-lifecycle (rule check → simulation → envelope gate with suspend/resume → fresh check + permit → release), day-ahead-situation, day-ahead-bid, TriggerService (scheduled/event/manual), LlmPort (Mastra/Scripted/Null), Case Desk HTTP API, dev entry point. - Tests: all eight docs/01 invariants, docs/07 06:00→08:30 end to end with LLM down, restart survival of a suspended approval, permit expiry and revocation, replay of a released proposal, trigger scheduling. 141 TS + 60 Python tests. - Known gaps: ledger not yet persisted (replayed on restart); STALE ends the run instead of looping to rule check; synthetic data stands in for historical replay. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 06:55:55 -04:00
import type { SkillClient } from '@vpp/services'
M2: skill contracts, Python skill service, L2 eval harness with baseline - packages/domain: ForecastRequest, BidOptimizationRequest/Result, ReportRequest, SkillReport (+ golden and invalid fixtures, exported to contracts/ and regenerated as pydantic models). - skills-py/vpp_skills: FastAPI service with versioned registry; load/PV/ price forecasts (same-day-type EWM point forecast, conformal residual quantiles — coverage test as acceptance gate); bid-optimization MILP on HiGHS (binary block participation, hard ledger energy bounds, exact Decimal fit of the rounded curve inside the bounds, revenue distribution over quantile paths); report generator whose every figure is a {tool_call_id, path} reference, with a verifier. 48 tests incl. hypothesis property test that bids respect ledger constraints. - packages/services: LedgerService.dayAheadBounds (the P7 cascade band handed to the optimizer); Decimal resolved once for CJS/ESM interop. - packages/evals: L2 metrics (MAPE, nRMSE, coverage, direction accuracy, naive/hindsight revenue baselines), HTTP skill client, rolling-origin harness that pushes each bid through the real ledger, CLI with --check/--write-baseline; committed baseline on the SYNTHETIC dataset (no historical Hubei data yet — baselines measure the harness, not KPI). - CI: evals job boots the skill service and fails on baseline digest drift. - docs/open-questions: A6 (flexibility marginal cost = offer floor); A4/B6 wired as placeholders. README/CLAUDE.md status → M2 done, M3 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 06:29:08 -04:00
import { loadDataset } from '../src/dataset.js'
import { DEFAULT_CONFIG, runL2 } from '../src/harness.js'
const datasetPath = fileURLToPath(new URL('../datasets/synthetic-hubei-v0.json', import.meta.url))
/**
* Stub skills with trivially checkable behaviour: forecast = last history
* day ± 10%; bid = flat quantity hitting the lower energy bound at offer 0.
* Lets the harness itself be tested without the Python service running.
*/
class StubClient implements SkillClient {
calls = 0
async skills() {
return [
{ id: 'load-forecast', version: '0.0.1', endpoint: '/stub' },
{ id: 'bid-optimization-milp', version: '0.0.1', endpoint: '/stub' },
]
}
async forecast(kind: ForecastKind, req: ForecastRequest): Promise<ForecastBundle> {
this.calls++
const last = req.history[req.history.length - 1]!
const scaled = (f: number) => ({
interval_minutes: 15 as const,
date: req.market_date,
values: last.values.map((v) => (Number(v) * f).toFixed(3)),
})
return {
id: `stub-${kind}`,
kind,
market_date: req.market_date,
unit: req.unit,
quantiles: { p10: scaled(0.9), p50: scaled(1), p90: scaled(1.1) },
model: { name: 'stub', version: '0.0.1' },
features_snapshot_ref: req.features_snapshot_ref,
generated_at: '2026-01-01T00:00:00Z',
}
}
async optimizeBid(req: BidOptimizationRequest): Promise<BidOptimizationResult> {
this.calls++
const per = (Number(req.position_bounds.daily_energy_min_mwh) / 96).toFixed(3)
const flat = { interval_minutes: 15 as const, date: req.market_date, values: Array(96).fill(per) as string[] }
const energy = (Number(per) * 96).toFixed(3)
return {
market_date: req.market_date,
prices_yuan_per_mwh: { ...flat, values: Array(96).fill('0.00') },
quantities_mwh: flat,
daily_energy_mwh: energy,
expected_revenue_yuan: '0.00',
revenue_distribution_yuan: { p10: '0.00', p50: '0.00', p90: '0.00' },
position_bounds: req.position_bounds,
solver: { name: 'stub', version: '0', status: 'OPTIMAL', objective_value: '0.00', wall_time_ms: 0 },
binding_constraints: ['daily_energy_min'],
skill_version: '0.0.1',
}
}
async report(_req: ReportRequest): Promise<SkillReport> {
throw new Error('not used')
}
}
const cfg = { ...DEFAULT_CONFIG, window: 7, holdoutFrom: 10, holdoutDays: 5 }
describe('L2 harness', () => {
it('scores every held-out day and pushes each bid through the real ledger', async () => {
const { dataset, sha256 } = loadDataset(datasetPath)
const client = new StubClient()
const run = await runL2(dataset, sha256, client, cfg, () => '2026-09-01T00:00:00Z')
expect(run.per_day).toHaveLength(5)
expect(client.calls).toBe(5 * 4)
expect(run.metrics.bid.optimal_days).toBe(5)
expect(run.metrics.bid.ledger_accepted_days).toBe(5) // stub bids sit exactly on the lower bound
expect(run.metrics.load.coverage_p10_p90).toBeGreaterThan(0)
expect(run.metrics.price.direction_accuracy).toBeGreaterThan(0.5) // yesterday's shape is informative
expect(run.metrics.bid.revenue_hindsight_yuan).toBeGreaterThanOrEqual(run.metrics.bid.revenue_skill_yuan)
expect(run.metrics.bid.revenue_hindsight_yuan).toBeGreaterThanOrEqual(run.metrics.bid.revenue_naive_yuan)
expect(run.skill_versions['load-forecast']).toBe('0.0.1')
expect(run.dataset.synthetic).toBe(true)
})
it('is reproducible: same inputs → same digest, independent of run time', async () => {
const { dataset, sha256 } = loadDataset(datasetPath)
const a = await runL2(dataset, sha256, new StubClient(), cfg, () => '2026-09-01T00:00:00Z')
const b = await runL2(dataset, sha256, new StubClient(), cfg, () => '2026-09-02T00:00:00Z')
expect(a.digest).toBe(b.digest)
expect(a.id).not.toBe(b.id)
})
it('rejects a holdout that starts before the window is filled', async () => {
const { dataset, sha256 } = loadDataset(datasetPath)
await expect(runL2(dataset, sha256, new StubClient(), { ...cfg, holdoutFrom: 3 })).rejects.toThrow(/window/)
})
})