vpp-ai-platform/packages/evals/test/harness.test.ts
Thomas Bayes 1cc21e0dc6
Some checks are pending
ci / typescript (push) Waiting to run
ci / python (push) Waiting to run
ci / evals (push) Blocked by required conditions
M3: Mastra runtime, safety chain, two agents, Case Desk v1
- packages/domain: safety-chain objects (ValidationResult, SimulationResult,
  EnvelopeMatch, StaleDenial, ExecutionReceipt, LineageRef/BidProposalDraft,
  RouterDecision, BidExportFile) + fixtures on both sides.
- packages/services: proposal digest; PolicyEngine + hubei-spot-bidding pack
  (digest-valid, bid-format, price-limits, quantity-non-negative,
  ledger-consistency, lineage-integrity, originator-permission — each with
  pass/fail tests); EnvelopeService; AuthorityService (fresh check, permits,
  revoke, gateway validate); FileExportGateway (idempotent receipts);
  Memory/File EventBus; LineageRecorder + P2 assembler; RevenueScenario
  simulator; CaseDeskService; FsRepository; skill HTTP client moved here.
- packages/runtime: createRuntime (LibSQL storage, per-runtime workflow
  factories), proposal-lifecycle (rule check → simulation → envelope gate
  with suspend/resume → fresh check + permit → release), day-ahead-situation,
  day-ahead-bid, TriggerService (scheduled/event/manual), LlmPort
  (Mastra/Scripted/Null), Case Desk HTTP API, dev entry point.
- Tests: all eight docs/01 invariants, docs/07 06:00→08:30 end to end with
  LLM down, restart survival of a suspended approval, permit expiry and
  revocation, replay of a released proposal, trigger scheduling. 141 TS +
  60 Python tests.
- Known gaps: ledger not yet persisted (replayed on restart); STALE ends the
  run instead of looping to rule check; synthetic data stands in for
  historical replay.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 06:55:55 -04:00

105 lines
4.3 KiB
TypeScript

import { fileURLToPath } from 'node:url'
import { describe, expect, it } from 'vitest'
import type {
BidOptimizationRequest,
BidOptimizationResult,
ForecastBundle,
ForecastKind,
ForecastRequest,
ReportRequest,
SkillReport,
} from '@vpp/domain'
import type { SkillClient } from '@vpp/services'
import { loadDataset } from '../src/dataset.js'
import { DEFAULT_CONFIG, runL2 } from '../src/harness.js'
const datasetPath = fileURLToPath(new URL('../datasets/synthetic-hubei-v0.json', import.meta.url))
/**
* Stub skills with trivially checkable behaviour: forecast = last history
* day ± 10%; bid = flat quantity hitting the lower energy bound at offer 0.
* Lets the harness itself be tested without the Python service running.
*/
class StubClient implements SkillClient {
calls = 0
async skills() {
return [
{ id: 'load-forecast', version: '0.0.1', endpoint: '/stub' },
{ id: 'bid-optimization-milp', version: '0.0.1', endpoint: '/stub' },
]
}
async forecast(kind: ForecastKind, req: ForecastRequest): Promise<ForecastBundle> {
this.calls++
const last = req.history[req.history.length - 1]!
const scaled = (f: number) => ({
interval_minutes: 15 as const,
date: req.market_date,
values: last.values.map((v) => (Number(v) * f).toFixed(3)),
})
return {
id: `stub-${kind}`,
kind,
market_date: req.market_date,
unit: req.unit,
quantiles: { p10: scaled(0.9), p50: scaled(1), p90: scaled(1.1) },
model: { name: 'stub', version: '0.0.1' },
features_snapshot_ref: req.features_snapshot_ref,
generated_at: '2026-01-01T00:00:00Z',
}
}
async optimizeBid(req: BidOptimizationRequest): Promise<BidOptimizationResult> {
this.calls++
const per = (Number(req.position_bounds.daily_energy_min_mwh) / 96).toFixed(3)
const flat = { interval_minutes: 15 as const, date: req.market_date, values: Array(96).fill(per) as string[] }
const energy = (Number(per) * 96).toFixed(3)
return {
market_date: req.market_date,
prices_yuan_per_mwh: { ...flat, values: Array(96).fill('0.00') },
quantities_mwh: flat,
daily_energy_mwh: energy,
expected_revenue_yuan: '0.00',
revenue_distribution_yuan: { p10: '0.00', p50: '0.00', p90: '0.00' },
position_bounds: req.position_bounds,
solver: { name: 'stub', version: '0', status: 'OPTIMAL', objective_value: '0.00', wall_time_ms: 0 },
binding_constraints: ['daily_energy_min'],
skill_version: '0.0.1',
}
}
async report(_req: ReportRequest): Promise<SkillReport> {
throw new Error('not used')
}
}
const cfg = { ...DEFAULT_CONFIG, window: 7, holdoutFrom: 10, holdoutDays: 5 }
describe('L2 harness', () => {
it('scores every held-out day and pushes each bid through the real ledger', async () => {
const { dataset, sha256 } = loadDataset(datasetPath)
const client = new StubClient()
const run = await runL2(dataset, sha256, client, cfg, () => '2026-09-01T00:00:00Z')
expect(run.per_day).toHaveLength(5)
expect(client.calls).toBe(5 * 4)
expect(run.metrics.bid.optimal_days).toBe(5)
expect(run.metrics.bid.ledger_accepted_days).toBe(5) // stub bids sit exactly on the lower bound
expect(run.metrics.load.coverage_p10_p90).toBeGreaterThan(0)
expect(run.metrics.price.direction_accuracy).toBeGreaterThan(0.5) // yesterday's shape is informative
expect(run.metrics.bid.revenue_hindsight_yuan).toBeGreaterThanOrEqual(run.metrics.bid.revenue_skill_yuan)
expect(run.metrics.bid.revenue_hindsight_yuan).toBeGreaterThanOrEqual(run.metrics.bid.revenue_naive_yuan)
expect(run.skill_versions['load-forecast']).toBe('0.0.1')
expect(run.dataset.synthetic).toBe(true)
})
it('is reproducible: same inputs → same digest, independent of run time', async () => {
const { dataset, sha256 } = loadDataset(datasetPath)
const a = await runL2(dataset, sha256, new StubClient(), cfg, () => '2026-09-01T00:00:00Z')
const b = await runL2(dataset, sha256, new StubClient(), cfg, () => '2026-09-02T00:00:00Z')
expect(a.digest).toBe(b.digest)
expect(a.id).not.toBe(b.id)
})
it('rejects a holdout that starts before the window is filled', async () => {
const { dataset, sha256 } = loadDataset(datasetPath)
await expect(runL2(dataset, sha256, new StubClient(), { ...cfg, holdoutFrom: 3 })).rejects.toThrow(/window/)
})
})