vpp-ai-platform/packages/evals/test/harness.test.ts
Thomas Bayes 23fdf49ff2
Some checks are pending
ci / typescript (push) Waiting to run
ci / python (push) Waiting to run
ci / evals (push) Blocked by required conditions
M4: resource agent, envelopes live, review loop, insight cards
- packages/domain: AwardNotice, DISPATCH_PLAN proposal payload, ExecutionReport,
  MeteringRecord, potential-assessment and dispatch-optimization contracts,
  ReviewFinding (+ typed writebacks), SemanticMemoryEntry,
  EnvelopeChangeRequest, InsightCard — exported with fixtures on both sides.
- skills-py: potential-assessment (certified × rolling fulfilment, evidence
  days) and dispatch-optimization (per-interval LP on HiGHS, shortfall
  reported) skills + routes + tests.
- packages/services: dispatch rules in the policy pack (over-allocation,
  award anchor, lineage integrity for allocations); PowerBalanceSimulator;
  SimulationGateway (permit-only, idempotent, seeded execute → ExecutionReports);
  envelope deviation-streak suspension + apply(); ReviewService (attribution,
  reliability EWMA writeback, semantic memory, envelope recommendations as
  change requests); dispatch assembler; skill client methods.
- packages/runtime: resource agent; award-decomposition, review and
  envelope-review workflows; lifecycle selects simulator/gateway by proposal
  type; trigger hooks for awards, execution reports, metering; decide()
  resumes either lifecycle or envelope-review runs; insight cards API.
- Tests: docs/07 D-1 16:00 and D+1 end to end; reliability score 0.9 → 0.880
  and the next assessment de-rates capacity; envelope suspension on a seeded
  3-day streak; WIDEN request applied only by a human. 184 TS + 80 Python.
- docs/open-questions: B10 (reliability/potential parameters). README and
  CLAUDE.md status → M4 done, M5 next.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 19:31:12 -04:00

111 lines
4.5 KiB
TypeScript

import { fileURLToPath } from 'node:url'
import { describe, expect, it } from 'vitest'
import type {
BidOptimizationRequest,
BidOptimizationResult,
ForecastBundle,
ForecastKind,
ForecastRequest,
ReportRequest,
SkillReport,
} from '@vpp/domain'
import type { SkillClient } from '@vpp/services'
import { loadDataset } from '../src/dataset.js'
import { DEFAULT_CONFIG, runL2 } from '../src/harness.js'
const datasetPath = fileURLToPath(new URL('../datasets/synthetic-hubei-v0.json', import.meta.url))
/**
* Stub skills with trivially checkable behaviour: forecast = last history
* day ± 10%; bid = flat quantity hitting the lower energy bound at offer 0.
* Lets the harness itself be tested without the Python service running.
*/
class StubClient implements SkillClient {
calls = 0
async skills() {
return [
{ id: 'load-forecast', version: '0.0.1', endpoint: '/stub' },
{ id: 'bid-optimization-milp', version: '0.0.1', endpoint: '/stub' },
]
}
async forecast(kind: ForecastKind, req: ForecastRequest): Promise<ForecastBundle> {
this.calls++
const last = req.history[req.history.length - 1]!
const scaled = (f: number) => ({
interval_minutes: 15 as const,
date: req.market_date,
values: last.values.map((v) => (Number(v) * f).toFixed(3)),
})
return {
id: `stub-${kind}`,
kind,
market_date: req.market_date,
unit: req.unit,
quantiles: { p10: scaled(0.9), p50: scaled(1), p90: scaled(1.1) },
model: { name: 'stub', version: '0.0.1' },
features_snapshot_ref: req.features_snapshot_ref,
generated_at: '2026-01-01T00:00:00Z',
}
}
async optimizeBid(req: BidOptimizationRequest): Promise<BidOptimizationResult> {
this.calls++
const per = (Number(req.position_bounds.daily_energy_min_mwh) / 96).toFixed(3)
const flat = { interval_minutes: 15 as const, date: req.market_date, values: Array(96).fill(per) as string[] }
const energy = (Number(per) * 96).toFixed(3)
return {
market_date: req.market_date,
prices_yuan_per_mwh: { ...flat, values: Array(96).fill('0.00') },
quantities_mwh: flat,
daily_energy_mwh: energy,
expected_revenue_yuan: '0.00',
revenue_distribution_yuan: { p10: '0.00', p50: '0.00', p90: '0.00' },
position_bounds: req.position_bounds,
solver: { name: 'stub', version: '0', status: 'OPTIMAL', objective_value: '0.00', wall_time_ms: 0 },
binding_constraints: ['daily_energy_min'],
skill_version: '0.0.1',
}
}
async report(_req: ReportRequest): Promise<SkillReport> {
throw new Error('not used')
}
async assessPotential(): Promise<never> {
throw new Error('not used')
}
async optimizeDispatch(): Promise<never> {
throw new Error('not used')
}
}
const cfg = { ...DEFAULT_CONFIG, window: 7, holdoutFrom: 10, holdoutDays: 5 }
describe('L2 harness', () => {
it('scores every held-out day and pushes each bid through the real ledger', async () => {
const { dataset, sha256 } = loadDataset(datasetPath)
const client = new StubClient()
const run = await runL2(dataset, sha256, client, cfg, () => '2026-09-01T00:00:00Z')
expect(run.per_day).toHaveLength(5)
expect(client.calls).toBe(5 * 4)
expect(run.metrics.bid.optimal_days).toBe(5)
expect(run.metrics.bid.ledger_accepted_days).toBe(5) // stub bids sit exactly on the lower bound
expect(run.metrics.load.coverage_p10_p90).toBeGreaterThan(0)
expect(run.metrics.price.direction_accuracy).toBeGreaterThan(0.5) // yesterday's shape is informative
expect(run.metrics.bid.revenue_hindsight_yuan).toBeGreaterThanOrEqual(run.metrics.bid.revenue_skill_yuan)
expect(run.metrics.bid.revenue_hindsight_yuan).toBeGreaterThanOrEqual(run.metrics.bid.revenue_naive_yuan)
expect(run.skill_versions['load-forecast']).toBe('0.0.1')
expect(run.dataset.synthetic).toBe(true)
})
it('is reproducible: same inputs → same digest, independent of run time', async () => {
const { dataset, sha256 } = loadDataset(datasetPath)
const a = await runL2(dataset, sha256, new StubClient(), cfg, () => '2026-09-01T00:00:00Z')
const b = await runL2(dataset, sha256, new StubClient(), cfg, () => '2026-09-02T00:00:00Z')
expect(a.digest).toBe(b.digest)
expect(a.id).not.toBe(b.id)
})
it('rejects a holdout that starts before the window is filled', async () => {
const { dataset, sha256 } = loadDataset(datasetPath)
await expect(runL2(dataset, sha256, new StubClient(), { ...cfg, holdoutFrom: 3 })).rejects.toThrow(/window/)
})
})