vpp-ai-platform/skills-py/tests/test_forecast.py

82 lines
3.4 KiB
Python
Raw Normal View History

M2: skill contracts, Python skill service, L2 eval harness with baseline - packages/domain: ForecastRequest, BidOptimizationRequest/Result, ReportRequest, SkillReport (+ golden and invalid fixtures, exported to contracts/ and regenerated as pydantic models). - skills-py/vpp_skills: FastAPI service with versioned registry; load/PV/ price forecasts (same-day-type EWM point forecast, conformal residual quantiles — coverage test as acceptance gate); bid-optimization MILP on HiGHS (binary block participation, hard ledger energy bounds, exact Decimal fit of the rounded curve inside the bounds, revenue distribution over quantile paths); report generator whose every figure is a {tool_call_id, path} reference, with a verifier. 48 tests incl. hypothesis property test that bids respect ledger constraints. - packages/services: LedgerService.dayAheadBounds (the P7 cascade band handed to the optimizer); Decimal resolved once for CJS/ESM interop. - packages/evals: L2 metrics (MAPE, nRMSE, coverage, direction accuracy, naive/hindsight revenue baselines), HTTP skill client, rolling-origin harness that pushes each bid through the real ledger, CLI with --check/--write-baseline; committed baseline on the SYNTHETIC dataset (no historical Hubei data yet — baselines measure the harness, not KPI). - CI: evals job boots the skill service and fails on baseline digest drift. - docs/open-questions: A6 (flexibility marginal cost = offer floor); A4/B6 wired as placeholders. README/CLAUDE.md status → M2 done, M3 next. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UoYoGYzHkFyv3ALenkRPhA
2026-09-02 06:29:08 -04:00
"""Forecast skill: contract shape, determinism, and the calibration test the
roadmap names as the M2 acceptance gate (coverage of the P10–P90 band)."""
from __future__ import annotations
from datetime import datetime, timezone
from decimal import Decimal
import numpy as np
import pytest
from vpp_contracts.forecast_request import ForecastRequest
from vpp_skills.forecast import forecast
from .conftest import forecast_request
CLOCK = lambda: datetime(2026, 3, 14, 6, 0, tzinfo=timezone.utc) # noqa: E731
def _arr(curve) -> np.ndarray:
return np.array([float(Decimal(v.root)) for v in curve.values])
@pytest.mark.parametrize("kind", ["LOAD", "PV", "PRICE"])
def test_bundle_shape_and_ordering(dataset, kind):
req = ForecastRequest.model_validate(forecast_request(dataset, kind, 40))
b = forecast(req, CLOCK)
assert b.kind.value == kind and b.market_date == req.market_date
p10, p50, p90 = (_arr(getattr(b.quantiles, q)) for q in ("p10", "p50", "p90"))
assert np.all(p10 <= p50 + 1e-9) and np.all(p50 <= p90 + 1e-9)
if kind != "PRICE":
assert np.all(p10 >= 0)
assert b.generated_at.isoformat() == "2026-03-14T06:00:00+00:00"
assert b.model.name == f"{kind.lower()}-forecast"
def test_deterministic(dataset):
req = ForecastRequest.model_validate(forecast_request(dataset, "LOAD", 50))
a, b = forecast(req, CLOCK), forecast(req, CLOCK)
assert a.model_dump() == b.model_dump()
def test_pv_night_stays_zero(dataset):
req = ForecastRequest.model_validate(forecast_request(dataset, "PV", 45))
b = forecast(req, CLOCK)
assert _arr(b.quantiles.p90)[:20].sum() == 0.0 # 00:00–05:00
assert _arr(b.quantiles.p50)[44:52].sum() > 0.0 # midday
@pytest.mark.parametrize("kind", ["LOAD", "PV", "PRICE"])
def test_interval_calibration(dataset, kind):
"""Nominal 80% band must cover roughly 80% of held-out actuals. Materially
under-covering (optimistic bands) is the failure mode docs/12 flags as more
dangerous than point error."""
field = {"LOAD": "load_mw", "PV": "pv_mw", "PRICE": "price_yuan_per_mwh"}[kind]
hits = total = 0
for idx in range(35, 90):
req = ForecastRequest.model_validate(forecast_request(dataset, kind, idx))
b = forecast(req, CLOCK)
actual = np.array([float(Decimal(v)) for v in dataset["days"][idx][field]])
p10, p90 = _arr(b.quantiles.p10), _arr(b.quantiles.p90)
if kind == "PV": # night intervals are trivially covered; score daylight only
mask = actual > 0
actual, p10, p90 = actual[mask], p10[mask], p90[mask]
hits += int(np.sum((actual >= p10) & (actual <= p90)))
total += len(actual)
coverage = hits / total
assert 0.70 <= coverage <= 0.92, f"{kind} P10–P90 coverage {coverage:.3f} off nominal 0.80"
def test_rejects_bad_history(dataset):
base = forecast_request(dataset, "LOAD", 40)
unsorted = {**base, "history": list(reversed(base["history"]))}
with pytest.raises(ValueError, match="ascending"):
forecast(ForecastRequest.model_validate(unsorted), CLOCK)
future = {**base, "market_date": base["history"][0]["date"]}
with pytest.raises(ValueError, match="strictly before"):
forecast(ForecastRequest.model_validate(future), CLOCK)
wrong_unit = {**base, "unit": "yuan_per_mwh"}
with pytest.raises(ValueError, match="unit"):
forecast(ForecastRequest.model_validate(wrong_unit), CLOCK)