Files
challenge_ai/tests/test_validation.py
T
mattandClaude Sonnet 5 afa02864a7
CI / test (push) Successful in 12s
Fix test suite hangs caused by degenerate MILP fixtures
Two tests were spinning up real CBC solves over 96-half-hour windows
with long runs of exactly-repeated prices (e.g. [0.0] * 40). That gives
the LP relaxation a huge set of economically indistinguishable ways to
spread a trade, which the MILP fallback's branch-and-bound then wastes
enormous effort disambiguating (47k+ nodes without closing the gap,
confirmed by running CBC verbosely). Real Attachment 2 data has no such
flat runs and solves in ~0.1s/window; a synthetic sine wiggle wasn't
enough either, since neighbouring half-hours stayed too similar.

Fixes, matched to what each test actually needs:
- The 9 validator tests only need *some* structurally valid schedule to
  mutate; they were deriving it by running the real optimiser once per
  test. Replaced with make_valid_schedule(), built directly from the
  model's own energy-balance formulas -- no solver involved, and it's
  now a true unit test of validate_schedule() in isolation.
- The rolling-horizon carry-forward test was exercising the mechanism
  at full production scale (48h window / 24h commit) when a 2h/1h
  window proves the same boundary-carrying behaviour with a trivial
  MILP, regardless of price structure.
- Fixed a genuine tie in test_optimum_uses_whichever_market_pays_more:
  two equal-price hours with just enough stored energy for one meant
  either market was a valid optimum. Sized the charge phase so delivery
  must split across both hours, pinning a unique answer.

Full suite: 38 passed in ~1.3s (previously hung indefinitely on CI and
locally within seconds of the same wall-clock variance CBC shows on
degenerate MIPs).

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-09-24 16:10:55 +01:00

125 lines
4.4 KiB
Python

"""The validator must catch each violation type when it is injected."""
from __future__ import annotations
import numpy as np
import pandas as pd
import pytest
from conftest import make_valid_schedule
from battery_dispatch.validation import validate_schedule
@pytest.fixture
def schedule(spec) -> pd.DataFrame:
"""A genuine, valid schedule to mutate in each test. See
``conftest.make_valid_schedule`` for why this isn't a real optimiser run.
"""
return make_valid_schedule(spec)
def test_a_valid_schedule_passes_every_check(schedule, spec):
report = validate_schedule(schedule, spec)
assert report.ok, report.summary()
assert len(report.checks) == 7
assert "all 7 validation checks passed" in report.summary()
def test_catches_charging_above_the_power_limit(schedule, spec):
schedule.loc[schedule.index[0], "charge_market_1_mw"] = 5.0
report = validate_schedule(schedule, spec)
assert not report.ok
assert report.checks["charge_within_limit"] is False
assert any("combined charge" in failure for failure in report.failures)
def test_catches_discharging_above_the_power_limit(schedule, spec):
schedule.loc[schedule.index[30], "discharge_market_2_mw"] = 3.5
report = validate_schedule(schedule, spec)
assert not report.ok
assert report.checks["discharge_within_limit"] is False
def test_catches_simultaneous_charge_and_discharge(schedule, spec):
row = schedule.index[0]
schedule.loc[row, "charge_market_1_mw"] = 1.0
schedule.loc[row, "discharge_market_1_mw"] = 1.0
report = validate_schedule(schedule, spec)
assert not report.ok
assert report.checks["no_simultaneous_charge_discharge"] is False
assert report.details["simultaneous_half_hours"] == 1.0
def test_catches_market_2_power_varying_within_an_hour(schedule, spec):
# Break only the second half-hour of hour zero.
schedule.loc[schedule.index[1], "discharge_market_2_mw"] = 1.0
schedule.loc[schedule.index[0], "discharge_market_2_mw"] = 0.0
report = validate_schedule(schedule, spec)
assert not report.ok
assert report.checks["hourly_commitment_constant"] is False
def test_catches_a_state_of_charge_that_does_not_follow_the_power_flows(
schedule, spec
):
schedule.loc[schedule.index[10], "soc_mwh"] += 1.5
report = validate_schedule(schedule, spec)
assert not report.ok
assert report.checks["soc_matches_power_flows"] is False
def test_catches_a_state_of_charge_above_capacity(schedule, spec):
"""Energy appearing from nowhere shows up as an out-of-bounds recompute."""
# Charge hard enough to overfill, and keep the reported SoC consistent so
# only the bounds check can fail.
schedule["charge_market_1_mw"] = 2.0
schedule["discharge_market_1_mw"] = 0.0
schedule["charge_market_2_mw"] = 0.0
schedule["discharge_market_2_mw"] = 0.0
delta = 0.5 * spec.charge_efficiency * schedule["charge_market_1_mw"]
schedule["soc_mwh"] = delta.cumsum()
for name in ("market_1", "market_2"):
schedule[f"revenue_{name}_gbp"] = (
0.5
* schedule[f"{name}_price"]
* (schedule[f"discharge_{name}_mw"] - schedule[f"charge_{name}_mw"])
)
report = validate_schedule(schedule, spec)
assert not report.ok
assert report.checks["soc_within_bounds"] is False
assert report.checks["soc_matches_power_flows"] is True
def test_catches_revenue_that_does_not_match_prices_and_powers(schedule, spec):
schedule.loc[schedule.index[5], "revenue_market_1_gbp"] += 25.0
report = validate_schedule(schedule, spec)
assert not report.ok
assert report.checks["revenue_matches_prices_and_powers"] is False
assert report.details["max_revenue_error_gbp"] == pytest.approx(25.0)
def test_catches_a_sign_error_in_the_revenue_convention(schedule, spec):
"""Paying for exports instead of being paid is the classic sign slip."""
schedule["revenue_market_1_gbp"] *= -1
report = validate_schedule(schedule, spec)
assert not report.ok
assert report.checks["revenue_matches_prices_and_powers"] is False
def test_reports_are_falsy_only_when_something_failed(schedule, spec):
good = validate_schedule(schedule, spec)
assert good.ok and not good.failures
schedule.loc[schedule.index[0], "charge_market_1_mw"] = np.float64(9.0)
bad = validate_schedule(schedule, spec)
assert not bad.ok and bad.failures
assert "FAILED" in bad.summary()