Fix test suite hangs caused by degenerate MILP fixtures
CI / test (push) Successful in 12s

Two tests were spinning up real CBC solves over 96-half-hour windows
with long runs of exactly-repeated prices (e.g. [0.0] * 40). That gives
the LP relaxation a huge set of economically indistinguishable ways to
spread a trade, which the MILP fallback's branch-and-bound then wastes
enormous effort disambiguating (47k+ nodes without closing the gap,
confirmed by running CBC verbosely). Real Attachment 2 data has no such
flat runs and solves in ~0.1s/window; a synthetic sine wiggle wasn't
enough either, since neighbouring half-hours stayed too similar.

Fixes, matched to what each test actually needs:
- The 9 validator tests only need *some* structurally valid schedule to
  mutate; they were deriving it by running the real optimiser once per
  test. Replaced with make_valid_schedule(), built directly from the
  model's own energy-balance formulas -- no solver involved, and it's
  now a true unit test of validate_schedule() in isolation.
- The rolling-horizon carry-forward test was exercising the mechanism
  at full production scale (48h window / 24h commit) when a 2h/1h
  window proves the same boundary-carrying behaviour with a trivial
  MILP, regardless of price structure.
- Fixed a genuine tie in test_optimum_uses_whichever_market_pays_more:
  two equal-price hours with just enough stored energy for one meant
  either market was a valid optimum. Sized the charge phase so delivery
  must split across both hours, pinning a unique answer.

Full suite: 38 passed in ~1.3s (previously hung indefinitely on CI and
locally within seconds of the same wall-clock variance CBC shows on
degenerate MIPs).

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
2026-09-24 16:10:55 +01:00
co-authored by Claude Sonnet 5
parent 298724a9d3
commit afa02864a7
3 changed files with 91 additions and 27 deletions
+58 -1
View File
@@ -2,11 +2,13 @@
from __future__ import annotations from __future__ import annotations
import math
import numpy as np import numpy as np
import pandas as pd import pandas as pd
import pytest import pytest
from battery_dispatch.config import BatterySpec, RunConfig from battery_dispatch.config import HALF_HOUR, BatterySpec, RunConfig
from battery_dispatch.markets import MARKET_1, MARKET_2, Market from battery_dispatch.markets import MARKET_1, MARKET_2, Market
# Equal to Attachment 1, but stated literally so the tests do not depend on the # Equal to Attachment 1, but stated literally so the tests do not depend on the
@@ -73,3 +75,58 @@ def prices_frame(
}, },
index=index, index=index,
) )
def make_valid_schedule(spec: BatterySpec, n: int = 48) -> pd.DataFrame:
"""A schedule (shaped like the optimiser's output) that satisfies every
invariant ``validate_schedule`` checks -- built directly from the model's
own energy-balance and revenue formulas, not by running the optimiser.
``validate_schedule`` is tested in isolation: it only needs *some*
structurally valid schedule to mutate one violation into. Deriving that
from a real rolling-horizon solve would mean spinning up CBC once per
test for no reason connected to what is under test, so this constructs
one by hand instead -- charge for two hours, sit idle, discharge for two
hours, comfortably within every power and capacity bound.
"""
charge1 = np.zeros(n)
charge2 = np.zeros(n)
discharge1 = np.zeros(n)
discharge2 = np.zeros(n)
charge1[0:4] = 1.0
charge2[0:4] = 0.5
discharge1[36:40] = 0.5
discharge2[36:40] = 0.3
total_charge = charge1 + charge2
total_discharge = discharge1 + discharge2
delta = HALF_HOUR * (
spec.charge_efficiency * total_charge - total_discharge / spec.discharge_efficiency
)
soc = np.cumsum(delta)
assert soc.min() >= -1e-9 and soc.max() <= spec.capacity_mwh - 1e-9, (
"fixture parameters must stay clear of the battery's bounds"
)
price1 = np.array([40.0 + 10.0 * math.sin(i / 5) for i in range(n)])
price2_hourly = np.array([35.0 + 8.0 * math.sin(h / 4) for h in range(n // 2)])
price2 = np.repeat(price2_hourly, 2)
frame = pd.DataFrame(
{
"market_1_price": price1,
"market_2_price": price2,
"charge_market_1_mw": charge1,
"discharge_market_1_mw": discharge1,
"charge_market_2_mw": charge2,
"discharge_market_2_mw": discharge2,
"soc_mwh": soc,
"capacity_mwh": np.full(n, spec.capacity_mwh),
},
index=pd.date_range("2018-01-01", periods=n, freq="30min"),
)
frame.index.name = "timestamp"
for name in (MARKET_1, MARKET_2):
net_export = frame[f"discharge_{name}_mw"] - frame[f"charge_{name}_mw"]
frame[f"revenue_{name}_gbp"] = HALF_HOUR * frame[f"{name}_price"] * net_export
return frame
+28 -16
View File
@@ -98,16 +98,22 @@ def test_combined_power_respects_the_two_megawatt_cap(spec, config):
def test_optimum_uses_whichever_market_pays_more_in_each_hour(spec, config): def test_optimum_uses_whichever_market_pays_more_in_each_hour(spec, config):
"""Across hours the model switches markets rather than favouring one.""" """Across hours the model switches markets rather than favouring one.
# Hour 0 cheap in both (charge); hour 1 Market 1 pays best; hour 2 Market 2 does.
Two cheap hours fill the battery to capacity (3.8 MWh stored). Delivering
that much exceeds what a single hour can carry at the 2 MW cap, so the
discharge must split across both peak hours regardless of price -- what
is under test is *which market* gets each hour's discharge, which each
hour's own prices pin unambiguously (300 vs 10 in both cases, not a tie).
"""
markets = markets_from( markets = markets_from(
[0.0, 0.0, 300.0, 300.0, 10.0, 10.0], [0.0, 0.0, 0.0, 0.0, 300.0, 300.0, 10.0, 10.0],
[0.0, 10.0, 300.0], [0.0, 0.0, 10.0, 300.0],
) )
solution, _ = solve_window(markets, spec, 0.0, spec.capacity_mwh, config) solution, _ = solve_window(markets, spec, 0.0, spec.capacity_mwh, config)
assert solution.discharge_mw[MARKET_1][2:4].sum() > 0, "hour 1 should sell into M1" assert solution.discharge_mw[MARKET_1][4:6].sum() > 0, "hour 2 should sell into M1"
assert solution.discharge_mw[MARKET_2][4:6].sum() > 0, "hour 2 should sell into M2" assert solution.discharge_mw[MARKET_2][6:8].sum() > 0, "hour 3 should sell into M2"
assert total_power(solution, "discharge").max() <= spec.max_discharge_mw + 1e-6 assert total_power(solution, "discharge").max() <= spec.max_discharge_mw + 1e-6
@@ -137,21 +143,27 @@ def test_degradation_cost_suppresses_marginal_cycling(spec):
def test_rolling_horizon_carries_state_across_windows(spec): def test_rolling_horizon_carries_state_across_windows(spec):
"""State of charge at a commit boundary is the next window's starting point.""" """State of charge at a commit boundary is the next window's starting point.
# Two days: charge cheaply late on day 1, sell into the day-2 morning peak.
day = [0.0] * 40 + [5.0] * 8
day_2 = [200.0] * 8 + [50.0] * 40
prices = prices_frame(day + day_2, [40.0] * 48)
config = RunConfig(window_hours=48, commit_hours=24) Small on purpose: this is testing ``run_rolling_horizon``'s bookkeeping
(does the committed end-of-window SoC become the next window's start?),
not the optimiser's economics, so it uses a 2-hour window / 1-hour commit
rather than the production 48h/24h -- a real rolling-horizon MILP solve
at production scale belongs in a slower, separately-run integration
check, not in the unit suite.
"""
# Cheap for two hours (charge), expensive for two hours (discharge), the
# split falling on the commit boundary.
prices = prices_frame([0.0] * 4 + [100.0] * 4, [10.0] * 4)
config = RunConfig(window_hours=4, commit_hours=2)
result, state, stats = run_rolling_horizon(prices, spec, config) result, state, stats = run_rolling_horizon(prices, spec, config)
assert stats.windows == 2 assert stats.windows == 2
assert len(result) == 96 assert len(result) == 8
# The lookahead must carry energy over midnight to reach the day-2 peak. # The lookahead must carry energy over the boundary to reach the peak.
soc_at_midnight = result["soc_mwh"].iloc[47] soc_at_boundary = result["soc_mwh"].iloc[3]
assert soc_at_midnight > 1.0 assert soc_at_boundary > 1.0
assert state.equivalent_full_cycles > 0 assert state.equivalent_full_cycles > 0
+5 -10
View File
@@ -5,22 +5,17 @@ from __future__ import annotations
import numpy as np import numpy as np
import pandas as pd import pandas as pd
import pytest import pytest
from conftest import prices_frame from conftest import make_valid_schedule
from battery_dispatch.config import RunConfig
from battery_dispatch.optimiser import run_rolling_horizon
from battery_dispatch.validation import validate_schedule from battery_dispatch.validation import validate_schedule
@pytest.fixture @pytest.fixture
def schedule(spec) -> pd.DataFrame: def schedule(spec) -> pd.DataFrame:
"""A genuine, valid two-day schedule to mutate in each test.""" """A genuine, valid schedule to mutate in each test. See
prices = prices_frame( ``conftest.make_valid_schedule`` for why this isn't a real optimiser run.
[0.0] * 20 + [90.0] * 8 + [10.0] * 20 + [5.0] * 16 + [120.0] * 8 + [40.0] * 24, """
[30.0] * 48, return make_valid_schedule(spec)
)
result, _, _ = run_rolling_horizon(prices, spec, RunConfig())
return result
def test_a_valid_schedule_passes_every_check(schedule, spec): def test_a_valid_schedule_passes_every_check(schedule, spec):