Files
challenge_ai/tests/test_optimiser.py
T
mattandClaude Sonnet 5 afa02864a7
CI / test (push) Successful in 12s
Fix test suite hangs caused by degenerate MILP fixtures
Two tests were spinning up real CBC solves over 96-half-hour windows
with long runs of exactly-repeated prices (e.g. [0.0] * 40). That gives
the LP relaxation a huge set of economically indistinguishable ways to
spread a trade, which the MILP fallback's branch-and-bound then wastes
enormous effort disambiguating (47k+ nodes without closing the gap,
confirmed by running CBC verbosely). Real Attachment 2 data has no such
flat runs and solves in ~0.1s/window; a synthetic sine wiggle wasn't
enough either, since neighbouring half-hours stayed too similar.

Fixes, matched to what each test actually needs:
- The 9 validator tests only need *some* structurally valid schedule to
  mutate; they were deriving it by running the real optimiser once per
  test. Replaced with make_valid_schedule(), built directly from the
  model's own energy-balance formulas -- no solver involved, and it's
  now a true unit test of validate_schedule() in isolation.
- The rolling-horizon carry-forward test was exercising the mechanism
  at full production scale (48h window / 24h commit) when a 2h/1h
  window proves the same boundary-carrying behaviour with a trivial
  MILP, regardless of price structure.
- Fixed a genuine tie in test_optimum_uses_whichever_market_pays_more:
  two equal-price hours with just enough stored energy for one meant
  either market was a valid optimum. Sized the charge phase so delivery
  must split across both hours, pinning a unique answer.

Full suite: 38 passed in ~1.3s (previously hung indefinitely on CI and
locally within seconds of the same wall-clock variance CBC shows on
degenerate MIPs).

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-09-24 16:10:55 +01:00

175 lines
7.5 KiB
Python

"""Optimiser tests built on small, hand-checkable cases."""
from __future__ import annotations
import numpy as np
import pytest
from conftest import markets_from, prices_frame
from battery_dispatch.config import RunConfig
from battery_dispatch.markets import MARKET_1, MARKET_2
from battery_dispatch.optimiser import (
POWER_TOLERANCE,
run_rolling_horizon,
solve_window,
)
def total_power(solution, kind: str) -> np.ndarray:
values = solution.charge_mw if kind == "charge" else solution.discharge_mw
return sum(values.values())
def test_two_period_arbitrage_matches_analytic_revenue(spec, config):
"""£0 then £100 in Market 1: revenue is pinned by the round-trip efficiency.
Charging at 2 MW for half an hour imports 1 MWh, of which 0.95 MWh is
stored; discharging all of it delivers 0.95 x 0.95 = 0.9025 MWh to the grid
at £100/MWh. Market 2 is priced at £50 for the hour, which is strictly
worse in both directions, so it must stay out of the optimum.
"""
markets = markets_from([0.0, 100.0], [50.0])
solution, _ = solve_window(markets, spec, 0.0, spec.capacity_mwh, config)
assert solution.objective_gbp == pytest.approx(90.25, abs=0.01)
# The whole trade sits in Market 1.
assert solution.charge_mw[MARKET_1] == pytest.approx([2.0, 0.0], abs=1e-6)
assert solution.discharge_mw[MARKET_1] == pytest.approx([0.0, 1.805], abs=1e-6)
assert solution.charge_mw[MARKET_2] == pytest.approx([0.0, 0.0], abs=1e-6)
assert solution.discharge_mw[MARKET_2] == pytest.approx([0.0, 0.0], abs=1e-6)
# Energy delivered to the grid, cross-checked independently.
delivered = 0.5 * solution.discharge_mw[MARKET_1][1]
assert delivered == pytest.approx(1.0 * 0.95 * 0.95, abs=1e-6)
def test_market_2_power_is_constant_within_an_hour(spec, config):
"""Opposite Market 1 signals inside one hour cannot bend Market 2's power."""
# Market 1 pays to charge in the first half-hour and pays well to discharge
# in the second; Market 2 sits at a flat £50 across the whole hour.
markets = markets_from([-100.0, 200.0, 0.0, 0.0], [50.0, 10.0])
solution, _ = solve_window(markets, spec, 2.0, spec.capacity_mwh, config)
for series in (solution.charge_mw[MARKET_2], solution.discharge_mw[MARKET_2]):
assert series[0] == pytest.approx(series[1], abs=1e-9)
assert series[2] == pytest.approx(series[3], abs=1e-9)
def test_milp_fallback_removes_cross_market_simultaneous_flow(spec):
"""Paid to charge in one market and to discharge in the other at once.
The LP relaxation takes both sides of that trade, which no single battery
can do. The MILP fallback must detect and eliminate it.
"""
markets = markets_from([-50.0, -50.0], [50.0])
lp_solution, _ = solve_window(
markets, spec, 0.0, spec.capacity_mwh, RunConfig(solver_mode="lp-only")
)
cheating = (total_power(lp_solution, "charge") > POWER_TOLERANCE) & (
total_power(lp_solution, "discharge") > POWER_TOLERANCE
)
assert cheating.any(), "expected the relaxation to charge and discharge at once"
solution, used_milp = solve_window(
markets, spec, 0.0, spec.capacity_mwh, RunConfig(solver_mode="auto")
)
assert used_milp, "the fallback should have been triggered"
honest = (total_power(solution, "charge") > POWER_TOLERANCE) & (
total_power(solution, "discharge") > POWER_TOLERANCE
)
assert not honest.any()
assert solution.objective_gbp <= lp_solution.objective_gbp + 1e-6
def test_combined_power_respects_the_two_megawatt_cap(spec, config):
"""Both markets attractive at once: the cap applies to their sum, not each."""
# A full battery and a high price in both markets. How the power splits
# between them is a tie, but the total is capped at 2 MW either way.
markets = markets_from([1000.0, 1000.0], [1000.0])
solution, _ = solve_window(markets, spec, spec.capacity_mwh, spec.capacity_mwh, config)
combined = total_power(solution, "discharge")
assert combined.max() <= spec.max_discharge_mw + POWER_TOLERANCE
assert combined[0] == pytest.approx(spec.max_discharge_mw, abs=1e-6)
assert total_power(solution, "charge").max() <= POWER_TOLERANCE
def test_optimum_uses_whichever_market_pays_more_in_each_hour(spec, config):
"""Across hours the model switches markets rather than favouring one.
Two cheap hours fill the battery to capacity (3.8 MWh stored). Delivering
that much exceeds what a single hour can carry at the 2 MW cap, so the
discharge must split across both peak hours regardless of price -- what
is under test is *which market* gets each hour's discharge, which each
hour's own prices pin unambiguously (300 vs 10 in both cases, not a tie).
"""
markets = markets_from(
[0.0, 0.0, 0.0, 0.0, 300.0, 300.0, 10.0, 10.0],
[0.0, 0.0, 10.0, 300.0],
)
solution, _ = solve_window(markets, spec, 0.0, spec.capacity_mwh, config)
assert solution.discharge_mw[MARKET_1][4:6].sum() > 0, "hour 2 should sell into M1"
assert solution.discharge_mw[MARKET_2][6:8].sum() > 0, "hour 3 should sell into M2"
assert total_power(solution, "discharge").max() <= spec.max_discharge_mw + 1e-6
def test_negative_prices_make_charging_profitable(spec, config):
"""A negative price is an inducement to import, not merely a cheap one."""
markets = markets_from([-80.0, -80.0], [-80.0])
solution, _ = solve_window(markets, spec, 0.0, spec.capacity_mwh, config)
assert total_power(solution, "charge")[0] == pytest.approx(2.0, abs=1e-6)
assert solution.objective_gbp > 0
def test_degradation_cost_suppresses_marginal_cycling(spec):
"""Pricing cycle life makes a thin spread not worth taking."""
# A £12/MWh gross spread: profitable at zero degradation cost, not at £40.
prices = [0.0, 0.0, 12.0, 12.0]
markets = markets_from(prices, [0.0, 12.0])
free, _ = solve_window(markets, spec, 0.0, spec.capacity_mwh, RunConfig())
priced, _ = solve_window(
markets, spec, 0.0, spec.capacity_mwh,
RunConfig(degradation_cost_gbp_per_mwh=40.0),
)
assert total_power(free, "discharge").sum() > 0
assert total_power(priced, "discharge").sum() == pytest.approx(0.0, abs=1e-6)
def test_rolling_horizon_carries_state_across_windows(spec):
"""State of charge at a commit boundary is the next window's starting point.
Small on purpose: this is testing ``run_rolling_horizon``'s bookkeeping
(does the committed end-of-window SoC become the next window's start?),
not the optimiser's economics, so it uses a 2-hour window / 1-hour commit
rather than the production 48h/24h -- a real rolling-horizon MILP solve
at production scale belongs in a slower, separately-run integration
check, not in the unit suite.
"""
# Cheap for two hours (charge), expensive for two hours (discharge), the
# split falling on the commit boundary.
prices = prices_frame([0.0] * 4 + [100.0] * 4, [10.0] * 4)
config = RunConfig(window_hours=4, commit_hours=2)
result, state, stats = run_rolling_horizon(prices, spec, config)
assert stats.windows == 2
assert len(result) == 8
# The lookahead must carry energy over the boundary to reach the peak.
soc_at_boundary = result["soc_mwh"].iloc[3]
assert soc_at_boundary > 1.0
assert state.equivalent_full_cycles > 0
def test_window_start_must_align_to_hour_blocks(spec):
"""Slicing a window mid-hour would silently break Market 2's commitment."""
markets = markets_from([1.0, 2.0, 3.0, 4.0], [1.0, 2.0])
with pytest.raises(ValueError, match="block boundary"):
markets[1].slice(1, 3)