"""Tests for loading, cleaning and aligning the two price series.""" from __future__ import annotations from pathlib import Path import numpy as np import openpyxl import pandas as pd import pytest from battery_dispatch.config import BatterySpec from battery_dispatch.data import ( HALF_HOURLY_SHEET, HOURLY_SHEET, build_markets, load_prices, slice_whole_days, ) from battery_dispatch.markets import MARKET_1, MARKET_2 REPO_ROOT = Path(__file__).resolve().parents[1] PRICE_FILE = REPO_ROOT / "data" / "Attachment 2.xlsx" SPEC_FILE = REPO_ROOT / "data" / "Attachment 1.xlsx" requires_data = pytest.mark.skipif( not PRICE_FILE.exists(), reason="Attachment 2.xlsx not present" ) def write_workbook( path: Path, half_hourly: list[tuple], hourly: list[tuple], ) -> Path: """Build a miniature Attachment 2 with the same sheet names and layout.""" workbook = openpyxl.Workbook() sheet = workbook.active sheet.title = HALF_HOURLY_SHEET sheet.append([None, "Market 1 Price [£/MWh]"]) for row in half_hourly: sheet.append(list(row)) second = workbook.create_sheet(HOURLY_SHEET) second.append([None, "Market 2 Price [£/MWh]"]) for row in hourly: second.append(list(row)) workbook.save(path) return path def synthetic_day(tmp_path: Path, trailing_blanks: int = 0) -> Path: """One clean day: 48 half-hours and 24 hours, plus optional blank rows.""" half_hourly = [ (pd.Timestamp("2018-01-01") + pd.Timedelta(minutes=30 * i), float(i)) for i in range(48) ] hourly = [ (pd.Timestamp("2018-01-01") + pd.Timedelta(hours=h), 100.0 + h) for h in range(24) ] hourly += [(None, None)] * trailing_blanks return write_workbook(tmp_path / "prices.xlsx", half_hourly, hourly) def test_trailing_blank_rows_are_dropped(tmp_path): path = synthetic_day(tmp_path, trailing_blanks=24) prices = load_prices(path, expected_days=1) assert len(prices) == 48 def test_hourly_prices_are_broadcast_onto_half_hours(tmp_path): prices = load_prices(synthetic_day(tmp_path), expected_days=1) # Each hourly price covers exactly two consecutive half-hours. assert prices["market_2_price"].iloc[0] == 100.0 assert prices["market_2_price"].iloc[1] == 100.0 assert prices["market_2_price"].iloc[2] == 101.0 assert prices["market_2_price"].iloc[47] == 123.0 pairs = prices["market_2_price"].to_numpy().reshape(-1, 2) assert np.array_equal(pairs[:, 0], pairs[:, 1]) # Market 1 is untouched, and hour_index groups half-hours in twos. assert prices["market_1_price"].tolist() == [float(i) for i in range(48)] assert prices["hour_index"].tolist() == [i // 2 for i in range(48)] def test_row_count_mismatch_is_an_error_not_a_guess(tmp_path): half_hourly = [ (pd.Timestamp("2018-01-01") + pd.Timedelta(minutes=30 * i), 1.0) for i in range(47) # one short ] hourly = [ (pd.Timestamp("2018-01-01") + pd.Timedelta(hours=h), 1.0) for h in range(24) ] path = write_workbook(tmp_path / "short.xlsx", half_hourly, hourly) with pytest.raises(ValueError, match="expected 48 rows"): load_prices(path, expected_days=1) def test_index_is_rebuilt_regularly_despite_clock_change_labels(tmp_path, caplog): """Duplicated 02:00 labels are logged and replaced by a regular index.""" stamps = [ pd.Timestamp("2018-03-25") + pd.Timedelta(minutes=30 * i) for i in range(48) ] # Reproduce the spring-forward labelling: 01:00/01:30 skipped, 02:00/02:30 twice. stamps[2] = pd.Timestamp("2018-03-25 02:00") stamps[3] = pd.Timestamp("2018-03-25 02:30") half_hourly = [(ts, 10.0) for ts in stamps] hourly = [ (pd.Timestamp("2018-03-25") + pd.Timedelta(hours=h), 20.0) for h in range(24) ] path = write_workbook(tmp_path / "clocks.xlsx", half_hourly, hourly) with caplog.at_level("INFO"): prices = load_prices(path, expected_days=1) assert "do not match a regular half-hourly index" in caplog.text expected = pd.date_range("2018-01-01", periods=48, freq="30min") assert prices.index.equals(expected) assert not prices.index.duplicated().any() def test_slice_whole_days_includes_the_whole_end_day(tmp_path): half_hourly = [ (pd.Timestamp("2018-01-01") + pd.Timedelta(minutes=30 * i), float(i)) for i in range(96) ] hourly = [ (pd.Timestamp("2018-01-01") + pd.Timedelta(hours=h), 1.0) for h in range(48) ] prices = load_prices(write_workbook(tmp_path / "two.xlsx", half_hourly, hourly), expected_days=2) first_day = slice_whole_days(prices, "2018-01-01", "2018-01-01") assert len(first_day) == 48 assert first_day.index[-1] == pd.Timestamp("2018-01-01 23:30") # hour_index is re-based so a slice can be optimised standalone. assert first_day["hour_index"].iloc[0] == 0 second_day = slice_whole_days(prices, "2018-01-02", "2018-01-02") assert second_day["hour_index"].iloc[0] == 0 def test_build_markets_sets_the_commitment_block_lengths(tmp_path): prices = load_prices(synthetic_day(tmp_path), expected_days=1) market_1, market_2 = build_markets(prices) assert market_1.name == MARKET_1 and market_1.block_half_hours == 1 assert market_2.name == MARKET_2 and market_2.block_half_hours == 2 assert market_2.block_of(0) == market_2.block_of(1) == 0 assert market_2.block_of(2) == 1 @requires_data def test_real_workbook_loads_with_the_expected_shape(): prices = load_prices(PRICE_FILE) assert len(prices) == 1096 * 48 assert prices.index[0] == pd.Timestamp("2018-01-01 00:00") assert prices.index[-1] == pd.Timestamp("2020-12-31 23:30") assert prices.index.is_monotonic_increasing assert not prices.index.duplicated().any() assert prices.notna().all().all() pairs = prices["market_2_price"].to_numpy().reshape(-1, 2) assert np.array_equal(pairs[:, 0], pairs[:, 1]) @requires_data def test_spec_reads_losses_as_efficiencies(): """Attachment 1 quotes losses; the model must store 1 - loss.""" spec = BatterySpec.from_excel(SPEC_FILE) assert spec.max_charge_mw == 2.0 assert spec.max_discharge_mw == 2.0 assert spec.capacity_mwh == 4.0 assert spec.charge_efficiency == pytest.approx(0.95) assert spec.discharge_efficiency == pytest.approx(0.95) assert spec.round_trip_efficiency == pytest.approx(0.9025) assert spec.lifetime_years == 10.0 assert spec.lifetime_cycles == 5000.0 assert spec.degradation_fraction_per_cycle == pytest.approx(1e-5) assert spec.capex_gbp == 500_000.0 assert spec.fixed_opex_gbp_per_year == 5_000.0