From adef12f6f23849b3632e6f93cd5c7b180ff41cf3 Mon Sep 17 00:00:00 2001 From: Fredrik Ahlgren Date: Thu, 1 Oct 2026 19:10:02 +0200 Subject: [PATCH] fix(forecast): score forecasts per pipeline, not per Core build MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The evaluation cohort hashed the Core version together with the learning revision, the Energyplan worker and the pipeline policy. Error bands (NewCalibrator) and the previous-day and persistence baselines only use samples from the current cohort, so every update started them from nothing. Empirical bands need 48 samples over 7 days; with weekly betas the forecast margin never left its cold-start widths (35–50 % of load). The cohort now keys on the learning revision, the worker build and forecastPipelinePolicy. Bump the policy when Core changes what reaches the planner. Fixes srcfl/ftw#1489 Co-Authored-By: Claude Opus 5.5 Signed-off-by: Fredrik Ahlgren --- .changeset/forecast-bands-survive-updates.md | 5 +++ docs/architecture.md | 3 +- go/cmd/ftw/forecast_site.go | 7 +++- go/cmd/ftw/forecast_site_test.go | 35 ++++++++++++++++++++ 4 files changed, 48 insertions(+), 2 deletions(-) create mode 100644 .changeset/forecast-bands-survive-updates.md diff --git a/.changeset/forecast-bands-survive-updates.md b/.changeset/forecast-bands-survive-updates.md new file mode 100644 index 000000000..6c7c42f37 --- /dev/null +++ b/.changeset/forecast-bands-survive-updates.md @@ -0,0 +1,5 @@ +--- +"ftw": patch +--- + +Forecast error bands and baselines now carry over across Core updates that leave forecasting alone. Before, each update started them over, so the planner's forecast margin always used its widest cold-start bands. diff --git a/docs/architecture.md b/docs/architecture.md index 3207421b7..4f316eb7d 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -255,7 +255,8 @@ state has been saved, and startup restores that saved state. The learning revision binds state to forecast inputs and stable hardware identities. A binding or input change starts fresh learning, while a compatible program upgrade can reuse the state. Issued forecasts use a stricter revision that also -includes the Core build, worker bytes and pipeline policy. +includes the worker bytes and pipeline policy, so a Core update that leaves +forecasting alone keeps its scored errors and bands. Core keeps issued forecasts, frozen inputs, model-state references and qualified truth in a bounded local archive. The read-only `ftw-forecast-evaluate` source diff --git a/go/cmd/ftw/forecast_site.go b/go/cmd/ftw/forecast_site.go index fdf994a7a..cde0343aa 100644 --- a/go/cmd/ftw/forecast_site.go +++ b/go/cmd/ftw/forecast_site.go @@ -46,6 +46,11 @@ type forecastIdentityReceipt struct { } const forecastIdentityReceiptKey = "forecast/live_identity_v1" + +// forecastPipelinePolicy names how Core composes the forecast it plans with. +// The evaluation cohort keys on it rather than on the Core version, so error +// bands and baselines survive updates that leave forecasting alone. Bump it +// whenever Core changes what reaches the planner. const forecastPipelinePolicy = "energyplan-primary-v2" func newForecastSiteConfig(st *state.Store) *forecastSiteConfig { @@ -212,7 +217,7 @@ func (s *forecastSiteConfig) RefreshIdentity(now time.Time) bool { learningBase = s.accepted.LearningBaseRevision } learning := fmt.Sprintf("site-v2:%x", sha256.Sum256([]byte(learningBase+"/"+string(data)))) - cohort := learning + "/" + Version + "/" + s.engineVersion + "/" + forecastPipelinePolicy + cohort := learning + "/" + s.engineVersion + "/" + forecastPipelinePolicy revision := fmt.Sprintf("forecast-v1:%x", sha256.Sum256([]byte(cohort))) opts := s.baseOptions if pending && opts.HouseholdInvalidReason == "" { diff --git a/go/cmd/ftw/forecast_site_test.go b/go/cmd/ftw/forecast_site_test.go index 4314bb305..de8c4e316 100644 --- a/go/cmd/ftw/forecast_site_test.go +++ b/go/cmd/ftw/forecast_site_test.go @@ -437,3 +437,38 @@ func TestForecastLearningIDCarriesAcrossPortablePathPolicy(t *testing.T) { t.Fatalf("receipt learning base = %q, want %q", s.accepted.LearningBaseRevision, old) } } + +// Error bands and baselines are scored per evaluation cohort. A Core update +// that leaves the site, the worker and the pipeline policy alone must keep +// the cohort, or the bands never leave their cold-start width (#1489). +func TestForecastEvaluationCohortSurvivesCoreUpdate(t *testing.T) { + st, err := state.Open(filepath.Join(t.TempDir(), "state.db")) + if err != nil { + t.Fatal(err) + } + defer st.Close() + script := filepath.Join(t.TempDir(), "meter.lua") + if err := os.WriteFile(script, []byte("measurement code"), 0600); err != nil { + t.Fatal(err) + } + cfg := &config.Config{ + Drivers: []config.Driver{{Name: "meter", Lua: script, IsSiteMeter: true}}, + Weather: &config.Weather{Provider: "open_meteo", Latitude: 59, Longitude: 18}, + } + previous := Version + defer func() { Version = previous }() + s := newForecastSiteConfig(st) + Version = "v0.138.0-beta.1" + s.Configure(cfg, nil) + before := s.Snapshot() + Version = "v0.138.1-beta.1" + s.Configure(cfg, nil) + if got := s.Snapshot().Revision; got != before.Revision { + t.Fatal("a Core update without forecast changes started a new evaluation cohort") + } + s.engineVersion = "another-worker-build" + s.Configure(cfg, nil) + if s.Snapshot().Revision == before.Revision { + t.Fatal("a new forecast worker kept the old evaluation cohort") + } +}