From 594d2027f5dcb34cd4074dc517f781027b1ded86 Mon Sep 17 00:00:00 2001 From: Sivan Reddy Pushpagiri Date: Fri, 18 Sep 2026 16:32:05 -0400 Subject: [PATCH 1/2] Expose regime-conditional performance through the validation stack MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit regime_conditional_performance and RegimeResult were ported with the overfitting statistics but never reachable: exported from the domain package and called by nothing, with no tests. Wire them through the layers: - AnalyzeRegimes use case in validation/application, converting EquityCurve/pandas/array input via the existing returns bridge - AlgoSystem.analyze_regimes() and an `algosystem regimes` CLI command - RegimeResult.to_metrics(), keyed by ValidationMetricKey The statistics degrade quietly on thin input: a series no longer than vol_window collapses to a single "all" regime, and any bucket under five observations scores 0.0, so a data shortage reads like a finding ("Sharpe 0.0, regime-dependent"). AnalyzeRegimes rejects input it cannot describe instead. Regimes are assigned positionally, so dated series are checked for matching indices rather than matching length alone; the CLI restricts strategy and benchmark to the dates they share. Add REGIME_COUNTS, WORST_REGIME_NAME and WORKS_IN_ALL_REGIMES to ValidationMetricKey so to_metrics() is lossless — the counts are what distinguish an empty bucket from a genuine loss. The ported numerics are unchanged. --- CHANGELOG.md | 5 + README.md | 1 + algosystem/__init__.py | 5 + algosystem/interfaces/api.py | 30 +- algosystem/interfaces/cli/main.py | 93 +++++- algosystem/validation/__init__.py | 10 + algosystem/validation/application/__init__.py | 2 + .../validation/application/analyze_regimes.py | 199 ++++++++++++ .../domain/statistics/robustness.py | 26 ++ .../validation/domain/validation_metric.py | 3 + docs/API_GUIDE.md | 29 ++ docs/CLI_GUIDE.md | 26 ++ docs/VALIDATION_GUIDE.md | 74 +++++ tests/test_cli.py | 11 +- tests/test_cli_integration.py | 171 +++++++++++ .../application/test_analyze_regimes.py | 284 ++++++++++++++++++ .../application/test_analyze_regimes_api.py | 110 +++++++ tests/validation/conftest.py | 27 ++ .../domain/test_robustness_regimes.py | 242 +++++++++++++++ 19 files changed, 1345 insertions(+), 3 deletions(-) create mode 100644 algosystem/validation/application/analyze_regimes.py create mode 100644 tests/validation/application/test_analyze_regimes.py create mode 100644 tests/validation/application/test_analyze_regimes_api.py create mode 100644 tests/validation/domain/test_robustness_regimes.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 9c6a02c..182c233 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,11 @@ ## Unreleased +- Exposed regime-conditional performance analysis through the validation stack: + an `AnalyzeRegimes` use case, `AlgoSystem.analyze_regimes()`, an `algosystem + regimes` CLI command, and a `RegimeResult.to_metrics()` accessor keyed by + `ValidationMetricKey`. The underlying `regime_conditional_performance` + statistics were already present but unreachable from any public entry point. - Added the validation context for overfitting detection, porting John Riley's statistical core, shipped strategy archetypes, chart/report adapters, CLI commands, and facade methods into the diff --git a/README.md b/README.md index 269f5bd..be17310 100644 --- a/README.md +++ b/README.md @@ -79,6 +79,7 @@ algosystem backtest strategy.csv --price-column Strategy --detailed algosystem tearsheet strategy.csv --price-column Strategy --output tearsheet.html algosystem validate strategy.csv --strategy momentum --reps 200 --seed 7 --output overfit.html algosystem validate-strategies +algosystem regimes strategy.csv --price-column Strategy algosystem benchmarks algosystem db save strategy.csv --price-column Strategy --name strategy-v1 ``` diff --git a/algosystem/__init__.py b/algosystem/__init__.py index 44f7d15..01dfb65 100644 --- a/algosystem/__init__.py +++ b/algosystem/__init__.py @@ -31,6 +31,10 @@ "Ratio": ("algosystem.shared.values", "Ratio"), "RunId": ("algosystem.shared.values", "RunId"), "OverfitResults": ("algosystem.validation.domain.results", "OverfitResults"), + "RegimeResult": ( + "algosystem.validation.domain.statistics.robustness", + "RegimeResult", + ), "ParameterGrid": ("algosystem.validation.domain.strategy", "ParameterGrid"), "StrategySpec": ("algosystem.validation.domain.strategy", "StrategySpec"), "ValidationMetricKey": ( @@ -71,6 +75,7 @@ def __getattr__(name: str) -> object: "RepositoryError", "RunId", "OverfitResults", + "RegimeResult", "ParameterGrid", "StrategySpec", "ValidationError", diff --git a/algosystem/interfaces/api.py b/algosystem/interfaces/api.py index 6aab581..60f281b 100644 --- a/algosystem/interfaces/api.py +++ b/algosystem/interfaces/api.py @@ -4,7 +4,7 @@ import warnings from pathlib import Path -from typing import Mapping, Optional, Sequence +from typing import TYPE_CHECKING, Mapping, Optional, Sequence import pandas as pd from rich.console import Console @@ -36,6 +36,10 @@ from algosystem.shared.metric_key import MetricKey from algosystem.shared.values import RunId +if TYPE_CHECKING: + from algosystem.validation.application.equity_curve_bridge import SeriesKind + from algosystem.validation.domain.statistics.robustness import RegimeResult + console = Console() @@ -122,6 +126,30 @@ def detect_overfitting( n_workers=n_workers, ) + def analyze_regimes( + self, + returns: object, + *, + market_returns: object = None, + n_regimes: int = 3, + vol_window: int = 21, + annualize: float = 252.0, + input_kind: "SeriesKind" = "returns", + market_input_kind: Optional["SeriesKind"] = None, + ) -> "RegimeResult": + """Break realized performance down by volatility regime.""" + from algosystem.validation.application.analyze_regimes import AnalyzeRegimes + + return AnalyzeRegimes().execute( + returns, + market_returns=market_returns, + n_regimes=n_regimes, + vol_window=vol_window, + annualize=annualize, + input_kind=input_kind, + market_input_kind=market_input_kind, + ) + def validation_report( self, report: object, diff --git a/algosystem/interfaces/cli/main.py b/algosystem/interfaces/cli/main.py index d6e05eb..b8f00da 100644 --- a/algosystem/interfaces/cli/main.py +++ b/algosystem/interfaces/cli/main.py @@ -15,8 +15,9 @@ from algosystem import __version__ from algosystem.interfaces.api import AlgoSystem -from algosystem.shared.errors import AlgoSystemError +from algosystem.shared.errors import AlgoSystemError, ValidationError from algosystem.shared.values import RunId +from algosystem.validation.domain.validation_metric import ValidationMetricKey from .loaders import load_benchmark_input, load_prices @@ -184,6 +185,63 @@ def validate( raise click.ClickException(str(exc)) from exc +@cli.command() +@click.argument("input_file", type=click.Path(exists=True, dir_okay=False, path_type=Path)) +@click.option("--benchmark", help="Benchmark alias or CSV file used to define the regimes.") +@click.option( + "--regimes", + "n_regimes", + type=int, + default=3, + show_default=True, + help="Number of volatility buckets.", +) +@click.option( + "--vol-window", + type=int, + default=21, + show_default=True, + help="Rolling window for the volatility estimate.", +) +@click.option("--start", help="Start date, e.g. 2022-01-01.") +@click.option("--end", help="End date, e.g. 2023-01-01.") +@click.option("--price-column", help="Price/equity column for multi-column CSV files.") +def regimes( + input_file: Path, + benchmark: Optional[str], + n_regimes: int, + vol_window: int, + start: Optional[str], + end: Optional[str], + price_column: Optional[str], +) -> None: + """Break performance down by volatility regime from a CSV price/equity series.""" + try: + prices = load_prices(input_file) + series = _select_price_series(prices, price_column=price_column, start=start, end=end) + from algosystem.backtesting.domain.equity_curve import EquityCurve + + benchmark_series = load_benchmark_input(benchmark, start=start, end=end) + if benchmark_series is not None: + series, benchmark_aligned = _intersect_on_dates(series, benchmark_series) + market_curve = EquityCurve.from_series(benchmark_aligned) + else: + market_curve = None + equity_curve = EquityCurve.from_series(series) + + from algosystem.validation.application.analyze_regimes import AnalyzeRegimes + + result = AnalyzeRegimes().execute( + equity_curve, + market_returns=market_curve, + n_regimes=n_regimes, + vol_window=vol_window, + ) + _print_regime_summary(result) + except (AlgoSystemError, ValueError) as exc: + raise click.ClickException(str(exc)) from exc + + @cli.command("validate-strategies") def validate_strategies() -> None: """List shipped validation strategy archetypes and default grids.""" @@ -395,6 +453,39 @@ def _format_param_grid(grid: dict[str, list[object]]) -> str: return "; ".join(f"{name}={values}" for name, values in grid.items()) +def _intersect_on_dates(series, benchmark): + """Restrict both series to the dates they share, so regimes align positionally.""" + common = series.index.intersection(benchmark.index) + if len(common) < 2: + raise ValidationError( + "strategy and benchmark share fewer than two dates; " + "check the benchmark covers the selected range" + ) + return series.loc[common].sort_index(), benchmark.loc[common].sort_index() + + +def _print_regime_summary(result: object) -> None: + metrics = result.to_metrics() + sharpes = metrics[ValidationMetricKey.REGIME_SHARPES.value] + fracs = metrics[ValidationMetricKey.REGIME_FRAC.value] + counts = metrics[ValidationMetricKey.REGIME_COUNTS.value] + + table = Table(title="Regime-Conditional Performance") + table.add_column("Regime", style="cyan") + table.add_column("Sharpe", style="green", justify="right") + table.add_column("Time", style="green", justify="right") + table.add_column("Observations", style="green", justify="right") + for name in sharpes: + table.add_row( + name, + f"{sharpes[name]:.4f}", + _format_percent(fracs[name]), + str(counts[name]), + ) + console.print(table) + console.print(Panel("\n".join(result.summary()), title="Regime Report", expand=False)) + + def _print_validation_summary(results: object) -> None: table = Table(title="Validation Results") table.add_column("Metric", style="cyan") diff --git a/algosystem/validation/__init__.py b/algosystem/validation/__init__.py index 40a1e75..5b5d06c 100644 --- a/algosystem/validation/__init__.py +++ b/algosystem/validation/__init__.py @@ -5,6 +5,10 @@ from importlib import import_module _LAZY_EXPORTS = { + "AnalyzeRegimes": ( + "algosystem.validation.application.analyze_regimes", + "AnalyzeRegimes", + ), "DetectOverfitting": ( "algosystem.validation.application.detect_overfitting", "DetectOverfitting", @@ -13,6 +17,10 @@ "OverfitResults": ("algosystem.validation.domain.results", "OverfitResults"), "ParameterGrid": ("algosystem.validation.domain.strategy", "ParameterGrid"), "ParameterSet": ("algosystem.validation.domain.strategy", "ParameterSet"), + "RegimeResult": ( + "algosystem.validation.domain.statistics.robustness", + "RegimeResult", + ), "SignalAnalysisReport": ( "algosystem.validation.domain.statistics.signal_analyzer", "SignalAnalysisReport", @@ -38,11 +46,13 @@ def __getattr__(name: str) -> object: __all__ = [ + "AnalyzeRegimes", "DetectOverfitting", "OverfitDetector", "OverfitResults", "ParameterGrid", "ParameterSet", + "RegimeResult", "SignalAnalysisReport", "SignalAnalyzer", "StrategySpec", diff --git a/algosystem/validation/application/__init__.py b/algosystem/validation/application/__init__.py index 8e3e0bb..79e3c28 100644 --- a/algosystem/validation/application/__init__.py +++ b/algosystem/validation/application/__init__.py @@ -1,5 +1,6 @@ """Validation application use cases.""" +from .analyze_regimes import AnalyzeRegimes from .detect_overfitting import DetectOverfitting from .equity_curve_bridge import levels_from_returns, returns_from from .render_report import RenderValidationReport @@ -8,6 +9,7 @@ from .signal_analyzer import SignalAnalyzer __all__ = [ + "AnalyzeRegimes", "DetectOverfitting", "RenderValidationReport", "RunWalkForward", diff --git a/algosystem/validation/application/analyze_regimes.py b/algosystem/validation/application/analyze_regimes.py new file mode 100644 index 0000000..2a7d567 --- /dev/null +++ b/algosystem/validation/application/analyze_regimes.py @@ -0,0 +1,199 @@ +"""Regime-conditional performance use case.""" + +from __future__ import annotations + +import math + +import numpy as np +import pandas as pd + +from algosystem.shared.errors import ValidationError +from algosystem.validation.application.equity_curve_bridge import SeriesKind, returns_from +from algosystem.validation.domain.statistics.robustness import ( + RegimeResult, + regime_conditional_performance, +) + +MIN_OBSERVATIONS_PER_REGIME = 5 +"""Below this, the domain scores a bucket 0.0 rather than computing a Sharpe.""" + + +class AnalyzeRegimes: + """ + Split realized strategy performance by volatility regime. + + Answers "does this strategy only work in calm markets?" by classifying each + observation into a rolling-volatility bucket and reporting the Sharpe ratio + earned inside each one. This use case needs no parameter grid and no pass + runner: it describes a single realized return series rather than searching a + strategy's parameter space. + """ + + def execute( + self, + returns: object, + *, + market_returns: object | None = None, + n_regimes: int = 3, + vol_window: int = 21, + annualize: float = 252.0, + input_kind: SeriesKind = "returns", + market_input_kind: SeriesKind | None = None, + ) -> RegimeResult: + """ + Classify observations into volatility regimes and report Sharpe per regime. + + Args: + returns: Strategy series as an ``EquityCurve``, pandas Series/DataFrame, + numpy array, or sequence. + market_returns: Optional reference series used to define the regimes. + When omitted, the strategy's own volatility defines them. + n_regimes: Number of volatility buckets. Three yields low/med/high vol. + vol_window: Rolling window, in observations, for the volatility estimate. + annualize: Periods per year used to annualize each regime Sharpe. + input_kind: Whether raw array and pandas ``returns`` input holds + ``"returns"`` or ``"levels"``. ``EquityCurve`` always holds levels. + market_input_kind: The same choice for ``market_returns``. Defaults to + ``input_kind``; set it explicitly when the two series differ. + + Returns: + A ``RegimeResult`` holding per-regime Sharpe, observation counts and + time fractions, plus the worst regime and a works-everywhere flag. + + Raises: + ValidationError: If a parameter is out of range, a series cannot be + coerced to finite returns, the series is too short to classify, + or the two series cover different dates. + """ + n_regimes = _positive_int(n_regimes, name="n_regimes", minimum=2) + vol_window = _positive_int(vol_window, name="vol_window", minimum=2) + if not isinstance(annualize, (int, float)) or isinstance(annualize, bool): + raise ValidationError("annualize must be a number") + if not math.isfinite(annualize): + raise ValidationError("annualize must be finite") + if annualize <= 0: + raise ValidationError("annualize must be positive") + + strategy_array = _coerce(returns, input_kind, label="returns") + market_array = ( + None + if market_returns is None + else _coerce( + market_returns, + input_kind if market_input_kind is None else market_input_kind, + label="market_returns", + ) + ) + + if market_array is not None: + if len(market_array) != len(strategy_array): + raise ValidationError( + "market_returns must be the same length as returns " + f"({len(market_array)} != {len(strategy_array)})" + ) + _require_matching_dates( + returns, + market_returns, + input_kind=input_kind, + market_input_kind=input_kind if market_input_kind is None else market_input_kind, + ) + + _require_classifiable( + n_observations=len(strategy_array), + n_regimes=n_regimes, + vol_window=vol_window, + ) + + return regime_conditional_performance( + strategy_returns=strategy_array, + market_returns=market_array, + n_regimes=n_regimes, + vol_window=vol_window, + annualize=annualize, + ) + + +def _positive_int(value: object, *, name: str, minimum: int) -> int: + """Accept only a true integer, so a float does not escape as a raw TypeError.""" + if isinstance(value, bool) or not isinstance(value, (int, np.integer)): + raise ValidationError(f"{name} must be an integer") + coerced = int(value) + if coerced < minimum: + raise ValidationError(f"{name} must be at least {minimum}") + return coerced + + +def _coerce(source: object, input_kind: SeriesKind, *, label: str) -> np.ndarray: + """Convert one series, naming the offending argument if conversion fails.""" + try: + return returns_from(source, input_kind=input_kind) + except ValidationError as exc: + raise ValidationError(f"{label}: {exc}") from exc + + +def _require_classifiable(*, n_observations: int, n_regimes: int, vol_window: int) -> None: + """ + Reject series the classifier cannot describe. + + The domain degrades quietly rather than raising: a series no longer than + ``vol_window`` collapses to a single ``"all"`` regime, and any bucket holding + fewer than five observations scores 0.0. Both produce output that reads like + a finding ("Sharpe 0.0, regime-dependent") when it only means "not enough + data", so this use case refuses the input instead. + """ + classified = n_observations - vol_window + if classified <= 0: + raise ValidationError( + f"need more than vol_window={vol_window} observations to classify regimes, " + f"got {n_observations}" + ) + if classified < n_regimes * MIN_OBSERVATIONS_PER_REGIME: + raise ValidationError( + f"need at least {n_regimes * MIN_OBSERVATIONS_PER_REGIME} observations after the " + f"{vol_window}-observation warmup to fill {n_regimes} regimes, got {classified}; " + "reduce n_regimes or vol_window, or supply a longer series" + ) + + +def _require_matching_dates( + returns: object, + market_returns: object, + *, + input_kind: SeriesKind, + market_input_kind: SeriesKind, +) -> None: + """ + Reject two dated series that describe different dates. + + Regimes are assigned positionally, so equal-length series drawn from + different date ranges would be silently misaligned. + """ + strategy_index = _returns_index(returns, input_kind) + market_index = _returns_index(market_returns, market_input_kind) + if strategy_index is None or market_index is None: + return + if not strategy_index.equals(market_index): + raise ValidationError( + "returns and market_returns must cover the same dates; " + "align the two series before analyzing regimes" + ) + + +def _returns_index(source: object, input_kind: SeriesKind) -> pd.Index | None: + """Return the index the converted returns will carry, or None if undated.""" + if _is_equity_curve(source): + return source.values.index[1:] # type: ignore[union-attr] + if isinstance(source, (pd.Series, pd.DataFrame)): + return source.index[1:] if input_kind == "levels" else source.index + return None + + +def _is_equity_curve(source: object) -> bool: + return ( + source.__class__.__name__ == "EquityCurve" + and callable(getattr(source, "returns", None)) + and hasattr(source, "values") + ) + + +__all__ = ["MIN_OBSERVATIONS_PER_REGIME", "AnalyzeRegimes"] diff --git a/algosystem/validation/domain/statistics/robustness.py b/algosystem/validation/domain/statistics/robustness.py index c0920a9..64310ce 100644 --- a/algosystem/validation/domain/statistics/robustness.py +++ b/algosystem/validation/domain/statistics/robustness.py @@ -17,6 +17,8 @@ import numpy as np +from algosystem.validation.domain.validation_metric import ValidationMetricKey + # ----------------------------------------------------------------------- # 1. Bootstrap Confidence Interval on Sharpe Ratio # ----------------------------------------------------------------------- @@ -488,6 +490,30 @@ def summary(self) -> list[str]: ) return lines + def to_metrics(self) -> dict[str, object]: + """ + Return every field keyed by canonical validation metric names. + + Lossless: a caller can rebuild the whole breakdown from this mapping, + including the per-regime observation counts needed to tell an empty + bucket apart from a genuinely unprofitable one. + """ + return { + ValidationMetricKey.OVERALL_SHARPE.value: float(self.overall_sharpe), + ValidationMetricKey.REGIME_SHARPES.value: { + name: float(sharpe) for name, sharpe in zip(self.regime_names, self.regime_sharpes) + }, + ValidationMetricKey.REGIME_FRAC.value: { + name: float(frac) for name, frac in zip(self.regime_names, self.regime_frac) + }, + ValidationMetricKey.REGIME_COUNTS.value: { + name: int(count) for name, count in zip(self.regime_names, self.regime_counts) + }, + ValidationMetricKey.WORST_REGIME_SHARPE.value: float(self.worst_regime_sharpe), + ValidationMetricKey.WORST_REGIME_NAME.value: self.worst_regime_name, + ValidationMetricKey.WORKS_IN_ALL_REGIMES.value: bool(self.works_in_all_regimes), + } + def regime_conditional_performance( # noqa: C901 strategy_returns: np.ndarray, diff --git a/algosystem/validation/domain/validation_metric.py b/algosystem/validation/domain/validation_metric.py index 4057b23..444f021 100644 --- a/algosystem/validation/domain/validation_metric.py +++ b/algosystem/validation/domain/validation_metric.py @@ -102,8 +102,11 @@ class ValidationMetricKey(str, Enum): # Regime and surface metrics REGIME_SHARPES = "regime_sharpes" REGIME_FRAC = "regime_frac" + REGIME_COUNTS = "regime_counts" OVERALL_SHARPE = "overall_sharpe" WORST_REGIME_SHARPE = "worst_regime_sharpe" + WORST_REGIME_NAME = "worst_regime_name" + WORKS_IN_ALL_REGIMES = "works_in_all_regimes" ROBUSTNESS_RATIO = "robustness_ratio" FRAC_POSITIVE = "frac_positive" FRAC_ABOVE_HALF = "frac_above_half" diff --git a/docs/API_GUIDE.md b/docs/API_GUIDE.md index 1730d28..012260c 100644 --- a/docs/API_GUIDE.md +++ b/docs/API_GUIDE.md @@ -15,6 +15,7 @@ from algosystem import ( OverfitResults, ParameterGrid, PerformanceMetrics, + RegimeResult, RepositoryError, StrategySpec, ValidationError, @@ -67,6 +68,34 @@ algo.validation_report(report, output="overfit.html") The validation HTML report loads Plotly from a CDN when opened. +## Regimes + +```python +from algosystem.backtesting.domain.equity_curve import EquityCurve + +curve = EquityCurve.from_series(prices["Strategy"]) +result = algo.analyze_regimes(curve) + +print("\n".join(result.summary())) +print(result.works_in_all_regimes, result.worst_regime_name) +``` + +Classify on a benchmark instead of the strategy's own volatility by passing +`market_returns`. Both series must cover the same dates — when they carry a +`DatetimeIndex` this is checked, because regimes are assigned positionally: + +```python +result = algo.analyze_regimes(curve, market_returns=benchmark_curve) +``` + +Options: `n_regimes` (default 3), `vol_window` (default 21), `annualize` +(default 252), and `input_kind` / `market_input_kind` (`"returns"` or +`"levels"`) for raw array and pandas input. An `EquityCurve` always holds levels +and converts automatically. + +`result.to_metrics()` returns every field keyed by `ValidationMetricKey`, which +is the serialization-friendly form. + ## Persistence ```python diff --git a/docs/CLI_GUIDE.md b/docs/CLI_GUIDE.md index 0622f37..74d72dc 100644 --- a/docs/CLI_GUIDE.md +++ b/docs/CLI_GUIDE.md @@ -27,6 +27,32 @@ algosystem validate-strategies Validation reports are local HTML files, but the charts load Plotly from a CDN when opened in a browser. +## Regimes + +```bash +algosystem regimes strategy.csv --price-column Strategy +algosystem regimes strategy.csv --price-column Strategy --benchmark sp500 +algosystem regimes strategy.csv --price-column Strategy --regimes 4 --vol-window 63 +``` + +Breaks realized performance down by volatility regime and prints the Sharpe +earned in each. `--benchmark` classifies on a reference series instead of the +strategy's own volatility; the two series are restricted to the dates they +share, so a trading-day benchmark works against a calendar-day CSV. + +`--regimes 3` (the default) names the buckets `low_vol`/`med_vol`/`high_vol`; +any other count names them `regime_0` … `regime_N-1`. + +Because the two series are restricted to shared dates, a calendar mismatch +drops observations and widens the gaps between the survivors, so returns +span more elapsed time than before. `vol_window` counts observations, not +days, and the Sharpe annualization stays at 252 — treat a benchmarked run +on mismatched calendars as comparable within itself, not against an +unbenchmarked run. + +The command refuses a series too short to classify rather than printing an +all-zero breakdown that would read like a real finding. + ## Benchmarks ```bash diff --git a/docs/VALIDATION_GUIDE.md b/docs/VALIDATION_GUIDE.md index 23452f6..9ecae0b 100644 --- a/docs/VALIDATION_GUIDE.md +++ b/docs/VALIDATION_GUIDE.md @@ -62,6 +62,80 @@ Each archetype has a default `ParameterGrid` in `algosystem.validation.infrastructure.strategies.STRATEGY_REGISTRY`. The legacy name `vol_regime` is accepted as an alias for `volatility`. +## Regime-Conditional Performance + +Overfitting tests ask whether an edge is real. Regime analysis asks a different +question: *does the edge survive everywhere, or only in calm markets?* + +`AnalyzeRegimes` classifies every observation into a volatility bucket using a +rolling standard deviation, then reports the Sharpe ratio earned inside each +one. It takes a realized return series, so it needs no parameter grid and no +pass runner. + +```python +from algosystem import AlgoSystem +from algosystem.backtesting.domain.equity_curve import EquityCurve + +curve = EquityCurve.from_series(prices["Strategy"]) +result = AlgoSystem().analyze_regimes(curve) + +print("\n".join(result.summary())) +print(result.works_in_all_regimes) +``` + +Regimes default to terciles of the strategy's own volatility (`low_vol`, +`med_vol`, `high_vol`). Pass `market_returns` to classify on a benchmark +instead, which is usually what you want when asking whether a strategy depends +on the *market's* state rather than its own: + +```python +result = AlgoSystem().analyze_regimes(curve, market_returns=benchmark_curve) +``` + +Regimes are assigned **positionally**, so the two series must describe the same +observations. When both carry a `DatetimeIndex` this is verified and a mismatch +raises `ValidationError`; align the series yourself before calling if they sit +on different calendars. Undated input (plain arrays) can only be length-checked. + +Fields on `RegimeResult`: + +- `regime_names`, `regime_sharpes`, `regime_counts`, `regime_frac`: the + per-regime breakdown, index-aligned. +- `overall_sharpe`: Sharpe across every classified observation. +- `worst_regime_name` / `worst_regime_sharpe`: where the strategy is weakest. +- `works_in_all_regimes`: `True` only when Sharpe is positive in every regime. +- `summary()`: the report as a list of lines. +- `to_metrics()`: every field keyed by `ValidationMetricKey`, losslessly. + +Only the three-bucket default is named `low_vol`/`med_vol`/`high_vol`; any +other `n_regimes` names the buckets `regime_0` … `regime_N-1`. + +Options: `n_regimes` (default 3), `vol_window` (default 21 observations), +`annualize` (default 252), and `input_kind` / `market_input_kind` (`"returns"` +or `"levels"`) for raw array and pandas input. An `EquityCurve` always holds +levels and is converted automatically. + +### Insufficient data + +The statistics degrade quietly: a bucket holding fewer than five observations is +scored `0.0` rather than skipped, which makes a data shortage read like a +finding ("Sharpe 0.0, regime-dependent"). `AnalyzeRegimes` therefore refuses +input it cannot describe — a series no longer than `vol_window`, or one too +short to fill the requested number of regimes — and raises `ValidationError` +naming the offending parameter. + +That guard is necessary but not sufficient: percentile thresholds can still +leave one bucket thin. Check `regime_counts` (also carried by `to_metrics()`) +to tell an empty or thin bucket apart from a genuine loss. + +From the CLI: + +```bash +algosystem regimes strategy.csv --price-column Strategy +algosystem regimes strategy.csv --price-column Strategy --benchmark sp500 +algosystem regimes strategy.csv --price-column Strategy --regimes 4 --vol-window 63 +``` + ## Reading Results Important fields on `OverfitResults`: diff --git a/tests/test_cli.py b/tests/test_cli.py index e535ef9..24ca873 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -12,6 +12,7 @@ def test_cli_help_lists_phase_5_commands(): assert "tearsheet" in result.output assert "validate" in result.output assert "validate-strategies" in result.output + assert "regimes" in result.output assert "benchmarks" in result.output assert "db" in result.output @@ -24,7 +25,15 @@ def test_cli_version(): def test_top_level_subcommand_help(): - for command in ("backtest", "tearsheet", "validate", "validate-strategies", "benchmarks", "db"): + for command in ( + "backtest", + "tearsheet", + "validate", + "validate-strategies", + "regimes", + "benchmarks", + "db", + ): result = CliRunner().invoke(cli, [command, "--help"]) assert result.exit_code == 0 diff --git a/tests/test_cli_integration.py b/tests/test_cli_integration.py index 1b819ae..f7154e9 100644 --- a/tests/test_cli_integration.py +++ b/tests/test_cli_integration.py @@ -59,3 +59,174 @@ def print_summary(self, result, detailed=False): assert calls["kwargs"]["end"] == "2020-01-05" assert calls["kwargs"]["initial_capital"] == 5000 assert calls["summary"][1] is True + + +def test_regimes_command_reports_a_volatility_breakdown(tmp_path): + import numpy as np + + rng = np.random.default_rng(42) + returns = np.concatenate( + [rng.normal(0.0005, 0.001, size=300), rng.normal(0.0005, 0.02, size=300)] + ) + levels = np.concatenate([[100.0], 100.0 * np.cumprod(1.0 + returns)]) + frame = pd.DataFrame( + {"Strategy": levels}, + index=pd.date_range("2020-01-01", periods=len(levels), freq="D"), + ) + frame.index.name = "Date" + csv_path = tmp_path / "strategy.csv" + frame.to_csv(csv_path) + + result = CliRunner().invoke(cli, ["regimes", str(csv_path), "--price-column", "Strategy"]) + + assert result.exit_code == 0, result.output + assert "low_vol" in result.output + assert "high_vol" in result.output + assert "Overall Sharpe" in result.output + + +def test_regimes_command_rejects_an_unknown_price_column(tmp_path, sample_csv_file): + result = CliRunner().invoke(cli, ["regimes", sample_csv_file, "--price-column", "Missing"]) + + assert result.exit_code != 0 + assert "price column not found" in result.output + + +def _write_series(path, values, index, column): + frame = pd.DataFrame({column: values}, index=index) + frame.index.name = "Date" + frame.to_csv(path) + return str(path) + + +def _regime_levels(seed, size=600): + import numpy as np + + rng = np.random.default_rng(seed) + returns = np.concatenate( + [rng.normal(0.0005, 0.001, size=size // 2), rng.normal(0.0005, 0.02, size=size // 2)] + ) + return np.concatenate([[100.0], 100.0 * np.cumprod(1.0 + returns)]) + + +def _inverted_regime_levels(seed, size=600): + """Levels whose volatility break runs stormy-then-calm (opposite of _regime_levels).""" + import numpy as np + + rng = np.random.default_rng(seed) + returns = np.concatenate( + [rng.normal(0.0005, 0.02, size=size // 2), rng.normal(0.0005, 0.001, size=size // 2)] + ) + return np.concatenate([[100.0], 100.0 * np.cumprod(1.0 + returns)]) + + +def test_regimes_command_accepts_a_benchmark_on_a_different_calendar(tmp_path): + """A business-day benchmark against a calendar-day strategy must still work.""" + strategy = _regime_levels(seed=3) + strategy_index = pd.date_range("2020-01-01", periods=len(strategy), freq="D") + strategy_csv = _write_series(tmp_path / "strategy.csv", strategy, strategy_index, "Strategy") + + business_days = pd.date_range("2020-01-01", periods=len(strategy), freq="B") + benchmark = _regime_levels(seed=8, size=len(business_days) - 1) + benchmark_csv = _write_series(tmp_path / "benchmark.csv", benchmark, business_days, "Benchmark") + + result = CliRunner().invoke( + cli, + ["regimes", strategy_csv, "--price-column", "Strategy", "--benchmark", benchmark_csv], + ) + + assert result.exit_code == 0, result.output + assert "low_vol" in result.output + assert "high_vol" in result.output + + +def test_regimes_command_actually_classifies_on_the_benchmark(tmp_path): + """ + Without this, the --benchmark flag could be ignored and the happy-path test + would still pass. The benchmark's volatility break runs opposite to the + strategy's, so classifying on it must move the worst regime. + """ + strategy = _regime_levels(seed=3) + index = pd.date_range("2020-01-01", periods=len(strategy), freq="D") + strategy_csv = _write_series(tmp_path / "strategy.csv", strategy, index, "Strategy") + benchmark = _inverted_regime_levels(seed=8, size=len(strategy) - 1) + benchmark_csv = _write_series(tmp_path / "benchmark.csv", benchmark, index, "Benchmark") + + base = CliRunner().invoke(cli, ["regimes", strategy_csv, "--price-column", "Strategy"]) + with_benchmark = CliRunner().invoke( + cli, + ["regimes", strategy_csv, "--price-column", "Strategy", "--benchmark", benchmark_csv], + ) + + assert base.exit_code == 0, base.output + assert with_benchmark.exit_code == 0, with_benchmark.output + assert base.output != with_benchmark.output, "--benchmark did not change the classification" + + +def test_regimes_command_rejects_a_non_overlapping_benchmark(tmp_path): + strategy = _regime_levels(seed=3) + strategy_csv = _write_series( + tmp_path / "strategy.csv", + strategy, + pd.date_range("2020-01-01", periods=len(strategy), freq="D"), + "Strategy", + ) + benchmark = _regime_levels(seed=8) + benchmark_csv = _write_series( + tmp_path / "benchmark.csv", + benchmark, + pd.date_range("1990-01-01", periods=len(benchmark), freq="D"), + "Benchmark", + ) + + result = CliRunner().invoke( + cli, + ["regimes", strategy_csv, "--price-column", "Strategy", "--benchmark", benchmark_csv], + ) + + assert result.exit_code != 0 + assert "fewer than two dates" in result.output + + +def test_regimes_command_rejects_a_series_too_short_to_classify(tmp_path): + """A short series must not print a confident all-zero verdict.""" + levels = [100.0 + i for i in range(15)] + csv_path = _write_series( + tmp_path / "short.csv", + levels, + pd.date_range("2020-01-01", periods=len(levels), freq="D"), + "Strategy", + ) + + result = CliRunner().invoke(cli, ["regimes", csv_path, "--price-column", "Strategy"]) + + assert result.exit_code != 0 + assert "vol_window" in result.output + + +def test_regimes_command_honours_the_regime_and_window_options(tmp_path): + strategy = _regime_levels(seed=3) + csv_path = _write_series( + tmp_path / "strategy.csv", + strategy, + pd.date_range("2020-01-01", periods=len(strategy), freq="D"), + "Strategy", + ) + + result = CliRunner().invoke( + cli, + [ + "regimes", + csv_path, + "--price-column", + "Strategy", + "--regimes", + "4", + "--vol-window", + "10", + ], + ) + + assert result.exit_code == 0, result.output + assert "regime_0" in result.output + assert "regime_3" in result.output diff --git a/tests/validation/application/test_analyze_regimes.py b/tests/validation/application/test_analyze_regimes.py new file mode 100644 index 0000000..67f89ad --- /dev/null +++ b/tests/validation/application/test_analyze_regimes.py @@ -0,0 +1,284 @@ +"""Tests for the AnalyzeRegimes use case.""" + +import numpy as np +import pandas as pd +import pytest + +from algosystem.backtesting import EquityCurve +from algosystem.shared.errors import ValidationError +from algosystem.validation.application.analyze_regimes import AnalyzeRegimes +from algosystem.validation.domain.statistics.robustness import regime_conditional_performance +from algosystem.validation.domain.validation_metric import ValidationMetricKey +from tests.validation.conftest import regime_curve, regime_levels, regime_returns + +VOL_WINDOW = 21 + +_returns = regime_returns +_levels = regime_levels +_curve = regime_curve + + +class TestInputCoercion: + def test_array_input_matches_the_domain_function(self): + returns = _returns() + + result = AnalyzeRegimes().execute(returns, vol_window=VOL_WINDOW) + expected = regime_conditional_performance(returns, vol_window=VOL_WINDOW) + + assert result == expected + + def test_equity_curve_input_is_converted_to_returns(self): + """An EquityCurve holds levels; the use case must not treat them as returns.""" + returns = _returns() + curve = _curve(returns) + + result = AnalyzeRegimes().execute(curve, vol_window=VOL_WINDOW) + + assert sum(result.regime_counts) == len(returns) - VOL_WINDOW + direct = regime_conditional_performance(returns, vol_window=VOL_WINDOW) + assert result.regime_counts == direct.regime_counts + assert result.regime_sharpes == pytest.approx(direct.regime_sharpes, rel=1e-9) + assert result.regime_names == ["low_vol", "med_vol", "high_vol"] + + def test_series_of_levels_needs_the_explicit_input_kind(self): + returns = _returns() + levels = pd.Series(_levels(returns)) + + as_levels = AnalyzeRegimes().execute(levels, input_kind="levels", vol_window=VOL_WINDOW) + direct = regime_conditional_performance(returns, vol_window=VOL_WINDOW) + + assert as_levels.overall_sharpe == pytest.approx(direct.overall_sharpe, rel=1e-9) + + def test_returns_are_the_default_interpretation(self): + returns = _returns() + + default = AnalyzeRegimes().execute(returns, vol_window=VOL_WINDOW) + explicit = AnalyzeRegimes().execute(returns, input_kind="returns", vol_window=VOL_WINDOW) + + assert default == explicit + + def test_sequence_input_is_accepted(self): + returns = _returns() + + result = AnalyzeRegimes().execute(list(returns), vol_window=VOL_WINDOW) + + assert ( + result.regime_counts + == regime_conditional_performance(returns, vol_window=VOL_WINDOW).regime_counts + ) + + +class TestMarketReturns: + def test_market_series_defines_the_regimes(self): + rng = np.random.default_rng(13) + strategy = rng.normal(0.0005, 0.01, size=600) + market = _returns() + + on_strategy = AnalyzeRegimes().execute(strategy, vol_window=VOL_WINDOW) + on_market = AnalyzeRegimes().execute(strategy, market_returns=market, vol_window=VOL_WINDOW) + + assert on_strategy.regime_sharpes != on_market.regime_sharpes + + def test_market_equity_curve_is_converted_the_same_way(self): + strategy = _returns(seed=2) + market = _returns(seed=9) + + from_curve = AnalyzeRegimes().execute( + _curve(strategy), market_returns=_curve(market), vol_window=VOL_WINDOW + ) + from_arrays = AnalyzeRegimes().execute( + np.asarray(_curve(strategy).returns()), + market_returns=np.asarray(_curve(market).returns()), + vol_window=VOL_WINDOW, + ) + + assert from_curve == from_arrays + + def test_length_mismatch_raises(self): + with pytest.raises(ValidationError, match="same length"): + AnalyzeRegimes().execute(_returns(size=600), market_returns=_returns(size=400)) + + +class TestParameterValidation: + @pytest.mark.parametrize( + "kwargs, message", + [ + ({"n_regimes": 1}, "n_regimes must be at least 2"), + ({"n_regimes": 0}, "n_regimes must be at least 2"), + ({"vol_window": 1}, "vol_window must be at least 2"), + ({"annualize": 0.0}, "annualize must be positive"), + ({"annualize": -252.0}, "annualize must be positive"), + ], + ) + def test_out_of_range_parameters_raise(self, kwargs, message): + with pytest.raises(ValidationError, match=message): + AnalyzeRegimes().execute(_returns(), **kwargs) + + def test_parameters_are_checked_before_the_series_is_converted(self): + """A bad parameter must surface as itself, not as a conversion failure.""" + with pytest.raises(ValidationError, match="n_regimes must be at least 2"): + AnalyzeRegimes().execute([], n_regimes=1) + + @pytest.mark.parametrize( + "source, message", + [ + ([], "must not be empty"), + ([0.01], "at least two observations"), + ([0.01, np.nan], "finite"), + ], + ) + def test_unusable_series_raise(self, source, message): + with pytest.raises(ValidationError, match=message): + AnalyzeRegimes().execute(source) + + def test_invalid_input_kind_raises(self): + with pytest.raises(ValidationError, match="input_kind"): + AnalyzeRegimes().execute(_returns(), input_kind="prices") + + +class TestDateAlignment: + def test_disjoint_dated_series_are_rejected(self): + """Regimes are assigned positionally, so equal length is not enough.""" + strategy = pd.Series( + _returns(seed=2), index=pd.date_range("2020-01-01", periods=600, freq="D") + ) + market = pd.Series( + _returns(seed=9), index=pd.date_range("2010-01-01", periods=600, freq="D") + ) + + with pytest.raises(ValidationError, match="same dates"): + AnalyzeRegimes().execute(strategy, market_returns=market) + + def test_shifted_dates_are_rejected(self): + index = pd.date_range("2020-01-01", periods=600, freq="D") + strategy = pd.Series(_returns(seed=2), index=index) + market = pd.Series(_returns(seed=9), index=index.shift(1)) + + with pytest.raises(ValidationError, match="same dates"): + AnalyzeRegimes().execute(strategy, market_returns=market) + + def test_matching_dates_are_accepted(self): + index = pd.date_range("2020-01-01", periods=600, freq="D") + strategy = pd.Series(_returns(seed=2), index=index) + market = pd.Series(_returns(seed=9), index=index) + + result = AnalyzeRegimes().execute(strategy, market_returns=market) + + assert sum(result.regime_counts) == 600 - VOL_WINDOW + + def test_equity_curve_pair_aligns_on_the_shared_dates(self): + strategy = _returns(seed=2) + market = _returns(seed=9) + + result = AnalyzeRegimes().execute(_curve(strategy), market_returns=_curve(market)) + + assert sum(result.regime_counts) == len(strategy) - VOL_WINDOW + + def test_undated_input_skips_the_date_check(self): + """Plain arrays carry no dates, so only the length guard can apply.""" + result = AnalyzeRegimes().execute(_returns(seed=2), market_returns=_returns(seed=9)) + + assert sum(result.regime_counts) == 600 - VOL_WINDOW + + +class TestMixedInputKinds: + def test_market_input_kind_defaults_to_input_kind(self): + strategy = _returns(seed=2) + market = _returns(seed=9) + + default = AnalyzeRegimes().execute(strategy, market_returns=market) + explicit = AnalyzeRegimes().execute( + strategy, market_returns=market, market_input_kind="returns" + ) + + assert default == explicit + + def test_strategy_levels_can_pair_with_market_returns(self): + strategy = _returns(seed=2) + market = _returns(seed=9) + + mixed = AnalyzeRegimes().execute( + _levels(strategy), + market_returns=market, + input_kind="levels", + market_input_kind="returns", + ) + plain = AnalyzeRegimes().execute(strategy, market_returns=market) + + assert mixed.regime_counts == plain.regime_counts + assert mixed.regime_sharpes == pytest.approx(plain.regime_sharpes, rel=1e-9) + + +class TestInsufficientData: + def test_series_no_longer_than_the_window_is_rejected(self): + """The domain would quietly collapse to a single 'all' regime scored 0.0.""" + with pytest.raises(ValidationError, match="more than vol_window"): + AnalyzeRegimes().execute(np.full(15, 0.001), vol_window=VOL_WINDOW) + + def test_series_too_short_to_fill_the_buckets_is_rejected(self): + """Buckets under five observations score 0.0, which reads as a false verdict.""" + returns = np.linspace(0.001, 0.02, 30) + + with pytest.raises(ValidationError, match="at least 15 observations"): + AnalyzeRegimes().execute(returns, vol_window=VOL_WINDOW) + + def test_a_shorter_window_makes_the_same_series_classifiable(self): + returns = np.linspace(0.001, 0.02, 30) + + result = AnalyzeRegimes().execute(returns, vol_window=5) + + assert sum(result.regime_counts) == 25 + + def test_thin_buckets_remain_possible_and_are_visible_in_the_counts(self): + """ + The guard is necessary, not sufficient: percentile thresholds can still + leave a bucket under five observations, which the domain scores 0.0. The + counts are what let a caller tell that apart from a real loss, which is + why to_metrics() carries them. + """ + result = AnalyzeRegimes().execute(np.linspace(0.001, 0.02, 30), vol_window=5) + + thin = [name for name, count in zip(result.regime_names, result.regime_counts) if count < 5] + assert thin, "this fixture is chosen to produce a thin bucket" + for name in thin: + index = result.regime_names.index(name) + assert result.regime_sharpes[index] == 0.0 + assert result.to_metrics()[ValidationMetricKey.REGIME_COUNTS.value] == dict( + zip(result.regime_names, result.regime_counts) + ) + + +class TestNumericParameterTypes: + @pytest.mark.parametrize("value", [3.5, "3", None]) + def test_non_integer_n_regimes_raises_a_typed_error(self, value): + with pytest.raises(ValidationError, match="n_regimes must be an integer"): + AnalyzeRegimes().execute(_returns(), n_regimes=value) + + @pytest.mark.parametrize("value", [2.5, "21", None]) + def test_non_integer_vol_window_raises_a_typed_error(self, value): + with pytest.raises(ValidationError, match="vol_window must be an integer"): + AnalyzeRegimes().execute(_returns(), vol_window=value) + + @pytest.mark.parametrize("value", [float("nan"), float("inf"), float("-inf")]) + def test_non_finite_annualize_raises(self, value): + """nan slips past a bare `annualize <= 0` check and poisons every Sharpe.""" + with pytest.raises(ValidationError, match="annualize must be finite"): + AnalyzeRegimes().execute(_returns(), annualize=value) + + def test_annualize_changes_the_reported_sharpe(self): + returns = _returns() + + daily = AnalyzeRegimes().execute(returns, annualize=1.0, vol_window=VOL_WINDOW) + annual = AnalyzeRegimes().execute(returns, annualize=252.0, vol_window=VOL_WINDOW) + + assert annual.overall_sharpe == pytest.approx(daily.overall_sharpe * np.sqrt(252.0)) + + +class TestErrorAttribution: + def test_a_bad_market_series_names_the_offending_argument(self): + with pytest.raises(ValidationError, match="market_returns:"): + AnalyzeRegimes().execute(_returns(), market_returns=[0.01, np.nan]) + + def test_a_bad_strategy_series_names_the_offending_argument(self): + with pytest.raises(ValidationError, match="returns:"): + AnalyzeRegimes().execute([0.01, np.nan]) diff --git a/tests/validation/application/test_analyze_regimes_api.py b/tests/validation/application/test_analyze_regimes_api.py new file mode 100644 index 0000000..e59039b --- /dev/null +++ b/tests/validation/application/test_analyze_regimes_api.py @@ -0,0 +1,110 @@ +"""Tests for regime analysis reached through the public AlgoSystem facade.""" + +from pathlib import Path + +import numpy as np +import pytest + +from algosystem.backtesting.domain.equity_curve import EquityCurve +from algosystem.backtesting.infrastructure.fake_calculator import FakeMetricsCalculator +from algosystem.interfaces.api import AlgoSystem +from algosystem.shared.errors import ValidationError +from algosystem.validation.domain.statistics.robustness import ( + RegimeResult, + regime_conditional_performance, +) +from tests.validation.conftest import regime_curve, regime_levels, regime_returns + +VOL_WINDOW = 21 + +_returns = regime_returns +_levels = regime_levels +_curve = regime_curve + + +class StubTearsheetRenderer: + """Satisfies the TearsheetRenderer port without importing quantstats.""" + + def render(self, result, output_path, title, mode="html", rf=0.0, periods_per_year=252): + return Path(output_path) + + +@pytest.fixture +def algo(): + """AlgoSystem with vendor adapters stubbed, so the facade is testable offline.""" + return AlgoSystem(calculator=FakeMetricsCalculator(), renderer=StubTearsheetRenderer()) + + +def test_facade_returns_a_regime_result(algo): + result = algo.analyze_regimes(_returns(), vol_window=VOL_WINDOW) + + assert isinstance(result, RegimeResult) + assert result.regime_names == ["low_vol", "med_vol", "high_vol"] + + +def test_facade_matches_the_domain_function(algo): + returns = _returns() + + result = algo.analyze_regimes(returns, vol_window=VOL_WINDOW) + + assert result == regime_conditional_performance(returns, vol_window=VOL_WINDOW) + + +def test_facade_accepts_an_equity_curve(algo): + returns = _returns() + + result = algo.analyze_regimes(_curve(returns), vol_window=VOL_WINDOW) + + assert sum(result.regime_counts) == len(returns) - VOL_WINDOW + + +def test_facade_forwards_every_option(algo): + returns = _returns() + + result = algo.analyze_regimes(returns, n_regimes=4, vol_window=10, annualize=52.0) + expected = regime_conditional_performance(returns, n_regimes=4, vol_window=10, annualize=52.0) + + assert result == expected + assert result.regime_names == ["regime_0", "regime_1", "regime_2", "regime_3"] + + +def test_facade_forwards_the_market_series(algo): + strategy = _returns(seed=2) + market = _returns(seed=9) + + result = algo.analyze_regimes(strategy, market_returns=market, vol_window=VOL_WINDOW) + + assert result == regime_conditional_performance( + strategy, market_returns=market, vol_window=VOL_WINDOW + ) + + +def test_facade_supports_mixed_input_kinds(algo): + strategy = _returns(seed=2) + market = _returns(seed=9) + levels = np.concatenate([[100.0], 100.0 * np.cumprod(1.0 + strategy)]) + + result = algo.analyze_regimes( + levels, + market_returns=market, + input_kind="levels", + market_input_kind="returns", + vol_window=VOL_WINDOW, + ) + + assert ( + result.regime_counts + == regime_conditional_performance( + strategy, market_returns=market, vol_window=VOL_WINDOW + ).regime_counts + ) + + +def test_facade_surfaces_validation_errors(algo): + with pytest.raises(ValidationError, match="n_regimes must be at least 2"): + algo.analyze_regimes(_returns(), n_regimes=1) + + +def test_facade_rejects_a_series_too_short_to_classify(algo): + with pytest.raises(ValidationError, match="more than vol_window"): + algo.analyze_regimes(np.full(15, 0.001), vol_window=VOL_WINDOW) diff --git a/tests/validation/conftest.py b/tests/validation/conftest.py index 8eb0aff..c4191f6 100644 --- a/tests/validation/conftest.py +++ b/tests/validation/conftest.py @@ -63,3 +63,30 @@ def quick_detector(noise_returns, small_grid): n_workers=1, seed=42, ) + + +def regime_returns(seed: int = 7, size: int = 600) -> np.ndarray: + """Returns with a sharp volatility break: calm for the first half, stormy after.""" + rng = np.random.default_rng(seed) + return np.concatenate( + [ + rng.normal(0.0005, 0.001, size=size // 2), + rng.normal(0.0005, 0.02, size=size // 2), + ] + ) + + +def regime_levels(returns: np.ndarray, initial_value: float = 100.0) -> np.ndarray: + """Levels whose pct_change reproduces `returns` exactly (one extra opening point).""" + return np.concatenate([[initial_value], initial_value * np.cumprod(1.0 + returns)]) + + +def regime_curve(returns: np.ndarray): + """An EquityCurve over `regime_levels`, dated daily from 2020-01-01.""" + import pandas as pd + + from algosystem.backtesting.domain.equity_curve import EquityCurve + + levels = regime_levels(returns) + index = pd.date_range("2020-01-01", periods=len(levels), freq="D") + return EquityCurve.from_series(pd.Series(levels, index=index)) diff --git a/tests/validation/domain/test_robustness_regimes.py b/tests/validation/domain/test_robustness_regimes.py new file mode 100644 index 0000000..1109adf --- /dev/null +++ b/tests/validation/domain/test_robustness_regimes.py @@ -0,0 +1,242 @@ +"""Tests for volatility-regime classification and regime-conditional performance.""" + +import dataclasses + +import numpy as np +import pytest + +from algosystem.validation.domain.statistics.robustness import ( + RegimeResult, + regime_conditional_performance, +) +from algosystem.validation.domain.validation_metric import ValidationMetricKey + +VOL_WINDOW = 21 +CALM_END = 300 + + +def _regime_labels(returns: np.ndarray, *, vol_window: int) -> np.ndarray: + """Recompute the classifier's labels so tests can inspect what landed where.""" + n = len(returns) + rolling = np.full(n, np.nan) + for i in range(vol_window, n): + rolling[i] = np.std(returns[i - vol_window : i], ddof=1) + valid = ~np.isnan(rolling) + thresholds = np.percentile(rolling[valid], [100 / 3, 200 / 3]) + labels = np.full(n, -1, dtype=int) + labels[valid] = np.searchsorted(thresholds, rolling[valid], side="left") + return labels + + +def _calm_then_stormy(seed: int = 7) -> np.ndarray: + """Returns with a sharp volatility break: calm for 300 steps, then stormy.""" + rng = np.random.default_rng(seed) + calm = rng.normal(0.0, 0.001, size=CALM_END) + stormy = rng.normal(0.0, 0.02, size=CALM_END) + return np.concatenate([calm, stormy]) + + +class TestRegimeClassification: + def test_names_are_low_med_high_for_three_regimes(self): + result = regime_conditional_performance(_calm_then_stormy(), vol_window=VOL_WINDOW) + + assert result.regime_names == ["low_vol", "med_vol", "high_vol"] + + def test_names_are_indexed_when_not_three_regimes(self): + result = regime_conditional_performance( + _calm_then_stormy(), n_regimes=4, vol_window=VOL_WINDOW + ) + + assert result.regime_names == ["regime_0", "regime_1", "regime_2", "regime_3"] + + def test_counts_cover_every_observation_past_the_warmup(self): + returns = _calm_then_stormy() + + result = regime_conditional_performance(returns, vol_window=VOL_WINDOW) + + assert sum(result.regime_counts) == len(returns) - VOL_WINDOW + + def test_time_fractions_sum_to_one(self): + result = regime_conditional_performance(_calm_then_stormy(), vol_window=VOL_WINDOW) + + assert sum(result.regime_frac) == pytest.approx(1.0) + + def test_low_and_high_regimes_track_the_actual_volatility_break(self): + """ + The calm half must land in low_vol and the stormy half in high_vol. + + Asserting bucket sizes proves nothing — terciles are a third each by + construction, even on IID noise with no regime structure at all. So this + measures the realized volatility inside each bucket instead. + """ + returns = _calm_then_stormy() + + result = regime_conditional_performance(returns, vol_window=VOL_WINDOW) + labels = _regime_labels(returns, vol_window=VOL_WINDOW) + + low_vol = float(np.std(returns[labels == 0], ddof=1)) + high_vol = float(np.std(returns[labels == 2], ddof=1)) + + # The fixture's halves differ in sigma by 20x (0.001 vs 0.02), but the + # classifier is causal: the label at i comes from returns before i, so + # the low bucket legitimately holds the ~21 stormy returns whose + # preceding window was still calm. That caps realized separation near 8x. + assert high_vol > 5 * low_vol + # The extreme buckets must be drawn from one side of the break each. + assert np.mean(np.flatnonzero(labels == 0) < CALM_END) > 0.95 + assert np.mean(np.flatnonzero(labels == 2) >= CALM_END) > 0.95 + + def test_regimes_can_be_defined_by_a_separate_market_series(self): + """Passing market_returns classifies on the market, not the strategy.""" + rng = np.random.default_rng(11) + strategy = rng.normal(0.0, 0.01, size=600) + market = _calm_then_stormy() + + on_strategy = regime_conditional_performance(strategy, vol_window=VOL_WINDOW) + on_market = regime_conditional_performance( + strategy, market_returns=market, vol_window=VOL_WINDOW + ) + + assert on_strategy.regime_sharpes != on_market.regime_sharpes + + +class TestRegimeVerdict: + def test_calm_only_edge_is_flagged_regime_dependent(self): + """A strategy that earns only in calm markets must not pass.""" + rng = np.random.default_rng(5) + market = _calm_then_stormy() + strategy = np.concatenate( + [ + rng.normal(0.002, 0.001, size=CALM_END), # earns while calm + rng.normal(-0.002, 0.001, size=CALM_END), # bleeds once vol arrives + ] + ) + + result = regime_conditional_performance( + strategy, market_returns=market, vol_window=VOL_WINDOW + ) + + assert result.works_in_all_regimes is False + assert result.worst_regime_name == "high_vol" + assert result.worst_regime_sharpe < 0 + + def test_uniformly_profitable_strategy_works_in_all_regimes(self): + rng = np.random.default_rng(3) + market = _calm_then_stormy() + strategy = rng.normal(0.002, 0.001, size=600) + + result = regime_conditional_performance( + strategy, market_returns=market, vol_window=VOL_WINDOW + ) + + assert result.works_in_all_regimes is True + assert all(sharpe > 0 for sharpe in result.regime_sharpes) + + +class TestDegenerateInput: + def test_series_shorter_than_the_window_returns_a_single_regime(self): + returns = np.full(10, 0.001) + + result = regime_conditional_performance(returns, vol_window=VOL_WINDOW) + + assert result.regime_names == ["all"] + assert result.regime_counts == [10] + assert result.works_in_all_regimes is False + + +class TestRegimeResultOutput: + def test_result_is_frozen(self): + result = regime_conditional_performance(_calm_then_stormy(), vol_window=VOL_WINDOW) + + assert isinstance(result, RegimeResult) + with pytest.raises(dataclasses.FrozenInstanceError): + result.overall_sharpe = 1.0 + + def test_summary_returns_lines_and_never_prints(self, capsys): + result = regime_conditional_performance(_calm_then_stormy(), vol_window=VOL_WINDOW) + + lines = result.summary() + + assert isinstance(lines, list) + assert all(isinstance(line, str) for line in lines) + assert capsys.readouterr().out == "" + + def test_summary_warns_when_a_regime_loses_money(self): + rng = np.random.default_rng(5) + market = _calm_then_stormy() + strategy = np.concatenate( + [ + rng.normal(0.002, 0.001, size=CALM_END), + rng.normal(-0.002, 0.001, size=CALM_END), + ] + ) + + lines = regime_conditional_performance( + strategy, market_returns=market, vol_window=VOL_WINDOW + ).summary() + + assert any("WARNING" in line for line in lines) + assert "REGIME-DEPENDENT" in lines[0] + + def test_to_metrics_uses_canonical_validation_metric_names(self): + result = regime_conditional_performance(_calm_then_stormy(), vol_window=VOL_WINDOW) + + metrics = result.to_metrics() + + assert set(metrics) == { + ValidationMetricKey.OVERALL_SHARPE.value, + ValidationMetricKey.REGIME_SHARPES.value, + ValidationMetricKey.REGIME_FRAC.value, + ValidationMetricKey.REGIME_COUNTS.value, + ValidationMetricKey.WORST_REGIME_SHARPE.value, + ValidationMetricKey.WORST_REGIME_NAME.value, + ValidationMetricKey.WORKS_IN_ALL_REGIMES.value, + } + assert metrics[ValidationMetricKey.OVERALL_SHARPE.value] == result.overall_sharpe + assert metrics[ValidationMetricKey.REGIME_SHARPES.value] == dict( + zip(result.regime_names, result.regime_sharpes) + ) + assert metrics[ValidationMetricKey.REGIME_FRAC.value] == dict( + zip(result.regime_names, result.regime_frac) + ) + assert metrics[ValidationMetricKey.REGIME_COUNTS.value] == dict( + zip(result.regime_names, result.regime_counts) + ) + assert metrics[ValidationMetricKey.WORST_REGIME_SHARPE.value] == result.worst_regime_sharpe + assert metrics[ValidationMetricKey.WORST_REGIME_NAME.value] == result.worst_regime_name + assert ( + metrics[ValidationMetricKey.WORKS_IN_ALL_REGIMES.value] == result.works_in_all_regimes + ) + assert all( + type(value) is float + for value in metrics[ValidationMetricKey.REGIME_FRAC.value].values() + ), "to_metrics must emit plain floats so results serialize cleanly" + assert all( + type(value) is int + for value in metrics[ValidationMetricKey.REGIME_COUNTS.value].values() + ) + + def test_to_metrics_is_lossless(self): + """Every RegimeResult field must be recoverable from the mapping.""" + result = regime_conditional_performance(_calm_then_stormy(), vol_window=VOL_WINDOW) + + metrics = result.to_metrics() + + # regime_names is carried by the per-regime dict keys, not its own entry. + names = list(metrics[ValidationMetricKey.REGIME_SHARPES.value]) + assert names == result.regime_names + + rebuilt = RegimeResult( + regime_names=names, + regime_sharpes=[metrics[ValidationMetricKey.REGIME_SHARPES.value][n] for n in names], + regime_counts=[metrics[ValidationMetricKey.REGIME_COUNTS.value][n] for n in names], + regime_frac=[metrics[ValidationMetricKey.REGIME_FRAC.value][n] for n in names], + overall_sharpe=metrics[ValidationMetricKey.OVERALL_SHARPE.value], + worst_regime_sharpe=metrics[ValidationMetricKey.WORST_REGIME_SHARPE.value], + worst_regime_name=metrics[ValidationMetricKey.WORST_REGIME_NAME.value], + works_in_all_regimes=metrics[ValidationMetricKey.WORKS_IN_ALL_REGIMES.value], + ) + assert rebuilt == result + + # Guard against a field being added later without a matching key. + assert len(dataclasses.fields(result)) == len(metrics) + 1 From 94e0ec9b512ed3e1ded85c52862880bd5c46c57e Mon Sep 17 00:00:00 2001 From: Sivan Reddy Pushpagiri Date: Tue, 22 Sep 2026 09:05:42 -0400 Subject: [PATCH 2/2] Add regime detection demo and implementation notes The demo builds two strategies calibrated to the same overall Sharpe and shows that only the regime breakdown separates them: one holds its edge across all volatility buckets, the other earns everything while calm and gives it back once volatility arrives. It also exercises both guardrails (short series, mismatched dates). The notes record what was already in the tree before this work, the design decisions behind the new use case, the defects found in review, and the known follow-ups. --- REGIME_DETECTION_HANDOFF.md | 223 ++++++++++++++++++++++++++++++ examples/regime_detection_demo.py | 101 ++++++++++++++ 2 files changed, 324 insertions(+) create mode 100644 REGIME_DETECTION_HANDOFF.md create mode 100644 examples/regime_detection_demo.py diff --git a/REGIME_DETECTION_HANDOFF.md b/REGIME_DETECTION_HANDOFF.md new file mode 100644 index 0000000..2c68445 --- /dev/null +++ b/REGIME_DETECTION_HANDOFF.md @@ -0,0 +1,223 @@ +# Regime Detection — Implementation Handoff + +**Branch:** `feat/regime-detection` · **Commit:** `594d202` · **Base:** `main` (`d929556`) +**Scope:** 19 files, +1345 / −3 + +--- + +## 1. What the task turned out to be + +The ticket read "implement regime detection." The important discovery is that **most of it already existed.** + +`algosystem/validation/domain/statistics/robustness.py` already contained +`regime_conditional_performance()` and `RegimeResult` — volatility-tercile classification +answering *"does this strategy only work in calm markets?"* It arrived with John Riley's +overfitting port in phase 7 (commit `d929556`). + +But it was **dead code**: + +- exported from `validation/domain/__init__.py` and referenced by nothing else in the package +- no use case, no `AlgoSystem` method, no CLI command +- zero tests +- absent from every guide + +So the work was **not** "write a regime detector from scratch." It was: make the existing, +correct statistics reachable, safe to call, and tested. That framing matters — `AGENTS.md` +and `.codex/phases/phase-7.md` both forbid altering Riley's numerics. + +> The alternative reading — build a *new* detector (HMM / Markov-switching) — would have been +> a multi-day design job duplicating working code. If that was actually intended, this branch +> is still the right foundation for it. + +--- + +## 2. What was built + +Wired through the existing DDD layering (`interfaces → application → domain`). + +| Layer | File | Change | +|---|---|---| +| Domain | `validation/domain/statistics/robustness.py` | `RegimeResult.to_metrics()` — lossless export keyed by `ValidationMetricKey`. **+26 lines, 0 deletions — no numerics touched.** | +| Domain | `validation/domain/validation_metric.py` | Added `REGIME_COUNTS`, `WORST_REGIME_NAME`, `WORKS_IN_ALL_REGIMES` so every computed field has a canonical name (rule R4/V3). | +| Application | `validation/application/analyze_regimes.py` *(new, 199 lines)* | `AnalyzeRegimes` use case: input coercion, parameter validation, date-alignment and sufficient-data guards. | +| Interfaces | `interfaces/api.py` | `AlgoSystem.analyze_regimes()` | +| Interfaces | `interfaces/cli/main.py` | `algosystem regimes` command + rich rendering | +| Exports | `algosystem/__init__.py`, `validation/__init__.py`, `validation/application/__init__.py` | `AnalyzeRegimes`, `RegimeResult` | +| Docs | `VALIDATION_GUIDE.md`, `API_GUIDE.md`, `CLI_GUIDE.md`, `README.md`, `CHANGELOG.md` | New sections | +| Tests | 3 new files + 2 extended | **58 new test functions** | + +### Design decisions worth defending + +**No `facade.py` entry point.** `validation/facade.py` exists to compose *infrastructure* +(`detect_overfitting` builds a `PassRunner`, resolves strategies from +`validation.infrastructure.strategies`). `AnalyzeRegimes` composes nothing — it takes a +realized return series and calls one pure function. A facade wrapper would be a pure +pass-through. + +**The CLI calls the use case directly, not `AlgoSystem`.** `AlgoSystem.__init__` eagerly +builds a quantstats metrics calculator that regime analysis never uses. Routing the CLI +around it keeps the command usable. Precedent exists — `validate-strategies` and +`benchmarks` also import their dependencies directly. *(See §6 — the underlying eagerness +is a pre-existing issue worth its own ticket.)* + +**`AnalyzeRegimes` is a stateless class with one `execute`.** Effectively a namespaced +function, but it preserves the uniform `X(...).execute(...)` protocol used by +`DetectOverfitting`, `RunWalkForward`, and `ScreenSignals`. + +--- + +## 3. What code review caught (and how it was fixed) + +I put the first version through three review passes — correctness, architecture/spec +compliance, and test/doc quality. **The first pass had real defects.** All are fixed; this +section records them so the fixes can be scrutinised rather than taken on trust. + +| # | Defect | Fix | +|---|---|---| +| 1 | **Silent date misalignment.** Regimes are assigned *positionally*; only series *length* was checked. A strategy dated 2020 with a benchmark dated 2010 produced confident, meaningless output — and the docs recommended exactly that call. | `_require_matching_dates()` compares the converted returns indices when both inputs carry a `DatetimeIndex`. Undated arrays still fall back to the length check. | +| 2 | **Data shortage reported as a finding.** A bucket under 5 observations scores `0.0`. A 30-row CSV printed "every regime 0.0, REGIME-DEPENDENT", exit 0 — for a strategy whose true Sharpe was ~93. | `_require_classifiable()` rejects a series no longer than `vol_window`, or too short to fill `n_regimes`, naming the fixable parameter. | +| 3 | **`nan` annualize poisoned every Sharpe silently** (`nan <= 0` is `False`). Non-integer `n_regimes` leaked a raw `TypeError` past the typed-error boundary. | Explicit finite/integer checks raising `ValidationError`. | +| 4 | **Tautological test.** `test_low_and_high_regimes_track_the_actual_volatility_break` asserted only bucket *sizes* — a third each by construction. It passed on pure IID noise with no regime structure. | Now measures realized volatility inside each bucket. Verified: **8.01×** separation on the regime fixture vs **0.97×** on IID noise. | +| 5 | **The branch was red.** `to_metrics()` was expanded from 4 keys to 7 without updating its test. | Test asserts all 7 keys plus a round-trip reconstruction of the whole `RegimeResult`. | +| 6 | **`--benchmark` was a dead end** for calendar-day files (a business-day benchmark always errored), and its happy-path test passed even with the flag ignored entirely. | Date **intersection** instead of strict reindex; added a test that runs with and without `--benchmark` and asserts the classification changes. | +| 7 | `to_metrics()` was lossy (4 of 8 fields) and had no consumer. | Added the missing enum members; the CLI now renders from it. | + +Also fixed: shared test fixtures moved to `tests/validation/conftest.py`; `pytest.approx` +tolerances tightened from `1e-6` to `1e-9` (measured error is ~`1e-16`); exact float equality +on `sum(regime_frac)` replaced with `approx`; `FrozenInstanceError` used instead of a +`try/except/else`; `CLI_GUIDE.md` and `API_GUIDE.md` updated (originally missed). + +--- + +## 4. Verification + +``` +python -m pytest tests/ +→ 22 failed, 301 passed, 1 error in 479s +``` + +**All 22 failures and the 1 collection error reproduce identically on `main`.** Confirmed by +running the same selection in a clean `git worktree` at `main`. They are `quantstats` not +being installed in this environment (no poetry venv on this machine), plus one sandbox +`parquet_cache` filesystem failure. **Zero regressions.** + +| Check | Result | +|---|---| +| `black --check` / `isort --check` | clean, 132 files | +| `import algosystem` under `-W error` | clean — no side effects, no warnings | +| `validation_domain_forbidden` contract | KEPT across 16 domain modules | +| Domain purity (subprocess test) | passes | +| Layering contracts R1/R2/R5/R6/R7 | verified by AST across all 90 modules | +| New tests | 58 functions, all green | + +> **Environment caveat:** `ruff` and `import-linter` are not installed on this machine, so +> contracts were verified by AST analysis rather than the real tool. **CI is the authoritative +> check** — run it before merging. + +--- + +## 5. How to demo it + +### A. The script — strongest for a non-engineering audience + +```bash +python examples/regime_detection_demo.py +``` + +Builds two strategies calibrated to the **identical** headline Sharpe of 1.62, then separates +them: + +``` +Strategy A (robust) Sharpe : 1.62 Strategy B (fragile) Sharpe : 1.62 + ↓ ↓ + low_vol 0.3333 low_vol 4.8624 + med_vol 3.4279 med_vol 1.6555 + high_vol 0.8719 high_vol -1.3092 ← WARNING + [ALL REGIMES] [REGIME-DEPENDENT] + works_in_all_regimes = True works_in_all_regimes = False +``` + +**The pitch in one screen: a tearsheet cannot tell these apart. This can.** + +The script also demonstrates both guardrails rejecting bad input rather than producing +confident nonsense. + +### B. The CLI — live demo, no Python + +```bash +algosystem regimes strategy.csv --price-column Strategy +algosystem regimes strategy.csv --price-column Strategy --benchmark market.csv +algosystem regimes strategy.csv --price-column Strategy --regimes 4 --vol-window 63 +``` + +**Worth knowing when demoing this.** Without `--benchmark`, regimes come from the strategy's +*own* volatility. A strategy with constant volatility but changing drift reports +`ALL REGIMES` (1.48 / 0.34 / 2.03) — its own vol cannot see the regimes. Add `--benchmark` +and the same data reads **4.44 / 2.15 / −3.05, REGIME-DEPENDENT**. + +> That is the answer to "why didn't it catch it?" — classify on the **market** when asking +> whether the market's state breaks the strategy. + +### C. The tests — engineering proof + +```bash +python -m pytest tests/validation/ tests/test_cli.py tests/test_cli_integration.py +``` + +The strongest evidence that these tests are meaningful is item 4 in §3: the classification +test now fails on scrambled data. The first version of it didn't. + +### D. Python API + +```python +from algosystem import AlgoSystem +from algosystem.backtesting.domain.equity_curve import EquityCurve + +curve = EquityCurve.from_series(prices["Strategy"]) +result = AlgoSystem().analyze_regimes(curve, market_returns=benchmark_curve) + +print("\n".join(result.summary())) +print(result.works_in_all_regimes, result.worst_regime_name) +print(result.to_metrics()) # keyed by ValidationMetricKey +``` + +--- + +## 6. Known gaps and follow-ups + +Recording these up front rather than leaving them to surface in review. + +1. **Not integrated into the HTML validation report.** `html_report.py::_add_robustness` + already dispatches on a `robustness` mapping, and `RenderValidationReport.execute` takes + `robustness=`. Regimes were left out **because that parameter is currently dead for all + five robustness statistics** — wiring only regimes would be inconsistent. ~6 lines if + wanted. *Defensible scope boundary, but expect the question.* + +2. **`AlgoSystem.__init__` eagerly builds a quantstats calculator**, so `AlgoSystem()` cannot + be constructed without quantstats even for regime analysis, which uses none of it. Worked + around (CLI routes direct; facade tests inject `FakeMetricsCalculator`). The real fix is + lazy adapter construction — **a pre-existing issue, deliberately out of scope here.** + +3. **Thin buckets remain possible.** The sufficient-data guard is necessary, not sufficient: + percentile thresholds can still leave a bucket under 5 observations, scored `0.0`. This is + why `regime_counts` is exposed in `to_metrics()` — it distinguishes an empty bucket from a + genuine loss. Documented and tested. + +4. **Calendar mismatch shifts the Sharpe.** Restricting to shared dates drops observations and + widens gaps between survivors, while `vol_window` counts observations and annualization + stays at 252. Documented in `CLI_GUIDE.md`; no `--annualize` flag was added. + +5. **Two return-coercion paths coexist.** `AnalyzeRegimes` uses the public `returns_from`; + `run_walk_forward.py` and `screen_signals.py` use the private `_coerce_returns_array`. + The new code picked the better one; collapsing the siblings is a follow-up. + +--- + +## 7. Reverting + +```bash +git checkout main # walk away, branch intact +git branch -D feat/regime-detection # delete outright +``` + +Untracked helper files not in the commit: `examples/regime_detection_demo.py`, and this file. diff --git a/examples/regime_detection_demo.py b/examples/regime_detection_demo.py new file mode 100644 index 0000000..0016f06 --- /dev/null +++ b/examples/regime_detection_demo.py @@ -0,0 +1,101 @@ +""" +Demo: regime detection catches what the headline Sharpe ratio hides. + +Two strategies are calibrated to the SAME overall Sharpe. +One earns its return evenly across all market conditions. The other earns +everything in calm markets and gives it back when volatility arrives. + +The headline number cannot tell them apart. Regime analysis can. +""" + +import numpy as np +import pandas as pd + +from algosystem.backtesting.domain.equity_curve import EquityCurve +from algosystem.shared.errors import ValidationError +from algosystem.validation.application.analyze_regimes import AnalyzeRegimes + +N = 750 +BREAK = N // 2 +rng = np.random.default_rng(20260918) + +# A market that is calm for the first half and stormy for the second. +market = np.concatenate( + [rng.normal(0.0004, 0.004, BREAK), rng.normal(-0.0002, 0.025, N - BREAK)] +) + +# Strategy A: a steady edge, indifferent to the market's state. +robust = rng.normal(0.00075, 0.009, N) + +# Strategy B: a big edge while calm, bleeding once volatility arrives. +_fragile = np.concatenate( + [rng.normal(0.0026, 0.009, BREAK), rng.normal(-0.0011, 0.009, N - BREAK)] +) + + +def curve(returns): + levels = np.concatenate([[100.0], 100.0 * np.cumprod(1.0 + returns)]) + index = pd.date_range("2023-01-02", periods=len(levels), freq="B") + return EquityCurve.from_series(pd.Series(levels, index=index)) + + +def sharpe(returns): + return float(np.mean(returns) / np.std(returns, ddof=1) * np.sqrt(252)) + + +def matched_to(target, returns): + """Shift a return series so its overall Sharpe equals `target` exactly. + + A constant shift leaves the standard deviation untouched, so this changes + only the headline number -- the regime *structure* is preserved. + """ + sd = np.std(returns, ddof=1) + return returns + (target * sd / np.sqrt(252) - np.mean(returns)) + + +# Calibrate B to the SAME overall Sharpe as A, so only regime analysis separates them. +fragile = matched_to(sharpe(robust), _fragile) + + +market_curve = curve(market) +analyzer = AnalyzeRegimes() + +print("=" * 66) +print("HEADLINE NUMBERS -- these are what a tearsheet shows you") +print("=" * 66) +print(f" Strategy A (robust) annualized Sharpe : {sharpe(robust):.2f}") +print(f" Strategy B (fragile) annualized Sharpe : {sharpe(fragile):.2f}") +print("\n Same number. A tearsheet cannot separate these. Now split by regime.\n") + +for name, returns in [("A (robust)", robust), ("B (fragile)", fragile)]: + result = analyzer.execute(curve(returns), market_returns=market_curve) + print("=" * 66) + print(f"STRATEGY {name}") + print("=" * 66) + print("\n".join(" " + line for line in result.summary())) + verdict = "SAFE" if result.works_in_all_regimes else "DO NOT TRADE AS SIZED" + print(f"\n works_in_all_regimes = {result.works_in_all_regimes} -> {verdict}\n") + +print("=" * 66) +print("GUARDRAILS") +print("=" * 66) + +# Too little data to classify -- refuses rather than printing zeros. +try: + analyzer.execute(np.full(15, 0.001)) +except ValidationError as exc: + print(f" short series -> ValidationError: {exc}") + +# Benchmark from a different period -- refuses rather than misaligning. +stale = EquityCurve.from_series( + pd.Series( + np.concatenate([[100.0], 100.0 * np.cumprod(1.0 + market)]), + index=pd.date_range("2010-01-04", periods=N + 1, freq="B"), + ) +) +try: + analyzer.execute(curve(fragile), market_returns=stale) +except ValidationError as exc: + print(f" mismatched dates -> ValidationError: {exc}") + +print("\n Both would otherwise have produced confident, meaningless output.")