Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
48 commits
Select commit Hold shift + click to select a range
a68d670
feat(sim): add Python task routing evaluation
ayushag-nv Sep 30, 2026
9b936f1
fix(sim): preserve valid trials beside malformed artifacts
ayushag-nv Sep 30, 2026
74ed6a6
fix(sim): retain routing diagnostics and validate CLI options early
ayushag-nv Sep 30, 2026
dd574d9
feat(sim): support ATIF trajectories and custom converters
ayushag-nv Sep 30, 2026
ba256f7
fix(python): preserve decision headers and redact invalid credentials
ayushag-nv Sep 30, 2026
dd36d12
fix(sim): validate and snapshot CLI configuration before output
ayushag-nv Sep 30, 2026
a734f16
fix(sim): release routing session state after task decisions
ayushag-nv Sep 30, 2026
6394e14
fix(runner): reject ambiguous completion target identities
ayushag-nv Sep 30, 2026
01eb72b
docs(sim): clarify recovery and input ownership
ayushag-nv Sep 30, 2026
adb258b
fix(python): join native decision workers on completion
ayushag-nv Sep 30, 2026
33c6b41
docs(sim): show policy comparison and held-out evaluation
ayushag-nv Sep 30, 2026
0d86f53
test(python): cover response-based decision outcomes
ayushag-nv Sep 30, 2026
f0c84dd
fix(libsy): stop cleanup workers when routers are dropped
ayushag-nv Sep 30, 2026
3e2b02d
test(python): cover provider-specific decision usage
ayushag-nv Sep 30, 2026
b076858
fix(python): mark public package as typed
ayushag-nv Sep 30, 2026
ed78294
fix(sim): reject invalid task checksum evidence
ayushag-nv Sep 30, 2026
d08fea4
test(python): cover usage retained after a later call fails
ayushag-nv Sep 30, 2026
37cd3f2
docs(python): show configured routing decisions
ayushag-nv Sep 30, 2026
bf470e6
fix(sim): reject blank rejected-trial identities
ayushag-nv Sep 30, 2026
c56ab87
docs(sim): clarify result callback and write ownership
ayushag-nv Sep 30, 2026
fbb6b5e
test(python): smoke public simulation imports in wheels
ayushag-nv Sep 30, 2026
148e606
test(sim): guard native-free data preparation
ayushag-nv Sep 30, 2026
e66b9f8
fix(python): accept mapping headers in routing decisions
ayushag-nv Sep 30, 2026
8b2bc85
docs(sim): clarify repeat inputs and baseline validation
ayushag-nv Sep 30, 2026
f1aeab4
fix(sim): detect incomplete Harbor job copies
ayushag-nv Sep 30, 2026
7eb48cf
fix(runner): scope target alias validation to decisions
ayushag-nv Sep 30, 2026
5b1ae69
docs(sim): retain rows for reproducible policy comparisons
ayushag-nv Sep 30, 2026
13120b3
test(sim): cover native completion prompt scoring guards
ayushag-nv Sep 30, 2026
cd2e62c
perf(runner): precompute decision target validation
ayushag-nv Sep 30, 2026
41dc55b
fix(sim): preserve context when supplying missing task input
ayushag-nv Sep 30, 2026
9742f8e
fix(sim): keep CLI output outside recorded runs
ayushag-nv Sep 30, 2026
3256e44
fix(sim): reject duplicate Harbor trial directory aliases
ayushag-nv Sep 30, 2026
4474e45
docs(sim): explain importer and dataset memory ownership
ayushag-nv Sep 30, 2026
ca7f236
perf(sim): release excluded trials before CLI routing
ayushag-nv Sep 30, 2026
155bcb2
fix(sim): preserve finite means and signed reward totals
ayushag-nv Sep 30, 2026
3f2e66e
docs(sim): clarify source setup and classifier input context
ayushag-nv Sep 30, 2026
496bbd5
test(python): cover composite child routing decisions
ayushag-nv Sep 30, 2026
39dba2b
docs(sim): clarify provider configuration scoring boundary
ayushag-nv Sep 30, 2026
6242afe
fix(sim): retain import options in evaluation manifests
ayushag-nv Sep 30, 2026
f54655a
test(python): cover concurrent decision request ownership
ayushag-nv Sep 30, 2026
cd23e00
perf(sim): avoid copying rows before JSON encoding
ayushag-nv Sep 30, 2026
1880a60
docs(sim): document architecture and extension boundaries
ayushag-nv Sep 30, 2026
5556178
fix(python): redact invalid Unicode header values
ayushag-nv Sep 30, 2026
41371ba
docs(sim): explain unscored tasks and routing failures
ayushag-nv Sep 30, 2026
c5352fd
fix(python): redact invalid Unicode configuration sources
ayushag-nv Sep 30, 2026
fe1f2ab
fix(sim): retain issues for excessively nested JSON
ayushag-nv Sep 30, 2026
e20b2d4
test(sim): respect interpreter JSON nesting limits
ayushag-nv Sep 30, 2026
8201d1a
fix(python): bound recursive native conversion
ayushag-nv Sep 30, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 16 additions & 0 deletions .github/workflows/package-portability.yml
Original file line number Diff line number Diff line change
Expand Up @@ -84,16 +84,24 @@ jobs:
import os
import sys
import tempfile
from importlib.resources import files

os.chdir(tempfile.mkdtemp(prefix="switchyard-wheel-smoke-"))
workspace = os.environ.get("GITHUB_WORKSPACE")
sys.path = [entry for entry in sys.path if entry not in ("", workspace)]

import switchyard
import switchyard_rust
from switchyard.runner import Runner
from switchyard.sim import Trajectory
from switchyard_rust import _switchyard_rust

for package in ("switchyard", "switchyard_rust"):
assert files(package).joinpath("py.typed").is_file(), package

print("switchyard", switchyard.__version__)
print("runner", Runner)
print("trajectory", Trajectory)
print("switchyard_rust", switchyard_rust.__file__)
print("rust extension", _switchyard_rust.__name__)
PY
Expand All @@ -109,16 +117,24 @@ jobs:
import os
import sys
import tempfile
from importlib.resources import files

os.chdir(tempfile.mkdtemp(prefix="switchyard-wheel-smoke-"))
workspace = os.environ.get("GITHUB_WORKSPACE")
sys.path = [entry for entry in sys.path if entry not in ("", workspace)]

import switchyard
import switchyard_rust
from switchyard.runner import Runner
from switchyard.sim import Trajectory
from switchyard_rust import _switchyard_rust

for package in ("switchyard", "switchyard_rust"):
assert files(package).joinpath("py.typed").is_file(), package

print("switchyard", switchyard.__version__)
print("runner", Runner)
print("trajectory", Trajectory)
print("switchyard_rust", switchyard_rust.__file__)
print("rust extension", _switchyard_rust.__name__)
PY
8 changes: 8 additions & 0 deletions .github/workflows/publish.yml
Original file line number Diff line number Diff line change
Expand Up @@ -278,10 +278,14 @@ jobs:

import switchyard
import switchyard_rust
from switchyard.runner import Runner
from switchyard.sim import Trajectory
from switchyard_rust import _switchyard_rust
from switchyard_rust.server import Server

print("switchyard", switchyard.__version__)
print("runner", Runner)
print("trajectory", Trajectory)
print("switchyard_rust", switchyard_rust.__file__)
print("rust extension", _switchyard_rust.__name__)
print("server", Server)
Expand All @@ -307,10 +311,14 @@ jobs:

import switchyard
import switchyard_rust
from switchyard.runner import Runner
from switchyard.sim import Trajectory
from switchyard_rust import _switchyard_rust
from switchyard_rust.server import Server

print("switchyard", switchyard.__version__)
print("runner", Runner)
print("trajectory", Trajectory)
print("switchyard_rust", switchyard_rust.__file__)
print("rust extension", _switchyard_rust.__name__)
print("server", Server)
Expand Down
2 changes: 2 additions & 0 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

6 changes: 5 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -41,7 +41,9 @@ Agent-specific guides are available for [pi](docs/integrations/pi.md) and

### Embed the library in your harness

[Embed the library in your harness](docs/getting_started.md#library-path) to run routing inside your Rust application. For Python, see the [embedding example](examples/libsy.py).
[Embed the library in your harness](docs/getting_started.md#library-path) to run routing inside your Rust application.
For Python, use the [configured decision API](docs/simulation.md#use-the-python-decision-api)
or the [algorithm stream example](examples/libsy.py).

## Routing algorithms

Expand All @@ -63,6 +65,8 @@ The [routing overview](docs/routing_algorithms/overview.md) retains the full cat
![Task completion versus cost for Switchyard classification, stage, and escalation routing, compared with Opus 4.8 and GLM 5.2 single-model baselines.](assets/switchyard-cost-accuracy.png)

Results depend on the benchmark, model pool, serving stack, and routing configuration.
Use the Python [task routing evaluator](docs/simulation.md) to compare routing decisions against ATIF trajectories, custom recordings, or Harbor Claude and Codex runs.
See its [architecture](docs/simulation-architecture.md) for the ATIF converter boundary, routing and scoring flow, and future replay extension.
For latency and routing overhead testing, see [Soak Testing](docs/operations/soak_test.md).

### Further reading
Expand Down
2 changes: 1 addition & 1 deletion crates/libsy-llm-client/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,7 @@ pub use client::{AuxiliaryOperation, ModelConfig, TranslatingLlmClient};
pub use error::{LlmClientError, Result};
pub use observation::{LlmCallObservation, RunObservation, RunObserver};
pub use raw::RawResponse;
pub use run::{ClientRouter, decide, run};
pub use run::{ClientRouter, decide, decide_with_observer, run};
pub use switchyard_translation::RawEventStream;

/// Registers process-wide compatibility gauges with the global meter provider.
Expand Down
4 changes: 2 additions & 2 deletions crates/libsy-llm-client/src/observation.rs
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,7 @@ pub struct LlmCallObservation {
pub usage: Option<Usage>,
}

/// Events emitted inline while [`crate::run`] serves a routing request.
/// Events emitted inline by [`crate::run`] and [`crate::decide_with_observer`].
#[derive(Clone, Debug)]
pub enum RunObservation {
/// Metadata attached to the completed routing outcome.
Expand All @@ -31,7 +31,7 @@ pub enum RunObservation {
LlmCall(LlmCallObservation),
/// A completed terminal model call made from the routing outcome.
AnswerCall(LlmCallObservation),
/// Routing time recorded by the `switchyard.routing_overhead_ms` metric.
/// Elapsed routing time, including model calls required by the algorithm.
RoutingOverhead(Duration),
}

Expand Down
127 changes: 115 additions & 12 deletions crates/libsy-llm-client/src/run.rs
Original file line number Diff line number Diff line change
Expand Up @@ -131,23 +131,57 @@ pub async fn decide(
request: Request,
models: Arc<RuntimeModels>,
) -> Result<RoutingOutcome> {
decide_with_observer(algorithm, clients, request, models, None).await
}

/// Resolve a decision and report its completed calls and elapsed routing time.
///
/// Observations belong to this invocation, including calls completed before a failure.
/// A routing-time response is reported as an answer call when it becomes the outcome.
pub async fn decide_with_observer(
algorithm: Arc<dyn Algorithm>,
clients: ClientRouter,
request: Request,
models: Arc<RuntimeModels>,
observer: Option<RunObserver>,
) -> Result<RoutingOutcome> {
let started = Instant::now();
let routing_clients = clients.clone();
let mut outcome = match clients.stored_state_owner(&request) {
Some(owner) => continue_on(owner, algorithm.name(), request),
let observations = observer.as_ref().map(|_| Arc::new(Mutex::new(Vec::new())));
let outcome = match clients.stored_state_owner(&request) {
Some(owner) => Ok(continue_on(owner, algorithm.name(), request)),
None => {
drive(algorithm, request, models, move |call| {
serve(routing_clients.clone(), call, None)
drive(algorithm, request, models, {
let observations = observations.clone();
move |call| serve(routing_clients.clone(), call, observations.clone())
})
.await?
.await
}
};
let selected_model_id = outcome.selected_model_id()?.clone();
outcome.request = clients.prepare_completion_request(outcome.request, &selected_model_id);
outcome.response = outcome
.response
.map(|response| clients.remember_state_owner(&outcome.request, response))
.transpose()?;
Ok(outcome)
let answered_model = outcome
.as_ref()
.ok()
.and_then(|outcome| outcome.response.as_ref())
.and_then(Response::served_model);
emit_routing_observations(&observer, &observations, answered_model);
let outcome = outcome.and_then(|mut outcome| {
let selected_model_id = outcome.selected_model_id()?.clone();
outcome.request = clients.prepare_completion_request(outcome.request, &selected_model_id);
outcome.response = outcome
.response
.map(|response| clients.remember_state_owner(&outcome.request, response))
.transpose()?;
Ok(outcome)
});
if let Some(observer) = observer {
if let Ok(outcome) = &outcome
&& let Some(metadata) = &outcome.metadata
{
observer(RunObservation::Outcome(metadata.clone()));
}
observer(RunObservation::RoutingOverhead(started.elapsed()));
}
outcome
}

/// Emits completed routing calls after the outcome reveals whether one response became the answer.
Expand Down Expand Up @@ -566,6 +600,11 @@ enum Routing {
}

impl ClientRouter {
/// Configured target that may produce an answer during routing.
pub fn routing_answer_target(&self) -> Option<&ModelId> {
self.inner.routing_answer_target.as_ref()
}

/// Build a router over `model name -> client`, for targets spread across providers.
pub fn new(by_model: HashMap<ModelId, Arc<dyn RoutedLlmClient>>) -> Self {
Self::new_with_target_prompts(by_model, HashMap::new(), None)
Expand Down Expand Up @@ -1817,6 +1856,70 @@ mod tests {
Ok(())
}

#[tokio::test]
async fn observed_decision_does_not_call_a_selected_completion() -> Result<()> {
let client = Arc::new(CandidateClient {
calls: Mutex::new(Vec::new()),
requests: Mutex::new(Vec::new()),
first: FirstOutcome::StreamSuccess,
});
let observations = Arc::new(Mutex::new(Vec::new()));
let captured = Arc::clone(&observations);
let observer: RunObserver = Arc::new(move |event| captured.lock().push(event));

let outcome = decide_with_observer(
Arc::new(CandidateAlgorithm {}),
ClientRouter::single(client.clone()),
request(),
to_category_map(&["weak", "strong"]),
Some(observer),
)
.await?;

assert_eq!(outcome.selected_model_id()?, "weak");
assert!(outcome.response.is_none());
assert!(client.calls.lock().is_empty());
assert!(matches!(
&observations.lock()[..],
[
RunObservation::Outcome(_),
RunObservation::RoutingOverhead(_)
]
));
Ok(())
}

#[tokio::test]
async fn observed_decision_reports_completed_calls_on_failure() {
let client = Arc::new(CandidateClient {
calls: Mutex::new(Vec::new()),
requests: Mutex::new(Vec::new()),
first: FirstOutcome::Unauthorized,
});
let observations = Arc::new(Mutex::new(Vec::new()));
let captured = Arc::clone(&observations);
let observer: RunObserver = Arc::new(move |event| captured.lock().push(event));

let result = decide_with_observer(
Arc::new(AnsweredAlgorithm {
model: "weak".into(),
}),
ClientRouter::single(client.clone()),
request(),
to_category_map(&["weak", "strong"]),
Some(observer),
)
.await;

assert!(result.is_err());
assert_eq!(&*client.calls.lock(), &[ModelId::from("weak")]);
assert!(matches!(
&observations.lock()[..],
[RunObservation::LlmCall(call), RunObservation::RoutingOverhead(_)]
if call.selected_model == "weak" && !call.is_success && call.usage.is_none()
));
}

#[tokio::test]
async fn decision_prompts_a_routing_response_target() -> Result<()> {
let client = Arc::new(CandidateClient {
Expand Down
Loading
Loading