Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions crates/libsy-llm-client/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,7 @@ parking_lot.workspace = true
http.workspace = true
httpdate.workspace = true
serde_json.workspace = true
serde.workspace = true
tokio.workspace = true
tracing.workspace = true
tracing-opentelemetry.workspace = true
Expand Down
4 changes: 2 additions & 2 deletions crates/libsy-llm-client/src/client.rs
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
// SPDX-License-Identifier: Apache-2.0

//! [`TranslatingLlmClient`] — the crate's single public entry point: encode a neutral
//! [`TranslatingLlmClient`]: encode a neutral
//! request, call the configured backend over HTTP, decode the neutral response.

use std::collections::{BTreeMap, BTreeSet, HashMap};
Expand Down Expand Up @@ -896,7 +896,7 @@ fn record_gen_ai_request(url: &str, model: &str, streaming: bool) {
}
}

fn convert_reqwest_error(error: reqwest::Error) -> LlmClientError {
pub(crate) fn convert_reqwest_error(error: reqwest::Error) -> LlmClientError {
// Reqwest labels truncated or otherwise unreadable response bodies as decode
// errors, so distinguish them from serde JSON failures at the call site.
let error = error.without_url();
Expand Down
7 changes: 6 additions & 1 deletion crates/libsy-llm-client/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,9 @@
//! back to a [`switchyard_protocol::Response`] — supporting both buffered and
//! streamed responses.
//!
//! [`SystemOneClient`] serves typed decision requests. Register decision targets
//! with [`ClientRouter::with_decision_clients`] to serve them alongside LLM calls.
//!
//! [`run()`] pairs the client with a libsy algorithm: it drives
//! [`switchyard_libsy::Algorithm::run_stream`], serves routing-time calls, and makes the terminal
//! answer call from the routing outcome when needed. A host that just wants the answer does not
Expand All @@ -24,14 +27,16 @@ mod observability;
mod observation;
pub mod raw;
pub mod run;
mod system_one;

pub use backend::{Backend, DEFAULT_MAX_RETRIES, HttpBackendConfig};
pub use client::{AuxiliaryOperation, ModelConfig, TranslatingLlmClient};
pub use error::{LlmClientError, Result};
pub use observation::{LlmCallObservation, RunObservation, RunObserver};
pub use observation::{LlmCallObservation, ModelCallObservation, RunObservation, RunObserver};
pub use raw::RawResponse;
pub use run::{ClientRouter, decide, run};
pub use switchyard_translation::RawEventStream;
pub use system_one::SystemOneClient;

/// Registers process-wide compatibility gauges with the global meter provider.
pub fn initialize_metrics() {
Expand Down
9 changes: 7 additions & 2 deletions crates/libsy-llm-client/src/observation.rs
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,7 @@ use switchyard_protocol::{ModelId, Usage};

/// One completed model call observed while serving an algorithm run.
#[derive(Clone, Debug)]
pub struct LlmCallObservation {
pub struct ModelCallObservation {
/// Model selected for the completed call.
pub selected_model: ModelId,
/// Whether the call completed successfully.
Expand All @@ -22,13 +22,18 @@ pub struct LlmCallObservation {
pub usage: Option<Usage>,
}

/// Events emitted inline while [`crate::run`] serves a routing request.
/// An LLM call's observation, sharing the fields used by decision calls.
pub type LlmCallObservation = ModelCallObservation;

/// Events emitted inline while [`crate::run()`] serves a routing request.
#[derive(Clone, Debug)]
pub enum RunObservation {
/// Metadata attached to the completed routing outcome.
Outcome(OutcomeMetadata),
/// A completed model call requested by the algorithm for routing work.
LlmCall(LlmCallObservation),
/// A completed decision call requested by the algorithm for routing work.
DecisionCall(ModelCallObservation),
/// A completed terminal model call made from the routing outcome.
AnswerCall(LlmCallObservation),
/// Routing time recorded by the `switchyard.routing_overhead_ms` metric.
Expand Down
Loading
Loading