Skip to content
1 change: 1 addition & 0 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions crates/libsy-llm-client/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,7 @@ parking_lot.workspace = true
http.workspace = true
httpdate.workspace = true
serde_json.workspace = true
serde.workspace = true
tokio.workspace = true
tracing.workspace = true
tracing-opentelemetry.workspace = true
Expand Down
4 changes: 2 additions & 2 deletions crates/libsy-llm-client/src/client.rs
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
// SPDX-License-Identifier: Apache-2.0

//! [`TranslatingLlmClient`] — the crate's single public entry point: encode a neutral
//! [`TranslatingLlmClient`]: encode a neutral
//! request, call the configured backend over HTTP, decode the neutral response.

use std::collections::{BTreeMap, BTreeSet, HashMap};
Expand Down Expand Up @@ -896,7 +896,7 @@ fn record_gen_ai_request(url: &str, model: &str, streaming: bool) {
}
}

fn convert_reqwest_error(error: reqwest::Error) -> LlmClientError {
pub(crate) fn convert_reqwest_error(error: reqwest::Error) -> LlmClientError {
// Reqwest labels truncated or otherwise unreadable response bodies as decode
// errors, so distinguish them from serde JSON failures at the call site.
let error = error.without_url();
Expand Down
7 changes: 6 additions & 1 deletion crates/libsy-llm-client/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,9 @@
//! back to a [`switchyard_protocol::Response`] — supporting both buffered and
//! streamed responses.
//!
//! [`SystemOneClient`] serves typed decision requests. Register decision targets
//! with [`ClientRouter::with_decision_clients`] to serve them alongside LLM calls.
//!
//! [`run()`] pairs the client with a libsy algorithm: it drives
//! [`switchyard_libsy::Algorithm::run_stream`], serves routing-time calls, and makes the terminal
//! answer call from the routing outcome when needed. A host that just wants the answer does not
Expand All @@ -24,14 +27,16 @@ mod observability;
mod observation;
pub mod raw;
pub mod run;
mod system_one;

pub use backend::{Backend, DEFAULT_MAX_RETRIES, HttpBackendConfig};
pub use client::{AuxiliaryOperation, ModelConfig, TranslatingLlmClient};
pub use error::{LlmClientError, Result};
pub use observation::{LlmCallObservation, RunObservation, RunObserver};
pub use observation::{ModelCallObservation, RunObservation, RunObserver};
Comment thread
ayushag-nv marked this conversation as resolved.
pub use raw::RawResponse;
pub use run::{ClientRouter, decide, run};
pub use switchyard_translation::RawEventStream;
pub use system_one::SystemOneClient;

/// Registers process-wide compatibility gauges with the global meter provider.
pub fn initialize_metrics() {
Expand Down
10 changes: 6 additions & 4 deletions crates/libsy-llm-client/src/observation.rs
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,7 @@ use switchyard_protocol::{ModelId, Usage};

/// One completed model call observed while serving an algorithm run.
#[derive(Clone, Debug)]
pub struct LlmCallObservation {
pub struct ModelCallObservation {
/// Model selected for the completed call.
pub selected_model: ModelId,
/// Whether the call completed successfully.
Expand All @@ -22,15 +22,17 @@ pub struct LlmCallObservation {
pub usage: Option<Usage>,
}

/// Events emitted inline while [`crate::run`] serves a routing request.
/// Events emitted inline while [`crate::run()`] serves a routing request.
#[derive(Clone, Debug)]
pub enum RunObservation {
/// Metadata attached to the completed routing outcome.
Outcome(OutcomeMetadata),
/// A completed model call requested by the algorithm for routing work.
LlmCall(LlmCallObservation),
LlmCall(ModelCallObservation),
/// A completed decision call requested by the algorithm for routing work.
DecisionCall(ModelCallObservation),
/// A completed terminal model call made from the routing outcome.
AnswerCall(LlmCallObservation),
AnswerCall(ModelCallObservation),
/// Routing time recorded by the `switchyard.routing_overhead_ms` metric.
RoutingOverhead(Duration),
}
Expand Down
Loading
Loading