From a608c33f5be5d084892e9bc3713ada6ad04659fc Mon Sep 17 00:00:00 2001 From: Bryan Bednarski Date: Wed, 19 Aug 2026 14:12:46 -0700 Subject: [PATCH 1/4] feat(relay): add Switchyard-owned HTTP dynamic plugin Signed-off-by: Bryan Bednarski --- CHANGELOG.md | 15 + Cargo.lock | 499 ++++++---- Cargo.toml | 3 + README.md | 2 + .../switchyard-nemo-relay-plugin/Cargo.toml | 30 + crates/switchyard-nemo-relay-plugin/README.md | 400 ++++++++ .../config.schema.json | 217 +++++ .../relay-plugin.toml | 31 + .../scripts/package_bundle.py | 99 ++ .../src/client.rs | 463 ++++++++++ .../src/config.rs | 587 ++++++++++++ .../src/config/tests.rs | 545 +++++++++++ .../src/executor.rs | 137 +++ .../switchyard-nemo-relay-plugin/src/ffi.rs | 444 +++++++++ .../switchyard-nemo-relay-plugin/src/lib.rs | 442 +++++++++ .../src/runtime.rs | 675 ++++++++++++++ .../src/runtime/tests.rs | 874 ++++++++++++++++++ .../src/translation.rs | 116 +++ .../tests/test_package_bundle.py | 76 ++ docs/index.md | 2 + 20 files changed, 5480 insertions(+), 177 deletions(-) create mode 100644 crates/switchyard-nemo-relay-plugin/Cargo.toml create mode 100644 crates/switchyard-nemo-relay-plugin/README.md create mode 100644 crates/switchyard-nemo-relay-plugin/config.schema.json create mode 100644 crates/switchyard-nemo-relay-plugin/relay-plugin.toml create mode 100644 crates/switchyard-nemo-relay-plugin/scripts/package_bundle.py create mode 100644 crates/switchyard-nemo-relay-plugin/src/client.rs create mode 100644 crates/switchyard-nemo-relay-plugin/src/config.rs create mode 100644 crates/switchyard-nemo-relay-plugin/src/config/tests.rs create mode 100644 crates/switchyard-nemo-relay-plugin/src/executor.rs create mode 100644 crates/switchyard-nemo-relay-plugin/src/ffi.rs create mode 100644 crates/switchyard-nemo-relay-plugin/src/lib.rs create mode 100644 crates/switchyard-nemo-relay-plugin/src/runtime.rs create mode 100644 crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs create mode 100644 crates/switchyard-nemo-relay-plugin/src/translation.rs create mode 100644 crates/switchyard-nemo-relay-plugin/tests/test_package_bundle.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 6a5e3cd1a..f64803eb6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,21 @@ adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ### Added +- **NeMo Relay native plugin** — a dynamically loaded integration that runs + libsy's weighted-random, LLM-classifier, escalation, and stage-router + algorithms in process while Switchyard owns provider HTTP dispatch, + credentials, translation, retries, and fallback. Managed calls require NeMo + Relay 0.7 or newer and do not depend on `switchyard-server`. Target bindings + accept non-secret `extra_body` provider defaults, preserved requests are + re-encoded after routing mutations, and synthetic Relay gateway identities do + not become shared router session state. + +- **NeMo Relay routing-model usage marks** — classifier judges, escalation + judges and discarded weak candidates, and failed routing candidates now emit + `switchyard.routing.llm_call` ATOF marks with normalized token usage and + latency. The final serving call remains represented only by Relay's outer LLM + lifecycle event to prevent double-counting. + - **Advisor-gate routing** — new `advisor` route type pairing the serving executor with a stronger judge-only advisor that reviews terminal turns: APPROVE releases the buffered turn, REDO discards it and feeds the advisor's diff --git a/Cargo.lock b/Cargo.lock index 405dda9da..b655d02b4 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -18,9 +18,9 @@ dependencies = [ [[package]] name = "aho-corasick" -version = "1.1.4" +version = "1.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ddd31a130427c27518df266943a5308ed92d4b226cc639f5a8f1002816174301" +checksum = "c982642fa9e8606056828ee9a8505737230110bb1099153c79efe865c59d12ba" dependencies = [ "memchr", ] @@ -106,6 +106,18 @@ dependencies = [ "serde_json", ] +[[package]] +name = "async-channel" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "924ed96dd52d1b75e9c1a3e6275715fd320f5f9439fb5a4a11fa51f4221158d2" +dependencies = [ + "concurrent-queue", + "event-listener-strategy", + "futures-core", + "pin-project-lite", +] + [[package]] name = "async-stream" version = "0.3.6" @@ -130,13 +142,13 @@ dependencies = [ [[package]] name = "async-trait" -version = "0.1.89" +version = "0.1.92" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9035ad2d096bed7955a320ee7e2230574d28fd3c3a0f186cbea1ff3c7eed5dbb" +checksum = "82f6aeea286b8eb4dd3431a1be1b59d290ace00f5bfd8e2a159bc2a05e2c1667" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.3", ] [[package]] @@ -153,9 +165,9 @@ checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" [[package]] name = "aws-lc-rs" -version = "1.17.1" +version = "1.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4342d8937fc7e5dd9b1c60292261c0670c882a2cd1719cfc11b1af41731e32ad" +checksum = "ce2b2dcc879c3bae0d371e77c99f2238400ef24ec001394befa67b6e543add9e" dependencies = [ "aws-lc-sys", "zeroize", @@ -163,9 +175,9 @@ dependencies = [ [[package]] name = "aws-lc-sys" -version = "0.42.0" +version = "0.44.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6d9ceb1da931507a12f4fccea479dccd00da1943e1b4ae72d8e502d707361444" +checksum = "f09fae7be8bb3174e05c6afdb34199e6dc0c7c04ba9fa237b1967adfbde27483" dependencies = [ "cc", "cmake", @@ -274,6 +286,9 @@ name = "bitflags" version = "2.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" +dependencies = [ + "serde_core", +] [[package]] name = "borrow-or-share" @@ -301,9 +316,9 @@ checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" [[package]] name = "cc" -version = "1.2.67" +version = "1.4.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e17dd265a7d0f31ef544e1b20e03add05d3b45b491b633b10d67145d2acc1a38" +checksum = "509591b7bcd67f4ef775afad7662703b4935daaa6ec0e5605cfb1090b32a2b6d" dependencies = [ "find-msvc-tools", "jobserver", @@ -334,11 +349,21 @@ dependencies = [ "rand_core 0.10.1", ] +[[package]] +name = "chrono" +version = "0.4.45" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" +dependencies = [ + "num-traits", + "serde", +] + [[package]] name = "clap" -version = "4.6.2" +version = "4.6.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dd059f9da4f5c36b3787f65d38ccaab1cc315f07b01f89abc8359ee6a8205011" +checksum = "473c7e07f409a8d772161724aa8db6a765a2532a70f9667eeb7b49d3d02fbdca" dependencies = [ "clap_builder", "clap_derive", @@ -346,9 +371,9 @@ dependencies = [ [[package]] name = "clap_builder" -version = "4.6.2" +version = "4.6.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f09628afdcc538b57f3c6341e9c8e9970f18e4a481690a64974d7023bd33548b" +checksum = "7b48fea5a88e9ae728a2dcbedbfc0e730f7d60da42e1cb049a83c9fb8b789889" dependencies = [ "anstream", "anstyle", @@ -358,14 +383,14 @@ dependencies = [ [[package]] name = "clap_derive" -version = "4.6.1" +version = "4.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2ce8604710f6733aa641a2b3731eaa1e8b3d9973d5e3565da11800813f997a9" +checksum = "d012d2b9d65aca7f18f4d9878a045bc17899bba951561ba5ec3c2ba1eed9a061" dependencies = [ "heck", "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.3", ] [[package]] @@ -399,6 +424,15 @@ dependencies = [ "memchr", ] +[[package]] +name = "concurrent-queue" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4ca0197aee26d1ae37445ee532fefce43251d24cc7c166799f4d46817f1d3973" +dependencies = [ + "crossbeam-utils", +] + [[package]] name = "core-foundation" version = "0.10.1" @@ -424,6 +458,12 @@ dependencies = [ "libc", ] +[[package]] +name = "crossbeam-utils" +version = "0.8.22" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "61803da095bee82a81bb1a452ecc25d3b2f1416d1897eb86430c6159ef717c17" + [[package]] name = "data-encoding" version = "2.11.1" @@ -456,13 +496,13 @@ checksum = "56254986775e3233ffa9c4d7d3faaf6d36a2c09d30b20687e9f88bc8bafc16c8" [[package]] name = "displaydoc" -version = "0.2.6" +version = "0.2.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ac70aa55017e108007fbaf5aa0f54b021c98f92ff8af59d42eda9da96e3dd4f" +checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.3", ] [[package]] @@ -473,9 +513,9 @@ checksum = "92773504d58c093f6de2459af4af33faa518c13451eb8f2b5698ed3d36e7c813" [[package]] name = "either" -version = "1.16.0" +version = "1.17.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "91622ff5e7162018101f2fea40d6ebf4a78bbe5a49736a2020649edf9693679e" +checksum = "9e5e8f6c15a24b9a3ee5efec809ccd006d3b30e8b3bb63c39af737c7f87daa1d" [[package]] name = "email_address" @@ -502,11 +542,31 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "event-listener" +version = "5.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5a23add41df1562121a9393cb065eab5146a1242410f23a644851e90cfd669d2" +dependencies = [ + "parking", + "pin-project-lite", +] + +[[package]] +name = "event-listener-strategy" +version = "0.5.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8be9f3dfaaffdae2972880079a491a1a8bb7cbed0b8dd7a347f668b4150a3b93" +dependencies = [ + "event-listener", + "pin-project-lite", +] + [[package]] name = "fancy-regex" -version = "0.18.0" +version = "0.19.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e1e1dacd0d2082dfcf1351c4bdd566bbe89a2b263235a2b50058f1e130a47277" +checksum = "476de73bddf2ef8490aa4ee8f1cf40b430bf1d56c48c22080e5186952cd580e6" dependencies = [ "bit-set", "regex-automata", @@ -521,9 +581,9 @@ checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" [[package]] name = "find-msvc-tools" -version = "0.1.9" +version = "0.1.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" +checksum = "d45db016d36b838f563236e9193d0ee6ce38f3f68b6c94e914b4929c96bbb890" [[package]] name = "fluent-uri" @@ -585,9 +645,9 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c" [[package]] name = "futures" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b147ee9d1f6d097cef9ce628cd2ee62288d963e16fb287bd9286455b241382d" +checksum = "9a31d2a3fbaaeb2af2368bbdd904aa8e812d3c04a1ee10d3171f52d556e5d0a3" dependencies = [ "futures-channel", "futures-core", @@ -600,9 +660,9 @@ dependencies = [ [[package]] name = "futures-channel" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "07bbe89c50d7a535e539b8c17bc0b49bdb77747034daa8087407d655f3f7cc1d" +checksum = "b1f9e3d69d39e4862ffed03ed071a76f9a13ba1d9109d355b0f0aa6b15e393c4" dependencies = [ "futures-core", "futures-sink", @@ -610,15 +670,15 @@ dependencies = [ [[package]] name = "futures-core" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e3450815272ef58cec6d564423f6e755e25379b217b0bc688e295ba24df6b1d" +checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" [[package]] name = "futures-executor" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "baf29c38818342a3b26b5b923639e7b1f4a61fc5e76102d4b1981c6dc7a7579d" +checksum = "031b47cf1a3c6cc8bc2fc76cd437f521619387907d469316e7c0bc278f1f5432" dependencies = [ "futures-core", "futures-task", @@ -627,38 +687,38 @@ dependencies = [ [[package]] name = "futures-io" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cecba35d7ad927e23624b22ad55235f2239cfa44fd10428eecbeba6d6a717718" +checksum = "53c0fa8157de1303bfffdaa1cc2a673bfffb60102f76b0ef4441659124373fed" [[package]] name = "futures-macro" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e835b70203e41293343137df5c0664546da5745f82ec9b84d40be8336958447b" +checksum = "9fb9654ba8355388abeb8dcb4fc62f511300867002afc858860463bdd9fe0c44" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.3", ] [[package]] name = "futures-sink" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c39754e157331b013978ec91992bde1ac089843443c49cbc7f46150b0fad0893" +checksum = "1944426bf7d03f1d14f708785e4b33efd750b36d48a157b836b3efc15ede8e1d" [[package]] name = "futures-task" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "037711b3d59c33004d3856fbdc83b99d4ff37a24768fa1be9ce3538a1cde4393" +checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" [[package]] name = "futures-util" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "389ca41296e6190b48053de0321d02a77f32f8a5d2461dd38762c0593805c6d6" +checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" dependencies = [ "futures-channel", "futures-core", @@ -714,9 +774,9 @@ dependencies = [ [[package]] name = "h2" -version = "0.4.15" +version = "0.4.17" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6cb093c84e8bd9b188d4c4a8cb6579fc016968d14c99882163cd3ff402a4f155" +checksum = "9f877e75f39e9827ec50a572dd592684ac28c029578726c85f1b2aa6ab807449" dependencies = [ "atomic-waker", "bytes", @@ -756,9 +816,9 @@ checksum = "fc0fef456e4baa96da950455cd02c081ca953b141298e41db3fc7e36b1da849c" [[package]] name = "http" -version = "1.4.2" +version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6970f50e31d6fc17d3fa27329444bfa74e196cf62e95052a3f6fee181dba6425" +checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" dependencies = [ "bytes", "itoa", @@ -776,9 +836,9 @@ dependencies = [ [[package]] name = "http-body-util" -version = "0.1.4" +version = "0.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e9f41fd6a08e4d4ec69df65976da761afd5ad5e58a9d4acb46bd1c953a9e3ff2" +checksum = "23169fe34a5fbcdd3f3862e78fb9b6fccd5f02a6dc6f732547005d45631ce71c" dependencies = [ "bytes", "futures-core", @@ -807,9 +867,9 @@ checksum = "15cdd26707701c53297e2fa6afb323d55fbc1d0810c3aec078ae3ef0424c3c15" [[package]] name = "hyper" -version = "1.10.1" +version = "1.11.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "55281c53a1894c864990125767da440a4e630446785086f52523b20033b74498" +checksum = "d22053281f852e11534f5198498373cbb59295120a20771d90f7ed1897490a72" dependencies = [ "atomic-waker", "bytes", @@ -867,9 +927,9 @@ dependencies = [ [[package]] name = "icu_collections" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2984d1cd16c883d7935b9e07e44071dca8d917fd52ecc02c04d5fa0b5a3f191c" +checksum = "fa68d21081c4a05d5a901a1c62add574c77048b6a1c67be3b50ce0b60d4ca513" dependencies = [ "displaydoc", "potential_utf", @@ -881,9 +941,9 @@ dependencies = [ [[package]] name = "icu_locale_core" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92219b62b3e2b4d88ac5119f8904c10f8f61bf7e95b640d25ba3075e6cac2c29" +checksum = "d56e28588da92eee5c3201a6eff33fabdd49b62269c8938d4ff050ce4d900deb" dependencies = [ "displaydoc", "litemap", @@ -894,9 +954,9 @@ dependencies = [ [[package]] name = "icu_normalizer" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c56e5ee99d6e3d33bd91c5d85458b6005a22140021cc324cea84dd0e72cff3b4" +checksum = "12f9cf5f235641ed274641dd81c3f28d870e276763d0797aeeab72317b1c646f" dependencies = [ "icu_collections", "icu_normalizer_data", @@ -908,16 +968,17 @@ dependencies = [ [[package]] name = "icu_normalizer_data" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "da3be0ae77ea334f4da67c12f149704f19f81d1adf7c51cf482943e84a2bad38" +checksum = "1563da1ed3e0b3bf3d74c9b85917ac9c56464d2f57242270c09c9e752f8021a0" [[package]] name = "icu_properties" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bee3b67d0ea5c2cca5003417989af8996f8604e34fb9ddf96208a033901e70de" +checksum = "7e7ca276ad3145661a65914e6daf131ca5120cd3dcee8f8f3214b8875184a148" dependencies = [ + "displaydoc", "icu_collections", "icu_locale_core", "icu_properties_data", @@ -928,15 +989,15 @@ dependencies = [ [[package]] name = "icu_properties_data" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8e2bbb201e0c04f7b4b3e14382af113e17ba4f63e2c9d2ee626b720cbce54a14" +checksum = "e590f038c1464a96894fd6d10127e90a8be4509f56ff7ecef851b15cee0b7caa" [[package]] name = "icu_provider" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "139c4cf31c8b5f33d7e199446eff9c1e02decfc2f0eec2c8d71f65befa45b421" +checksum = "92a7ed671a6aad807a8651a2e1782a6598fda9ce5185dd8158549e95a91c6428" dependencies = [ "displaydoc", "icu_locale_core", @@ -980,9 +1041,9 @@ dependencies = [ [[package]] name = "ipnet" -version = "2.12.0" +version = "2.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d98f6fed1fde3f8c21bc40a1abb88dd75e67924f9cffc3ef95607bad8017f8e2" +checksum = "6a756c3fac73139e83f14c2d742155dd2b78d3ee56597b419a0579b7bdd6dd78" [[package]] name = "is_terminal_polyfill" @@ -1017,7 +1078,7 @@ dependencies = [ "jni-sys", "log", "simd_cesu8", - "thiserror 2.0.18", + "thiserror 2.0.20", "walkdir", "windows-link", ] @@ -1066,9 +1127,9 @@ dependencies = [ [[package]] name = "js-sys" -version = "0.3.103" +version = "0.3.104" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102" +checksum = "0e0c1080212aad755ea003d18543e8768dd432c48819efd73a7bf1e39b7a5a3a" dependencies = [ "cfg-if", "futures-util", @@ -1087,9 +1148,9 @@ dependencies = [ [[package]] name = "jsonschema" -version = "0.49.4" +version = "0.49.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "257549c093d8f3d043337ba0d507e8eeace2447dfae3f0ee37eb0f57d3df7655" +checksum = "59ec8a241beed129f06114aa68007e905ca350e7baeb6e17a7631bb7978d91b2" dependencies = [ "ahash", "bytecount", @@ -1116,18 +1177,18 @@ dependencies = [ [[package]] name = "jsonschema-regex" -version = "0.49.4" +version = "0.49.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "474790e948498099d61ec85c10dc72d459d61298b7975eda7fb163af1934b134" +checksum = "91994f45017ed5e66aa8e59b8415f4cb033a6380d7200387b7cf117595fbdf85" dependencies = [ "regex-syntax", ] [[package]] name = "jsonschema-value" -version = "0.49.4" +version = "0.49.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8212315eb8e0bc44959d1af653cd5ec515771a3cdca966cfc0a10999bff144b9" +checksum = "7ec7637f83e510868ae6ed625f7ebfbbde4554ee8ce49854caa5126a8b9b9ecb" dependencies = [ "ahash", "bytecount", @@ -1145,9 +1206,9 @@ checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe" [[package]] name = "libc" -version = "0.2.186" +version = "0.2.189" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68ab91017fe16c622486840e4c83c9a37afeff978bd239b5293d61ece587de66" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" [[package]] name = "linux-raw-sys" @@ -1157,9 +1218,9 @@ checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" [[package]] name = "litemap" -version = "0.8.2" +version = "0.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92daf443525c4cce67b150400bc2316076100ce0b3686209eb8cf3c31612e6f0" +checksum = "47d9d19d1d6efa0109d2f65ff4c85cddd50bd572e5a00127ab10987290bcefae" [[package]] name = "lock_api" @@ -1226,6 +1287,32 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "nemo-relay-plugin" +version = "0.8.0" +source = "git+https://github.com/NVIDIA/NeMo-Relay.git?rev=ca08901629e6058c2d5d65cc7708ec5264073d7b#ca08901629e6058c2d5d65cc7708ec5264073d7b" +dependencies = [ + "futures", + "nemo-relay-types", + "serde", + "serde_json", + "tokio", + "tokio-util", +] + +[[package]] +name = "nemo-relay-types" +version = "0.8.0" +source = "git+https://github.com/NVIDIA/NeMo-Relay.git?rev=ca08901629e6058c2d5d65cc7708ec5264073d7b#ca08901629e6058c2d5d65cc7708ec5264073d7b" +dependencies = [ + "bitflags", + "chrono", + "serde", + "serde_json", + "typed-builder", + "uuid", +] + [[package]] name = "nu-ansi-term" version = "0.50.3" @@ -1276,9 +1363,9 @@ dependencies = [ [[package]] name = "num-integer" -version = "0.1.46" +version = "0.1.47" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7969661fd2958a5cb096e56c8e1ad0444ac2bbcd0061bd28660485a44879858f" +checksum = "7ce2d95d4b3734dc35aa2f45e1aa22cd416814592a4f9d9205e11affd5b8e10b" dependencies = [ "num-traits", ] @@ -1351,7 +1438,7 @@ dependencies = [ "futures-sink", "js-sys", "pin-project-lite", - "thiserror 2.0.18", + "thiserror 2.0.20", "tracing", ] @@ -1381,7 +1468,7 @@ dependencies = [ "opentelemetry_sdk", "prost", "reqwest", - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] @@ -1421,7 +1508,7 @@ dependencies = [ "percent-encoding", "portable-atomic", "rand 0.9.5", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", ] @@ -1431,6 +1518,12 @@ version = "0.5.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1a80800c0488c3a21695ea981a54918fbb37abf04f4d0720c453632255e2ff0e" +[[package]] +name = "parking" +version = "2.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f38d5652c16fde515bb1ecef450ab0f6a219d619a7274976324d5e377f7dceba" + [[package]] name = "parking_lot" version = "0.12.5" @@ -1468,21 +1561,21 @@ checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" [[package]] name = "pkg-config" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e" +checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" [[package]] name = "portable-atomic" -version = "1.13.1" +version = "1.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c33a9471896f1c69cecef8d20cbe2f7accd12527ce60845ff44c153bb2a21b49" +checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" [[package]] name = "potential_utf" -version = "0.1.5" +version = "0.1.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0103b1cef7ec0cf76490e969665504990193874ea05c85ff9bab8b911d0a0564" +checksum = "d83eb9bc6d8e5cf568e7a1101d60ee05e81ed50ea106026f3d18deeb046d7661" dependencies = [ "zerovec", ] @@ -1508,9 +1601,9 @@ dependencies = [ [[package]] name = "proc-macro2" -version = "1.0.106" +version = "1.0.107" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fd00f0bb2e90d81d1044c2b32617f68fcb9fa3bb7640c23e9c748e53fb30934" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" dependencies = [ "unicode-ident", ] @@ -1527,7 +1620,7 @@ dependencies = [ "memchr", "parking_lot", "protobuf", - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] @@ -1669,7 +1762,7 @@ dependencies = [ "rustc-hash", "rustls", "socket2", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tracing", "web-time", @@ -1677,9 +1770,9 @@ dependencies = [ [[package]] name = "quinn-proto" -version = "0.11.16" +version = "0.11.17" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2f4bfc015262b9df63c8845072ce59068853ff5872180c2ce2f13038b970e560" +checksum = "04759210543be93709136e28212294a659ef5001836ff4eab4d663e4529bba83" dependencies = [ "aws-lc-rs", "bytes", @@ -1692,7 +1785,7 @@ dependencies = [ "rustls", "rustls-pki-types", "slab", - "thiserror 2.0.18", + "thiserror 2.0.20", "tinyvec", "tracing", "web-time", @@ -1714,9 +1807,9 @@ dependencies = [ [[package]] name = "quote" -version = "1.0.46" +version = "1.0.47" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dfbc457d0c7a0759a614551b11a6409e5951f6c7537be1f1b7682b9ae9230368" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" dependencies = [ "proc-macro2", ] @@ -1799,18 +1892,18 @@ dependencies = [ [[package]] name = "ref-cast" -version = "1.0.26" +version = "1.0.27" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "216e8f773d7923bcba9ceb86a86c93cabb3903a11872fc3f138c49630e50b96d" +checksum = "7e440fb4e4b4147295338efb76001ab9e4efc0e5839df2c47fc5ac2381d365c3" dependencies = [ "ref-cast-impl", ] [[package]] name = "ref-cast-impl" -version = "1.0.26" +version = "1.0.27" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2c9283685feec7d69af75fb0e858d5e7378f33fe4fc699383b2916ab9273e03c" +checksum = "92ecd8964f8453721699a1ed72037b0db49ce2f5a5138486ee89bed6f67cdf3a" dependencies = [ "proc-macro2", "quote", @@ -1819,9 +1912,9 @@ dependencies = [ [[package]] name = "referencing" -version = "0.49.4" +version = "0.49.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2bf1a74e036be2546c0c81cdfad6b2def6a25bf31c4e34a468037058d3bd05c6" +checksum = "6efa2154ea6f5ce0fdecdd2a8d18f2fa1a39a8fbba91564f555a592e4dce8278" dependencies = [ "ahash", "fluent-uri", @@ -1848,9 +1941,9 @@ dependencies = [ [[package]] name = "regex-automata" -version = "0.4.16" +version = "0.4.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fcfdb36bda0c880c5931cdc7a2bcdc8ba4556847b9d912bca70bc94708711ad" +checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" dependencies = [ "aho-corasick", "memchr", @@ -1948,9 +2041,9 @@ dependencies = [ [[package]] name = "rustls" -version = "0.23.42" +version = "0.23.43" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3c54fcab019b409d04215d3a17cb438fd7fbf192ee61461f20f4fe18704bc138" +checksum = "0283386ce02abc0151e1761d08802dfe86c173b0b494af5cbc086574e453da06" dependencies = [ "aws-lc-rs", "once_cell", @@ -1974,9 +2067,9 @@ dependencies = [ [[package]] name = "rustls-pki-types" -version = "1.15.0" +version = "1.15.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "764899a24af3980067ee14bc143654f297b22eaebfe3c7b6b211920a5a59b046" +checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" dependencies = [ "web-time", "zeroize", @@ -2011,9 +2104,9 @@ checksum = "f87165f0995f63a9fbeea62b64d10b4d9d8e78ec6d7d51fb2125fda7bb36788f" [[package]] name = "rustls-webpki" -version = "0.103.13" +version = "0.103.14" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "61c429a8649f110dddef65e2a5ad240f747e85f7758a6bccc7e5777bd33f756e" +checksum = "0527518605e68109d875e248ea259b6758801cf165e4b2c2733ae3b51f12535a" dependencies = [ "aws-lc-rs", "ring", @@ -2088,9 +2181,9 @@ checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd" [[package]] name = "serde" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" dependencies = [ "serde_core", "serde_derive", @@ -2098,29 +2191,29 @@ dependencies = [ [[package]] name = "serde_core" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" dependencies = [ "serde_derive", ] [[package]] name = "serde_derive" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.3", ] [[package]] name = "serde_json" -version = "1.0.150" +version = "1.0.151" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e8014e44b4736ed0538adeecded0fce2a272f22dc9578a7eb6b2d9993c74cfb9" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" dependencies = [ "itoa", "memchr", @@ -2280,7 +2373,7 @@ dependencies = [ "serde", "serde_json", "switchyard-protocol", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tokio-stream", "tracing", @@ -2305,7 +2398,7 @@ dependencies = [ "switchyard-libsy", "switchyard-protocol", "switchyard-translation", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tracing", "tracing-opentelemetry", @@ -2313,6 +2406,24 @@ dependencies = [ "wiremock", ] +[[package]] +name = "switchyard-nemo-relay-plugin" +version = "0.2.0" +dependencies = [ + "async-channel", + "async-trait", + "futures-util", + "http", + "nemo-relay-plugin", + "serde", + "serde_json", + "switchyard-libsy", + "switchyard-llm-client", + "switchyard-protocol", + "switchyard-translation", + "tokio", +] + [[package]] name = "switchyard-protocol" version = "0.2.0" @@ -2322,7 +2433,7 @@ dependencies = [ "http", "serde", "serde_json", - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] @@ -2384,7 +2495,7 @@ dependencies = [ "async-trait", "serde", "serde_json", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", ] @@ -2398,7 +2509,7 @@ dependencies = [ "serde", "serde_json", "switchyard-protocol", - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] @@ -2473,11 +2584,11 @@ dependencies = [ [[package]] name = "thiserror" -version = "2.0.18" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4288b5bcbc7920c07a1149a35cf9590a2aa808e0bc1eafaade0b80947865fbc4" +checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f" dependencies = [ - "thiserror-impl 2.0.18", + "thiserror-impl 2.0.20", ] [[package]] @@ -2493,13 +2604,13 @@ dependencies = [ [[package]] name = "thiserror-impl" -version = "2.0.18" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ebc4ee7f67670e9b64d05fa4253e753e016c6c95ff35b89b7941d6b856dec1d5" +checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.3", ] [[package]] @@ -2513,9 +2624,9 @@ dependencies = [ [[package]] name = "tinystr" -version = "0.8.3" +version = "0.8.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8323304221c2a851516f22236c5722a72eaa19749016521d6dff0824447d96d" +checksum = "b1e27c91459209c2986af3dcf603a5a74a4368754ce37414f59acc971167f643" dependencies = [ "displaydoc", "zerovec", @@ -2538,9 +2649,9 @@ checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" [[package]] name = "tokio" -version = "1.52.4" +version = "1.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "317fafbbe3f02fc663dad00ea6186197de963cd4190e86a26d8d0fae095539af" +checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" dependencies = [ "bytes", "libc", @@ -2555,13 +2666,13 @@ dependencies = [ [[package]] name = "tokio-macros" -version = "2.7.0" +version = "2.7.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "385a6cb71ab9ab790c5fe8d67f1645e6c450a7ce006a33de03daa956cf70a496" +checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.3", ] [[package]] @@ -2576,9 +2687,9 @@ dependencies = [ [[package]] name = "tokio-stream" -version = "0.1.18" +version = "0.1.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32da49809aab5c3bc678af03902d4ccddea2a87d028d86392a4b1560c6906c70" +checksum = "a3d06f0b082ba57c26b79407372e57cf2a1e28124f78e9479fe80322cf53420b" dependencies = [ "futures-core", "pin-project-lite", @@ -2587,22 +2698,24 @@ dependencies = [ [[package]] name = "tokio-util" -version = "0.7.18" +version = "0.7.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ae9cec805b01e8fc3fd2fe289f89149a9b66dd16786abd8b19cfa7b48cb0098" +checksum = "494815d09bf52b5548659851081238f0ca39ff638363907596da739561c62c52" dependencies = [ "bytes", "futures-core", "futures-sink", + "futures-util", + "libc", "pin-project-lite", "tokio", ] [[package]] name = "toml" -version = "1.1.3+spec-1.1.0" +version = "1.1.4+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53c96ecdfa941c8fc4fcaed14f99ada8ebed502eef533015095a07e3301d4c3c" +checksum = "3aace63f4bbcdfc2c965b059de67119c89c4017a70d633be6c104910f67056f5" dependencies = [ "indexmap", "serde_core", @@ -2624,9 +2737,9 @@ dependencies = [ [[package]] name = "toml_parser" -version = "1.1.2+spec-1.1.0" +version = "1.1.3+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a2abe9b86193656635d2411dc43050282ca48aa31c2451210f4202550afb7526" +checksum = "1d38ac1cf9b95face32296c0a3ede1fdc270627c9d9c02a7274dd6d960dc4d56" dependencies = [ "winnow", ] @@ -2767,6 +2880,26 @@ version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" +[[package]] +name = "typed-builder" +version = "0.23.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "31aa81521b70f94402501d848ccc0ecaa8f93c8eb6999eb9747e72287757ffda" +dependencies = [ + "typed-builder-macro", +] + +[[package]] +name = "typed-builder-macro" +version = "0.23.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "076a02dc54dd46795c2e9c8282ed40bcfb1e22747e955de9389a1de28190fb26" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + [[package]] name = "unicode-general-category" version = "1.1.0" @@ -2809,6 +2942,18 @@ version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" +[[package]] +name = "uuid" +version = "1.18.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2f87b8aa10b915a06587d0dec516c282ff295b475d94abf425d62b57710070a2" +dependencies = [ + "getrandom 0.3.4", + "js-sys", + "serde", + "wasm-bindgen", +] + [[package]] name = "uuid-simd" version = "0.8.0" @@ -2873,9 +3018,9 @@ dependencies = [ [[package]] name = "wasm-bindgen" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4" +checksum = "1b70935747edd64d89de3efa29d73789b806c15798f8e7dca4d8ac356b50ce70" dependencies = [ "cfg-if", "once_cell", @@ -2886,9 +3031,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-futures" -version = "0.4.76" +version = "0.4.77" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c62df1340f32221cb9c54d6a27b030e3dba64361d4a95bed55f9aacb44da291d" +checksum = "6b7777d5cc23d0e91404e53ce2d5e8ec7acae3026b16233dba62cd3246457950" dependencies = [ "js-sys", "wasm-bindgen", @@ -2896,9 +3041,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1" +checksum = "77775f8f3f7217702089053b94958f8f54061a3f663417df76e19cbdcca29bc1" dependencies = [ "quote", "wasm-bindgen-macro-support", @@ -2906,9 +3051,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro-support" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e" +checksum = "e11d33f857dc2fb11b8bc75aee111aa9cbeb12cd9f25efd3d4c2a3dd4e235284" dependencies = [ "bumpalo", "proc-macro2", @@ -2919,9 +3064,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-shared" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24" +checksum = "7ef64dbcc55df09c7e5a46182d181c2cfa3e925f3da937ea764728b4bbb9dcbf" dependencies = [ "unicode-ident", ] @@ -2941,9 +3086,9 @@ dependencies = [ [[package]] name = "web-sys" -version = "0.3.103" +version = "0.3.104" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8622dcb61c0bcc9fffa6938bed81210af2da9a7e4a1a834b2e37a59b6dfb6141" +checksum = "c435338968042f4f59a557f690a253676d47ce13ceb55d70100e7facf6620a30" dependencies = [ "js-sys", "wasm-bindgen", @@ -3102,9 +3247,9 @@ checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" [[package]] name = "writeable" -version = "0.6.3" +version = "0.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ffae5123b2d3fc086436f8834ae3ab053a283cfac8fe0a0b8eaae044768a4c4" +checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc" [[package]] name = "yansi" @@ -3137,18 +3282,18 @@ dependencies = [ [[package]] name = "zerocopy" -version = "0.8.54" +version = "0.8.56" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b7cbbc0a705a0fd05cc3676525980d2bf5a9bc4adac6d6475209a7887cf59d19" +checksum = "556764e583adb45a9f8d413c2a147fa7e8d821e48e12b14fd560b607998b75eb" dependencies = [ "zerocopy-derive", ] [[package]] name = "zerocopy-derive" -version = "0.8.54" +version = "0.8.56" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e2e817b7b52d0c7358d3246da9d69935ebb18116b2b102b4230dac079b4862f5" +checksum = "f2ab42fc20575779bd240faa45f94a74256f755c0fa9e89f0ede20d91d0cdfc1" dependencies = [ "proc-macro2", "quote", @@ -3184,9 +3329,9 @@ checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" [[package]] name = "zerotrie" -version = "0.2.4" +version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0f9152d31db0792fa83f70fb2f83148effb5c1f5b8c7686c3459e361d9bc20bf" +checksum = "4ea269c3bd32f0a32c321907a2ae912ba6f4649bb0fc764a15627e99a7095a3f" dependencies = [ "displaydoc", "yoke", @@ -3195,9 +3340,9 @@ dependencies = [ [[package]] name = "zerovec" -version = "0.11.6" +version = "0.11.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "90f911cbc359ab6af17377d242225f4d75119aec87ea711a880987b18cd7b239" +checksum = "94b5c6b5976d66c1d703c4fd17d3f5e43c8cedaacf604961b171adc7130896d8" dependencies = [ "yoke", "zerofrom", @@ -3206,13 +3351,13 @@ dependencies = [ [[package]] name = "zerovec-derive" -version = "0.11.3" +version = "0.11.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" +checksum = "9f212a141d820099d57ffafb9569be9617a6f27d3dc881fbee8fb56642f917a9" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.3", ] [[package]] diff --git a/Cargo.toml b/Cargo.toml index 98433d8c1..a8bba2c80 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -8,6 +8,7 @@ members = [ "crates/libsy-llm-client", "crates/switchyard-py", "crates/protocol", + "crates/switchyard-nemo-relay-plugin", "crates/switchyard-server", "crates/switchyard-skill-distillation", "crates/switchyard-translation", @@ -22,6 +23,7 @@ repository = "https://github.com/NVIDIA-NeMo/Switchyard" rust-version = "1.96.1" [workspace.dependencies] +async-channel = "2" async-stream = "0.3" async-trait = "0.1" futures = "0.3" @@ -30,6 +32,7 @@ http = "1" httpdate = "1" jsonschema = { version = "0.49.4", default-features = false } jsonptr = { version = "0.8.1", default-features = false, features = ["std", "json", "resolve"] } +nemo-relay-plugin = { git = "https://github.com/NVIDIA/NeMo-Relay.git", rev = "ca08901629e6058c2d5d65cc7708ec5264073d7b" } parking_lot = "0.12" rand = "0.10" regex = "1" diff --git a/README.md b/README.md index afeb70248..25e0a33ad 100644 --- a/README.md +++ b/README.md @@ -21,6 +21,7 @@ algorithm you write yourself. - **Protocol Translation**: convert between OpenAI Chat, Anthropic Messages, and OpenAI Responses formats - **Multi-Backend Routing**: random routing, LLM-as-classifier routing, signal-driven stage-router, or your own algorithm - **Operational Metrics**: Prometheus metrics cover requests, errors, latency, tokens, and routing overhead +- **NeMo Relay Plugin**: run random, classifier, escalation, or stage routing in Relay while Switchyard owns provider HTTP dispatch ## Maturity @@ -154,6 +155,7 @@ configured LLM client selects one upstream format. - **[`switchyard-libsy`](crates/libsy/README.md)**: embed routing algorithms in a Rust application - **[`switchyard-protocol`](crates/protocol/README.md)**: provider-neutral request, response, and streaming types - **[`switchyard-translation`](crates/switchyard-translation/README.md)**: request, response, and stream translation +- **[`switchyard-nemo-relay-plugin`](crates/switchyard-nemo-relay-plugin/README.md)**: install Switchyard as a native NeMo Relay plugin ## Community diff --git a/crates/switchyard-nemo-relay-plugin/Cargo.toml b/crates/switchyard-nemo-relay-plugin/Cargo.toml new file mode 100644 index 000000000..243e66cc0 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/Cargo.toml @@ -0,0 +1,30 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +[package] +name = "switchyard-nemo-relay-plugin" +version.workspace = true +description = "Switchyard-owned HTTP routing plugin for NeMo Relay" +authors.workspace = true +edition.workspace = true +license.workspace = true +repository.workspace = true +rust-version.workspace = true +publish = false + +[lib] +crate-type = ["cdylib"] + +[dependencies] +async-channel.workspace = true +async-trait.workspace = true +futures-util.workspace = true +http.workspace = true +nemo-relay-plugin.workspace = true +serde.workspace = true +serde_json.workspace = true +switchyard-libsy.workspace = true +switchyard-llm-client.workspace = true +switchyard-protocol.workspace = true +switchyard-translation.workspace = true +tokio.workspace = true diff --git a/crates/switchyard-nemo-relay-plugin/README.md b/crates/switchyard-nemo-relay-plugin/README.md new file mode 100644 index 000000000..d718b4a68 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/README.md @@ -0,0 +1,400 @@ + + +# Switchyard NeMo Relay Dynamic Plugin + +This crate builds the external `nvidia.switchyard` native plugin. It embeds +`switchyard-libsy`, drives it through `switchyard-llm-client::run`, and uses +`switchyard-llm-client` for provider HTTP calls. Managed calls use Relay's +completion-based asynchronous middleware hooks and do not require a targeted +provider continuation from Relay. + +The plugin uses NeMo Relay native API v1. It depends on the small +`nemo-relay-plugin` authoring SDK, not the Relay runtime, and does not start +`switchyard-server`. Managed provider calls do not use Relay's provider +continuation. + +## Ownership boundary + +For a managed LLM call: + +1. Relay invokes the native LLM execution intercept. +2. The plugin decodes the caller JSON through `switchyard-translation`. +3. The plugin passes the configured algorithm and its target-to-client map to + `switchyard-llm-client::run`, using the library's public execution and + observation boundary. +4. For every routed call, the selected target client translates the neutral + request, applies its URL and credentials, and performs the HTTP request. +5. `switchyard-llm-client` drives libsy to its final response while the plugin + records decisions and routing-only model usage. +6. The plugin encodes the final neutral response into the caller's protocol. + +Relay still owns the outer LLM lifecycle, dynamic-plugin loading, plugin +configuration, and event substrate. Relay's downstream LLM continuation is +used only for calls whose inbound protocol is not managed by this plugin. + +```mermaid +flowchart LR + A["Caller JSON"] --> B["Relay LLM execution intercept"] + B --> C["Switchyard decode"] + C --> D["switchyard-llm-client run"] + D --> E["libsy algorithm"] + E --> F["target ClientRouter"] + F --> G["Provider HTTP endpoint"] + G --> H["Switchyard response or event decode"] + H --> I["libsy final response"] + I --> J["routing observations"] + J --> K["Switchyard encode"] + K --> A + + U["Unmanaged profile"] -.-> V["Relay v1 continuation"] +``` + +This boundary has two important consequences: + +- Managed provider calls do not traverse Relay middleware registered after the + Switchyard intercept and do not use the host's provider callback. Provider + transport activity is therefore not represented as nested Relay LLM + lifecycle events. Relay records the outer managed call and the plugin emits + Switchyard routing marks; bridging Switchyard transport spans into Relay is + future work. The adapter captures the active Relay scope before returning + `Pending`, so asynchronous routing marks retain their event parent. +- Switchyard owns provider URLs, credentials, HTTP retry behavior, and + translation for managed calls. Relay neither validates nor transports those + target details. + +## Native API v1 and asynchronous execution + +The manifest remains `compat.native_api = "1"`, but the plugin requires the +generic host-table v3 extension shipped by Relay 0.7. It registers through +v3's completion-based buffered and incremental streaming hooks, returns +`Pending` immediately, and performs libsy and provider HTTP work on a +plugin-owned Tokio runtime. Relay workers therefore do not wait synchronously +for provider I/O. + +The stream adapter forwards the plugin's bounded 32-message channel into +Relay's bounded output queue. It retries a logical event when the host queue is +full and checks cancellation between attempts. Managed HTTP work is selected +against Relay caller cancellation, so cancelling a buffered or streaming call +drops its in-flight provider future. + +Unmanaged profiles use the same v3 continuation hooks for pass-through. V3's +downstream stream callback has continue/cancel control but no asynchronous +acknowledgement, so the adapter uses a nonblocking bridge capped at 8 MiB of +queued encoded payloads and 256 events. A pass-through stream that outruns +either bound is rejected rather than consuming unbounded memory. + +This is a raw C boundary: Switchyard contains a small ownership adapter for +host strings, completion and stream handles, continuation handles, and captured +scope handles because Relay 0.7 does not expose a safe Rust facade for its +generic asynchronous surface. The HTTP, routing, and translation behavior +remains in Switchyard. NeMo Relay plans to provide the equivalent safe typed +surface in 0.8.0; once that is available, the raw-FFI compatibility adapter +should be removable. + +## Supported routers + +The plugin supports four libsy routing modes: + +- seeded, weighted `random` routing; and +- capability-based `llm_classifier` routing, where a judge selects the weak or + strong target before the final provider call; +- escalation-mode `llm_classifier` routing, where a judge evaluates the weak + model's completed turn and latches a session to the strong target after a + configured confirmation streak; and +- signal-driven `stage_router` routing, with optional handoff notes, tier + prompts, and a capability-classifier fallback for ambiguous turns. + +Unsupported algorithm kinds are rejected instead of being approximated. + +## Compatibility Matrix + +The following matrix describes the algorithm behavior implemented by the +plugin. `Conditional` means that the feature is implemented with the constraint +shown in the table; it does not mean that the feature falls back to a different +algorithm. + +| Compatibility Area | `random` | `llm_classifier` (`capability`) | `llm_classifier` (`escalation`) | `stage_router` | +|---|---|---|---|---| +| Version-2 configuration and static validation | Supported | Supported | Supported | Supported | +| Caller protocols | OpenAI Chat, OpenAI Responses, Anthropic Messages | OpenAI Chat, OpenAI Responses, Anthropic Messages | OpenAI Chat, OpenAI Responses, Anthropic Messages | OpenAI Chat, OpenAI Responses, Anthropic Messages | +| Serving-target protocols | OpenAI Chat, OpenAI Responses, Anthropic Messages | OpenAI Chat, OpenAI Responses, Anthropic Messages | OpenAI Chat, OpenAI Responses, Anthropic Messages | OpenAI Chat, OpenAI Responses, Anthropic Messages | +| Structured-output judge protocols | Not applicable | OpenAI Chat or OpenAI Responses | OpenAI Chat or OpenAI Responses | OpenAI Chat or OpenAI Responses for the optional classifier | +| Buffered responses | Supported | Supported | Supported | Supported | +| Streaming responses | Supported | Supported after the judge selects a target | Conditional: an unlatched weak stream is aggregated before the judge runs | Supported after the signal cascade selects a target | +| Retained routing state | No selection affinity; context-overflow eviction can use session identity | Optional session affinity and message-hash fallback | Confirmation streak and strong latch require stable session identity | No classifier affinity; context-overflow eviction can use session identity | +| Router-specific prompts | Not applicable | Optional judge prompt | Optional escalation-judge prompt | Optional tier prompts, handoff notes, and classifier prompt | +| Relay decision marks | Algorithm, attempt, selected target, reasoning, answer-call status, and identity | Algorithm, attempt, selected target, reasoning, answer-call status, and identity | Algorithm, attempt, selected target, reasoning, answer-call status, and identity | Algorithm, attempt, selected target, reasoning, answer-call status, and identity | +| ATOF routing-LLM usage | Not applicable unless a failed candidate is replaced | Judge calls, plus failed candidates | Judge calls and discarded weak candidates | Optional classifier judge calls, plus failed candidates | + +Anthropic Messages is supported for callers and serving targets, but not for a +structured-output judge. That restriction is intentional and fails during +static configuration loading. Same-protocol streaming preserves parsed provider +events when the router does not aggregate or replace them; raw SSE bytes and +framing are not part of the compatibility contract. + +### Known issue: OpenAI Responses structured-output judges + +OpenAI Responses targets are accepted for structured-output judges, but the +shared Responses request encoder currently emits the Chat-compatible JSON +Schema object directly under `text.format`. This places `name`, `schema`, and +`strict` under `text.format.json_schema`; conforming Responses endpoints expect +those fields directly under `text.format`. InferenceHub therefore returns HTTP +400 with `Missing required parameter: 'text.format.name'`, and the affected +router follows its existing judge-failure or fall-open path. + +OpenAI Responses remains supported as a caller and ordinary serving-target +protocol. Until the shared `switchyard-translation` encoder is corrected, +configure structured-output judges with `protocol = "openai_chat"`. Follow-up +work must add the inverse of the existing Responses-to-neutral schema conversion +plus core and process-level regression coverage for all three affected router +paths. + +Managed inner provider calls also do not re-enter Relay's downstream provider +middleware. This behavior is part of the current ownership boundary, not an +automatic compatibility fallback. + +Each completed routing-only model call emits a +`switchyard.routing.llm_call` ATOF mark. Its data identifies the algorithm, +attempt, call order, target, role (`judge` or discarded `candidate`), outcome, +latency, and normalized provider token `usage`. The +successful call that serves the caller is deliberately excluded because +Relay's outer LLM end event already records that usage. A failed call, or a +provider response that omits usage, has `usage = null`. Consumers can therefore +add these marks to the outer LLM usage to measure total request compute without +double-counting the serving model. + +The plugin owns the outer routing retry loop. Each retry starts a fresh libsy +run. Random routing draws again; an algorithm configured with persistent state, +such as classifier session affinity, may intentionally retain its assignment. +Each target's built-in HTTP retry count is set to zero to avoid retrying a +failed target invisibly before reselection. A random target with `weight = 0` +is fallback-only and is not considered by the algorithm. Trusted fallback is +attempted at most once and, for streaming responses, only before the first +caller event is emitted. Outer routing retries use exponential backoff starting +at 250 milliseconds and capped at 2 seconds. They do not currently honor +provider `Retry-After` headers because the client error contract does not expose +that metadata to the routing loop. + +## Translation and stream fidelity + +`switchyard-translation` is the only request, response, and event translation +layer. It decodes caller JSON into Switchyard's neutral protocol, encodes each +selected call for the target protocol, decodes provider results, and encodes +`ReturnToAgent` back to the caller protocol. Relay codecs are not used. + +The streaming contract carries each parsed provider JSON event in a preservation +envelope alongside its normalized `LlmResponseChunk` representation. +Same-protocol routes replay the preserved JSON unchanged, including +provider-specific fields; this preserves parsed events, not raw SSE bytes or +framing. Cross-protocol routes encode only normalized chunks, and the streaming +helpers still do not expose the buffered translation engine's reject-lossy +diagnostics, so unsupported fields may be normalized or omitted. Replacing +normalized stream content or folding a stream into an aggregate drops the +per-event preservation envelope. + +## Configuration + +The manifest declares `compat.native_api = "1"` and Relay `>=0.7.0,<0.8`, and +the Rust SDK uses the exact published `0.7.0` crate. The manifest API value +selects Relay's released native plugin contract; the binary also requires the +v3 C host table shipped on the Relay 0.7 line. Rebuild the bundle when changing +SDK versions rather than assuming Rust dynamic-library compatibility from the +manifest value alone. + +A Relay project can configure a seeded weighted-random router as follows: + +```toml +version = 1 + +[[plugins.dynamic]] +manifest = "/opt/switchyard-relay-plugin/relay-plugin.toml" + +[plugins.dynamic.config] +version = 2 +priority = 0 +max_retries = 3 + +[plugins.dynamic.config.algorithm] +kind = "random" +seed = 42 + +[plugins.dynamic.config.default_targets] +openai_chat = "fast" + +[plugins.dynamic.config.targets.fast] +model = "provider/model" +protocol = "openai_chat" +endpoint = "/v1/chat/completions" +base_url = "https://provider.example.com" +weight = 1 +drop_caller_extra_body = true + +[plugins.dynamic.config.targets.fast.header_env] +authorization = "PROVIDER_AUTHORIZATION" +``` + +Target map keys such as `fast` are stable semantic names visible to libsy. The +target binding is authoritative for the provider model, protocol, endpoint, +base URL, weight, and environment-backed headers. Each `default_targets` key +both enables that inbound protocol and names its trusted fallback. + +`header_env` is the only custom provider-header source. It resolves values in +the plugin process at registration time so literal header values never appear +in configuration. Environment values must not appear in errors, routing marks, +spans, or debug output. The plugin does not inherit caller credentials for +managed calls. Each variable supplies the complete header value, so an +`authorization` value must include its scheme, such as `Bearer`. Literal +`headers` configuration is rejected; non-secret routing or tenancy headers must +also use `header_env`. + +Relay may intercept an OpenAI SDK call before the SDK materializes its +`extra_body` option into a provider request. Targets that reject this +caller-specific wrapper can set `drop_caller_extra_body = true`. The plugin +then drops the wrapper and its contents; it does not promote those values to +top-level provider fields. The default is `false` so lossless same-format +forwarding remains unchanged for targets that consume the extension. + +`extra_body` supplies non-secret provider defaults for a target. It is useful +for provider-specific controls such as disabling reasoning on a dedicated +judge model. Fields already present on the caller's request take precedence. +Do not put credentials in `extra_body`; use `header_env` for secrets. + +For `kind = "llm_classifier"`, the classifier target must use `openai_chat` or +`openai_responses`; libsy's judge request uses a JSON-schema response format +that cannot be represented losslessly by Anthropic Messages. Omitting `mode` +selects `capability`, preserving the original version-2 configuration shape. + +Escalation mode evaluates the weak model's completed response before returning +it or replacing it with a strong-model response: + +```toml +[plugins.dynamic.config.algorithm] +kind = "llm_classifier" +mode = "escalation" +classifier_target = "judge" +weak_target = "weak" +strong_target = "strong" +prompt = "Judge whether the weak model is stuck." +max_output_tokens = 512 + +[plugins.dynamic.config.algorithm.escalation] +confirmations = 2 +recent_turn_window = 28 +window_message_chars = 500 +``` + +`judge`, `weak`, and `strong` are keys in +`plugins.dynamic.config.targets`, configured with the same model, protocol, +URL, and `header_env` fields shown above. The judge must use `openai_chat` or +`openai_responses`; the serving targets may use any supported protocol. + +Use a dedicated, non-reasoning model for the judge when possible. Providers +that expose a reasoning switch can configure it on that target, for example: + +```toml +[plugins.dynamic.config.targets.judge] +model = "provider/non-reasoning-judge" +protocol = "openai_chat" +base_url = "https://provider.example.com" +extra_body = { think = false } +``` + +The packaged escalation rubric is intentionally detailed and can consume +roughly two thousand or more input tokens depending on the tokenizer. Every +unlatched request also pays for a complete judge call. A custom `prompt` can +reduce that cost, but should be evaluated against representative trajectories +before deployment. Reasoning models may spend `max_output_tokens` on hidden or +visible reasoning before returning the structured verdict; disable reasoning +with provider-supported `extra_body` controls or raise the cap after measuring. + +An unlatched streaming escalation request is intentionally buffered. Libsy must +read the complete weak response before asking the judge, so caller first-token +delivery waits for the weak call and judge verdict. A declined escalation is +reconstructed as a stream from the aggregate response, which drops the +provider-event preservation envelope. A confirmed escalation discards that +weak response and serves the strong target. + +The default `confirmations = 2` retains a streak per Switchyard session. Callers +must send a stable `x-switchyard-session-id` header for the streak and strong +latch to survive across turns. Without session identity each request has +isolated state and a multi-confirmation escalation cannot latch. + +A full stage router can combine tool-result signals, model-specific prompts, +handoff notes, and an optional judge for ambiguous turns: + +```toml +[plugins.dynamic.config.algorithm] +kind = "stage_router" +capable_target = "strong" +efficient_target = "weak" +picker = "efficient_first" +confidence_threshold = 0.5 +recent_turn_window = 3 +capable_system_prompt = "Diagnose before editing." +efficient_system_prompt = "Follow the settled plan." + +[plugins.dynamic.config.algorithm.handoff_notes] +escalation_note = "The previous model was stalling; pick up the diagnosis." +deescalation_note = "The task is settled; continue with the mechanical work." +only_on_wrong_signal_escalation = true + +[plugins.dynamic.config.algorithm.classifier] +target = "judge" +base_threshold = 0.5 +threshold_step = 0.1 +recent_turn_window = 3 +prompt = "Estimate whether the efficient target can finish this turn." +max_output_tokens = 512 +``` + +Stage routing reads normalized tool calls and tool results from OpenAI Chat, +OpenAI Responses, and Anthropic Messages traffic. When the signals do not cross +`confidence_threshold`, the optional classifier decides; if it is absent or +cannot decide, the configured picker's default tier serves the turn. The +classifier target has the same structured-output protocol restriction as the +standalone classifier. + +Ambiguous turns that reach the optional classifier add one judge call; +decisive tool signals do not. Decision marks report the selected model, +reasoning, and answer-call status exposed by libsy's `Decision` API. + +Version-1 service configuration, decision-only execution, and observe-only +mode are rejected. + +## Build and bundle + +The crate is a non-publishable member of the Switchyard Cargo workspace. +Operators install a binary bundle rather than a Rust crate: + +```bash +cargo build --release \ + --manifest-path crates/switchyard-nemo-relay-plugin/Cargo.toml +python3 crates/switchyard-nemo-relay-plugin/scripts/package_bundle.py \ + --library target/release/libswitchyard_nemo_relay_plugin.so \ + --output build/switchyard-nemo-relay-plugin-linux-x86_64 \ + --archive dist/switchyard-nemo-relay-plugin-0.2.0-linux-x86_64.tar.gz +``` + +On macOS the library suffix is `.dylib`; Windows builds use `.dll`. The bundle +builder creates the Relay package: the shared library, a materialized manifest +with Relay's inline SHA-256 integrity digest, the JSON schema, and the project +license files. Use `.tar.gz` archives on Linux and macOS and `.zip` on Windows. +The archive's top-level directory is always `switchyard-nemo-relay-plugin`. + +The release archive convention is +`switchyard-nemo-relay-plugin--.`. A future Actions +matrix should upload each archive under the artifact name +`switchyard-nemo-relay-plugin-`, matching Switchyard's existing +platform-qualified artifact convention. + +Install the materialized bundle with Relay's normal lifecycle commands: + +```bash +nemo-relay plugins validate /opt/switchyard-relay-plugin/relay-plugin.toml +nemo-relay plugins add /opt/switchyard-relay-plugin/relay-plugin.toml +nemo-relay plugins enable nvidia.switchyard +nemo-relay plugins inspect nvidia.switchyard +``` diff --git a/crates/switchyard-nemo-relay-plugin/config.schema.json b/crates/switchyard-nemo-relay-plugin/config.schema.json new file mode 100644 index 000000000..63db7b40c --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/config.schema.json @@ -0,0 +1,217 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "title": "Switchyard NeMo Relay Plugin", + "description": "In-process Switchyard routing with Switchyard-owned provider HTTP dispatch.", + "type": "object", + "additionalProperties": false, + "required": ["version", "algorithm", "targets", "default_targets"], + "properties": { + "version": { + "const": 2, + "description": "Library-only Switchyard configuration version." + }, + "priority": { + "type": "integer", + "default": 0 + }, + "max_retries": { + "type": "integer", + "minimum": 0, + "maximum": 10, + "default": 3, + "description": "Routing retries after the initial libsy run. Every retry starts a fresh run." + }, + "algorithm": { + "description": "In-process random, capability, escalation, or stage-router configuration.", + "oneOf": [ + { + "type": "object", + "additionalProperties": false, + "required": ["kind"], + "properties": { + "kind": { "const": "random" }, + "seed": { "type": ["integer", "null"], "minimum": 0 } + } + }, + { + "type": "object", + "additionalProperties": false, + "required": [ + "kind", + "classifier_target", + "weak_target", + "strong_target", + "base_threshold" + ], + "properties": { + "kind": { "const": "llm_classifier" }, + "mode": { "const": "capability", "default": "capability" }, + "classifier_target": { + "type": "string", + "minLength": 1, + "description": "Semantic target name for the judge. The target must use openai_chat or openai_responses because the judge requires a JSON-schema response format." + }, + "weak_target": { "type": "string", "minLength": 1 }, + "strong_target": { "type": "string", "minLength": 1 }, + "base_threshold": { "type": "number", "minimum": 0, "maximum": 1 }, + "threshold_step": { "type": "number", "minimum": 0, "default": 0 }, + "recent_turn_window": { + "type": ["integer", "null"], + "minimum": 0 + }, + "max_output_tokens": { + "type": "integer", + "minimum": 1, + "default": 4096 + }, + "prompt": { "type": "string" }, + "session_affinity": { "type": "boolean", "default": false }, + "message_hash_fallback": { "type": "boolean", "default": false } + } + }, + { + "type": "object", + "additionalProperties": false, + "required": [ + "kind", + "mode", + "classifier_target", + "weak_target", + "strong_target", + "escalation" + ], + "properties": { + "kind": { "const": "llm_classifier" }, + "mode": { "const": "escalation" }, + "classifier_target": { + "type": "string", + "minLength": 1, + "description": "Semantic target name for the trajectory judge. The target must use openai_chat or openai_responses." + }, + "weak_target": { "type": "string", "minLength": 1 }, + "strong_target": { "type": "string", "minLength": 1 }, + "prompt": { "type": "string" }, + "max_output_tokens": { + "type": "integer", + "minimum": 1, + "default": 4096 + }, + "escalation": { + "type": "object", + "additionalProperties": false, + "properties": { + "confirmations": { "type": "integer", "minimum": 1, "default": 2 }, + "recent_turn_window": { "type": "integer", "minimum": 1, "default": 28 }, + "window_message_chars": { "type": "integer", "minimum": 50, "default": 500 } + } + } + } + }, + { + "type": "object", + "additionalProperties": false, + "required": [ + "kind", + "capable_target", + "efficient_target", + "picker", + "confidence_threshold" + ], + "properties": { + "kind": { "const": "stage_router" }, + "capable_target": { "type": "string", "minLength": 1 }, + "efficient_target": { "type": "string", "minLength": 1 }, + "picker": { "enum": ["capable_first", "efficient_first"] }, + "confidence_threshold": { "type": "number", "minimum": 0, "maximum": 1 }, + "recent_turn_window": { + "type": ["integer", "null"], + "minimum": 0, + "description": "Trailing tool results scored for the current turn. Null uses libsy's default window." + }, + "capable_system_prompt": { "type": "string" }, + "efficient_system_prompt": { "type": "string" }, + "handoff_notes": { + "type": "object", + "additionalProperties": false, + "required": ["escalation_note"], + "properties": { + "escalation_note": { "type": "string", "minLength": 1 }, + "deescalation_note": { "type": ["string", "null"], "minLength": 1 }, + "only_on_wrong_signal_escalation": { "type": "boolean", "default": true } + } + }, + "classifier": { + "type": "object", + "additionalProperties": false, + "required": ["target", "base_threshold"], + "properties": { + "target": { + "type": "string", + "minLength": 1, + "description": "Judge target used only when stage signals are ambiguous. It must use openai_chat or openai_responses." + }, + "base_threshold": { "type": "number", "minimum": 0, "maximum": 1 }, + "threshold_step": { "type": "number", "minimum": 0, "default": 0 }, + "recent_turn_window": { "type": ["integer", "null"], "minimum": 0 }, + "prompt": { "type": "string" }, + "max_output_tokens": { "type": "integer", "minimum": 1, "default": 4096 } + } + } + } + } + ] + }, + "targets": { + "type": "object", + "minProperties": 1, + "additionalProperties": { + "type": "object", + "additionalProperties": false, + "required": ["model", "protocol", "base_url"], + "properties": { + "model": { "type": "string", "minLength": 1 }, + "protocol": { + "enum": ["openai_chat", "openai_responses", "anthropic_messages"] + }, + "endpoint": { + "type": "string", + "pattern": "^$|^/", + "description": "Optional provider endpoint override. The resolved URL must end in the canonical route for the selected protocol." + }, + "base_url": { + "type": "string", + "pattern": "^https?://" + }, + "weight": { "type": "number", "minimum": 0, "default": 1 }, + "drop_caller_extra_body": { + "type": "boolean", + "default": false, + "description": "Drop an intercepted OpenAI SDK extra_body wrapper instead of forwarding it to targets that reject caller-specific extensions." + }, + "extra_body": { + "type": "object", + "default": {}, + "description": "Non-secret provider request defaults, such as judge reasoning controls. Caller-provided fields take precedence.", + "additionalProperties": true + }, + "header_env": { + "type": "object", + "description": "Sole custom provider-header source. Maps header names to environment-variable names resolved by the plugin process so literal values are never stored in configuration.", + "additionalProperties": { "type": "string", "minLength": 1 } + } + } + } + }, + "default_targets": { + "type": "object", + "description": "Maps each managed inbound protocol to its trusted fallback target.", + "minProperties": 1, + "additionalProperties": false, + "properties": { + "openai_chat": { "type": "string", "minLength": 1 }, + "openai_responses": { "type": "string", "minLength": 1 }, + "anthropic_messages": { "type": "string", "minLength": 1 } + } + } + } +} diff --git a/crates/switchyard-nemo-relay-plugin/relay-plugin.toml b/crates/switchyard-nemo-relay-plugin/relay-plugin.toml new file mode 100644 index 000000000..ba5992bd6 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/relay-plugin.toml @@ -0,0 +1,31 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +manifest_version = 1 + +[plugin] +id = "nvidia.switchyard" +kind = "rust_dynamic" + +[compat] +relay = ">=0.8.0,<1.0" +native_api = "1" + +[defaults] +enabled = false + +[capabilities] +items = ["plugin_native", "config_schema"] + +[config_schema] +path = "config.schema.json" + +[source] +artifact = "" + +[integrity] +sha256 = "sha256:" + +[load] +library = "" +symbol = "nemo_relay_register_plugin" diff --git a/crates/switchyard-nemo-relay-plugin/scripts/package_bundle.py b/crates/switchyard-nemo-relay-plugin/scripts/package_bundle.py new file mode 100644 index 000000000..c5ea9e946 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/scripts/package_bundle.py @@ -0,0 +1,99 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Materialize the minimal Relay plugin bundle from a compiled cdylib.""" + +from __future__ import annotations + +import argparse +import hashlib +import shutil +import tarfile +import zipfile +from pathlib import Path + +CRATE_ROOT = Path(__file__).resolve().parents[1] +REPOSITORY_ROOT = CRATE_ROOT.parents[1] +PACKAGE_NAME = "switchyard-nemo-relay-plugin" + + +def digest(path: Path) -> str: + """Return the lowercase SHA-256 digest for a file.""" + value = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b""): + value.update(chunk) + return value.hexdigest() + + +def archive_bundle(bundle: Path, archive: Path) -> None: + """Archive a materialized bundle under the stable package directory name.""" + archive.parent.mkdir(parents=True, exist_ok=True) + if archive.exists(): + raise ValueError(f"bundle archive already exists: {archive}") + + if archive.name.endswith(".tar.gz"): + with tarfile.open(archive, "w:gz") as stream: + stream.add(bundle, arcname=PACKAGE_NAME) + return + + if archive.suffix == ".zip": + with zipfile.ZipFile(archive, "w", compression=zipfile.ZIP_DEFLATED) as stream: + for path in sorted(bundle.rglob("*")): + if path.is_file(): + stream.write(path, Path(PACKAGE_NAME) / path.relative_to(bundle)) + return + + raise ValueError("bundle archive must end in .tar.gz or .zip") + + +def main() -> None: + """Materialize a Relay-loadable plugin bundle in an empty directory.""" + parser = argparse.ArgumentParser() + parser.add_argument("--library", required=True, type=Path) + parser.add_argument("--output", required=True, type=Path) + parser.add_argument("--archive", type=Path) + args = parser.parse_args() + + library = args.library.resolve() + if not library.is_file(): + parser.error(f"compiled plugin library does not exist: {library}") + + manifest = (CRATE_ROOT / "relay-plugin.toml").read_text(encoding="utf-8") + placeholders = ("", "") + missing = [placeholder for placeholder in placeholders if placeholder not in manifest] + if missing: + parser.error(f"plugin manifest is missing placeholders: {', '.join(missing)}") + + output = args.output.resolve() + if output.exists() and not output.is_dir(): + parser.error(f"bundle output exists and is not a directory: {output}") + if output.is_dir() and any(output.iterdir()): + parser.error(f"bundle output directory must be empty: {output}") + output.mkdir(parents=True, exist_ok=True) + + artifact = output / library.name + shutil.copy2(library, artifact) + shutil.copy2(CRATE_ROOT / "config.schema.json", output / "config.schema.json") + for filename in ("LICENSE", "NOTICE"): + shutil.copy2(REPOSITORY_ROOT / filename, output / filename) + + artifact_digest = digest(artifact) + manifest = manifest.replace("", artifact.name) + manifest = manifest.replace("", artifact_digest) + (output / "relay-plugin.toml").write_text(manifest, encoding="utf-8") + + if args.archive is None: + print(output) + return + + archive = args.archive.resolve() + try: + archive_bundle(output, archive) + except ValueError as error: + parser.error(str(error)) + print(archive) + + +if __name__ == "__main__": + main() diff --git a/crates/switchyard-nemo-relay-plugin/src/client.rs b/crates/switchyard-nemo-relay-plugin/src/client.rs new file mode 100644 index 000000000..bc197bfba --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/client.rs @@ -0,0 +1,463 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +//! Switchyard-owned HTTP clients bound to one semantic routing target. + +use std::collections::BTreeMap; + +use async_trait::async_trait; +use serde_json::Value as Json; +use switchyard_llm_client::{Backend, HttpBackendConfig, ModelConfig, TranslatingLlmClient}; +use switchyard_protocol::{ + ContentBlock, Decision, LlmClientError, Message, Request, Response, Role, RoutedLlmClient, + ToolCall, ToolResult, WireFormat, +}; +use switchyard_translation::TranslationEngine; + +use crate::translation; + +/// A provider client bound to one configured Switchyard target. +/// +/// libsy routes with a stable semantic name (for example `fast`). The provider +/// still expects its own model id (for example `meta/llama-3.1-8b-instruct`). +/// Keeping that mapping here prevents an algorithm's semantic labels from +/// leaking into provider requests. +pub(crate) struct TargetClient { + provider_model: String, + target_format: WireFormat, + drop_caller_extra_body: bool, + inner: TranslatingLlmClient, + translation: TranslationEngine, +} + +impl TargetClient { + pub(crate) fn new( + provider_model: String, + target_format: WireFormat, + dispatch_url: String, + headers: BTreeMap, + extra_body: BTreeMap, + drop_caller_extra_body: bool, + ) -> Result { + let backend_config = HttpBackendConfig { + // `dispatch_url` is already resolved by configuration. Backend URL + // joining accepts a complete canonical endpoint as well as a base + // URL/prefix. + base_url: dispatch_url, + api_key: None, + extra_headers: headers, + extra_body, + // Routing retries belong to the plugin: every retry must start a + // fresh libsy run and obtain a fresh decision. + max_retries: 0, + }; + let backend = match target_format { + WireFormat::OpenAiChat => Backend::OpenAiChat(backend_config), + WireFormat::OpenAiResponses => Backend::OpenAiResponses(backend_config), + WireFormat::AnthropicMessages => Backend::Anthropic(backend_config), + }; + let model = ModelConfig::new(provider_model.clone(), backend, None); + let inner = TranslatingLlmClient::new(&[model])?; + Ok(Self { + provider_model, + target_format, + drop_caller_extra_body, + inner, + translation: TranslationEngine::default(), + }) + } + + /// Retargets only the provider-facing transport metadata. + /// + /// Correlation and agent identity remain available to libsy, while inbound + /// HTTP headers are deliberately removed. Provider credentials come solely + /// from this target's `header_env` configuration. + fn prepare_request(&self, mut request: Request, decision: &Decision) -> Request { + if !decision.is_answer_call() { + sanitize_judge_request(&mut request); + } + if decision.reasoning() == Some("escalation classifier: efficient tier") { + // Escalation always buffers this draft before judging it. Asking the + // provider for a buffered response preserves normalized usage for ATOF; + // libsy reconstructs a caller stream when the weak draft wins. + request.llm_request.stream = false; + request.llm_request.preservation.requests.clear(); + } + let metadata = request.metadata.get_or_insert_default(); + metadata.wire_format = Some(self.target_format); + metadata.http_headers = None; + if self.drop_caller_extra_body { + request.llm_request.extensions.fields.remove("extra_body"); + for preserved in request.llm_request.preservation.requests.values_mut() { + if let Some(body) = preserved.as_object_mut() { + body.remove("extra_body"); + } + } + } + request + } +} + +#[async_trait] +impl RoutedLlmClient for TargetClient { + async fn call(&self, request: Request, decision: Decision) -> Result { + let request = self.prepare_request(request, &decision); + translation::validate_target_request( + &self.translation, + self.target_format, + &request.llm_request, + ) + .map_err(LlmClientError::RequestEncoding)?; + self.inner + .call_rewrite_model(request, Some(&self.provider_model)) + .await + } +} + +/// Maximum plain-text context retained from one native tool block in a judge request. +const MAX_JUDGE_TOOL_CONTEXT_CHARS: usize = 4_096; + +/// Keep judge requests provider-neutral. Native tool turns without their original +/// definitions are rejected by some OpenAI-compatible Bedrock gateways, while the +/// text evidence is still valuable to the classifier. +fn sanitize_judge_request(request: &mut Request) { + request.llm_request.messages = request + .llm_request + .messages + .drain(..) + .map(|message| Message { + role: if message.role == Role::Tool { + Role::User + } else { + message.role + }, + content: message + .content + .into_iter() + .map(|block| match block { + ContentBlock::ToolCall(call) => ContentBlock::Text { + text: bounded_tool_context(tool_call_text(call)), + }, + ContentBlock::ToolResult(result) => ContentBlock::Text { + text: bounded_tool_context(tool_result_text(result)), + }, + ordinary => ordinary, + }) + .collect(), + }) + .collect(); + request.llm_request.tools.clear(); + request.llm_request.tool_choice = None; + if let Some(response_format) = request.llm_request.output.response_format.as_mut() { + remove_numeric_schema_bounds(response_format); + } + request.llm_request.preservation.requests.clear(); +} + +fn tool_call_text(call: ToolCall) -> String { + format!( + "[tool call]\nid: {}\nname: {}\narguments: {}", + Json::String(call.id), + Json::String(call.name), + call.arguments + ) +} + +fn tool_result_text(result: ToolResult) -> String { + let content = result + .content + .into_iter() + .map(tool_content_text) + .collect::>() + .join("\n"); + format!( + "[tool result]\ncall_id: {}\nis_error: {}\ncontent:\n{}", + Json::String(result.tool_call_id), + result + .is_error + .map_or_else(|| "unknown".to_string(), |value| value.to_string()), + content + ) +} + +fn tool_content_text(block: ContentBlock) -> String { + match block { + ContentBlock::Text { text } + | ContentBlock::Reasoning { text, .. } + | ContentBlock::Refusal { text } => text, + ContentBlock::ToolCall(call) => tool_call_text(call), + ContentBlock::ToolResult(result) => tool_result_text(result), + ContentBlock::Image { .. } => "[image omitted]".to_string(), + ContentBlock::Audio { .. } => "[audio omitted]".to_string(), + ContentBlock::Video { .. } => "[video omitted]".to_string(), + ContentBlock::File { .. } => "[file omitted]".to_string(), + ContentBlock::Unknown { provider, .. } => { + format!("[unsupported {provider} content omitted]") + } + } +} + +fn bounded_tool_context(text: String) -> String { + const TRUNCATED: &str = "\n[truncated]"; + let keep = MAX_JUDGE_TOOL_CONTEXT_CHARS - TRUNCATED.chars().count(); + let mut chars = text.chars(); + let prefix = chars.by_ref().take(keep).collect::(); + if chars.next().is_some() { + prefix + TRUNCATED + } else { + prefix + } +} + +fn remove_numeric_schema_bounds(value: &mut Json) { + match value { + Json::Object(object) => { + for key in ["minimum", "maximum", "exclusiveMinimum", "exclusiveMaximum"] { + object.remove(key); + } + for child in object.values_mut() { + remove_numeric_schema_bounds(child); + } + } + Json::Array(values) => { + for child in values { + remove_numeric_schema_bounds(child); + } + } + _ => {} + } +} + +#[cfg(test)] +mod tests { + use super::*; + use serde_json::json; + use switchyard_protocol::{ + LlmRequest, Metadata, PreservationMetadata, ProviderExtensions, ToolChoice, ToolDefinition, + }; + + fn decision() -> Decision { + Decision::new("target", None, true) + } + + fn client(format: WireFormat) -> TargetClient { + TargetClient::new( + "provider/model".into(), + format, + match format { + WireFormat::OpenAiChat => "https://provider.example/v1/chat/completions".into(), + WireFormat::OpenAiResponses => "https://provider.example/v1/responses".into(), + WireFormat::AnthropicMessages => "https://provider.example/v1/messages".into(), + }, + BTreeMap::new(), + BTreeMap::new(), + false, + ) + .unwrap() + } + + #[test] + fn target_preparation_forces_format_and_removes_inbound_headers() { + let client = client(WireFormat::AnthropicMessages); + let request = Request { + metadata: Some(Metadata { + correlation_id: Some("request-123".into()), + wire_format: Some(WireFormat::OpenAiChat), + http_headers: Some(http::HeaderMap::from_iter([ + ( + http::HeaderName::from_static("authorization"), + http::HeaderValue::from_static("Bearer caller-secret"), + ), + ( + http::HeaderName::from_static("x-caller-only"), + http::HeaderValue::from_static("must-not-forward"), + ), + ])), + ..Metadata::default() + }), + ..Request::default() + }; + + let prepared = client.prepare_request(request, &decision()); + let metadata = prepared.metadata.unwrap(); + assert_eq!(metadata.wire_format, Some(WireFormat::AnthropicMessages)); + assert_eq!(metadata.correlation_id.as_deref(), Some("request-123")); + assert!(metadata.http_headers.is_none()); + } + + #[test] + fn missing_metadata_is_created_for_the_target_format() { + let client = client(WireFormat::OpenAiResponses); + let prepared = client.prepare_request(Request::default(), &decision()); + assert_eq!( + prepared.metadata.and_then(|metadata| metadata.wire_format), + Some(WireFormat::OpenAiResponses) + ); + } + + #[test] + fn configured_target_drops_intercepted_caller_extra_body() { + let client = TargetClient::new( + "provider/model".into(), + WireFormat::OpenAiChat, + "https://provider.example/v1/chat/completions".into(), + BTreeMap::new(), + BTreeMap::new(), + true, + ) + .unwrap(); + let request = Request { + llm_request: LlmRequest { + extensions: ProviderExtensions { + fields: serde_json::Map::from_iter([( + "extra_body".into(), + json!({"reasoning": {"effort": "medium"}}), + )]), + }, + preservation: PreservationMetadata { + requests: BTreeMap::from([( + WireFormat::OpenAiChat.into(), + json!({ + "model": "route", + "messages": [{"role": "user", "content": "hello"}], + "extra_body": { + "reasoning": {"effort": "medium"}, + "session_id": "hermes-session" + } + }), + )]), + ..PreservationMetadata::default() + }, + ..LlmRequest::default() + }, + ..Request::default() + }; + + let prepared = client.prepare_request(request, &decision()); + assert!( + !prepared + .llm_request + .extensions + .fields + .contains_key("extra_body") + ); + assert!( + prepared + .llm_request + .preservation + .requests + .values() + .all(|body| body.get("extra_body").is_none()) + ); + } + + #[test] + fn judge_preparation_sanitizes_tool_history_and_schema_dialect() { + let client = client(WireFormat::OpenAiChat); + let request = Request { + llm_request: LlmRequest { + messages: vec![ + Message::text(Role::User, "inspect the workspace"), + Message { + role: Role::Assistant, + content: vec![ContentBlock::ToolCall(ToolCall { + id: "call-1".into(), + name: "terminal".into(), + arguments: json!({"command": "pwd"}), + })], + }, + Message { + role: Role::Tool, + content: vec![ContentBlock::ToolResult(ToolResult { + tool_call_id: "call-1".into(), + content: vec![ContentBlock::Text { + text: format!("result {} TAIL", "x".repeat(5_000)), + }], + is_error: Some(false), + })], + }, + ], + tools: vec![ToolDefinition { + name: "terminal".into(), + description: None, + parameters: json!({"type": "object"}), + strict: None, + }], + tool_choice: Some(ToolChoice::Required), + output: switchyard_protocol::OutputParams { + max_output_tokens: Some(64), + response_format: Some(json!({ + "type": "json_schema", + "json_schema": { + "schema": { + "properties": { + "p_solve": { + "type": "number", + "minimum": 0.0, + "maximum": 1.0 + } + } + } + } + })), + }, + ..LlmRequest::default() + }, + ..Request::default() + }; + + let prepared = client.prepare_request( + request, + &Decision::new("judge", Some("structured judge".into()), false), + ); + + assert!(prepared.llm_request.tools.is_empty()); + assert_eq!(prepared.llm_request.tool_choice, None); + assert!( + prepared + .llm_request + .messages + .iter() + .all(|message| message.role != Role::Tool) + ); + let text = prepared + .llm_request + .messages + .iter() + .filter_map(|message| message.text_content("\n")) + .collect::>() + .join("\n"); + assert!(text.contains("[tool call]")); + assert!(text.contains("terminal")); + assert!(text.contains("[tool result]")); + assert!(text.contains("[truncated]")); + assert!(!text.contains("TAIL")); + let schema = prepared.llm_request.output.response_format.unwrap(); + let p_solve = schema + .pointer("/json_schema/schema/properties/p_solve") + .unwrap(); + assert!(p_solve.get("minimum").is_none()); + assert!(p_solve.get("maximum").is_none()); + } + + #[test] + fn escalation_candidate_is_buffered_for_usage_accounting() { + let client = client(WireFormat::OpenAiChat); + let mut request = Request::default(); + request.llm_request.stream = true; + request.llm_request.preservation.requests.insert( + WireFormat::OpenAiChat.into(), + json!({"model": "route", "stream": true}), + ); + let decision = Decision::new( + "weak", + Some("escalation classifier: efficient tier".into()), + true, + ); + + let prepared = client.prepare_request(request, &decision); + + assert!(!prepared.llm_request.stream); + assert!(prepared.llm_request.preservation.requests.is_empty()); + } +} diff --git a/crates/switchyard-nemo-relay-plugin/src/config.rs b/crates/switchyard-nemo-relay-plugin/src/config.rs new file mode 100644 index 000000000..ff3668150 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/config.rs @@ -0,0 +1,587 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +use std::collections::{BTreeMap, BTreeSet}; +use std::sync::Arc; + +use http::Uri; +use http::header::{HeaderName, HeaderValue}; +use serde::Deserialize; +use serde_json::Value as Json; +use switchyard_libsy::{ + Algorithm, ClassifierContractConfig, EscalationJudgeConfig, HandoffNoteConfig, + LlmClassifierConfig, LlmFallback, LlmTarget, LlmTargetSet, LlmTaskClassifier, PickerMode, + Random, StageRouter, StageRouterConfig, TargetPrompts, TaskClassifierConfig, +}; +use switchyard_protocol::{RoutedLlmClient, WireFormat}; + +use crate::client::TargetClient; + +pub(crate) fn protocol_from_call(name: &str) -> Option { + match name { + "openai.chat_completions" => Some(WireFormat::OpenAiChat), + "openai.responses" => Some(WireFormat::OpenAiResponses), + "anthropic.messages" => Some(WireFormat::AnthropicMessages), + _ => None, + } +} + +const fn default_endpoint(protocol: WireFormat) -> &'static str { + match protocol { + WireFormat::OpenAiChat => "/v1/chat/completions", + WireFormat::OpenAiResponses => "/v1/responses", + WireFormat::AnthropicMessages => "/v1/messages", + } +} + +#[derive(Deserialize)] +#[serde(deny_unknown_fields)] +struct TargetBinding { + model: String, + protocol: WireFormat, + #[serde(default)] + endpoint: String, + base_url: String, + #[serde(default = "default_weight")] + weight: f64, + #[serde(default)] + drop_caller_extra_body: bool, + #[serde(default)] + header_env: BTreeMap, + #[serde(default)] + extra_body: BTreeMap, +} + +impl TargetBinding { + fn dispatch_url(&self) -> String { + let base = self.base_url.trim_end_matches('/'); + let default = default_endpoint(self.protocol); + if self.endpoint.is_empty() && base.ends_with(default) { + return base.to_string(); + } + let endpoint = if self.endpoint.is_empty() { + default + } else { + &self.endpoint + }; + let endpoint = if base.ends_with("/v1") && endpoint.starts_with("/v1/") { + &endpoint[3..] + } else { + endpoint + }; + format!("{base}{endpoint}") + } + + fn validate(&self, name: &str) -> Result<(), String> { + if self.model.trim().is_empty() { + return Err(format!("target {name:?} model must be non-empty")); + } + if !self.endpoint.is_empty() && !self.endpoint.starts_with('/') { + return Err(format!( + "target {name:?} endpoint must be empty or begin with '/'" + )); + } + if !self.weight.is_finite() || self.weight < 0.0 { + return Err(format!( + "target {name:?} weight must be finite and nonnegative" + )); + } + validate_dispatch_url(name, self.protocol, &self.dispatch_url())?; + self.validate_headers(name) + } + + fn validate_headers(&self, target_name: &str) -> Result<(), String> { + let mut normalized = BTreeSet::new(); + for (name, variable) in &self.header_env { + let canonical = validate_header_name(name)?; + if !normalized.insert(canonical) { + return Err(format!( + "target {target_name:?} configures header {name:?} more than once (header names are case-insensitive)" + )); + } + if variable.trim().is_empty() { + return Err(format!( + "environment variable name for target header {name:?} must not be empty" + )); + } + if variable.as_bytes().contains(&b'=') || variable.as_bytes().contains(&b'\0') { + return Err(format!( + "environment variable name for target header {name:?} must not contain '=' or NUL" + )); + } + } + Ok(()) + } + + fn prepare(&self) -> Result { + let mut headers = BTreeMap::new(); + for (name, variable) in &self.header_env { + let value = std::env::var(variable) + .map_err(|_| format!("environment variable {variable:?} is not set"))?; + validate_header(name, &value)?; + headers.insert(name.clone(), value); + } + let dispatch_url = self.dispatch_url(); + let client = TargetClient::new( + self.model.clone(), + self.protocol, + dispatch_url, + headers, + self.extra_body.clone(), + self.drop_caller_extra_body, + ) + .map_err(|error| format!("failed to create target HTTP client: {error}"))?; + Ok(PreparedTargetBinding { + client: Arc::new(client), + }) + } +} + +pub(crate) struct PreparedTargetBinding { + pub(crate) client: Arc, +} + +#[derive(Clone, Copy, Default, Deserialize)] +#[serde(rename_all = "snake_case")] +enum LlmClassifierMode { + #[default] + Capability, + Escalation, +} + +#[derive(Clone, Deserialize)] +#[serde(deny_unknown_fields)] +struct LlmClassifierAlgorithmConfig { + #[serde(default)] + mode: LlmClassifierMode, + classifier_target: String, + weak_target: String, + strong_target: String, + #[serde(default)] + base_threshold: Option, + #[serde(default)] + threshold_step: Option, + #[serde(default)] + session_affinity: Option, + #[serde(default)] + message_hash_fallback: Option, + #[serde(default)] + recent_turn_window: Option, + #[serde(default)] + prompt: Option, + #[serde(default = "default_classifier_max_output_tokens")] + max_output_tokens: u64, + #[serde(default)] + escalation: Option, +} + +impl LlmClassifierAlgorithmConfig { + fn capability_config(&self) -> Result { + if self.escalation.is_some() { + return Err( + "llm_classifier capability mode does not accept escalation settings".into(), + ); + } + let base_threshold = self + .base_threshold + .ok_or_else(|| "llm_classifier capability mode requires base_threshold".to_string())?; + let mut contract = ClassifierContractConfig::default(); + if let Some(prompt) = &self.prompt { + contract = contract.with_prompt(prompt.clone()); + } + Ok(TaskClassifierConfig { + base_threshold, + threshold_step: self.threshold_step.unwrap_or_default(), + session_affinity: self.session_affinity.unwrap_or_default(), + message_hash_fallback: self.message_hash_fallback.unwrap_or_default(), + recent_turn_window: self.recent_turn_window, + contract, + max_output_tokens: self.max_output_tokens, + }) + } + + fn escalation_config( + &self, + ) -> Result<(ClassifierContractConfig, EscalationJudgeConfig), String> { + if self.base_threshold.is_some() + || self.threshold_step.is_some() + || self.session_affinity.is_some() + || self.message_hash_fallback.is_some() + || self.recent_turn_window.is_some() + { + return Err( + "llm_classifier escalation mode does not accept capability settings".into(), + ); + } + let config = self.escalation.clone().ok_or_else(|| { + "llm_classifier escalation mode requires escalation settings".to_string() + })?; + let mut contract = ClassifierContractConfig::default(); + if let Some(prompt) = &self.prompt { + contract = contract.with_prompt(prompt.clone()); + } + Ok((contract, config)) + } +} + +#[derive(Clone, Deserialize)] +#[serde(deny_unknown_fields)] +struct StageFallbackConfig { + target: String, + base_threshold: f64, + #[serde(default)] + threshold_step: f64, + #[serde(default)] + recent_turn_window: Option, + #[serde(default)] + prompt: Option, + #[serde(default = "default_classifier_max_output_tokens")] + max_output_tokens: u64, +} + +impl StageFallbackConfig { + fn classifier_config(&self) -> TaskClassifierConfig { + let mut contract = ClassifierContractConfig::default(); + if let Some(prompt) = &self.prompt { + contract = contract.with_prompt(prompt.clone()); + } + TaskClassifierConfig { + base_threshold: self.base_threshold, + threshold_step: self.threshold_step, + session_affinity: false, + message_hash_fallback: false, + recent_turn_window: self.recent_turn_window, + contract, + max_output_tokens: self.max_output_tokens, + } + } +} + +#[derive(Deserialize)] +#[serde(tag = "kind", rename_all = "snake_case", deny_unknown_fields)] +enum AlgorithmConfig { + Random { + #[serde(default)] + seed: Option, + }, + LlmClassifier { + #[serde(flatten)] + config: LlmClassifierAlgorithmConfig, + }, + StageRouter { + capable_target: String, + efficient_target: String, + picker: PickerMode, + confidence_threshold: f64, + #[serde(default)] + recent_turn_window: Option, + #[serde(default)] + capable_system_prompt: Option, + #[serde(default)] + efficient_system_prompt: Option, + #[serde(default)] + handoff_notes: Option, + #[serde(default)] + classifier: Option, + }, +} + +#[derive(Deserialize)] +#[serde(deny_unknown_fields)] +pub(crate) struct SwitchyardConfig { + version: u32, + #[serde(default)] + pub(crate) priority: i32, + #[serde(default = "default_max_retries")] + max_retries: u32, + algorithm: AlgorithmConfig, + targets: BTreeMap, + default_targets: BTreeMap, +} + +pub(crate) struct PreparedConfig { + pub(crate) max_retries: u32, + pub(crate) algorithm: Arc, + pub(crate) targets: BTreeMap, + pub(crate) default_targets: BTreeMap, +} + +impl SwitchyardConfig { + pub(crate) fn validate(&self) -> Result<(), String> { + self.validate_structure()?; + self.build_algorithm(None).map(drop) + } + + fn validate_structure(&self) -> Result<(), String> { + if self.version != 2 { + return Err(format!( + "unsupported Switchyard config version {}; version 1 used switchyard-server; migrate to version = 2", + self.version + )); + } + if self.max_retries > 10 { + return Err("max_retries must not exceed 10".into()); + } + if self.targets.is_empty() { + return Err("targets must not be empty".into()); + } + if self.default_targets.is_empty() { + return Err("default_targets must not be empty".into()); + } + for (name, target) in &self.targets { + if name.trim().is_empty() { + return Err("target names must be non-empty".into()); + } + target.validate(name)?; + } + for (protocol, fallback) in &self.default_targets { + let target = self + .targets + .get(fallback) + .ok_or_else(|| format!("default target {fallback:?} is not configured"))?; + if target.protocol != *protocol { + return Err(format!( + "default target {fallback:?} must use protocol {}", + protocol.as_str() + )); + } + } + Ok(()) + } + + pub(crate) fn prepare(self) -> Result { + self.validate_structure()?; + let targets = self + .targets + .iter() + .map(|(name, target)| target.prepare().map(|prepared| (name.clone(), prepared))) + .collect::, _>>()?; + let algorithm = self.build_algorithm(Some(&targets))?; + Ok(PreparedConfig { + max_retries: self.max_retries, + algorithm, + targets, + default_targets: self.default_targets, + }) + } + + fn build_algorithm( + &self, + prepared: Option<&BTreeMap>, + ) -> Result, String> { + let target = |name: &str| { + if !self.targets.contains_key(name) { + return Err(format!("algorithm target {name:?} is not configured")); + } + Ok(match prepared { + Some(targets) => { + targets + .get(name) + .ok_or_else(|| format!("algorithm target {name:?} was not prepared"))?; + LlmTarget { + semantic_name: name.to_string(), + } + } + None => LlmTarget { + semantic_name: name.to_string(), + }, + }) + }; + + match &self.algorithm { + AlgorithmConfig::Random { seed } => { + let routable = self + .targets + .iter() + .filter(|(_, binding)| binding.weight > 0.0) + .collect::>(); + if routable.is_empty() { + return Err( + "random routing requires at least one positive target weight".into(), + ); + } + let targets = routable + .iter() + .map(|(name, _)| target(name)) + .collect::, _>>()?; + let weights = routable + .iter() + .map(|(_, binding)| binding.weight) + .collect::>(); + Random::new(LlmTargetSet::new(targets), Some(weights), *seed) + .map(|algorithm| Arc::new(algorithm) as Arc) + .map_err(|error| error.to_string()) + } + AlgorithmConfig::LlmClassifier { config } => { + self.validate_judge_target(&config.classifier_target)?; + let algorithm = match config.mode { + LlmClassifierMode::Capability => LlmClassifierConfig::Capability { + judge_target: target(&config.classifier_target)?, + efficient_target: target(&config.weak_target)?, + capable_target: target(&config.strong_target)?, + config: config.capability_config()?, + }, + LlmClassifierMode::Escalation => { + let (contract, escalation) = config.escalation_config()?; + LlmClassifierConfig::Escalation { + judge_target: target(&config.classifier_target)?, + efficient_target: target(&config.weak_target)?, + capable_target: target(&config.strong_target)?, + contract, + config: escalation, + max_output_tokens: config.max_output_tokens, + } + } + }; + LlmTaskClassifier::new(algorithm) + .map(|algorithm| Arc::new(algorithm) as Arc) + .map_err(|error| error.to_string()) + } + AlgorithmConfig::StageRouter { + capable_target, + efficient_target, + picker, + confidence_threshold, + recent_turn_window, + capable_system_prompt, + efficient_system_prompt, + handoff_notes, + classifier, + } => { + let capable = target(capable_target)?; + let efficient = target(efficient_target)?; + let mut config = StageRouterConfig::new(*picker, *confidence_threshold); + config.recent_window = *recent_turn_window; + config.handoff_notes = handoff_notes.clone(); + let mut prompts = TargetPrompts::default(); + if let Some(prompt) = capable_system_prompt { + prompts = prompts.with(capable_target, prompt); + } + if let Some(prompt) = efficient_system_prompt { + prompts = prompts.with(efficient_target, prompt); + } + config.tier_prompts = prompts; + if let Some(classifier) = classifier { + self.validate_judge_target(&classifier.target)?; + config.llm_fallback = Some(LlmFallback { + judge_target: target(&classifier.target)?, + config: classifier.classifier_config(), + }); + } + StageRouter::new(capable, efficient, config) + .map(|algorithm| Arc::new(algorithm) as Arc) + .map_err(|error| error.to_string()) + } + } + } + + fn validate_judge_target(&self, name: &str) -> Result<(), String> { + let binding = self + .targets + .get(name) + .ok_or_else(|| format!("algorithm target {name:?} is not configured"))?; + if binding.protocol == WireFormat::AnthropicMessages { + return Err(format!( + "classifier target {name:?} uses anthropic_messages, which cannot encode the required JSON-schema response format without loss; use an openai_chat or openai_responses target" + )); + } + Ok(()) + } +} + +fn validate_dispatch_url( + target_name: &str, + protocol: WireFormat, + dispatch_url: &str, +) -> Result<(), String> { + let uri = dispatch_url + .parse::() + .map_err(|error| format!("target {target_name:?} has invalid URL: {error}"))?; + if !matches!(uri.scheme_str(), Some("http" | "https")) { + return Err(format!( + "target {target_name:?} base_url must use http or https" + )); + } + let authority = uri + .authority() + .ok_or_else(|| format!("target {target_name:?} URL must include a host"))?; + if authority.host().is_empty() { + return Err(format!("target {target_name:?} URL must include a host")); + } + if authority.as_str().contains('@') { + return Err(format!( + "target {target_name:?} URL must not contain embedded credentials" + )); + } + if uri.query().is_some() { + return Err(format!( + "target {target_name:?} URL query parameters are not supported" + )); + } + + // The current switchyard-llm-client accepts provider base URLs and complete + // canonical endpoints. Reject a custom terminal route to avoid + // allowing Backend::url() to append another provider suffix silently. + let expected_suffix = match protocol { + WireFormat::OpenAiChat => "/chat/completions", + WireFormat::OpenAiResponses => "/responses", + WireFormat::AnthropicMessages => "/v1/messages", + }; + if !uri.path().ends_with(expected_suffix) { + return Err(format!( + "target {target_name:?} endpoint must resolve to a canonical {protocol} route ending in {expected_suffix:?}" + )); + } + Ok(()) +} + +fn validate_header_name(name: &str) -> Result { + let parsed = HeaderName::from_bytes(name.as_bytes()) + .map_err(|error| format!("invalid target header name {name:?}: {error}"))?; + let canonical = parsed.as_str().to_ascii_lowercase(); + if is_forbidden_target_header(&canonical) { + return Err(format!( + "target header {name:?} is controlled by the HTTP transport and cannot be configured" + )); + } + Ok(canonical) +} + +fn validate_header(name: &str, value: &str) -> Result { + let canonical = validate_header_name(name)?; + HeaderValue::from_str(value) + .map_err(|error| format!("invalid target header value for {name:?}: {error}"))?; + Ok(canonical) +} + +fn is_forbidden_target_header(name: &str) -> bool { + matches!( + name, + "connection" + | "content-length" + | "host" + | "keep-alive" + | "proxy-connection" + | "proxy-authenticate" + | "proxy-authorization" + | "te" + | "trailer" + | "transfer-encoding" + | "upgrade" + ) || name.starts_with("x-nemo-relay-internal-") +} + +const fn default_max_retries() -> u32 { + 3 +} + +const fn default_weight() -> f64 { + 1.0 +} + +fn default_classifier_max_output_tokens() -> u64 { + TaskClassifierConfig::default().max_output_tokens +} + +#[cfg(test)] +mod tests; diff --git a/crates/switchyard-nemo-relay-plugin/src/config/tests.rs b/crates/switchyard-nemo-relay-plugin/src/config/tests.rs new file mode 100644 index 000000000..1a1153e9f --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/config/tests.rs @@ -0,0 +1,545 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +use super::*; +use serde_json::{Value, json}; + +fn binding(protocol: WireFormat, model: &str) -> TargetBinding { + TargetBinding { + model: model.into(), + protocol, + endpoint: String::new(), + base_url: "https://provider.example/v1".into(), + weight: 1.0, + drop_caller_extra_body: false, + header_env: BTreeMap::new(), + extra_body: BTreeMap::new(), + } +} + +fn config() -> SwitchyardConfig { + SwitchyardConfig { + version: 2, + priority: 0, + max_retries: 3, + algorithm: AlgorithmConfig::Random { seed: Some(42) }, + targets: BTreeMap::from([ + ( + "chat".into(), + binding(WireFormat::OpenAiChat, "provider/chat"), + ), + ( + "responses".into(), + binding(WireFormat::OpenAiResponses, "provider/responses"), + ), + ( + "anthropic".into(), + binding(WireFormat::AnthropicMessages, "provider/anthropic"), + ), + ]), + default_targets: BTreeMap::from([ + (WireFormat::OpenAiChat, "chat".into()), + (WireFormat::OpenAiResponses, "responses".into()), + (WireFormat::AnthropicMessages, "anthropic".into()), + ]), + } +} + +#[test] +fn version_two_random_configuration_builds_clients_without_a_service() { + let config = config(); + config.validate().unwrap(); + let prepared = config.prepare().unwrap(); + assert_eq!(prepared.algorithm.name(), "random"); + assert_eq!(prepared.targets.len(), 3); + assert!( + prepared + .targets + .values() + .all(|target| Arc::strong_count(&target.client) == 1) + ); +} + +#[test] +fn target_endpoints_must_be_canonical_for_the_current_http_client() { + let mut config = config(); + config.targets.get_mut("chat").unwrap().endpoint = "/custom/chat".into(); + let error = config.validate().unwrap_err(); + assert!(error.contains("ending in \"/chat/completions\"")); + + config.targets.get_mut("chat").unwrap().endpoint = "/custom/chat/completions".into(); + config.validate().unwrap(); + assert_eq!( + config.targets["chat"].dispatch_url(), + "https://provider.example/v1/custom/chat/completions" + ); +} + +#[test] +fn complete_provider_endpoint_is_not_appended_twice() { + let mut config = config(); + let chat = config.targets.get_mut("chat").unwrap(); + chat.base_url = "https://provider.example/v1/chat/completions/".into(); + assert_eq!( + chat.dispatch_url(), + "https://provider.example/v1/chat/completions" + ); + config.validate().unwrap(); +} + +#[test] +fn absolute_urls_cannot_embed_credentials_or_query_parameters() { + let mut config = config(); + config.targets.get_mut("chat").unwrap().base_url = + "https://user:password@provider.example/v1".into(); + assert!( + config + .validate() + .unwrap_err() + .contains("embedded credentials") + ); + + config.targets.get_mut("chat").unwrap().base_url = + "https://provider.example/v1?api-version=1".into(); + assert!(config.validate().unwrap_err().contains("query parameters")); +} + +#[test] +fn transport_owned_and_case_duplicate_environment_headers_are_rejected() { + let mut host_header_config = config(); + let chat = host_header_config.targets.get_mut("chat").unwrap(); + chat.header_env.insert("Host".into(), "TARGET_HOST".into()); + assert!( + host_header_config + .validate() + .unwrap_err() + .contains("HTTP transport") + ); + + let mut duplicate_config = config(); + let chat = duplicate_config.targets.get_mut("chat").unwrap(); + chat.header_env + .insert("X-Tenant".into(), "TARGET_TENANT_A".into()); + chat.header_env + .insert("x-tenant".into(), "TARGET_TENANT_B".into()); + assert!( + duplicate_config + .validate() + .unwrap_err() + .contains("more than once") + ); +} + +#[test] +fn only_canonical_relay_execution_names_resolve_protocols() { + assert_eq!( + protocol_from_call("openai.chat_completions"), + Some(WireFormat::OpenAiChat) + ); + assert_eq!( + protocol_from_call("openai.responses"), + Some(WireFormat::OpenAiResponses) + ); + assert_eq!( + protocol_from_call("anthropic.messages"), + Some(WireFormat::AnthropicMessages) + ); + assert_eq!(protocol_from_call("openai_chat"), None); +} + +#[test] +fn schema_required_contract_fields_do_not_default_during_deserialization() { + let base = json!({ + "version": 2, + "algorithm": {"kind": "random"}, + "targets": { + "chat": { + "model": "provider/chat", + "protocol": "openai_chat", + "base_url": "https://provider.example/v1" + } + }, + "default_targets": {"openai_chat": "chat"} + }); + for field in ["version", "algorithm", "default_targets"] { + let mut value = base.clone(); + value.as_object_mut().unwrap().remove(field); + let error = serde_json::from_value::(value) + .err() + .expect("required field must not default"); + assert!(error.to_string().contains(field), "field={field}: {error}"); + } +} + +#[test] +fn unknown_target_fields_are_rejected() { + let value = json!({ + "version": 2, + "algorithm": {"kind": "random"}, + "targets": { + "chat": { + "model": "provider/chat", + "protocol": "openai_chat", + "base_url": "https://provider.example/v1", + "unexpected_setting": true + } + }, + "default_targets": {"openai_chat": "chat"} + }); + let error = serde_json::from_value::(value) + .err() + .expect("unknown target field must be rejected"); + assert!(error.to_string().contains("unexpected_setting")); +} + +#[test] +fn literal_target_headers_are_rejected() { + let value = json!({ + "version": 2, + "algorithm": {"kind": "random"}, + "targets": { + "chat": { + "model": "provider/chat", + "protocol": "openai_chat", + "base_url": "https://provider.example/v1", + "headers": {"x-provider-token": "plaintext-secret"} + } + }, + "default_targets": {"openai_chat": "chat"} + }); + let error = serde_json::from_value::(value) + .err() + .expect("literal target headers must be rejected") + .to_string(); + assert!(error.contains("unknown field `headers`")); + assert!(!error.contains("plaintext-secret")); +} + +#[test] +fn unknown_algorithm_fields_are_rejected() { + let error = serde_json::from_value::(json!({ + "kind": "random", + "seed": 42, + "unexpected_setting": true + })) + .err() + .expect("unknown algorithm field must be rejected"); + assert!(error.to_string().contains("unexpected_setting")); +} + +#[test] +fn classifier_prepares_clients_for_judge_and_routed_targets() { + let mut config = config(); + config.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "classifier_target": "chat", + "weak_target": "responses", + "strong_target": "anthropic", + "base_threshold": 0.5, + "recent_turn_window": 4, + "max_output_tokens": 512 + })) + .unwrap(); + config.validate().unwrap(); + let prepared = config.prepare().unwrap(); + assert_eq!(prepared.algorithm.name(), "llm_task_classifier"); + assert!( + prepared + .targets + .values() + .all(|target| Arc::strong_count(&target.client) == 1) + ); +} + +#[test] +fn target_provider_defaults_are_accepted_for_judge_controls() { + let mut config = config(); + config.targets.get_mut("chat").unwrap().extra_body = + BTreeMap::from([("think".into(), json!(false))]); + + config.validate().unwrap(); + config.prepare().unwrap(); +} + +#[test] +fn classifier_rejects_anthropic_judge_targets_before_dispatch() { + let mut config = config(); + config.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "classifier_target": "anthropic", + "weak_target": "responses", + "strong_target": "chat", + "base_threshold": 0.5 + })) + .unwrap(); + + let error = config.validate().unwrap_err(); + assert!(error.contains("classifier target \"anthropic\" uses anthropic_messages")); +} + +#[test] +fn validation_does_not_resolve_environment_backed_headers() { + let mut config = config(); + config.targets.get_mut("chat").unwrap().header_env = BTreeMap::from([( + "authorization".into(), + "SWITCHYARD_TEST_ENVIRONMENT_VARIABLE_THAT_IS_NOT_SET".into(), + )]); + + config.validate().unwrap(); + let error = config + .prepare() + .err() + .expect("preparation must resolve headers"); + assert!(error.contains("SWITCHYARD_TEST_ENVIRONMENT_VARIABLE_THAT_IS_NOT_SET")); +} + +#[test] +fn invalid_environment_variable_names_are_rejected_before_resolution() { + for variable in ["INVALID=VARIABLE", "INVALID\0VARIABLE"] { + let mut config = config(); + config.targets.get_mut("chat").unwrap().header_env = + BTreeMap::from([("authorization".into(), variable.into())]); + + let error = config.validate().unwrap_err(); + assert!(error.contains("must not contain '=' or NUL")); + } +} + +#[test] +fn static_validation_preserves_algorithm_constructor_checks() { + let mut random = config(); + for target in random.targets.values_mut() { + target.weight = 0.0; + } + assert!( + random + .validate() + .unwrap_err() + .contains("at least one positive target weight") + ); + + let mut classifier = config(); + classifier.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "classifier_target": "chat", + "weak_target": "responses", + "strong_target": "anthropic", + "base_threshold": 1.1 + })) + .unwrap(); + assert!( + classifier + .validate() + .unwrap_err() + .contains("base_threshold must be between 0 and 1") + ); +} + +#[test] +fn escalation_classifier_builds_with_defaulted_policy_settings() { + let mut config = config(); + config.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "mode": "escalation", + "classifier_target": "chat", + "weak_target": "responses", + "strong_target": "anthropic", + "prompt": "Judge the completed trajectory.", + "max_output_tokens": 256, + "escalation": {} + })) + .unwrap(); + + config.validate().unwrap(); + let prepared = config.prepare().unwrap(); + assert_eq!(prepared.algorithm.name(), "llm_task_classifier"); + assert!( + prepared + .targets + .values() + .all(|target| Arc::strong_count(&target.client) == 1) + ); +} + +#[test] +fn classifier_modes_reject_mixed_or_missing_settings() { + let mut capability = config(); + capability.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "classifier_target": "chat", + "weak_target": "responses", + "strong_target": "anthropic", + "base_threshold": 0.5, + "escalation": {} + })) + .unwrap(); + assert!( + capability + .validate() + .unwrap_err() + .contains("capability mode does not accept escalation") + ); + + let mut escalation = config(); + escalation.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "mode": "escalation", + "classifier_target": "chat", + "weak_target": "responses", + "strong_target": "anthropic", + "base_threshold": 0.5, + "escalation": {} + })) + .unwrap(); + assert!( + escalation + .validate() + .unwrap_err() + .contains("escalation mode does not accept capability") + ); + + let mut missing = config(); + missing.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "mode": "escalation", + "classifier_target": "chat", + "weak_target": "responses", + "strong_target": "anthropic" + })) + .unwrap(); + assert!( + missing + .validate() + .unwrap_err() + .contains("requires escalation settings") + ); +} + +#[test] +fn escalation_settings_are_validated_by_the_libsy_constructor() { + for (settings, expected) in [ + ( + json!({"confirmations": 0}), + "confirmations must be at least 1", + ), + ( + json!({"recent_turn_window": 0}), + "recent_turn_window must be at least 1", + ), + ( + json!({"window_message_chars": 49}), + "window_message_chars must be at least 50", + ), + ] { + let mut config = config(); + config.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "mode": "escalation", + "classifier_target": "chat", + "weak_target": "responses", + "strong_target": "anthropic", + "escalation": settings + })) + .unwrap(); + assert!(config.validate().unwrap_err().contains(expected)); + } +} + +#[test] +fn full_stage_router_configuration_builds_all_clients() { + let mut config = config(); + config.algorithm = serde_json::from_value(json!({ + "kind": "stage_router", + "capable_target": "anthropic", + "efficient_target": "responses", + "picker": "efficient_first", + "confidence_threshold": 0.5, + "recent_turn_window": 3, + "capable_system_prompt": "Diagnose before editing.", + "efficient_system_prompt": "Follow the settled plan.", + "handoff_notes": { + "escalation_note": "The previous model was stalling.", + "deescalation_note": "The task is settled.", + "only_on_wrong_signal_escalation": true + }, + "classifier": { + "target": "chat", + "base_threshold": 0.5, + "threshold_step": 0.1, + "recent_turn_window": 3, + "prompt": "Can the efficient tier finish this turn?", + "max_output_tokens": 256 + } + })) + .unwrap(); + + config.validate().unwrap(); + let prepared = config.prepare().unwrap(); + assert_eq!(prepared.algorithm.name(), "stage_router"); + assert!( + prepared + .targets + .values() + .all(|target| Arc::strong_count(&target.client) == 1) + ); +} + +#[test] +fn stage_router_validates_threshold_targets_and_judge_protocol() { + let stage = |classifier: Value, threshold: f64| { + serde_json::from_value(json!({ + "kind": "stage_router", + "capable_target": "anthropic", + "efficient_target": "responses", + "picker": "capable_first", + "confidence_threshold": threshold, + "classifier": classifier + })) + .unwrap() + }; + + let mut invalid_threshold = config(); + invalid_threshold.algorithm = stage(Value::Null, 1.1); + assert!( + invalid_threshold + .validate() + .unwrap_err() + .contains("confidence_threshold must be between 0 and 1") + ); + + let mut missing_target = config(); + missing_target.algorithm = serde_json::from_value(json!({ + "kind": "stage_router", + "capable_target": "missing", + "efficient_target": "responses", + "picker": "capable_first", + "confidence_threshold": 0.5 + })) + .unwrap(); + assert!( + missing_target + .validate() + .unwrap_err() + .contains("algorithm target \"missing\" is not configured") + ); + + let mut anthropic_judge = config(); + anthropic_judge.algorithm = stage(json!({"target": "anthropic", "base_threshold": 0.5}), 0.5); + assert!( + anthropic_judge + .validate() + .unwrap_err() + .contains("classifier target \"anthropic\" uses anthropic_messages") + ); +} + +#[test] +fn zero_weight_random_targets_are_fallback_only() { + let mut config = config(); + config.targets.get_mut("anthropic").unwrap().weight = 0.0; + let prepared = config.prepare().unwrap(); + + assert_eq!(Arc::strong_count(&prepared.targets["anthropic"].client), 1); + assert_eq!(Arc::strong_count(&prepared.targets["chat"].client), 1); + assert_eq!(Arc::strong_count(&prepared.targets["responses"].client), 1); +} diff --git a/crates/switchyard-nemo-relay-plugin/src/executor.rs b/crates/switchyard-nemo-relay-plugin/src/executor.rs new file mode 100644 index 000000000..206c7aa99 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/executor.rs @@ -0,0 +1,137 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +use std::future::Future; +use std::sync::{Arc, Mutex, mpsc}; +use std::thread::{self, JoinHandle}; + +use tokio::runtime::{Builder, Handle}; +use tokio::sync::oneshot; +use tokio::task::AbortHandle; + +/// Plugin-owned async executor. +/// +/// The public native-plugin SDK uses synchronous Rust callbacks and pull-based +/// iterators at the dynamic-library boundary. Switchyard performs provider I/O +/// on this dedicated runtime rather than entering Relay's Tokio runtime from a +/// separately linked cdylib. +#[derive(Clone)] +pub(crate) struct PluginExecutor { + inner: Arc, +} + +struct ExecutorInner { + handle: Handle, + shutdown: Mutex>>, + thread: Mutex>>, +} + +impl PluginExecutor { + pub(crate) fn new() -> Result { + let (ready_tx, ready_rx) = mpsc::sync_channel(1); + let thread = thread::Builder::new() + .name("switchyard-relay-http".into()) + .spawn(move || { + let runtime = match Builder::new_multi_thread() + .worker_threads(2) + .thread_name("switchyard-relay-http-worker") + .enable_all() + .build() + { + Ok(runtime) => runtime, + Err(error) => { + let _ = ready_tx.send(Err(error.to_string())); + return; + } + }; + let (shutdown_tx, shutdown_rx) = oneshot::channel(); + if ready_tx + .send(Ok((runtime.handle().clone(), shutdown_tx))) + .is_err() + { + return; + } + runtime.block_on(async { + let _ = shutdown_rx.await; + }); + }) + .map_err(|error| format!("failed to start Switchyard HTTP runtime: {error}"))?; + let (handle, shutdown) = ready_rx + .recv() + .map_err(|_| "Switchyard HTTP runtime stopped during startup".to_string())??; + Ok(Self { + inner: Arc::new(ExecutorInner { + handle, + shutdown: Mutex::new(Some(shutdown)), + thread: Mutex::new(Some(thread)), + }), + }) + } + + pub(crate) fn spawn(&self, future: F) -> AbortHandle + where + F: Future + Send + 'static, + { + self.inner.handle.spawn(future).abort_handle() + } +} + +impl Drop for ExecutorInner { + fn drop(&mut self) { + if let Some(shutdown) = self + .shutdown + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .take() + { + let _ = shutdown.send(()); + } + if let Some(thread) = self + .thread + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .take() + { + if std::thread::current() + .name() + .is_some_and(|name| name.starts_with("switchyard-relay-http-worker")) + { + // The runtime owner will join this worker after the current + // task returns. Waiting here would deadlock that shutdown. + drop(thread); + } else { + let _ = thread.join(); + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn executor_spawns_work() { + let executor = PluginExecutor::new().unwrap(); + let (sender, receiver) = mpsc::sync_channel(1); + executor.spawn(async move { + sender.send("done").unwrap(); + }); + assert_eq!(receiver.recv().unwrap(), "done"); + } + + #[test] + fn last_reference_can_drop_on_a_worker() { + let executor = PluginExecutor::new().unwrap(); + let worker_reference = executor.clone(); + let (sender, receiver) = mpsc::sync_channel(1); + executor.spawn(async move { + drop(worker_reference); + sender.send(()).unwrap(); + }); + drop(executor); + receiver + .recv_timeout(std::time::Duration::from_secs(5)) + .expect("dropping the executor on its own worker must not deadlock"); + } +} diff --git a/crates/switchyard-nemo-relay-plugin/src/ffi.rs b/crates/switchyard-nemo-relay-plugin/src/ffi.rs new file mode 100644 index 000000000..f5ce094c1 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/ffi.rs @@ -0,0 +1,444 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +//! Small ownership wrapper around Relay's generic C host-table v3 hooks. +//! +//! The plugin manifest remains native API v1. Relay 0.7 supplies the appended +//! v3 host table to rebuilt v1 plugins, which lets this crate return `Pending` +//! and settle work from its own runtime without a targeted-continuation ABI. + +use std::ffi::c_void; +use std::ptr; +use std::sync::Arc; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::time::Duration; + +use nemo_relay_plugin::{ + Json, LlmRequest, NemoRelayNativeAsyncCompletion, NemoRelayNativeAsyncNext, + NemoRelayNativeAsyncNextStreamCb, NemoRelayNativeAsyncStream, NemoRelayNativeHostApiV1, + NemoRelayNativeHostApiV3, NemoRelayNativeScopeHandle, NemoRelayNativeString, NemoRelayStatus, +}; +use serde::Serialize; +use tokio::sync::{mpsc, oneshot}; + +const BACKPRESSURE_POLL: Duration = Duration::from_millis(1); +const CANCELLATION_POLL: Duration = Duration::from_millis(10); +const MAX_PASSTHROUGH_BUFFER_BYTES: usize = 8 * 1024 * 1024; +const MAX_PASSTHROUGH_BUFFER_EVENTS: usize = 256; + +pub(crate) struct HostString { + host: NemoRelayNativeHostApiV1, + ptr: *mut NemoRelayNativeString, +} + +// Host strings are immutable allocations owned by Relay's thread-safe host table. +unsafe impl Send for HostString {} + +impl HostString { + pub(crate) fn json( + host: &NemoRelayNativeHostApiV1, + value: &impl Serialize, + ) -> Result { + let value = serde_json::to_string(value).map_err(|error| error.to_string())?; + Self::text(host, &value) + } + + pub(crate) fn text(host: &NemoRelayNativeHostApiV1, value: &str) -> Result { + let mut ptr = ptr::null_mut(); + let status = unsafe { (host.string_new)(value.as_ptr(), value.len(), &mut ptr) }; + if status == NemoRelayStatus::Ok && !ptr.is_null() { + Ok(Self { host: *host, ptr }) + } else { + Err(format!("Relay host string allocation failed: {status:?}")) + } + } + + pub(crate) fn as_ptr(&self) -> *const NemoRelayNativeString { + self.ptr + } +} + +impl Drop for HostString { + fn drop(&mut self) { + unsafe { (self.host.string_free)(self.ptr) }; + } +} + +pub(crate) fn read_string( + host: &NemoRelayNativeHostApiV1, + value: *const NemoRelayNativeString, +) -> Result { + if value.is_null() { + return Err("Relay passed a null native string".into()); + } + let len = unsafe { (host.string_len)(value) }; + let data = unsafe { (host.string_data)(value) }; + if data.is_null() && len != 0 { + return Err("Relay passed an invalid native string".into()); + } + let bytes = if len == 0 { + &[][..] + } else { + unsafe { std::slice::from_raw_parts(data, len) } + }; + std::str::from_utf8(bytes) + .map(str::to_owned) + .map_err(|error| error.to_string()) +} + +pub(crate) fn read_json( + host: &NemoRelayNativeHostApiV1, + value: *const NemoRelayNativeString, +) -> Result { + serde_json::from_str(&read_string(host, value)?).map_err(|error| error.to_string()) +} + +/// Captures the current Relay scope as an explicit event parent. +/// +/// Async plugin work runs on a plugin-owned thread, so relying on thread-local +/// scope state would orphan its marks. The host handle is a cloned scope handle +/// and remains valid until this guard is dropped. +pub(crate) struct ParentScope { + host: NemoRelayNativeHostApiV1, + ptr: *mut NemoRelayNativeScopeHandle, +} + +unsafe impl Send for ParentScope {} +unsafe impl Sync for ParentScope {} + +impl ParentScope { + pub(crate) fn capture(host: &NemoRelayNativeHostApiV1) -> Option { + let mut ptr = ptr::null_mut(); + let status = unsafe { (host.scope_get_current)(&mut ptr) }; + (status == NemoRelayStatus::Ok && !ptr.is_null()).then_some(Self { host: *host, ptr }) + } + + pub(crate) fn emit_mark(&self, name: &str, data: &Json, metadata: &Json) -> Result<(), String> { + let name = HostString::text(&self.host, name)?; + let data = HostString::json(&self.host, data)?; + let metadata = HostString::json(&self.host, metadata)?; + let status = unsafe { + (self.host.emit_mark)( + name.as_ptr(), + self.ptr, + data.as_ptr(), + metadata.as_ptr(), + ptr::null(), + ) + }; + if status == NemoRelayStatus::Ok { + Ok(()) + } else { + Err(format!( + "Relay rejected Switchyard routing mark: {status:?}" + )) + } + } +} + +impl Drop for ParentScope { + fn drop(&mut self) { + unsafe { (self.host.scope_handle_free)(self.ptr) }; + } +} + +pub(crate) fn invoke_next_buffered( + host: &NemoRelayNativeHostApiV3, + next: usize, + completion: usize, + request: &LlmRequest, +) -> Result<(), String> { + let request = HostString::json(&host.v1, request)?; + let status = unsafe { + (host.async_next_invoke)( + next as *const NemoRelayNativeAsyncNext, + request.as_ptr(), + completion as *const NemoRelayNativeAsyncCompletion, + ) + }; + if status == NemoRelayStatus::Ok { + Ok(()) + } else { + Err(format!("Relay rejected buffered pass-through: {status:?}")) + } +} + +enum DownstreamStreamItem { + Chunk { value: Json, encoded_bytes: usize }, +} + +struct DownstreamStreamState { + host: NemoRelayNativeHostApiV1, + sender: mpsc::Sender, + terminal: Option>>, + queued_bytes: Arc, +} + +pub(crate) async fn invoke_next_stream( + host: &NemoRelayNativeHostApiV3, + next: usize, + output: usize, + request: &LlmRequest, +) -> Result<(), String> { + let request = HostString::json(&host.v1, request)?; + let (sender, mut receiver) = mpsc::channel(MAX_PASSTHROUGH_BUFFER_EVENTS); + let (terminal, terminal_result) = oneshot::channel(); + let queued_bytes = Arc::new(AtomicUsize::new(0)); + let state = Box::into_raw(Box::new(DownstreamStreamState { + host: host.v1, + sender, + terminal: Some(terminal), + queued_bytes: Arc::clone(&queued_bytes), + })) + .cast::(); + let status = unsafe { + (host.async_next_invoke_stream)( + next as *const NemoRelayNativeAsyncNext, + request.as_ptr(), + output as *const NemoRelayNativeAsyncStream, + downstream_stream_result as NemoRelayNativeAsyncNextStreamCb, + state, + ) + }; + if status != NemoRelayStatus::Ok { + unsafe { drop(Box::from_raw(state.cast::())) }; + return Err(format!("Relay rejected streaming pass-through: {status:?}")); + } + + while let Some(item) = receiver.recv().await { + match item { + DownstreamStreamItem::Chunk { + value, + encoded_bytes, + } => { + let result = push_stream(host, output, &value).await; + queued_bytes.fetch_sub(encoded_bytes, Ordering::AcqRel); + result?; + } + } + } + terminal_result + .await + .unwrap_or_else(|_| Err("Relay dropped the streaming pass-through callback".into())) +} + +unsafe extern "C" fn downstream_stream_result( + user_data: *mut c_void, + chunk_json: *const NemoRelayNativeString, + error: *const NemoRelayNativeString, + done: bool, +) -> bool { + if !error.is_null() { + let state = unsafe { Box::from_raw(user_data.cast::()) }; + let error = read_string(&state.host, error) + .unwrap_or_else(|_| "Relay streaming pass-through failed".into()); + settle_downstream_stream(state, Err(error)); + return false; + } + if done { + let state = unsafe { Box::from_raw(user_data.cast::()) }; + settle_downstream_stream(state, Ok(())); + return false; + } + + let state = unsafe { &*user_data.cast::() }; + let parsed = read_string(&state.host, chunk_json).and_then(|encoded| { + let encoded_bytes = encoded.len(); + let value = serde_json::from_str(&encoded).map_err(|error| error.to_string())?; + Ok((value, encoded_bytes)) + }); + let (value, encoded_bytes) = match parsed { + Ok(parsed) => parsed, + Err(error) => { + let state = unsafe { Box::from_raw(user_data.cast::()) }; + settle_downstream_stream(state, Err(error)); + return false; + } + }; + if !reserve_buffer_bytes(&state.queued_bytes, encoded_bytes) { + let state = unsafe { Box::from_raw(user_data.cast::()) }; + settle_downstream_stream( + state, + Err(format!( + "Relay streaming pass-through exceeded its {}-byte queued payload limit", + MAX_PASSTHROUGH_BUFFER_BYTES + )), + ); + return false; + } + + match state.sender.try_send(DownstreamStreamItem::Chunk { + value, + encoded_bytes, + }) { + Ok(()) => true, + Err(error) => { + let (item, message) = match error { + mpsc::error::TrySendError::Full(item) => ( + item, + format!( + "Relay streaming pass-through exceeded its {MAX_PASSTHROUGH_BUFFER_EVENTS}-event queue" + ), + ), + mpsc::error::TrySendError::Closed(item) => ( + item, + "Relay dropped the streaming pass-through receiver".into(), + ), + }; + let encoded_bytes = item.encoded_bytes(); + state + .queued_bytes + .fetch_sub(encoded_bytes, Ordering::AcqRel); + let state = unsafe { Box::from_raw(user_data.cast::()) }; + settle_downstream_stream(state, Err(message)); + false + } + } +} + +impl DownstreamStreamItem { + fn encoded_bytes(&self) -> usize { + match self { + Self::Chunk { encoded_bytes, .. } => *encoded_bytes, + } + } +} + +fn settle_downstream_stream(mut state: Box, result: Result<(), String>) { + if let Some(terminal) = state.terminal.take() { + let _ = terminal.send(result); + } +} + +fn reserve_buffer_bytes(queued: &AtomicUsize, encoded_bytes: usize) -> bool { + queued + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { + current + .checked_add(encoded_bytes) + .filter(|next| *next <= MAX_PASSTHROUGH_BUFFER_BYTES) + }) + .is_ok() +} + +pub(crate) async fn wait_for_completion_cancellation( + host: &NemoRelayNativeHostApiV3, + completion: usize, +) { + while !completion_cancelled(host, completion as *const NemoRelayNativeAsyncCompletion) { + tokio::time::sleep(CANCELLATION_POLL).await; + } +} + +pub(crate) async fn wait_for_stream_cancellation(host: &NemoRelayNativeHostApiV3, stream: usize) { + while !unsafe { (host.async_stream_is_cancelled)(stream as *const NemoRelayNativeAsyncStream) } + { + tokio::time::sleep(CANCELLATION_POLL).await; + } +} + +pub(crate) fn completion_cancelled( + host: &NemoRelayNativeHostApiV3, + completion: *const NemoRelayNativeAsyncCompletion, +) -> bool { + unsafe { (host.async_completion_is_cancelled)(completion) } +} + +pub(crate) fn resolve_completion( + host: &NemoRelayNativeHostApiV3, + completion: *const NemoRelayNativeAsyncCompletion, + value: &Json, +) -> NemoRelayStatus { + match HostString::json(&host.v1, value) { + Ok(value) => unsafe { (host.async_completion_resolve_json)(completion, value.as_ptr()) }, + Err(_) => NemoRelayStatus::Internal, + } +} + +pub(crate) fn reject_completion( + host: &NemoRelayNativeHostApiV3, + completion: *const NemoRelayNativeAsyncCompletion, + message: &str, +) -> NemoRelayStatus { + match HostString::text(&host.v1, message) { + Ok(message) => unsafe { (host.async_completion_reject)(completion, message.as_ptr()) }, + Err(_) => NemoRelayStatus::Internal, + } +} + +pub(crate) async fn push_stream( + host: &NemoRelayNativeHostApiV3, + stream: usize, + value: &Json, +) -> Result<(), String> { + let value = HostString::json(&host.v1, value)?; + loop { + if unsafe { (host.async_stream_is_cancelled)(stream as *const NemoRelayNativeAsyncStream) } + { + return Err("Relay caller cancelled the output stream".into()); + } + match unsafe { + (host.async_stream_push_json)( + stream as *const NemoRelayNativeAsyncStream, + value.as_ptr(), + ) + } { + NemoRelayStatus::Ok => return Ok(()), + // Native API v1 reports its bounded queue's WouldBlock state as Internal. + NemoRelayStatus::Internal => tokio::time::sleep(BACKPRESSURE_POLL).await, + status => return Err(format!("Relay rejected output stream event: {status:?}")), + } + } +} + +pub(crate) fn finish_stream( + host: &NemoRelayNativeHostApiV3, + stream: *const NemoRelayNativeAsyncStream, +) -> NemoRelayStatus { + unsafe { (host.async_stream_finish)(stream) } +} + +pub(crate) async fn reject_stream( + host: &NemoRelayNativeHostApiV3, + stream: usize, + message: &str, +) -> NemoRelayStatus { + let Ok(message) = HostString::text(&host.v1, message) else { + return NemoRelayStatus::Internal; + }; + loop { + if unsafe { (host.async_stream_is_cancelled)(stream as *const NemoRelayNativeAsyncStream) } + { + return NemoRelayStatus::InvalidArg; + } + match unsafe { + (host.async_stream_reject)( + stream as *const NemoRelayNativeAsyncStream, + message.as_ptr(), + ) + } { + NemoRelayStatus::Internal => tokio::time::sleep(BACKPRESSURE_POLL).await, + status => return status, + } + } +} + +pub(crate) unsafe fn release_completion( + host: &NemoRelayNativeHostApiV3, + completion: *const NemoRelayNativeAsyncCompletion, +) { + unsafe { (host.async_completion_release)(completion) }; +} + +pub(crate) unsafe fn release_next( + host: &NemoRelayNativeHostApiV3, + next: *const NemoRelayNativeAsyncNext, +) { + unsafe { (host.async_next_release)(next) }; +} + +pub(crate) unsafe fn release_stream( + host: &NemoRelayNativeHostApiV3, + stream: *const NemoRelayNativeAsyncStream, +) { + unsafe { (host.async_stream_release)(stream) }; +} diff --git a/crates/switchyard-nemo-relay-plugin/src/lib.rs b/crates/switchyard-nemo-relay-plugin/src/lib.rs new file mode 100644 index 000000000..f3a2fd7c6 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/lib.rs @@ -0,0 +1,442 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +mod client; +mod config; +mod executor; +mod ffi; +mod runtime; +mod translation; + +use std::ffi::c_void; +use std::mem; +use std::panic::AssertUnwindSafe; +use std::sync::Arc; + +use futures_util::FutureExt; +use nemo_relay_plugin::{ + ConfigDiagnostic, DiagnosticLevel, Json, NEMO_RELAY_NATIVE_ABI_VERSION_ASYNC_MIDDLEWARE, + NativePlugin, NemoRelayNativeAsyncCallbackState, NemoRelayNativeAsyncCompletion, + NemoRelayNativeAsyncMiddlewareKind, NemoRelayNativeAsyncNext, NemoRelayNativeAsyncStream, + NemoRelayNativeHostApiV3, NemoRelayNativeString, NemoRelayStatus, PluginContext, +}; +use serde::Deserialize; +use serde_json::Map; + +use crate::config::SwitchyardConfig; +use crate::executor::PluginExecutor; +use crate::runtime::{RoutingMark, StreamMessage, SwitchyardRuntime}; + +#[derive(Deserialize)] +struct Invocation { + name: String, + request: nemo_relay_plugin::LlmRequest, +} + +struct CallbackState { + host: NemoRelayNativeHostApiV3, + runtime: Arc, + executor: PluginExecutor, +} + +#[derive(Default)] +struct SwitchyardPlugin; + +impl NativePlugin for SwitchyardPlugin { + fn plugin_kind(&self) -> &str { + "nvidia.switchyard" + } + + fn allows_multiple_components(&self) -> bool { + false + } + + fn validate(&self, plugin_config: &Map) -> Vec { + match parse_config(plugin_config).and_then(|config| config.validate()) { + Ok(()) => Vec::new(), + Err(message) => vec![ConfigDiagnostic { + level: DiagnosticLevel::Error, + code: "switchyard.invalid_config".into(), + component: Some("nvidia.switchyard".into()), + field: Some("config".into()), + message, + }], + } + } + + fn register( + &mut self, + plugin_config: &Map, + ctx: &mut PluginContext<'_>, + ) -> nemo_relay_plugin::Result<()> { + let host_v1 = ctx.host_api(); + if host_v1.abi_version < NEMO_RELAY_NATIVE_ABI_VERSION_ASYNC_MIDDLEWARE + || host_v1.struct_size < mem::size_of::() + { + return Err( + "Switchyard requires Relay 0.7 or newer with the generic asynchronous native host table" + .into(), + ); + } + let host = unsafe { *(host_v1 as *const _ as *const NemoRelayNativeHostApiV3) }; + let config = parse_config(plugin_config)?; + let priority = config.priority; + let state = Arc::new(CallbackState { + host, + runtime: Arc::new(SwitchyardRuntime::new(config)?), + executor: PluginExecutor::new()?, + }); + + register_buffered(ctx, priority, Arc::clone(&state))?; + register_stream(ctx, priority, state)?; + Ok(()) + } +} + +fn register_buffered( + ctx: &mut PluginContext<'_>, + priority: i32, + state: Arc, +) -> Result<(), String> { + let user_data = Box::into_raw(Box::new(state)).cast::(); + let status = unsafe { + ctx.register_async_middleware_raw( + NemoRelayNativeAsyncMiddlewareKind::LlmExecutionIntercept, + "switchyard.run_stream.buffered", + priority, + false, + buffered_callback, + user_data, + Some(free_callback_state), + ) + }; + if status == NemoRelayStatus::Ok { + Ok(()) + } else { + Err(format!( + "failed to register Switchyard buffered execution: {status:?}" + )) + } +} + +fn register_stream( + ctx: &mut PluginContext<'_>, + priority: i32, + state: Arc, +) -> Result<(), String> { + let user_data = Box::into_raw(Box::new(state)).cast::(); + let status = unsafe { + ctx.register_async_stream_middleware_raw( + "switchyard.run_stream.streaming", + priority, + stream_callback, + user_data, + Some(free_callback_state), + ) + }; + if status == NemoRelayStatus::Ok { + Ok(()) + } else { + Err(format!( + "failed to register Switchyard streaming execution: {status:?}" + )) + } +} + +fn parse_config(plugin_config: &Map) -> Result { + match plugin_config.get("version").and_then(Json::as_u64) { + Some(2) => {} + Some(version) => { + return Err(format!( + "unsupported Switchyard config version {version}; version 1 used switchyard-server; migrate to version = 2" + )); + } + None => { + return Err("invalid Switchyard configuration: version must be the integer 2".into()); + } + } + serde_json::from_value(Json::Object(plugin_config.clone())) + .map_err(|error| format!("invalid Switchyard configuration: {error}")) +} + +fn emit_marks(parent: Option<&ffi::ParentScope>, marks: Vec) { + let Some(parent) = parent else { + return; + }; + for mark in marks { + if let Err(error) = parent.emit_mark(&mark.name, &mark.data, &mark.metadata) { + eprintln!( + "Switchyard could not emit routing mark {:?}: {error}", + mark.name + ); + } + } +} + +async fn execute_managed_stream( + state: &CallbackState, + output: usize, + inbound: switchyard_protocol::WireFormat, + request: switchyard_protocol::Request, + parent: Option<&ffi::ParentScope>, +) -> Result<(), String> { + let (sender, receiver) = async_channel::bounded(32); + let runtime = Arc::clone(&state.runtime); + let execution = async move { runtime.execute_stream(inbound, request, &sender).await }; + let forwarding = async { + while let Ok(message) = receiver.recv().await { + match message { + StreamMessage::Mark(mark) => emit_marks(parent, vec![mark]), + StreamMessage::Event(event) => { + ffi::push_stream(&state.host, output, &event).await? + } + } + } + Ok(()) + }; + tokio::try_join!(execution, forwarding)?; + Ok(()) +} + +unsafe extern "C" fn free_callback_state(user_data: *mut c_void) { + if !user_data.is_null() { + unsafe { drop(Box::from_raw(user_data.cast::>())) }; + } +} + +unsafe extern "C" fn buffered_callback( + user_data: *mut c_void, + invocation_json: *const NemoRelayNativeString, + next: *const NemoRelayNativeAsyncNext, + completion: *const NemoRelayNativeAsyncCompletion, +) -> u32 { + if user_data.is_null() || completion.is_null() || next.is_null() { + return NemoRelayNativeAsyncCallbackState::Complete as u32; + } + let state = unsafe { &*user_data.cast::>() }.clone(); + let invocation = ffi::read_json(&state.host.v1, invocation_json).and_then(|value| { + serde_json::from_value::(value).map_err(|error| error.to_string()) + }); + let next = next as usize; + let completion = completion as usize; + let invocation = match invocation { + Ok(invocation) => invocation, + Err(error) => { + let _ = ffi::reject_completion( + &state.host, + completion as *const NemoRelayNativeAsyncCompletion, + &format!("invalid Relay LLM invocation: {error}"), + ); + unsafe { + ffi::release_next(&state.host, next as *const NemoRelayNativeAsyncNext); + ffi::release_completion( + &state.host, + completion as *const NemoRelayNativeAsyncCompletion, + ); + } + return NemoRelayNativeAsyncCallbackState::Pending as u32; + } + }; + let Some(inbound) = state.runtime.managed_protocol(&invocation.name) else { + if let Err(error) = + ffi::invoke_next_buffered(&state.host, next, completion, &invocation.request) + { + let _ = ffi::reject_completion( + &state.host, + completion as *const NemoRelayNativeAsyncCompletion, + &error, + ); + } + unsafe { + ffi::release_next(&state.host, next as *const NemoRelayNativeAsyncNext); + ffi::release_completion( + &state.host, + completion as *const NemoRelayNativeAsyncCompletion, + ); + } + return NemoRelayNativeAsyncCallbackState::Pending as u32; + }; + let request = match state + .runtime + .decode_request(inbound, &invocation.request, false) + { + Ok(request) => request, + Err(error) => { + let _ = ffi::reject_completion( + &state.host, + completion as *const NemoRelayNativeAsyncCompletion, + &error, + ); + unsafe { + ffi::release_next(&state.host, next as *const NemoRelayNativeAsyncNext); + ffi::release_completion( + &state.host, + completion as *const NemoRelayNativeAsyncCompletion, + ); + } + return NemoRelayNativeAsyncCallbackState::Pending as u32; + } + }; + let parent = ffi::ParentScope::capture(&state.host.v1); + let task_state = Arc::clone(&state); + state.executor.spawn(async move { + let execution = AssertUnwindSafe(async { + let mut marks = Vec::new(); + let result = task_state + .runtime + .execute_buffered(inbound, request, &mut marks) + .await; + emit_marks(parent.as_ref(), marks); + result + }) + .catch_unwind(); + tokio::pin!(execution); + let result = tokio::select! { + biased; + () = ffi::wait_for_completion_cancellation(&task_state.host, completion) => None, + result = &mut execution => Some( + result.unwrap_or_else(|_| Err("Switchyard buffered execution panicked".into())) + ), + }; + + let completion_ptr = completion as *const NemoRelayNativeAsyncCompletion; + if let Some(result) = result { + match result { + Ok(response) => { + let _ = ffi::resolve_completion(&task_state.host, completion_ptr, &response); + } + Err(error) => { + let _ = ffi::reject_completion(&task_state.host, completion_ptr, &error); + } + } + } + unsafe { + ffi::release_next(&task_state.host, next as *const NemoRelayNativeAsyncNext); + ffi::release_completion(&task_state.host, completion_ptr); + } + }); + NemoRelayNativeAsyncCallbackState::Pending as u32 +} + +unsafe extern "C" fn stream_callback( + user_data: *mut c_void, + invocation_json: *const NemoRelayNativeString, + next: *const NemoRelayNativeAsyncNext, + output: *const NemoRelayNativeAsyncStream, +) -> u32 { + if user_data.is_null() || output.is_null() || next.is_null() { + return NemoRelayNativeAsyncCallbackState::Complete as u32; + } + let state = unsafe { &*user_data.cast::>() }.clone(); + let invocation = ffi::read_json(&state.host.v1, invocation_json).and_then(|value| { + serde_json::from_value::(value).map_err(|error| error.to_string()) + }); + let managed_protocol = invocation + .as_ref() + .ok() + .and_then(|invocation| state.runtime.managed_protocol(&invocation.name)); + let parent = managed_protocol.and_then(|_| ffi::ParentScope::capture(&state.host.v1)); + let next = next as usize; + let output = output as usize; + let task_state = Arc::clone(&state); + state.executor.spawn(async move { + let execution = AssertUnwindSafe(async { + match invocation { + Ok(invocation) => { + if let Some(inbound) = managed_protocol { + match task_state + .runtime + .decode_request(inbound, &invocation.request, true) + { + Ok(request) => { + execute_managed_stream( + &task_state, + output, + inbound, + request, + parent.as_ref(), + ) + .await + } + Err(error) => Err(error), + } + } else { + ffi::invoke_next_stream(&task_state.host, next, output, &invocation.request) + .await + } + } + Err(error) => Err(format!("invalid Relay LLM stream invocation: {error}")), + } + }) + .catch_unwind(); + tokio::pin!(execution); + let result = tokio::select! { + biased; + () = ffi::wait_for_stream_cancellation(&task_state.host, output) => None, + result = &mut execution => Some( + result.unwrap_or_else(|_| Err("Switchyard streaming execution panicked".into())) + ), + }; + + match result { + Some(Ok(())) => { + let _ = ffi::finish_stream( + &task_state.host, + output as *const NemoRelayNativeAsyncStream, + ); + } + Some(Err(error)) => { + let _ = ffi::reject_stream(&task_state.host, output, &error).await; + } + None => {} + } + unsafe { + ffi::release_next(&task_state.host, next as *const NemoRelayNativeAsyncNext); + ffi::release_stream( + &task_state.host, + output as *const NemoRelayNativeAsyncStream, + ); + } + }); + NemoRelayNativeAsyncCallbackState::Pending as u32 +} + +nemo_relay_plugin::nemo_relay_plugin!(nemo_relay_register_plugin, SwitchyardPlugin::default); + +#[cfg(test)] +mod tests { + use serde_json::json; + + use super::*; + + #[test] + fn version_one_service_config_gets_a_migration_error_before_v2_deserialization() { + let value = json!({ + "version": 1, + "service_url": "http://127.0.0.1:8080", + "health_endpoint": "/healthz" + }); + let plugin_config = value.as_object().unwrap(); + + let error = parse_config(plugin_config) + .err() + .expect("version one must be rejected"); + assert!(error.contains("version 1 used switchyard-server")); + assert!(error.contains("migrate to version = 2")); + assert!(!error.contains("unknown field")); + } + + #[test] + fn version_must_be_an_integer() { + let value = json!({"version": "2"}); + let plugin_config = value.as_object().unwrap(); + + let error = parse_config(plugin_config) + .err() + .expect("non-integer versions must be rejected"); + assert_eq!( + error, + "invalid Switchyard configuration: version must be the integer 2" + ); + } +} diff --git a/crates/switchyard-nemo-relay-plugin/src/runtime.rs b/crates/switchyard-nemo-relay-plugin/src/runtime.rs new file mode 100644 index 000000000..9615e91d8 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/runtime.rs @@ -0,0 +1,675 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +use std::collections::{BTreeMap, HashMap}; +use std::sync::{Arc, Mutex}; +use std::time::Duration; + +use futures_util::{StreamExt, stream}; +use nemo_relay_plugin::{Json, LlmRequest as RelayRequest}; +use serde_json::{Map, json}; +use switchyard_libsy::{Algorithm, LibsyError}; +use switchyard_llm_client::{ClientRouter, LlmCallObservation, RunObservation, RunObserver, run}; +use switchyard_protocol::{ + Decision, LlmClientError, LlmResponse, Metadata, Request, Response, WireFormat, +}; +use switchyard_translation::{TranslationEngine, encode_stream}; + +use crate::config::{PreparedTargetBinding, SwitchyardConfig, protocol_from_call}; +use crate::translation; + +const INITIAL_RETRY_BACKOFF: Duration = Duration::from_millis(250); +const MAX_RETRY_BACKOFF: Duration = Duration::from_secs(2); + +#[derive(Debug)] +pub(crate) struct RoutingMark { + pub(crate) name: String, + pub(crate) data: Json, + pub(crate) metadata: Json, +} + +#[derive(Debug)] +pub(crate) enum StreamMessage { + Mark(RoutingMark), + Event(Json), +} + +pub(crate) struct SwitchyardRuntime { + max_retries: u32, + algorithm: Arc, + targets: BTreeMap, + default_targets: BTreeMap, + translation: TranslationEngine, +} + +impl SwitchyardRuntime { + pub(crate) fn new(config: SwitchyardConfig) -> Result { + let prepared = config.prepare()?; + Ok(Self { + max_retries: prepared.max_retries, + algorithm: prepared.algorithm, + targets: prepared.targets, + default_targets: prepared.default_targets, + translation: TranslationEngine::default(), + }) + } + + pub(crate) fn managed_protocol(&self, name: &str) -> Option { + protocol_from_call(name).filter(|protocol| self.default_targets.contains_key(protocol)) + } + + pub(crate) fn decode_request( + &self, + inbound: WireFormat, + request: &RelayRequest, + streaming: bool, + ) -> Result { + let mut llm_request = translation::decode_request(&self.translation, inbound, request)?; + llm_request.stream = streaming; + let headers = string_headers(&request.headers); + let mut metadata = Metadata::from_headers(&headers); + let relay_gateway_placeholder = !headers.contains_key("x-switchyard-session-id") + && headers + .get("x-nemo-relay-source") + .and_then(|value| value.to_str().ok()) + == Some("gateway") + && metadata.session_id.as_deref() == Some("gateway-gateway"); + if relay_gateway_placeholder { + metadata.session_id = None; + } + // Keep identity/routing metadata, but target clients deliberately clear + // these caller headers before HTTP dispatch. + metadata.http_headers = Some(headers); + metadata.wire_format = Some(inbound); + Ok(Request { + llm_request, + raw_request: Some(request.content.clone()), + metadata: Some(metadata), + }) + } + + pub(crate) async fn execute_buffered( + &self, + inbound: WireFormat, + request: Request, + marks: &mut Vec, + ) -> Result { + let metadata = identity_metadata(request.metadata.as_ref()); + let max_attempts = self.max_retries + 1; + let mut attempt = 1; + loop { + self.mark( + marks, + "switchyard.routing.requested", + json!({"algorithm": self.algorithm.name(), "attempt": attempt}), + &metadata, + ); + let result = self + .drive(request.clone(), attempt, marks, &metadata) + .await + .and_then(|response| { + finalize_buffered_response(&self.translation, inbound, response) + .map_err(|source| LibsyError::client_call("return_to_agent", source)) + }); + match result { + Ok(response) => return Ok(response), + Err(failure) if libsy_error_retryable(&failure) && attempt < max_attempts => { + self.mark( + marks, + "switchyard.routing.retry", + failure_mark_data(attempt, &failure), + &metadata, + ); + sleep_before_retry(attempt).await; + attempt += 1; + } + Err(failure) => { + self.mark( + marks, + "switchyard.routing.error", + failure_mark_data(attempt, &failure), + &metadata, + ); + let response = self + .fallback_response(inbound, request, marks, &metadata) + .await?; + return finalize_buffered_response(&self.translation, inbound, response) + .map_err(|error| { + public_response_failure("trusted fallback response", &error) + }); + } + } + } + } + + pub(crate) async fn execute_stream( + &self, + inbound: WireFormat, + request: Request, + output: &async_channel::Sender, + ) -> Result<(), String> { + let metadata = identity_metadata(request.metadata.as_ref()); + let max_attempts = self.max_retries + 1; + let mut attempt = 1; + let mut marks = Vec::new(); + 'attempts: loop { + self.mark( + &mut marks, + "switchyard.routing.requested", + json!({"algorithm": self.algorithm.name(), "attempt": attempt}), + &metadata, + ); + let (response, mut fallback_used) = match self + .drive(request.clone(), attempt, &mut marks, &metadata) + .await + { + Ok(response) => (response, false), + Err(failure) if libsy_error_retryable(&failure) && attempt < max_attempts => { + self.mark( + &mut marks, + "switchyard.routing.retry", + failure_mark_data(attempt, &failure), + &metadata, + ); + send_marks(output, &mut marks).await?; + sleep_before_retry(attempt).await; + attempt += 1; + continue; + } + Err(failure) => { + self.mark( + &mut marks, + "switchyard.routing.error", + failure_mark_data(attempt, &failure), + &metadata, + ); + let fallback = self + .fallback_response(inbound, request.clone(), &mut marks, &metadata) + .await; + send_marks(output, &mut marks).await?; + (fallback?, true) + } + }; + send_marks(output, &mut marks).await?; + + let mut events = match returned_events(response, inbound).await { + Ok(events) => events, + Err(failure) + if !fallback_used + && libsy_error_retryable(&failure) + && attempt < max_attempts => + { + self.mark( + &mut marks, + "switchyard.routing.retry", + failure_mark_data(attempt, &failure), + &metadata, + ); + send_marks(output, &mut marks).await?; + sleep_before_retry(attempt).await; + attempt += 1; + continue; + } + Err(failure) if !fallback_used => { + self.mark( + &mut marks, + "switchyard.routing.error", + failure_mark_data(attempt, &failure), + &metadata, + ); + fallback_used = true; + let fallback = self + .fallback_response(inbound, request.clone(), &mut marks, &metadata) + .await; + send_marks(output, &mut marks).await?; + let fallback = fallback?; + returned_events(fallback, inbound) + .await + .map_err(|error| public_libsy_failure("trusted fallback stream", &error))? + } + Err(failure) => { + return Err(public_libsy_failure("trusted fallback stream", &failure)); + } + }; + + let mut committed = false; + while let Some(item) = events.next().await { + match item { + Ok(event) => { + send_event(output, event).await?; + committed = true; + } + Err(failure) + if !fallback_used + && !committed + && libsy_error_retryable(&failure) + && attempt < max_attempts => + { + self.mark( + &mut marks, + "switchyard.routing.retry", + failure_mark_data(attempt, &failure), + &metadata, + ); + send_marks(output, &mut marks).await?; + sleep_before_retry(attempt).await; + attempt += 1; + continue 'attempts; + } + Err(failure) if !fallback_used && !committed => { + self.mark( + &mut marks, + "switchyard.routing.error", + failure_mark_data(attempt, &failure), + &metadata, + ); + let fallback = self + .fallback_response(inbound, request.clone(), &mut marks, &metadata) + .await; + send_marks(output, &mut marks).await?; + let fallback = fallback?; + let mut fallback = + returned_events(fallback, inbound).await.map_err(|error| { + public_libsy_failure("trusted fallback stream", &error) + })?; + while let Some(item) = fallback.next().await { + let event = item.map_err(|error| { + public_libsy_failure("trusted fallback stream", &error) + })?; + send_event(output, event).await?; + } + return Ok(()); + } + Err(failure) if !committed => { + return Err(public_libsy_failure("trusted fallback stream", &failure)); + } + Err(failure) => { + self.mark( + &mut marks, + "switchyard.routing.error", + failure_mark_data(attempt, &failure), + &metadata, + ); + send_marks(output, &mut marks).await?; + return Err(public_libsy_failure( + "Switchyard stream failed after response commitment", + &failure, + )); + } + } + } + if committed { + return Ok(()); + } + return Err("Switchyard response stream produced no caller events".into()); + } + } + + async fn drive( + &self, + request: Request, + attempt: u32, + marks: &mut Vec, + mark_metadata: &Json, + ) -> Result { + let observations = Arc::new(Mutex::new(Vec::new())); + let observed_calls = observations.clone(); + let observer: RunObserver = Arc::new(move |observation| { + if let RunObservation::LlmCall(call) = observation { + observed_calls + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .push(call); + } + }); + let clients = ClientRouter::new( + self.targets + .iter() + .map(|(name, target)| (name.clone(), target.client.clone())) + .collect::>(), + ); + match run(self.algorithm.clone(), clients, request, Some(observer)).await { + Ok((decisions, response)) => { + for decision in decisions { + self.emit_decision(marks, &decision, attempt, mark_metadata); + } + self.emit_routing_llm_calls( + marks, + take_observed_calls(&observations), + attempt, + mark_metadata, + true, + ); + Ok(response) + } + Err(error) => { + self.emit_routing_llm_calls( + marks, + take_observed_calls(&observations), + attempt, + mark_metadata, + false, + ); + Err(error) + } + } + } + + async fn fallback_response( + &self, + inbound: WireFormat, + request: Request, + marks: &mut Vec, + metadata: &Json, + ) -> Result { + let target_name = self.default_target(inbound)?; + let target = self.target(target_name)?; + self.mark( + marks, + "switchyard.routing.fallback", + json!({"selected_target": target_name}), + metadata, + ); + let decision = Decision::new(target_name, Some("trusted fallback target".into()), true); + target + .client + .call(request, decision) + .await + .map_err(|error| public_client_failure("trusted fallback", &error)) + } + + fn target(&self, name: &str) -> Result<&PreparedTargetBinding, String> { + self.targets + .get(name) + .ok_or_else(|| format!("libsy selected unknown target {name:?}")) + } + + fn default_target(&self, protocol: WireFormat) -> Result<&str, String> { + self.default_targets + .get(&protocol) + .map(String::as_str) + .ok_or_else(|| format!("managed protocol {protocol} has no default target")) + } + + fn mark(&self, marks: &mut Vec, name: &str, data: Json, metadata: &Json) { + marks.push(RoutingMark { + name: name.to_string(), + data, + metadata: metadata.clone(), + }); + } + + fn emit_decision( + &self, + marks: &mut Vec, + decision: &Decision, + attempt: u32, + metadata: &Json, + ) { + self.mark( + marks, + "switchyard.routing.decision", + json!({ + "algorithm": self.algorithm.name(), + "attempt": attempt, + "selected_target": decision.selected_model_id(), + "reasoning": decision.reasoning(), + "is_answer_call": decision.is_answer_call(), + }), + metadata, + ); + } + + fn emit_routing_llm_calls( + &self, + marks: &mut Vec, + mut calls: Vec, + attempt: u32, + metadata: &Json, + successful_run: bool, + ) { + // The last successful routed call produced the response represented by Relay's + // outer LLM lifecycle event. Keep it out of these marks so consumers can add + // routing overhead without counting the serving call twice. Earlier routed calls + // are discarded candidates (for example, escalation's weak draft). + if successful_run + && let Some(position) = calls + .iter() + .rposition(|call| call.is_answer_call && call.is_success) + { + calls.remove(position); + } + + for (index, call) in calls.into_iter().enumerate() { + self.mark( + marks, + "switchyard.routing.llm_call", + json!({ + "algorithm": self.algorithm.name(), + "attempt": attempt, + "call_index": index + 1, + "selected_target": call.selected_model, + "call_role": if call.is_answer_call { "candidate" } else { "judge" }, + "outcome": if call.is_success { "ok" } else { "error" }, + "latency_ms": call.duration.as_secs_f64() * 1_000.0, + "usage": call.usage, + "contributes_to_routing_overhead": true, + }), + metadata, + ); + } + } +} + +fn take_observed_calls(observations: &Mutex>) -> Vec { + std::mem::take( + &mut *observations + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()), + ) +} + +async fn send_marks( + output: &async_channel::Sender, + marks: &mut Vec, +) -> Result<(), String> { + for mark in marks.drain(..) { + output + .send(StreamMessage::Mark(mark)) + .await + .map_err(|_| "Relay cancelled the Switchyard response stream".to_string())?; + } + Ok(()) +} + +async fn send_event( + output: &async_channel::Sender, + event: Json, +) -> Result<(), String> { + output + .send(StreamMessage::Event(event)) + .await + .map_err(|_| "Relay cancelled the Switchyard response stream".to_string()) +} + +type ReturnedEventStream = + std::pin::Pin> + Send>>; + +fn finalize_buffered_response( + translation_engine: &TranslationEngine, + inbound: WireFormat, + response: Response, +) -> Result { + let LlmResponse::Agg(response) = response.llm_response else { + return Err(LlmClientError::InvalidResponse { + source: Box::new(std::io::Error::other( + "libsy returned a stream for a buffered request", + )), + }); + }; + translation::encode_response(translation_engine, inbound, &response) + .map_err(LlmClientError::ResponseTranslation) +} + +async fn returned_events( + response: Response, + inbound: WireFormat, +) -> Result { + let chunks = match response.llm_response { + LlmResponse::Agg(response) => response.into_stream(), + LlmResponse::Stream(mut chunks) => { + let Some(first) = chunks.next().await else { + return Err(LibsyError::client_call( + "return_to_agent", + LlmClientError::InvalidResponse { + source: Box::new(std::io::Error::new( + std::io::ErrorKind::UnexpectedEof, + "provider returned an empty stream", + )), + }, + )); + }; + Box::pin(stream::once(async move { first }).chain(chunks)) + } + }; + let events = encode_stream(chunks, inbound, None) + .map_err(|error| LibsyError::client_call("return_to_agent", error))?; + Ok(Box::pin(events.map(|item| { + item.map_err(|source| match source.downcast::() { + Ok(source) => LibsyError::client_call("return_to_agent", *source), + Err(source) => LibsyError::client_call( + "return_to_agent", + LlmClientError::ResponseTranslation(source.to_string()), + ), + }) + }))) +} + +fn libsy_error_retryable(error: &LibsyError) -> bool { + let LibsyError::ClientCall { source, .. } = error else { + return false; + }; + match source { + LlmClientError::UpstreamHttp { status, .. } => { + matches!(*status, 408 | 425 | 429 | 500 | 502 | 503 | 504) + } + LlmClientError::Transport { .. } | LlmClientError::Timeout { .. } => true, + _ => false, + } +} + +fn retry_backoff(attempt: u32) -> Duration { + let exponent = attempt.saturating_sub(1).min(3); + INITIAL_RETRY_BACKOFF + .saturating_mul(1_u32 << exponent) + .min(MAX_RETRY_BACKOFF) +} + +async fn sleep_before_retry(attempt: u32) { + tokio::time::sleep(retry_backoff(attempt)).await; +} + +fn failure_mark_data(attempt: u32, failure: &LibsyError) -> Json { + let mut data = Map::from_iter([ + ("attempt".into(), Json::from(attempt)), + ( + "retryable".into(), + Json::from(libsy_error_retryable(failure)), + ), + ]); + match failure { + LibsyError::ClientCall { + source: LlmClientError::UpstreamHttp { status, .. }, + .. + } => { + data.insert("failure_kind".into(), Json::from("http")); + data.insert("http_status".into(), Json::from(*status)); + } + LibsyError::ClientCall { source, .. } => { + data.insert("failure_kind".into(), Json::from("non_http")); + data.insert( + "non_http_kind".into(), + Json::from(client_error_label(source)), + ); + } + _ => { + data.insert("failure_kind".into(), Json::from("algorithm")); + } + } + Json::Object(data) +} + +fn client_error_label(error: &LlmClientError) -> &'static str { + match error { + LlmClientError::InvalidRequest { .. } => "invalid_request", + LlmClientError::RequestTranslation(_) => "request_translation", + LlmClientError::RequestEncoding(_) => "request_encoding", + LlmClientError::ResponseTranslation(_) => "response_translation", + LlmClientError::Configuration { .. } => "configuration", + LlmClientError::Transport { .. } => "transport", + LlmClientError::Timeout { .. } => "timeout", + LlmClientError::ContextWindowExceeded { .. } => "context_window_exceeded", + LlmClientError::UpstreamHttp { .. } => "http", + LlmClientError::InvalidResponse { .. } => "invalid_response", + LlmClientError::Ffi { .. } => "ffi", + LlmClientError::General(_) => "general", + _ => "unknown", + } +} + +fn public_libsy_failure(prefix: &str, error: &LibsyError) -> String { + match error { + LibsyError::ClientCall { source, .. } => public_client_failure(prefix, source), + _ => format!("{prefix}: Switchyard algorithm failure"), + } +} + +fn public_response_failure(prefix: &str, error: &LlmClientError) -> String { + match error { + LlmClientError::InvalidResponse { .. } => format!("{prefix}: invalid response"), + LlmClientError::ResponseTranslation(_) => { + format!("{prefix}: response translation failure") + } + _ => format!("{prefix}: response finalization failure"), + } +} + +fn public_client_failure(prefix: &str, error: &LlmClientError) -> String { + match error { + LlmClientError::UpstreamHttp { status, .. } => { + format!("{prefix}: provider returned HTTP {status}") + } + _ => format!("{prefix}: provider {} failure", client_error_label(error)), + } +} + +fn string_headers(headers: &Map) -> http::HeaderMap { + let mut parsed = http::HeaderMap::with_capacity(headers.len()); + for (name, value) in headers { + let Some(value) = value.as_str() else { + continue; + }; + let (Ok(name), Ok(value)) = ( + http::HeaderName::from_bytes(name.as_bytes()), + http::HeaderValue::from_str(value), + ) else { + continue; + }; + parsed.insert(name, value); + } + parsed +} + +fn identity_metadata(metadata: Option<&Metadata>) -> Json { + json!({ + "session_id": metadata.and_then(|value| value.session_id.as_deref()), + "agent_id": metadata.and_then(|value| value.agent_id.as_deref()), + "parent_agent_id": metadata.and_then(|value| value.parent_agent_id.as_deref()), + "task_id": metadata.and_then(|value| value.task_id.as_deref()), + "turn_id": metadata.and_then(|value| value.turn_id.as_deref()), + "correlation_id": metadata.and_then(|value| value.correlation_id.as_deref()), + }) +} + +#[cfg(test)] +mod tests; diff --git a/crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs b/crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs new file mode 100644 index 000000000..54d383278 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs @@ -0,0 +1,874 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +use std::sync::atomic::{AtomicUsize, Ordering}; + +use switchyard_libsy::{ + ClassifierContractConfig, EscalationJudgeConfig, LlmClassifierConfig, LlmFallback, LlmTarget, + LlmTaskClassifier, Passthrough, PickerMode, StageRouter, StageRouterConfig, + TaskClassifierConfig, +}; +use switchyard_protocol::{LlmResponseStream, RoutedLlmClient, Usage, text_request, text_response}; + +use super::*; + +enum ScriptedBehavior { + Text(&'static str), + EmptyBuffered, + EmptyStream, + FailingStream, + TransportFailure(&'static str), +} + +struct ScriptedClient { + behavior: ScriptedBehavior, + calls: AtomicUsize, +} + +fn scripted(behavior: ScriptedBehavior) -> Arc { + Arc::new(ScriptedClient { + behavior, + calls: AtomicUsize::new(0), + }) +} + +#[async_trait::async_trait] +impl RoutedLlmClient for ScriptedClient { + async fn call( + &self, + request: Request, + _decision: Decision, + ) -> Result { + self.calls.fetch_add(1, Ordering::Relaxed); + match self.behavior { + ScriptedBehavior::Text(text) => { + let mut response = text_response(None, text); + response.usage = Usage { + input_tokens: Some(11), + output_tokens: Some(7), + total_tokens: Some(18), + ..Usage::default() + }; + Ok(Response { + llm_response: LlmResponse::Agg(response), + metadata: request.metadata, + }) + } + ScriptedBehavior::EmptyBuffered => Ok(Response { + llm_response: LlmResponse::Agg(Default::default()), + metadata: None, + }), + ScriptedBehavior::EmptyStream => Ok(Response { + llm_response: LlmResponse::Stream(Box::pin(stream::empty())), + metadata: None, + }), + ScriptedBehavior::FailingStream => { + let stream: LlmResponseStream = Box::pin(stream::once(async { + Err(LlmClientError::Transport { + source: Box::new(std::io::Error::other("fallback stream failed")), + }) + })); + Ok(Response { + llm_response: LlmResponse::Stream(stream), + metadata: None, + }) + } + ScriptedBehavior::TransportFailure(message) => Err(LlmClientError::Transport { + source: Box::new(std::io::Error::other(message)), + }), + } + } +} + +fn fixed_target(name: &str) -> LlmTarget { + LlmTarget { + semantic_name: name.to_string(), + } +} + +fn runtime_with_algorithm( + algorithm: Arc, + fallback: Arc, + protocol: WireFormat, +) -> SwitchyardRuntime { + runtime_with_algorithm_clients(algorithm, fallback, protocol, Vec::new()) +} + +fn runtime_with_algorithm_clients( + algorithm: Arc, + fallback: Arc, + protocol: WireFormat, + clients: Vec<(&str, Arc)>, +) -> SwitchyardRuntime { + let mut targets = BTreeMap::from([( + "fallback".into(), + PreparedTargetBinding { + client: fallback as Arc, + }, + )]); + for (name, client) in clients { + targets.insert( + name.to_string(), + PreparedTargetBinding { + client: client as Arc, + }, + ); + } + SwitchyardRuntime { + max_retries: 0, + algorithm, + targets, + default_targets: BTreeMap::from([(protocol, "fallback".into())]), + translation: TranslationEngine::default(), + } +} + +fn request_with_session(protocol: WireFormat, session: Option<&str>) -> Request { + Request { + llm_request: text_request(Some("auto".into()), "fix the build"), + raw_request: None, + metadata: Some(Metadata { + wire_format: Some(protocol), + session_id: session.map(str::to_string), + ..Metadata::default() + }), + } +} + +fn stage_signal_relay_request(protocol: WireFormat) -> RelayRequest { + let content = match protocol { + WireFormat::OpenAiChat => json!({ + "model": "auto", + "messages": [ + {"role": "user", "content": "fix the build"}, + { + "role": "assistant", + "content": null, + "tool_calls": [{ + "id": "call-1", + "type": "function", + "function": { + "name": "bash", + "arguments": "{\"cmd\":\"cargo test\"}" + } + }] + }, + { + "role": "tool", + "tool_call_id": "call-1", + "content": "fatal runtime error: out of memory" + } + ] + }), + WireFormat::OpenAiResponses => json!({ + "model": "auto", + "input": [ + {"type": "message", "role": "user", "content": "fix the build"}, + { + "type": "function_call", + "call_id": "call-1", + "name": "bash", + "arguments": "{\"cmd\":\"cargo test\"}" + }, + { + "type": "function_call_output", + "call_id": "call-1", + "output": "fatal runtime error: out of memory" + } + ] + }), + WireFormat::AnthropicMessages => json!({ + "model": "auto", + "max_tokens": 128, + "messages": [ + {"role": "user", "content": "fix the build"}, + { + "role": "assistant", + "content": [{ + "type": "tool_use", + "id": "call-1", + "name": "bash", + "input": {"cmd": "cargo test"} + }] + }, + { + "role": "user", + "content": [{ + "type": "tool_result", + "tool_use_id": "call-1", + "content": "fatal runtime error: out of memory", + "is_error": true + }] + } + ] + }), + }; + + RelayRequest { + headers: Map::from_iter([( + "x-switchyard-session-id".into(), + json!(format!("stage-{}", protocol.as_str())), + )]), + content, + } +} + +#[test] +fn relay_gateway_placeholder_session_is_not_retained() { + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let runtime = runtime_with_algorithm( + Arc::new(Passthrough::new(LlmTarget { + semantic_name: "selected".into(), + })), + fallback, + WireFormat::OpenAiChat, + ); + let request = RelayRequest { + headers: Map::from_iter([ + ("x-nemo-relay-source".into(), json!("gateway")), + ("x-nemo-relay-session-id".into(), json!("gateway-gateway")), + ("x-dynamo-session-id".into(), json!("gateway-gateway")), + ]), + content: json!({ + "model": "router", + "messages": [{"role": "user", "content": "hello"}] + }), + }; + + let decoded = runtime + .decode_request(WireFormat::OpenAiChat, &request, false) + .unwrap(); + + assert_eq!(decoded.metadata.unwrap().session_id, None); +} + +#[test] +fn explicit_switchyard_session_overrides_relay_gateway_placeholder() { + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let runtime = runtime_with_algorithm( + Arc::new(Passthrough::new(LlmTarget { + semantic_name: "selected".into(), + })), + fallback, + WireFormat::OpenAiChat, + ); + let request = RelayRequest { + headers: Map::from_iter([ + ("x-switchyard-session-id".into(), json!("caller-session")), + ("x-nemo-relay-source".into(), json!("gateway")), + ("x-nemo-relay-session-id".into(), json!("gateway-gateway")), + ]), + content: json!({ + "model": "router", + "messages": [{"role": "user", "content": "hello"}] + }), + }; + + let decoded = runtime + .decode_request(WireFormat::OpenAiChat, &request, false) + .unwrap(); + + assert_eq!( + decoded.metadata.unwrap().session_id.as_deref(), + Some("caller-session") + ); +} + +#[tokio::test] +async fn buffered_finalization_failure_uses_fallback_once() { + let selected = scripted(ScriptedBehavior::EmptyStream); + let fallback = scripted(ScriptedBehavior::EmptyBuffered); + let runtime = SwitchyardRuntime { + max_retries: 1, + algorithm: Arc::new(Passthrough::new(LlmTarget { + semantic_name: "selected".into(), + })), + targets: BTreeMap::from([ + ( + "selected".into(), + PreparedTargetBinding { + client: selected.clone(), + }, + ), + ( + "fallback".into(), + PreparedTargetBinding { + client: fallback.clone(), + }, + ), + ]), + default_targets: BTreeMap::from([(WireFormat::OpenAiChat, "fallback".into())]), + translation: TranslationEngine::default(), + }; + let mut marks = Vec::new(); + + let response = runtime + .execute_buffered(WireFormat::OpenAiChat, Request::default(), &mut marks) + .await + .expect("the buffered fallback response should be encoded"); + + assert!(response.is_object()); + assert_eq!(selected.calls.load(Ordering::Relaxed), 1); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 1); + assert!( + !marks + .iter() + .any(|mark| mark.name == "switchyard.routing.retry") + ); + let error = marks + .iter() + .find(|mark| mark.name == "switchyard.routing.error") + .expect("finalization failure should emit an error mark"); + assert_eq!(error.data["retryable"], false); + assert_eq!(error.data["non_http_kind"], "invalid_response"); + assert_eq!( + marks + .iter() + .filter(|mark| mark.name == "switchyard.routing.fallback") + .count(), + 1 + ); +} + +#[tokio::test] +async fn returned_events_replays_preserved_openai_chat_without_duplicate_terminal() { + let content = json!({ + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "model": "gpt-4o", + "system_fingerprint": "fp_provider_specific", + "choices": [{ + "index": 0, + "delta": {"content": "Hi"}, + "finish_reason": null + }] + }); + let terminal = json!({ + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "model": "gpt-4o", + "choices": [{ + "index": 0, + "delta": {}, + "finish_reason": "stop" + }] + }); + let body = format!("data: {content}\n\ndata: {terminal}\n\ndata: [DONE]\n\n").into_bytes(); + let stream = switchyard_translation::decode_stream( + stream::once(async move { Ok::<_, LlmClientError>(body) }), + WireFormat::OpenAiChat, + ) + .expect("provider SSE should decode"); + let response = Response { + llm_response: LlmResponse::Stream(stream), + metadata: None, + }; + + let replayed = returned_events(response, WireFormat::OpenAiChat) + .await + .expect("return stream should encode") + .collect::>() + .await + .into_iter() + .collect::, _>>() + .expect("return stream should not fail"); + + assert_eq!(replayed, vec![content, terminal]); +} + +#[tokio::test] +async fn invalid_selected_stream_does_not_invoke_failing_fallback_twice() { + let selected = scripted(ScriptedBehavior::EmptyStream); + let fallback = scripted(ScriptedBehavior::FailingStream); + let runtime = SwitchyardRuntime { + max_retries: 0, + algorithm: Arc::new(Passthrough::new(LlmTarget { + semantic_name: "selected".into(), + })), + targets: BTreeMap::from([ + ( + "selected".into(), + PreparedTargetBinding { + client: selected.clone(), + }, + ), + ( + "fallback".into(), + PreparedTargetBinding { + client: fallback.clone(), + }, + ), + ]), + default_targets: BTreeMap::from([(WireFormat::OpenAiChat, "fallback".into())]), + translation: TranslationEngine::default(), + }; + let (output, _messages) = async_channel::bounded(32); + + let error = runtime + .execute_stream(WireFormat::OpenAiChat, Request::default(), &output) + .await + .expect_err("the failing fallback stream must fail the request"); + + assert_eq!(error, "trusted fallback stream: provider transport failure"); + assert_eq!(selected.calls.load(Ordering::Relaxed), 1); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 1); +} + +#[tokio::test] +async fn failing_fallback_call_flushes_error_and_fallback_marks() { + let selected = scripted(ScriptedBehavior::EmptyStream); + let fallback = scripted(ScriptedBehavior::TransportFailure("fallback call failed")); + let runtime = SwitchyardRuntime { + max_retries: 0, + algorithm: Arc::new(Passthrough::new(LlmTarget { + semantic_name: "selected".into(), + })), + targets: BTreeMap::from([ + ( + "selected".into(), + PreparedTargetBinding { + client: selected.clone(), + }, + ), + ( + "fallback".into(), + PreparedTargetBinding { + client: fallback.clone(), + }, + ), + ]), + default_targets: BTreeMap::from([(WireFormat::OpenAiChat, "fallback".into())]), + translation: TranslationEngine::default(), + }; + let (output, messages) = async_channel::bounded(32); + + let error = runtime + .execute_stream(WireFormat::OpenAiChat, Request::default(), &output) + .await + .expect_err("the failing fallback call must fail the request"); + + assert_eq!(error, "trusted fallback: provider transport failure"); + assert_eq!(selected.calls.load(Ordering::Relaxed), 1); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 1); + let mut terminal_marks = Vec::new(); + while let Ok(message) = messages.try_recv() { + if let StreamMessage::Mark(mark) = message + && matches!( + mark.name.as_str(), + "switchyard.routing.error" | "switchyard.routing.fallback" + ) + { + terminal_marks.push(mark.name); + } + } + assert_eq!( + terminal_marks, + ["switchyard.routing.error", "switchyard.routing.fallback"] + ); +} + +#[test] +fn retry_backoff_increases_exponentially_and_is_capped() { + assert_eq!(retry_backoff(1), Duration::from_millis(250)); + assert_eq!(retry_backoff(2), Duration::from_millis(500)); + assert_eq!(retry_backoff(3), Duration::from_secs(1)); + assert_eq!(retry_backoff(4), Duration::from_secs(2)); + assert_eq!(retry_backoff(u32::MAX), Duration::from_secs(2)); +} + +#[tokio::test] +async fn capability_classifier_emits_judge_usage_without_serving_usage() { + let weak = scripted(ScriptedBehavior::Text("weak answer")); + let strong = scripted(ScriptedBehavior::Text("strong answer")); + let judge = scripted(ScriptedBehavior::Text( + r#"{"crux":"bounded","primary_rule":"SUP-1","capability_boundary":"supported","p_solve":0.9}"#, + )); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let algorithm = LlmTaskClassifier::new(LlmClassifierConfig::Capability { + judge_target: fixed_target("judge"), + efficient_target: fixed_target("weak"), + capable_target: fixed_target("strong"), + config: TaskClassifierConfig { + base_threshold: 0.5, + ..TaskClassifierConfig::default() + }, + }) + .unwrap(); + let runtime = runtime_with_algorithm_clients( + Arc::new(algorithm), + fallback, + WireFormat::OpenAiChat, + vec![ + ("weak", weak.clone()), + ("strong", strong.clone()), + ("judge", judge.clone()), + ], + ); + let mut marks = Vec::new(); + + runtime + .execute_buffered( + WireFormat::OpenAiChat, + request_with_session(WireFormat::OpenAiChat, Some("capability")), + &mut marks, + ) + .await + .unwrap(); + + assert_eq!(judge.calls.load(Ordering::Relaxed), 1); + assert_eq!(weak.calls.load(Ordering::Relaxed), 1); + assert_eq!(strong.calls.load(Ordering::Relaxed), 0); + let routing_calls = marks + .iter() + .filter(|mark| mark.name == "switchyard.routing.llm_call") + .collect::>(); + assert_eq!(routing_calls.len(), 1); + assert_eq!(routing_calls[0].data["selected_target"], "judge"); + assert_eq!(routing_calls[0].data["usage"]["total_tokens"], 18); +} + +#[tokio::test] +async fn escalation_buffers_weak_stream_then_latches_the_session_to_strong() { + let weak = scripted(ScriptedBehavior::Text("weak draft")); + let strong = scripted(ScriptedBehavior::Text("strong answer")); + let judge = scripted(ScriptedBehavior::Text( + r#"{"escalate":true,"reason":"stuck"}"#, + )); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let algorithm = LlmTaskClassifier::new(LlmClassifierConfig::Escalation { + judge_target: fixed_target("judge"), + efficient_target: fixed_target("weak"), + capable_target: fixed_target("strong"), + contract: ClassifierContractConfig::default(), + config: EscalationJudgeConfig { + confirmations: 1, + ..EscalationJudgeConfig::default() + }, + max_output_tokens: 128, + }) + .unwrap(); + let runtime = runtime_with_algorithm_clients( + Arc::new(algorithm), + fallback.clone(), + WireFormat::OpenAiChat, + vec![ + ("weak", weak.clone()), + ("strong", strong.clone()), + ("judge", judge.clone()), + ], + ); + + let mut first = request_with_session(WireFormat::OpenAiChat, Some("session-1")); + first.llm_request.stream = true; + let (output, messages) = async_channel::bounded(32); + runtime + .execute_stream(WireFormat::OpenAiChat, first, &output) + .await + .unwrap(); + let mut streamed = Vec::new(); + let mut routing_calls = Vec::new(); + while let Ok(message) = messages.try_recv() { + match message { + StreamMessage::Event(event) => streamed.push(event), + StreamMessage::Mark(mark) if mark.name == "switchyard.routing.llm_call" => { + routing_calls.push(mark.data) + } + StreamMessage::Mark(_) => {} + } + } + assert!(!streamed.is_empty()); + assert!( + streamed + .iter() + .any(|event| event.to_string().contains("strong answer")) + ); + assert_eq!(routing_calls.len(), 2); + assert_eq!(routing_calls[0]["selected_target"], "weak"); + assert_eq!(routing_calls[0]["call_role"], "candidate"); + assert_eq!(routing_calls[0]["usage"]["total_tokens"], 18); + assert_eq!(routing_calls[1]["selected_target"], "judge"); + assert_eq!(routing_calls[1]["call_role"], "judge"); + assert_eq!(routing_calls[1]["usage"]["total_tokens"], 18); + assert!( + routing_calls + .iter() + .all(|call| call["selected_target"] != "strong") + ); + + let mut marks = Vec::new(); + let response = runtime + .execute_buffered( + WireFormat::OpenAiChat, + request_with_session(WireFormat::OpenAiChat, Some("session-1")), + &mut marks, + ) + .await + .unwrap(); + assert!(response.to_string().contains("strong answer")); + assert_eq!(weak.calls.load(Ordering::Relaxed), 1); + assert_eq!(judge.calls.load(Ordering::Relaxed), 1); + assert_eq!(strong.calls.load(Ordering::Relaxed), 2); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 0); + assert!( + !marks + .iter() + .any(|mark| mark.name == "switchyard.routing.llm_call") + ); + assert!(marks.iter().any(|mark| { + mark.name == "switchyard.routing.decision" + && mark.data["selected_target"] == "strong" + && mark.data["is_answer_call"] == true + && mark.data["reasoning"].is_string() + && mark.metadata["session_id"] == "session-1" + })); +} + +#[tokio::test] +async fn escalation_judge_failure_falls_open_to_the_buffered_weak_response() { + let weak = scripted(ScriptedBehavior::Text("weak answer")); + let strong = scripted(ScriptedBehavior::Text("strong answer")); + let judge = scripted(ScriptedBehavior::TransportFailure("scripted failure")); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let algorithm = LlmTaskClassifier::new(LlmClassifierConfig::Escalation { + judge_target: fixed_target("judge"), + efficient_target: fixed_target("weak"), + capable_target: fixed_target("strong"), + contract: ClassifierContractConfig::default(), + config: EscalationJudgeConfig::default(), + max_output_tokens: 128, + }) + .unwrap(); + let runtime = runtime_with_algorithm_clients( + Arc::new(algorithm), + fallback.clone(), + WireFormat::OpenAiChat, + vec![ + ("weak", weak.clone()), + ("strong", strong.clone()), + ("judge", judge.clone()), + ], + ); + let mut marks = Vec::new(); + + let response = runtime + .execute_buffered( + WireFormat::OpenAiChat, + request_with_session(WireFormat::OpenAiChat, Some("session-1")), + &mut marks, + ) + .await + .unwrap(); + + assert!(response.to_string().contains("weak answer")); + assert_eq!(weak.calls.load(Ordering::Relaxed), 1); + assert_eq!(judge.calls.load(Ordering::Relaxed), 1); + assert_eq!(strong.calls.load(Ordering::Relaxed), 0); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 0); + let routing_calls = marks + .iter() + .filter(|mark| mark.name == "switchyard.routing.llm_call") + .collect::>(); + assert_eq!(routing_calls.len(), 1); + assert_eq!(routing_calls[0].data["selected_target"], "judge"); + assert_eq!(routing_calls[0].data["call_role"], "judge"); + assert_eq!(routing_calls[0].data["outcome"], "error"); + assert!(routing_calls[0].data["usage"].is_null()); +} + +#[tokio::test] +async fn escalation_without_session_identity_cannot_accumulate_confirmations() { + let weak = scripted(ScriptedBehavior::Text("weak answer")); + let strong = scripted(ScriptedBehavior::Text("strong answer")); + let judge = scripted(ScriptedBehavior::Text( + r#"{"escalate":true,"reason":"stuck"}"#, + )); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let algorithm = LlmTaskClassifier::new(LlmClassifierConfig::Escalation { + judge_target: fixed_target("judge"), + efficient_target: fixed_target("weak"), + capable_target: fixed_target("strong"), + contract: ClassifierContractConfig::default(), + config: EscalationJudgeConfig { + confirmations: 2, + ..EscalationJudgeConfig::default() + }, + max_output_tokens: 128, + }) + .unwrap(); + let runtime = runtime_with_algorithm_clients( + Arc::new(algorithm), + fallback.clone(), + WireFormat::OpenAiChat, + vec![ + ("weak", weak.clone()), + ("strong", strong.clone()), + ("judge", judge.clone()), + ], + ); + + for _ in 0..2 { + let mut marks = Vec::new(); + let response = runtime + .execute_buffered( + WireFormat::OpenAiChat, + request_with_session(WireFormat::OpenAiChat, None), + &mut marks, + ) + .await + .unwrap(); + assert!(response.to_string().contains("weak answer")); + } + assert_eq!(weak.calls.load(Ordering::Relaxed), 2); + assert_eq!(judge.calls.load(Ordering::Relaxed), 2); + assert_eq!(strong.calls.load(Ordering::Relaxed), 0); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 0); +} + +#[tokio::test] +async fn stage_router_uses_tool_signals_for_every_managed_protocol() { + for protocol in [ + WireFormat::OpenAiChat, + WireFormat::OpenAiResponses, + WireFormat::AnthropicMessages, + ] { + let capable = scripted(ScriptedBehavior::Text("capable answer")); + let efficient = scripted(ScriptedBehavior::Text("efficient answer")); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let algorithm = StageRouter::new( + fixed_target("strong"), + fixed_target("weak"), + StageRouterConfig::new(PickerMode::EfficientFirst, 0.5), + ) + .unwrap(); + let runtime = runtime_with_algorithm_clients( + Arc::new(algorithm), + fallback.clone(), + protocol, + vec![("strong", capable.clone()), ("weak", efficient.clone())], + ); + let mut marks = Vec::new(); + let relay_request = stage_signal_relay_request(protocol); + let request = runtime + .decode_request(protocol, &relay_request, false) + .unwrap(); + + let response = runtime + .execute_buffered(protocol, request, &mut marks) + .await + .unwrap(); + + assert!(response.to_string().contains("capable answer")); + assert_eq!(capable.calls.load(Ordering::Relaxed), 1); + assert_eq!(efficient.calls.load(Ordering::Relaxed), 0); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 0); + assert!(marks.iter().any(|mark| { + mark.name == "switchyard.routing.decision" + && mark.data["algorithm"] == "stage_router" + && mark.data["attempt"] == 1 + && mark.data["selected_target"] == "strong" + && mark.data["reasoning"].is_string() + && mark.data["is_answer_call"] == true + && mark.metadata["session_id"] == format!("stage-{}", protocol.as_str()) + })); + } +} + +#[tokio::test] +async fn stage_router_falls_open_to_each_picker_default_without_tool_history() { + for (picker, expected) in [ + (PickerMode::CapableFirst, "strong"), + (PickerMode::EfficientFirst, "weak"), + ] { + let capable = scripted(ScriptedBehavior::Text("strong")); + let efficient = scripted(ScriptedBehavior::Text("weak")); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let algorithm = StageRouter::new( + fixed_target("strong"), + fixed_target("weak"), + StageRouterConfig::new(picker, 0.5), + ) + .unwrap(); + let runtime = runtime_with_algorithm_clients( + Arc::new(algorithm), + fallback, + WireFormat::OpenAiChat, + vec![("strong", capable), ("weak", efficient)], + ); + let mut marks = Vec::new(); + + runtime + .execute_buffered( + WireFormat::OpenAiChat, + request_with_session(WireFormat::OpenAiChat, None), + &mut marks, + ) + .await + .unwrap(); + + assert!(marks.iter().any(|mark| { + mark.name == "switchyard.routing.decision" + && mark.data["selected_target"] == expected + && mark.data["reasoning"].is_string() + && mark.data["is_answer_call"] == true + })); + } +} + +#[tokio::test] +async fn stage_router_classifier_resolves_an_ambiguous_turn() { + let capable = scripted(ScriptedBehavior::Text("strong")); + let efficient = scripted(ScriptedBehavior::Text("weak")); + let judge = scripted(ScriptedBehavior::Text( + r#"{"crux":"bounded","primary_rule":"SUP-1","capability_boundary":"supported","p_solve":0.9}"#, + )); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let mut config = StageRouterConfig::new(PickerMode::CapableFirst, 0.5); + config.llm_fallback = Some(LlmFallback { + judge_target: fixed_target("judge"), + config: TaskClassifierConfig { + base_threshold: 0.5, + ..TaskClassifierConfig::default() + }, + }); + let algorithm = StageRouter::new(fixed_target("strong"), fixed_target("weak"), config).unwrap(); + let runtime = runtime_with_algorithm_clients( + Arc::new(algorithm), + fallback.clone(), + WireFormat::OpenAiChat, + vec![ + ("strong", capable.clone()), + ("weak", efficient.clone()), + ("judge", judge.clone()), + ], + ); + let mut marks = Vec::new(); + + runtime + .execute_buffered( + WireFormat::OpenAiChat, + request_with_session(WireFormat::OpenAiChat, Some("stage-classifier")), + &mut marks, + ) + .await + .unwrap(); + + assert_eq!(judge.calls.load(Ordering::Relaxed), 1); + assert_eq!(efficient.calls.load(Ordering::Relaxed), 1); + assert_eq!(capable.calls.load(Ordering::Relaxed), 0); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 0); + let routing_calls = marks + .iter() + .filter(|mark| mark.name == "switchyard.routing.llm_call") + .collect::>(); + assert_eq!(routing_calls.len(), 1); + assert_eq!(routing_calls[0].data["selected_target"], "judge"); + assert_eq!(routing_calls[0].data["call_role"], "judge"); + assert_eq!(routing_calls[0].data["outcome"], "ok"); + assert_eq!(routing_calls[0].data["usage"]["total_tokens"], 18); + assert!(marks.iter().any(|mark| { + mark.name == "switchyard.routing.decision" + && mark.data["selected_target"] == "weak" + && mark.data["reasoning"].is_string() + && mark.data["is_answer_call"] == true + })); +} diff --git a/crates/switchyard-nemo-relay-plugin/src/translation.rs b/crates/switchyard-nemo-relay-plugin/src/translation.rs new file mode 100644 index 000000000..6f6f965fe --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/translation.rs @@ -0,0 +1,116 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +use nemo_relay_plugin::LlmRequest as RelayRequest; +use serde_json::Value as Json; +use switchyard_protocol::{AggLlmResponse, LlmRequest, WireFormat}; +use switchyard_translation::{ + DeterministicIdPolicy, DiagnosticSeverity, LossyConversionPolicy, PreservationPolicy, + TargetCapabilities, TranslationDiagnostic, TranslationEngine, TranslationPolicy, + UnknownFieldPolicy, +}; + +pub(crate) fn decode_request( + engine: &TranslationEngine, + protocol: WireFormat, + request: &RelayRequest, +) -> Result { + let output = engine + .decode_request(protocol, &request.content, &policy()) + .map_err(error)?; + safe(&output.diagnostics)?; + Ok(output.request) +} + +pub(crate) fn validate_target_request( + engine: &TranslationEngine, + protocol: WireFormat, + request: &LlmRequest, +) -> Result<(), String> { + let output = engine + .encode_request(protocol, request, &request_policy(protocol)) + .map_err(error)?; + safe(&output.diagnostics) +} + +pub(crate) fn encode_response( + engine: &TranslationEngine, + protocol: WireFormat, + response: &AggLlmResponse, +) -> Result { + let output = engine + .encode_response(protocol, response, &policy()) + .map_err(error)?; + safe(&output.diagnostics)?; + Ok(output.body) +} + +fn policy() -> TranslationPolicy { + TranslationPolicy { + unknown_field_policy: UnknownFieldPolicy::Preserve, + lossy_conversion_policy: LossyConversionPolicy::Reject, + deterministic_ids: DeterministicIdPolicy::GenerateStable { + prefix: "relay".into(), + }, + preservation: PreservationPolicy::InMemory, + target_capabilities: TargetCapabilities::default(), + } +} + +fn request_policy(protocol: WireFormat) -> TranslationPolicy { + let mut policy = policy(); + if protocol == WireFormat::AnthropicMessages { + policy + .target_capabilities + .supports_json_schema_response_format = Some(false); + } + policy +} + +fn safe(diagnostics: &[TranslationDiagnostic]) -> Result<(), String> { + let unsafe_diagnostics = diagnostics + .iter() + .filter(|diagnostic| diagnostic.severity != DiagnosticSeverity::Info) + .collect::>(); + if unsafe_diagnostics.is_empty() { + Ok(()) + } else { + Err(format!( + "Switchyard translation was not lossless: {unsafe_diagnostics:?}" + )) + } +} + +fn error(error: switchyard_translation::TranslationError) -> String { + format!("Switchyard translation failed: {error}") +} + +#[cfg(test)] +mod tests { + use serde_json::{Map, json}; + + use super::*; + + #[test] + fn same_protocol_request_preserves_unknown_fields() { + let request = RelayRequest { + headers: Map::new(), + content: json!({ + "model": "route", + "messages": [{"role": "user", "content": "hello"}], + "provider_extension": {"exact": true} + }), + }; + let engine = TranslationEngine::default(); + let decoded = decode_request(&engine, WireFormat::OpenAiChat, &request).unwrap(); + validate_target_request(&engine, WireFormat::OpenAiChat, &decoded).unwrap(); + assert_eq!( + decoded + .preservation + .requests + .get(&WireFormat::OpenAiChat.into()) + .and_then(|body| body.get("provider_extension")), + Some(&json!({"exact": true})) + ); + } +} diff --git a/crates/switchyard-nemo-relay-plugin/tests/test_package_bundle.py b/crates/switchyard-nemo-relay-plugin/tests/test_package_bundle.py new file mode 100644 index 000000000..296cbc4b8 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/tests/test_package_bundle.py @@ -0,0 +1,76 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Tests for the native Relay plugin bundle packager.""" + +from __future__ import annotations + +import hashlib +import subprocess +import sys +import tarfile +import tempfile +import unittest +import zipfile +from pathlib import Path + +CRATE_ROOT = Path(__file__).resolve().parents[1] +PACKAGER = CRATE_ROOT / "scripts" / "package_bundle.py" +PACKAGE_NAME = "switchyard-nemo-relay-plugin" + + +class PackageBundleTest(unittest.TestCase): + """Verify materialized and archived plugin bundle contents.""" + + def test_materializes_and_archives_supported_formats(self) -> None: + for archive_suffix in (".tar.gz", ".zip"): + with self.subTest(archive_suffix=archive_suffix), tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + library = root / "libswitchyard_nemo_relay_plugin.so" + library.write_bytes(b"compiled plugin") + output = root / "bundle" + archive = root / f"{PACKAGE_NAME}-0.2.0-linux-x86_64{archive_suffix}" + + subprocess.run( + [ + sys.executable, + str(PACKAGER), + "--library", + str(library), + "--output", + str(output), + "--archive", + str(archive), + ], + check=True, + capture_output=True, + text=True, + ) + + expected = { + "LICENSE", + "NOTICE", + "config.schema.json", + library.name, + "relay-plugin.toml", + } + self.assertEqual({path.name for path in output.iterdir()}, expected) + manifest = (output / "relay-plugin.toml").read_text(encoding="utf-8") + self.assertIn(f'artifact = "{library.name}"', manifest) + self.assertIn(hashlib.sha256(library.read_bytes()).hexdigest(), manifest) + self.assertNotIn("", manifest) + self.assertNotIn("", manifest) + self.assertEqual(self.archive_members(archive), {f"{PACKAGE_NAME}/{name}" for name in expected}) + + @staticmethod + def archive_members(archive: Path) -> set[str]: + """Return regular-file paths from a supported bundle archive.""" + if archive.name.endswith(".tar.gz"): + with tarfile.open(archive) as stream: + return {member.name for member in stream.getmembers() if member.isfile()} + with zipfile.ZipFile(archive) as stream: + return {member.filename for member in stream.infolist() if not member.is_dir()} + + +if __name__ == "__main__": + unittest.main() diff --git a/docs/index.md b/docs/index.md index 1dc5056cd..93814e9cc 100644 --- a/docs/index.md +++ b/docs/index.md @@ -10,6 +10,7 @@ It supports OpenAI Chat Completions, OpenAI Responses, and Anthropic Messages. | Run Claude Code, Codex, or OpenClaw through Switchyard | Launcher Path | [Install and launch an agent](getting_started.md#launcher-path) | | Run Switchyard as a standalone proxy for API clients | Server Path | [Build and run the Rust server](getting_started.md#server-path) | | Add Switchyard routing to a Rust application | Library Path | [`switchyard-libsy`](../crates/libsy/README.md) | +| Add Switchyard routing to NeMo Relay | Native Plugin Path | [`switchyard-nemo-relay-plugin`](../crates/switchyard-nemo-relay-plugin/README.md) | The Launcher Path installs the `switchyard` CLI and hosts the native Rust server through its packaged PyO3 binding. The Server Path builds and runs the @@ -30,3 +31,4 @@ standalone `switchyard-server` binary. - [`switchyard-libsy`](reference/rust_api.md#switchyard-libsy): embeddable routing algorithms - [`switchyard-protocol`](reference/rust_api.md#switchyard-protocol): provider-neutral API types - [`switchyard-translation`](../crates/switchyard-translation/README.md): protocol translation +- [`switchyard-nemo-relay-plugin`](../crates/switchyard-nemo-relay-plugin/README.md): native NeMo Relay integration From 1f4dcd2ac7c4849521fad06fee42087b6a1eca55 Mon Sep 17 00:00:00 2001 From: Bryan Bednarski Date: Wed, 19 Aug 2026 14:18:38 -0700 Subject: [PATCH 2/4] refactor(relay): use current routing outcome API Signed-off-by: Bryan Bednarski --- crates/switchyard-nemo-relay-plugin/README.md | 6 +- .../src/client.rs | 262 +----------------- .../src/config.rs | 20 +- .../src/runtime.rs | 42 +-- .../src/runtime/tests.rs | 60 ++-- 5 files changed, 51 insertions(+), 339 deletions(-) diff --git a/crates/switchyard-nemo-relay-plugin/README.md b/crates/switchyard-nemo-relay-plugin/README.md index d718b4a68..4ec4faa6b 100644 --- a/crates/switchyard-nemo-relay-plugin/README.md +++ b/crates/switchyard-nemo-relay-plugin/README.md @@ -158,7 +158,7 @@ automatic compatibility fallback. Each completed routing-only model call emits a `switchyard.routing.llm_call` ATOF mark. Its data identifies the algorithm, -attempt, call order, target, role (`judge` or discarded `candidate`), outcome, +attempt, call order, target, routing role, outcome, latency, and normalized provider token `usage`. The successful call that serves the caller is deliberately excluded because Relay's outer LLM end event already records that usage. A failed call, or a @@ -358,8 +358,8 @@ classifier target has the same structured-output protocol restriction as the standalone classifier. Ambiguous turns that reach the optional classifier add one judge call; -decisive tool signals do not. Decision marks report the selected model, -reasoning, and answer-call status exposed by libsy's `Decision` API. +decisive tool signals do not. Decision marks report the selected model from +libsy's `RoutingOutcome`. Version-1 service configuration, decision-only execution, and observe-only mode are rejected. diff --git a/crates/switchyard-nemo-relay-plugin/src/client.rs b/crates/switchyard-nemo-relay-plugin/src/client.rs index bc197bfba..c3d345dad 100644 --- a/crates/switchyard-nemo-relay-plugin/src/client.rs +++ b/crates/switchyard-nemo-relay-plugin/src/client.rs @@ -9,8 +9,7 @@ use async_trait::async_trait; use serde_json::Value as Json; use switchyard_llm_client::{Backend, HttpBackendConfig, ModelConfig, TranslatingLlmClient}; use switchyard_protocol::{ - ContentBlock, Decision, LlmClientError, Message, Request, Response, Role, RoutedLlmClient, - ToolCall, ToolResult, WireFormat, + LlmClientError, ModelId, Request, Response, RoutedLlmClient, WireFormat, }; use switchyard_translation::TranslationEngine; @@ -23,7 +22,7 @@ use crate::translation; /// Keeping that mapping here prevents an algorithm's semantic labels from /// leaking into provider requests. pub(crate) struct TargetClient { - provider_model: String, + provider_model: ModelId, target_format: WireFormat, drop_caller_extra_body: bool, inner: TranslatingLlmClient, @@ -45,6 +44,7 @@ impl TargetClient { // URL/prefix. base_url: dispatch_url, api_key: None, + forward_auth: false, extra_headers: headers, extra_body, // Routing retries belong to the plugin: every retry must start a @@ -59,7 +59,7 @@ impl TargetClient { let model = ModelConfig::new(provider_model.clone(), backend, None); let inner = TranslatingLlmClient::new(&[model])?; Ok(Self { - provider_model, + provider_model: ModelId::from(provider_model), target_format, drop_caller_extra_body, inner, @@ -72,17 +72,7 @@ impl TargetClient { /// Correlation and agent identity remain available to libsy, while inbound /// HTTP headers are deliberately removed. Provider credentials come solely /// from this target's `header_env` configuration. - fn prepare_request(&self, mut request: Request, decision: &Decision) -> Request { - if !decision.is_answer_call() { - sanitize_judge_request(&mut request); - } - if decision.reasoning() == Some("escalation classifier: efficient tier") { - // Escalation always buffers this draft before judging it. Asking the - // provider for a buffered response preserves normalized usage for ATOF; - // libsy reconstructs a caller stream when the weak draft wins. - request.llm_request.stream = false; - request.llm_request.preservation.requests.clear(); - } + fn prepare_request(&self, mut request: Request) -> Request { let metadata = request.metadata.get_or_insert_default(); metadata.wire_format = Some(self.target_format); metadata.http_headers = None; @@ -100,8 +90,8 @@ impl TargetClient { #[async_trait] impl RoutedLlmClient for TargetClient { - async fn call(&self, request: Request, decision: Decision) -> Result { - let request = self.prepare_request(request, &decision); + async fn call(&self, request: Request) -> Result { + let request = self.prepare_request(request); translation::validate_target_request( &self.translation, self.target_format, @@ -114,131 +104,11 @@ impl RoutedLlmClient for TargetClient { } } -/// Maximum plain-text context retained from one native tool block in a judge request. -const MAX_JUDGE_TOOL_CONTEXT_CHARS: usize = 4_096; - -/// Keep judge requests provider-neutral. Native tool turns without their original -/// definitions are rejected by some OpenAI-compatible Bedrock gateways, while the -/// text evidence is still valuable to the classifier. -fn sanitize_judge_request(request: &mut Request) { - request.llm_request.messages = request - .llm_request - .messages - .drain(..) - .map(|message| Message { - role: if message.role == Role::Tool { - Role::User - } else { - message.role - }, - content: message - .content - .into_iter() - .map(|block| match block { - ContentBlock::ToolCall(call) => ContentBlock::Text { - text: bounded_tool_context(tool_call_text(call)), - }, - ContentBlock::ToolResult(result) => ContentBlock::Text { - text: bounded_tool_context(tool_result_text(result)), - }, - ordinary => ordinary, - }) - .collect(), - }) - .collect(); - request.llm_request.tools.clear(); - request.llm_request.tool_choice = None; - if let Some(response_format) = request.llm_request.output.response_format.as_mut() { - remove_numeric_schema_bounds(response_format); - } - request.llm_request.preservation.requests.clear(); -} - -fn tool_call_text(call: ToolCall) -> String { - format!( - "[tool call]\nid: {}\nname: {}\narguments: {}", - Json::String(call.id), - Json::String(call.name), - call.arguments - ) -} - -fn tool_result_text(result: ToolResult) -> String { - let content = result - .content - .into_iter() - .map(tool_content_text) - .collect::>() - .join("\n"); - format!( - "[tool result]\ncall_id: {}\nis_error: {}\ncontent:\n{}", - Json::String(result.tool_call_id), - result - .is_error - .map_or_else(|| "unknown".to_string(), |value| value.to_string()), - content - ) -} - -fn tool_content_text(block: ContentBlock) -> String { - match block { - ContentBlock::Text { text } - | ContentBlock::Reasoning { text, .. } - | ContentBlock::Refusal { text } => text, - ContentBlock::ToolCall(call) => tool_call_text(call), - ContentBlock::ToolResult(result) => tool_result_text(result), - ContentBlock::Image { .. } => "[image omitted]".to_string(), - ContentBlock::Audio { .. } => "[audio omitted]".to_string(), - ContentBlock::Video { .. } => "[video omitted]".to_string(), - ContentBlock::File { .. } => "[file omitted]".to_string(), - ContentBlock::Unknown { provider, .. } => { - format!("[unsupported {provider} content omitted]") - } - } -} - -fn bounded_tool_context(text: String) -> String { - const TRUNCATED: &str = "\n[truncated]"; - let keep = MAX_JUDGE_TOOL_CONTEXT_CHARS - TRUNCATED.chars().count(); - let mut chars = text.chars(); - let prefix = chars.by_ref().take(keep).collect::(); - if chars.next().is_some() { - prefix + TRUNCATED - } else { - prefix - } -} - -fn remove_numeric_schema_bounds(value: &mut Json) { - match value { - Json::Object(object) => { - for key in ["minimum", "maximum", "exclusiveMinimum", "exclusiveMaximum"] { - object.remove(key); - } - for child in object.values_mut() { - remove_numeric_schema_bounds(child); - } - } - Json::Array(values) => { - for child in values { - remove_numeric_schema_bounds(child); - } - } - _ => {} - } -} - #[cfg(test)] mod tests { use super::*; use serde_json::json; - use switchyard_protocol::{ - LlmRequest, Metadata, PreservationMetadata, ProviderExtensions, ToolChoice, ToolDefinition, - }; - - fn decision() -> Decision { - Decision::new("target", None, true) - } + use switchyard_protocol::{LlmRequest, Metadata, PreservationMetadata, ProviderExtensions}; fn client(format: WireFormat) -> TargetClient { TargetClient::new( @@ -278,7 +148,7 @@ mod tests { ..Request::default() }; - let prepared = client.prepare_request(request, &decision()); + let prepared = client.prepare_request(request); let metadata = prepared.metadata.unwrap(); assert_eq!(metadata.wire_format, Some(WireFormat::AnthropicMessages)); assert_eq!(metadata.correlation_id.as_deref(), Some("request-123")); @@ -288,7 +158,7 @@ mod tests { #[test] fn missing_metadata_is_created_for_the_target_format() { let client = client(WireFormat::OpenAiResponses); - let prepared = client.prepare_request(Request::default(), &decision()); + let prepared = client.prepare_request(Request::default()); assert_eq!( prepared.metadata.and_then(|metadata| metadata.wire_format), Some(WireFormat::OpenAiResponses) @@ -333,7 +203,7 @@ mod tests { ..Request::default() }; - let prepared = client.prepare_request(request, &decision()); + let prepared = client.prepare_request(request); assert!( !prepared .llm_request @@ -350,114 +220,4 @@ mod tests { .all(|body| body.get("extra_body").is_none()) ); } - - #[test] - fn judge_preparation_sanitizes_tool_history_and_schema_dialect() { - let client = client(WireFormat::OpenAiChat); - let request = Request { - llm_request: LlmRequest { - messages: vec![ - Message::text(Role::User, "inspect the workspace"), - Message { - role: Role::Assistant, - content: vec![ContentBlock::ToolCall(ToolCall { - id: "call-1".into(), - name: "terminal".into(), - arguments: json!({"command": "pwd"}), - })], - }, - Message { - role: Role::Tool, - content: vec![ContentBlock::ToolResult(ToolResult { - tool_call_id: "call-1".into(), - content: vec![ContentBlock::Text { - text: format!("result {} TAIL", "x".repeat(5_000)), - }], - is_error: Some(false), - })], - }, - ], - tools: vec![ToolDefinition { - name: "terminal".into(), - description: None, - parameters: json!({"type": "object"}), - strict: None, - }], - tool_choice: Some(ToolChoice::Required), - output: switchyard_protocol::OutputParams { - max_output_tokens: Some(64), - response_format: Some(json!({ - "type": "json_schema", - "json_schema": { - "schema": { - "properties": { - "p_solve": { - "type": "number", - "minimum": 0.0, - "maximum": 1.0 - } - } - } - } - })), - }, - ..LlmRequest::default() - }, - ..Request::default() - }; - - let prepared = client.prepare_request( - request, - &Decision::new("judge", Some("structured judge".into()), false), - ); - - assert!(prepared.llm_request.tools.is_empty()); - assert_eq!(prepared.llm_request.tool_choice, None); - assert!( - prepared - .llm_request - .messages - .iter() - .all(|message| message.role != Role::Tool) - ); - let text = prepared - .llm_request - .messages - .iter() - .filter_map(|message| message.text_content("\n")) - .collect::>() - .join("\n"); - assert!(text.contains("[tool call]")); - assert!(text.contains("terminal")); - assert!(text.contains("[tool result]")); - assert!(text.contains("[truncated]")); - assert!(!text.contains("TAIL")); - let schema = prepared.llm_request.output.response_format.unwrap(); - let p_solve = schema - .pointer("/json_schema/schema/properties/p_solve") - .unwrap(); - assert!(p_solve.get("minimum").is_none()); - assert!(p_solve.get("maximum").is_none()); - } - - #[test] - fn escalation_candidate_is_buffered_for_usage_accounting() { - let client = client(WireFormat::OpenAiChat); - let mut request = Request::default(); - request.llm_request.stream = true; - request.llm_request.preservation.requests.insert( - WireFormat::OpenAiChat.into(), - json!({"model": "route", "stream": true}), - ); - let decision = Decision::new( - "weak", - Some("escalation classifier: efficient tier".into()), - true, - ); - - let prepared = client.prepare_request(request, &decision); - - assert!(!prepared.llm_request.stream); - assert!(prepared.llm_request.preservation.requests.is_empty()); - } } diff --git a/crates/switchyard-nemo-relay-plugin/src/config.rs b/crates/switchyard-nemo-relay-plugin/src/config.rs index ff3668150..8e986c888 100644 --- a/crates/switchyard-nemo-relay-plugin/src/config.rs +++ b/crates/switchyard-nemo-relay-plugin/src/config.rs @@ -10,10 +10,10 @@ use serde::Deserialize; use serde_json::Value as Json; use switchyard_libsy::{ Algorithm, ClassifierContractConfig, EscalationJudgeConfig, HandoffNoteConfig, - LlmClassifierConfig, LlmFallback, LlmTarget, LlmTargetSet, LlmTaskClassifier, PickerMode, - Random, StageRouter, StageRouterConfig, TargetPrompts, TaskClassifierConfig, + LlmClassifierConfig, LlmFallback, LlmTaskClassifier, PickerMode, Random, StageRouter, + StageRouterConfig, TargetPrompts, TaskClassifierConfig, }; -use switchyard_protocol::{RoutedLlmClient, WireFormat}; +use switchyard_protocol::{ModelId, RoutedLlmClient, WireFormat}; use crate::client::TargetClient; @@ -378,13 +378,9 @@ impl SwitchyardConfig { targets .get(name) .ok_or_else(|| format!("algorithm target {name:?} was not prepared"))?; - LlmTarget { - semantic_name: name.to_string(), - } + ModelId::from(name) } - None => LlmTarget { - semantic_name: name.to_string(), - }, + None => ModelId::from(name), }) }; @@ -408,7 +404,7 @@ impl SwitchyardConfig { .iter() .map(|(_, binding)| binding.weight) .collect::>(); - Random::new(LlmTargetSet::new(targets), Some(weights), *seed) + Random::new(targets, Some(weights), *seed) .map(|algorithm| Arc::new(algorithm) as Arc) .map_err(|error| error.to_string()) } @@ -455,10 +451,10 @@ impl SwitchyardConfig { config.handoff_notes = handoff_notes.clone(); let mut prompts = TargetPrompts::default(); if let Some(prompt) = capable_system_prompt { - prompts = prompts.with(capable_target, prompt); + prompts = prompts.with(capable_target.as_str(), prompt); } if let Some(prompt) = efficient_system_prompt { - prompts = prompts.with(efficient_target, prompt); + prompts = prompts.with(efficient_target.as_str(), prompt); } config.tier_prompts = prompts; if let Some(classifier) = classifier { diff --git a/crates/switchyard-nemo-relay-plugin/src/runtime.rs b/crates/switchyard-nemo-relay-plugin/src/runtime.rs index 9615e91d8..cb15180d3 100644 --- a/crates/switchyard-nemo-relay-plugin/src/runtime.rs +++ b/crates/switchyard-nemo-relay-plugin/src/runtime.rs @@ -11,7 +11,7 @@ use serde_json::{Map, json}; use switchyard_libsy::{Algorithm, LibsyError}; use switchyard_llm_client::{ClientRouter, LlmCallObservation, RunObservation, RunObserver, run}; use switchyard_protocol::{ - Decision, LlmClientError, LlmResponse, Metadata, Request, Response, WireFormat, + LlmClientError, LlmResponse, Metadata, ModelId, Request, Response, WireFormat, }; use switchyard_translation::{TranslationEngine, encode_stream}; @@ -325,20 +325,17 @@ impl SwitchyardRuntime { let clients = ClientRouter::new( self.targets .iter() - .map(|(name, target)| (name.clone(), target.client.clone())) + .map(|(name, target)| (ModelId::from(name.as_str()), target.client.clone())) .collect::>(), ); match run(self.algorithm.clone(), clients, request, Some(observer)).await { - Ok((decisions, response)) => { - for decision in decisions { - self.emit_decision(marks, &decision, attempt, mark_metadata); - } + Ok((selected_model_id, response)) => { + self.emit_decision(marks, &selected_model_id, attempt, mark_metadata); self.emit_routing_llm_calls( marks, take_observed_calls(&observations), attempt, mark_metadata, - true, ); Ok(response) } @@ -348,7 +345,6 @@ impl SwitchyardRuntime { take_observed_calls(&observations), attempt, mark_metadata, - false, ); Err(error) } @@ -370,10 +366,9 @@ impl SwitchyardRuntime { json!({"selected_target": target_name}), metadata, ); - let decision = Decision::new(target_name, Some("trusted fallback target".into()), true); target .client - .call(request, decision) + .call(request) .await .map_err(|error| public_client_failure("trusted fallback", &error)) } @@ -402,7 +397,7 @@ impl SwitchyardRuntime { fn emit_decision( &self, marks: &mut Vec, - decision: &Decision, + selected_model_id: &ModelId, attempt: u32, metadata: &Json, ) { @@ -412,9 +407,7 @@ impl SwitchyardRuntime { json!({ "algorithm": self.algorithm.name(), "attempt": attempt, - "selected_target": decision.selected_model_id(), - "reasoning": decision.reasoning(), - "is_answer_call": decision.is_answer_call(), + "selected_target": selected_model_id, }), metadata, ); @@ -423,23 +416,10 @@ impl SwitchyardRuntime { fn emit_routing_llm_calls( &self, marks: &mut Vec, - mut calls: Vec, + calls: Vec, attempt: u32, metadata: &Json, - successful_run: bool, ) { - // The last successful routed call produced the response represented by Relay's - // outer LLM lifecycle event. Keep it out of these marks so consumers can add - // routing overhead without counting the serving call twice. Earlier routed calls - // are discarded candidates (for example, escalation's weak draft). - if successful_run - && let Some(position) = calls - .iter() - .rposition(|call| call.is_answer_call && call.is_success) - { - calls.remove(position); - } - for (index, call) in calls.into_iter().enumerate() { self.mark( marks, @@ -449,7 +429,7 @@ impl SwitchyardRuntime { "attempt": attempt, "call_index": index + 1, "selected_target": call.selected_model, - "call_role": if call.is_answer_call { "candidate" } else { "judge" }, + "call_role": "routing", "outcome": if call.is_success { "ok" } else { "error" }, "latency_ms": call.duration.as_secs_f64() * 1_000.0, "usage": call.usage, @@ -551,7 +531,7 @@ fn libsy_error_retryable(error: &LibsyError) -> bool { }; match source { LlmClientError::UpstreamHttp { status, .. } => { - matches!(*status, 408 | 425 | 429 | 500 | 502 | 503 | 504) + matches!(status.as_u16(), 408 | 425 | 429 | 500 | 502 | 503 | 504) } LlmClientError::Transport { .. } | LlmClientError::Timeout { .. } => true, _ => false, @@ -583,7 +563,7 @@ fn failure_mark_data(attempt: u32, failure: &LibsyError) -> Json { .. } => { data.insert("failure_kind".into(), Json::from("http")); - data.insert("http_status".into(), Json::from(*status)); + data.insert("http_status".into(), Json::from(status.as_u16())); } LibsyError::ClientCall { source, .. } => { data.insert("failure_kind".into(), Json::from("non_http")); diff --git a/crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs b/crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs index 54d383278..a4266764f 100644 --- a/crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs +++ b/crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs @@ -4,11 +4,13 @@ use std::sync::atomic::{AtomicUsize, Ordering}; use switchyard_libsy::{ - ClassifierContractConfig, EscalationJudgeConfig, LlmClassifierConfig, LlmFallback, LlmTarget, + ClassifierContractConfig, EscalationJudgeConfig, LlmClassifierConfig, LlmFallback, LlmTaskClassifier, Passthrough, PickerMode, StageRouter, StageRouterConfig, TaskClassifierConfig, }; -use switchyard_protocol::{LlmResponseStream, RoutedLlmClient, Usage, text_request, text_response}; +use switchyard_protocol::{ + LlmResponseStream, ModelId, RoutedLlmClient, Usage, text_request, text_response, +}; use super::*; @@ -34,11 +36,7 @@ fn scripted(behavior: ScriptedBehavior) -> Arc { #[async_trait::async_trait] impl RoutedLlmClient for ScriptedClient { - async fn call( - &self, - request: Request, - _decision: Decision, - ) -> Result { + async fn call(&self, request: Request) -> Result { self.calls.fetch_add(1, Ordering::Relaxed); match self.behavior { ScriptedBehavior::Text(text) => { @@ -80,10 +78,8 @@ impl RoutedLlmClient for ScriptedClient { } } -fn fixed_target(name: &str) -> LlmTarget { - LlmTarget { - semantic_name: name.to_string(), - } +fn fixed_target(name: &str) -> ModelId { + ModelId::from(name) } fn runtime_with_algorithm( @@ -217,9 +213,7 @@ fn stage_signal_relay_request(protocol: WireFormat) -> RelayRequest { fn relay_gateway_placeholder_session_is_not_retained() { let fallback = scripted(ScriptedBehavior::Text("fallback")); let runtime = runtime_with_algorithm( - Arc::new(Passthrough::new(LlmTarget { - semantic_name: "selected".into(), - })), + Arc::new(Passthrough::new(ModelId::from("selected"))), fallback, WireFormat::OpenAiChat, ); @@ -246,9 +240,7 @@ fn relay_gateway_placeholder_session_is_not_retained() { fn explicit_switchyard_session_overrides_relay_gateway_placeholder() { let fallback = scripted(ScriptedBehavior::Text("fallback")); let runtime = runtime_with_algorithm( - Arc::new(Passthrough::new(LlmTarget { - semantic_name: "selected".into(), - })), + Arc::new(Passthrough::new(ModelId::from("selected"))), fallback, WireFormat::OpenAiChat, ); @@ -280,9 +272,7 @@ async fn buffered_finalization_failure_uses_fallback_once() { let fallback = scripted(ScriptedBehavior::EmptyBuffered); let runtime = SwitchyardRuntime { max_retries: 1, - algorithm: Arc::new(Passthrough::new(LlmTarget { - semantic_name: "selected".into(), - })), + algorithm: Arc::new(Passthrough::new(ModelId::from("selected"))), targets: BTreeMap::from([ ( "selected".into(), @@ -382,9 +372,7 @@ async fn invalid_selected_stream_does_not_invoke_failing_fallback_twice() { let fallback = scripted(ScriptedBehavior::FailingStream); let runtime = SwitchyardRuntime { max_retries: 0, - algorithm: Arc::new(Passthrough::new(LlmTarget { - semantic_name: "selected".into(), - })), + algorithm: Arc::new(Passthrough::new(ModelId::from("selected"))), targets: BTreeMap::from([ ( "selected".into(), @@ -420,9 +408,7 @@ async fn failing_fallback_call_flushes_error_and_fallback_marks() { let fallback = scripted(ScriptedBehavior::TransportFailure("fallback call failed")); let runtime = SwitchyardRuntime { max_retries: 0, - algorithm: Arc::new(Passthrough::new(LlmTarget { - semantic_name: "selected".into(), - })), + algorithm: Arc::new(Passthrough::new(ModelId::from("selected"))), targets: BTreeMap::from([ ( "selected".into(), @@ -584,10 +570,10 @@ async fn escalation_buffers_weak_stream_then_latches_the_session_to_strong() { ); assert_eq!(routing_calls.len(), 2); assert_eq!(routing_calls[0]["selected_target"], "weak"); - assert_eq!(routing_calls[0]["call_role"], "candidate"); + assert_eq!(routing_calls[0]["call_role"], "routing"); assert_eq!(routing_calls[0]["usage"]["total_tokens"], 18); assert_eq!(routing_calls[1]["selected_target"], "judge"); - assert_eq!(routing_calls[1]["call_role"], "judge"); + assert_eq!(routing_calls[1]["call_role"], "routing"); assert_eq!(routing_calls[1]["usage"]["total_tokens"], 18); assert!( routing_calls @@ -617,8 +603,6 @@ async fn escalation_buffers_weak_stream_then_latches_the_session_to_strong() { assert!(marks.iter().any(|mark| { mark.name == "switchyard.routing.decision" && mark.data["selected_target"] == "strong" - && mark.data["is_answer_call"] == true - && mark.data["reasoning"].is_string() && mark.metadata["session_id"] == "session-1" })); } @@ -670,7 +654,7 @@ async fn escalation_judge_failure_falls_open_to_the_buffered_weak_response() { .collect::>(); assert_eq!(routing_calls.len(), 1); assert_eq!(routing_calls[0].data["selected_target"], "judge"); - assert_eq!(routing_calls[0].data["call_role"], "judge"); + assert_eq!(routing_calls[0].data["call_role"], "routing"); assert_eq!(routing_calls[0].data["outcome"], "error"); assert!(routing_calls[0].data["usage"].is_null()); } @@ -766,8 +750,6 @@ async fn stage_router_uses_tool_signals_for_every_managed_protocol() { && mark.data["algorithm"] == "stage_router" && mark.data["attempt"] == 1 && mark.data["selected_target"] == "strong" - && mark.data["reasoning"].is_string() - && mark.data["is_answer_call"] == true && mark.metadata["session_id"] == format!("stage-{}", protocol.as_str()) })); } @@ -806,10 +788,7 @@ async fn stage_router_falls_open_to_each_picker_default_without_tool_history() { .unwrap(); assert!(marks.iter().any(|mark| { - mark.name == "switchyard.routing.decision" - && mark.data["selected_target"] == expected - && mark.data["reasoning"].is_string() - && mark.data["is_answer_call"] == true + mark.name == "switchyard.routing.decision" && mark.data["selected_target"] == expected })); } } @@ -862,13 +841,10 @@ async fn stage_router_classifier_resolves_an_ambiguous_turn() { .collect::>(); assert_eq!(routing_calls.len(), 1); assert_eq!(routing_calls[0].data["selected_target"], "judge"); - assert_eq!(routing_calls[0].data["call_role"], "judge"); + assert_eq!(routing_calls[0].data["call_role"], "routing"); assert_eq!(routing_calls[0].data["outcome"], "ok"); assert_eq!(routing_calls[0].data["usage"]["total_tokens"], 18); assert!(marks.iter().any(|mark| { - mark.name == "switchyard.routing.decision" - && mark.data["selected_target"] == "weak" - && mark.data["reasoning"].is_string() - && mark.data["is_answer_call"] == true + mark.name == "switchyard.routing.decision" && mark.data["selected_target"] == "weak" })); } From 48815533576f2abd59f4ba9f09440d296197d77c Mon Sep 17 00:00:00 2001 From: Bryan Bednarski Date: Wed, 19 Aug 2026 18:18:08 -0700 Subject: [PATCH 3/4] refactor(relay): use typed middleware adapter Signed-off-by: Bryan Bednarski --- crates/switchyard-nemo-relay-plugin/README.md | 64 +-- .../src/executor.rs | 137 ------ .../switchyard-nemo-relay-plugin/src/ffi.rs | 444 ------------------ .../switchyard-nemo-relay-plugin/src/lib.rs | 406 ++++------------ 4 files changed, 130 insertions(+), 921 deletions(-) delete mode 100644 crates/switchyard-nemo-relay-plugin/src/executor.rs delete mode 100644 crates/switchyard-nemo-relay-plugin/src/ffi.rs diff --git a/crates/switchyard-nemo-relay-plugin/README.md b/crates/switchyard-nemo-relay-plugin/README.md index 4ec4faa6b..d5bc92be5 100644 --- a/crates/switchyard-nemo-relay-plugin/README.md +++ b/crates/switchyard-nemo-relay-plugin/README.md @@ -8,8 +8,8 @@ SPDX-License-Identifier: Apache-2.0 This crate builds the external `nvidia.switchyard` native plugin. It embeds `switchyard-libsy`, drives it through `switchyard-llm-client::run`, and uses `switchyard-llm-client` for provider HTTP calls. Managed calls use Relay's -completion-based asynchronous middleware hooks and do not require a targeted -provider continuation from Relay. +typed asynchronous middleware API and do not require a targeted provider +continuation from Relay. The plugin uses NeMo Relay native API v1. It depends on the small `nemo-relay-plugin` authoring SDK, not the Relay runtime, and does not start @@ -59,40 +59,29 @@ This boundary has two important consequences: transport activity is therefore not represented as nested Relay LLM lifecycle events. Relay records the outer managed call and the plugin emits Switchyard routing marks; bridging Switchyard transport spans into Relay is - future work. The adapter captures the active Relay scope before returning - `Pending`, so asynchronous routing marks retain their event parent. + future work. Relay's typed middleware adapter propagates the active scope so + asynchronous routing marks retain their event parent. - Switchyard owns provider URLs, credentials, HTTP retry behavior, and translation for managed calls. Relay neither validates nor transports those target details. -## Native API v1 and asynchronous execution - -The manifest remains `compat.native_api = "1"`, but the plugin requires the -generic host-table v3 extension shipped by Relay 0.7. It registers through -v3's completion-based buffered and incremental streaming hooks, returns -`Pending` immediately, and performs libsy and provider HTTP work on a -plugin-owned Tokio runtime. Relay workers therefore do not wait synchronously -for provider I/O. - -The stream adapter forwards the plugin's bounded 32-message channel into -Relay's bounded output queue. It retries a logical event when the host queue is -full and checks cancellation between attempts. Managed HTTP work is selected -against Relay caller cancellation, so cancelling a buffered or streaming call -drops its in-flight provider future. - -Unmanaged profiles use the same v3 continuation hooks for pass-through. V3's -downstream stream callback has continue/cancel control but no asynchronous -acknowledgement, so the adapter uses a nonblocking bridge capped at 8 MiB of -queued encoded payloads and 256 events. A pass-through stream that outruns -either bound is rejected rather than consuming unbounded memory. - -This is a raw C boundary: Switchyard contains a small ownership adapter for -host strings, completion and stream handles, continuation handles, and captured -scope handles because Relay 0.7 does not expose a safe Rust facade for its -generic asynchronous surface. The HTTP, routing, and translation behavior -remains in Switchyard. NeMo Relay plans to provide the equivalent safe typed -surface in 0.8.0; once that is available, the raw-FFI compatibility adapter -should be removable. +## Native API v1 and typed asynchronous execution + +The manifest remains `compat.native_api = "1"`, and the plugin requires Relay +`>=0.8.0,<1.0` with native host ABI v4. It registers typed buffered and +incremental streaming intercepts through `nemo-relay-plugin`. Relay owns the plugin executor, +continuation and output-stream lifecycle, cancellation, backpressure, and +scope propagation; Relay workers therefore do not wait synchronously for +provider I/O. + +The stream adapter preserves the plugin's bounded 32-message routing channel. +Relay's typed stream adapter forwards response events to its bounded output +queue and handles cancellation and backpressure. Cancelling a buffered or +streaming call drops its in-flight provider future. + +Unmanaged profiles use Relay's typed continuations for pass-through. The HTTP, +routing, and translation behavior remains in Switchyard; Relay owns the native +dynamic-library boundary. ## Supported routers @@ -197,12 +186,11 @@ per-event preservation envelope. ## Configuration -The manifest declares `compat.native_api = "1"` and Relay `>=0.7.0,<0.8`, and -the Rust SDK uses the exact published `0.7.0` crate. The manifest API value -selects Relay's released native plugin contract; the binary also requires the -v3 C host table shipped on the Relay 0.7 line. Rebuild the bundle when changing -SDK versions rather than assuming Rust dynamic-library compatibility from the -manifest value alone. +The manifest declares `compat.native_api = "1"` and Relay `>=0.8.0,<1.0`. +The manifest API value selects Relay's native plugin contract; the binary uses +Relay 0.8's typed middleware API and native host ABI v4. Rebuild the bundle +when changing SDK versions rather than assuming Rust dynamic-library +compatibility from the manifest value alone. A Relay project can configure a seeded weighted-random router as follows: diff --git a/crates/switchyard-nemo-relay-plugin/src/executor.rs b/crates/switchyard-nemo-relay-plugin/src/executor.rs deleted file mode 100644 index 206c7aa99..000000000 --- a/crates/switchyard-nemo-relay-plugin/src/executor.rs +++ /dev/null @@ -1,137 +0,0 @@ -// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -// SPDX-License-Identifier: Apache-2.0 - -use std::future::Future; -use std::sync::{Arc, Mutex, mpsc}; -use std::thread::{self, JoinHandle}; - -use tokio::runtime::{Builder, Handle}; -use tokio::sync::oneshot; -use tokio::task::AbortHandle; - -/// Plugin-owned async executor. -/// -/// The public native-plugin SDK uses synchronous Rust callbacks and pull-based -/// iterators at the dynamic-library boundary. Switchyard performs provider I/O -/// on this dedicated runtime rather than entering Relay's Tokio runtime from a -/// separately linked cdylib. -#[derive(Clone)] -pub(crate) struct PluginExecutor { - inner: Arc, -} - -struct ExecutorInner { - handle: Handle, - shutdown: Mutex>>, - thread: Mutex>>, -} - -impl PluginExecutor { - pub(crate) fn new() -> Result { - let (ready_tx, ready_rx) = mpsc::sync_channel(1); - let thread = thread::Builder::new() - .name("switchyard-relay-http".into()) - .spawn(move || { - let runtime = match Builder::new_multi_thread() - .worker_threads(2) - .thread_name("switchyard-relay-http-worker") - .enable_all() - .build() - { - Ok(runtime) => runtime, - Err(error) => { - let _ = ready_tx.send(Err(error.to_string())); - return; - } - }; - let (shutdown_tx, shutdown_rx) = oneshot::channel(); - if ready_tx - .send(Ok((runtime.handle().clone(), shutdown_tx))) - .is_err() - { - return; - } - runtime.block_on(async { - let _ = shutdown_rx.await; - }); - }) - .map_err(|error| format!("failed to start Switchyard HTTP runtime: {error}"))?; - let (handle, shutdown) = ready_rx - .recv() - .map_err(|_| "Switchyard HTTP runtime stopped during startup".to_string())??; - Ok(Self { - inner: Arc::new(ExecutorInner { - handle, - shutdown: Mutex::new(Some(shutdown)), - thread: Mutex::new(Some(thread)), - }), - }) - } - - pub(crate) fn spawn(&self, future: F) -> AbortHandle - where - F: Future + Send + 'static, - { - self.inner.handle.spawn(future).abort_handle() - } -} - -impl Drop for ExecutorInner { - fn drop(&mut self) { - if let Some(shutdown) = self - .shutdown - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) - .take() - { - let _ = shutdown.send(()); - } - if let Some(thread) = self - .thread - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) - .take() - { - if std::thread::current() - .name() - .is_some_and(|name| name.starts_with("switchyard-relay-http-worker")) - { - // The runtime owner will join this worker after the current - // task returns. Waiting here would deadlock that shutdown. - drop(thread); - } else { - let _ = thread.join(); - } - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn executor_spawns_work() { - let executor = PluginExecutor::new().unwrap(); - let (sender, receiver) = mpsc::sync_channel(1); - executor.spawn(async move { - sender.send("done").unwrap(); - }); - assert_eq!(receiver.recv().unwrap(), "done"); - } - - #[test] - fn last_reference_can_drop_on_a_worker() { - let executor = PluginExecutor::new().unwrap(); - let worker_reference = executor.clone(); - let (sender, receiver) = mpsc::sync_channel(1); - executor.spawn(async move { - drop(worker_reference); - sender.send(()).unwrap(); - }); - drop(executor); - receiver - .recv_timeout(std::time::Duration::from_secs(5)) - .expect("dropping the executor on its own worker must not deadlock"); - } -} diff --git a/crates/switchyard-nemo-relay-plugin/src/ffi.rs b/crates/switchyard-nemo-relay-plugin/src/ffi.rs deleted file mode 100644 index f5ce094c1..000000000 --- a/crates/switchyard-nemo-relay-plugin/src/ffi.rs +++ /dev/null @@ -1,444 +0,0 @@ -// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -// SPDX-License-Identifier: Apache-2.0 - -//! Small ownership wrapper around Relay's generic C host-table v3 hooks. -//! -//! The plugin manifest remains native API v1. Relay 0.7 supplies the appended -//! v3 host table to rebuilt v1 plugins, which lets this crate return `Pending` -//! and settle work from its own runtime without a targeted-continuation ABI. - -use std::ffi::c_void; -use std::ptr; -use std::sync::Arc; -use std::sync::atomic::{AtomicUsize, Ordering}; -use std::time::Duration; - -use nemo_relay_plugin::{ - Json, LlmRequest, NemoRelayNativeAsyncCompletion, NemoRelayNativeAsyncNext, - NemoRelayNativeAsyncNextStreamCb, NemoRelayNativeAsyncStream, NemoRelayNativeHostApiV1, - NemoRelayNativeHostApiV3, NemoRelayNativeScopeHandle, NemoRelayNativeString, NemoRelayStatus, -}; -use serde::Serialize; -use tokio::sync::{mpsc, oneshot}; - -const BACKPRESSURE_POLL: Duration = Duration::from_millis(1); -const CANCELLATION_POLL: Duration = Duration::from_millis(10); -const MAX_PASSTHROUGH_BUFFER_BYTES: usize = 8 * 1024 * 1024; -const MAX_PASSTHROUGH_BUFFER_EVENTS: usize = 256; - -pub(crate) struct HostString { - host: NemoRelayNativeHostApiV1, - ptr: *mut NemoRelayNativeString, -} - -// Host strings are immutable allocations owned by Relay's thread-safe host table. -unsafe impl Send for HostString {} - -impl HostString { - pub(crate) fn json( - host: &NemoRelayNativeHostApiV1, - value: &impl Serialize, - ) -> Result { - let value = serde_json::to_string(value).map_err(|error| error.to_string())?; - Self::text(host, &value) - } - - pub(crate) fn text(host: &NemoRelayNativeHostApiV1, value: &str) -> Result { - let mut ptr = ptr::null_mut(); - let status = unsafe { (host.string_new)(value.as_ptr(), value.len(), &mut ptr) }; - if status == NemoRelayStatus::Ok && !ptr.is_null() { - Ok(Self { host: *host, ptr }) - } else { - Err(format!("Relay host string allocation failed: {status:?}")) - } - } - - pub(crate) fn as_ptr(&self) -> *const NemoRelayNativeString { - self.ptr - } -} - -impl Drop for HostString { - fn drop(&mut self) { - unsafe { (self.host.string_free)(self.ptr) }; - } -} - -pub(crate) fn read_string( - host: &NemoRelayNativeHostApiV1, - value: *const NemoRelayNativeString, -) -> Result { - if value.is_null() { - return Err("Relay passed a null native string".into()); - } - let len = unsafe { (host.string_len)(value) }; - let data = unsafe { (host.string_data)(value) }; - if data.is_null() && len != 0 { - return Err("Relay passed an invalid native string".into()); - } - let bytes = if len == 0 { - &[][..] - } else { - unsafe { std::slice::from_raw_parts(data, len) } - }; - std::str::from_utf8(bytes) - .map(str::to_owned) - .map_err(|error| error.to_string()) -} - -pub(crate) fn read_json( - host: &NemoRelayNativeHostApiV1, - value: *const NemoRelayNativeString, -) -> Result { - serde_json::from_str(&read_string(host, value)?).map_err(|error| error.to_string()) -} - -/// Captures the current Relay scope as an explicit event parent. -/// -/// Async plugin work runs on a plugin-owned thread, so relying on thread-local -/// scope state would orphan its marks. The host handle is a cloned scope handle -/// and remains valid until this guard is dropped. -pub(crate) struct ParentScope { - host: NemoRelayNativeHostApiV1, - ptr: *mut NemoRelayNativeScopeHandle, -} - -unsafe impl Send for ParentScope {} -unsafe impl Sync for ParentScope {} - -impl ParentScope { - pub(crate) fn capture(host: &NemoRelayNativeHostApiV1) -> Option { - let mut ptr = ptr::null_mut(); - let status = unsafe { (host.scope_get_current)(&mut ptr) }; - (status == NemoRelayStatus::Ok && !ptr.is_null()).then_some(Self { host: *host, ptr }) - } - - pub(crate) fn emit_mark(&self, name: &str, data: &Json, metadata: &Json) -> Result<(), String> { - let name = HostString::text(&self.host, name)?; - let data = HostString::json(&self.host, data)?; - let metadata = HostString::json(&self.host, metadata)?; - let status = unsafe { - (self.host.emit_mark)( - name.as_ptr(), - self.ptr, - data.as_ptr(), - metadata.as_ptr(), - ptr::null(), - ) - }; - if status == NemoRelayStatus::Ok { - Ok(()) - } else { - Err(format!( - "Relay rejected Switchyard routing mark: {status:?}" - )) - } - } -} - -impl Drop for ParentScope { - fn drop(&mut self) { - unsafe { (self.host.scope_handle_free)(self.ptr) }; - } -} - -pub(crate) fn invoke_next_buffered( - host: &NemoRelayNativeHostApiV3, - next: usize, - completion: usize, - request: &LlmRequest, -) -> Result<(), String> { - let request = HostString::json(&host.v1, request)?; - let status = unsafe { - (host.async_next_invoke)( - next as *const NemoRelayNativeAsyncNext, - request.as_ptr(), - completion as *const NemoRelayNativeAsyncCompletion, - ) - }; - if status == NemoRelayStatus::Ok { - Ok(()) - } else { - Err(format!("Relay rejected buffered pass-through: {status:?}")) - } -} - -enum DownstreamStreamItem { - Chunk { value: Json, encoded_bytes: usize }, -} - -struct DownstreamStreamState { - host: NemoRelayNativeHostApiV1, - sender: mpsc::Sender, - terminal: Option>>, - queued_bytes: Arc, -} - -pub(crate) async fn invoke_next_stream( - host: &NemoRelayNativeHostApiV3, - next: usize, - output: usize, - request: &LlmRequest, -) -> Result<(), String> { - let request = HostString::json(&host.v1, request)?; - let (sender, mut receiver) = mpsc::channel(MAX_PASSTHROUGH_BUFFER_EVENTS); - let (terminal, terminal_result) = oneshot::channel(); - let queued_bytes = Arc::new(AtomicUsize::new(0)); - let state = Box::into_raw(Box::new(DownstreamStreamState { - host: host.v1, - sender, - terminal: Some(terminal), - queued_bytes: Arc::clone(&queued_bytes), - })) - .cast::(); - let status = unsafe { - (host.async_next_invoke_stream)( - next as *const NemoRelayNativeAsyncNext, - request.as_ptr(), - output as *const NemoRelayNativeAsyncStream, - downstream_stream_result as NemoRelayNativeAsyncNextStreamCb, - state, - ) - }; - if status != NemoRelayStatus::Ok { - unsafe { drop(Box::from_raw(state.cast::())) }; - return Err(format!("Relay rejected streaming pass-through: {status:?}")); - } - - while let Some(item) = receiver.recv().await { - match item { - DownstreamStreamItem::Chunk { - value, - encoded_bytes, - } => { - let result = push_stream(host, output, &value).await; - queued_bytes.fetch_sub(encoded_bytes, Ordering::AcqRel); - result?; - } - } - } - terminal_result - .await - .unwrap_or_else(|_| Err("Relay dropped the streaming pass-through callback".into())) -} - -unsafe extern "C" fn downstream_stream_result( - user_data: *mut c_void, - chunk_json: *const NemoRelayNativeString, - error: *const NemoRelayNativeString, - done: bool, -) -> bool { - if !error.is_null() { - let state = unsafe { Box::from_raw(user_data.cast::()) }; - let error = read_string(&state.host, error) - .unwrap_or_else(|_| "Relay streaming pass-through failed".into()); - settle_downstream_stream(state, Err(error)); - return false; - } - if done { - let state = unsafe { Box::from_raw(user_data.cast::()) }; - settle_downstream_stream(state, Ok(())); - return false; - } - - let state = unsafe { &*user_data.cast::() }; - let parsed = read_string(&state.host, chunk_json).and_then(|encoded| { - let encoded_bytes = encoded.len(); - let value = serde_json::from_str(&encoded).map_err(|error| error.to_string())?; - Ok((value, encoded_bytes)) - }); - let (value, encoded_bytes) = match parsed { - Ok(parsed) => parsed, - Err(error) => { - let state = unsafe { Box::from_raw(user_data.cast::()) }; - settle_downstream_stream(state, Err(error)); - return false; - } - }; - if !reserve_buffer_bytes(&state.queued_bytes, encoded_bytes) { - let state = unsafe { Box::from_raw(user_data.cast::()) }; - settle_downstream_stream( - state, - Err(format!( - "Relay streaming pass-through exceeded its {}-byte queued payload limit", - MAX_PASSTHROUGH_BUFFER_BYTES - )), - ); - return false; - } - - match state.sender.try_send(DownstreamStreamItem::Chunk { - value, - encoded_bytes, - }) { - Ok(()) => true, - Err(error) => { - let (item, message) = match error { - mpsc::error::TrySendError::Full(item) => ( - item, - format!( - "Relay streaming pass-through exceeded its {MAX_PASSTHROUGH_BUFFER_EVENTS}-event queue" - ), - ), - mpsc::error::TrySendError::Closed(item) => ( - item, - "Relay dropped the streaming pass-through receiver".into(), - ), - }; - let encoded_bytes = item.encoded_bytes(); - state - .queued_bytes - .fetch_sub(encoded_bytes, Ordering::AcqRel); - let state = unsafe { Box::from_raw(user_data.cast::()) }; - settle_downstream_stream(state, Err(message)); - false - } - } -} - -impl DownstreamStreamItem { - fn encoded_bytes(&self) -> usize { - match self { - Self::Chunk { encoded_bytes, .. } => *encoded_bytes, - } - } -} - -fn settle_downstream_stream(mut state: Box, result: Result<(), String>) { - if let Some(terminal) = state.terminal.take() { - let _ = terminal.send(result); - } -} - -fn reserve_buffer_bytes(queued: &AtomicUsize, encoded_bytes: usize) -> bool { - queued - .fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { - current - .checked_add(encoded_bytes) - .filter(|next| *next <= MAX_PASSTHROUGH_BUFFER_BYTES) - }) - .is_ok() -} - -pub(crate) async fn wait_for_completion_cancellation( - host: &NemoRelayNativeHostApiV3, - completion: usize, -) { - while !completion_cancelled(host, completion as *const NemoRelayNativeAsyncCompletion) { - tokio::time::sleep(CANCELLATION_POLL).await; - } -} - -pub(crate) async fn wait_for_stream_cancellation(host: &NemoRelayNativeHostApiV3, stream: usize) { - while !unsafe { (host.async_stream_is_cancelled)(stream as *const NemoRelayNativeAsyncStream) } - { - tokio::time::sleep(CANCELLATION_POLL).await; - } -} - -pub(crate) fn completion_cancelled( - host: &NemoRelayNativeHostApiV3, - completion: *const NemoRelayNativeAsyncCompletion, -) -> bool { - unsafe { (host.async_completion_is_cancelled)(completion) } -} - -pub(crate) fn resolve_completion( - host: &NemoRelayNativeHostApiV3, - completion: *const NemoRelayNativeAsyncCompletion, - value: &Json, -) -> NemoRelayStatus { - match HostString::json(&host.v1, value) { - Ok(value) => unsafe { (host.async_completion_resolve_json)(completion, value.as_ptr()) }, - Err(_) => NemoRelayStatus::Internal, - } -} - -pub(crate) fn reject_completion( - host: &NemoRelayNativeHostApiV3, - completion: *const NemoRelayNativeAsyncCompletion, - message: &str, -) -> NemoRelayStatus { - match HostString::text(&host.v1, message) { - Ok(message) => unsafe { (host.async_completion_reject)(completion, message.as_ptr()) }, - Err(_) => NemoRelayStatus::Internal, - } -} - -pub(crate) async fn push_stream( - host: &NemoRelayNativeHostApiV3, - stream: usize, - value: &Json, -) -> Result<(), String> { - let value = HostString::json(&host.v1, value)?; - loop { - if unsafe { (host.async_stream_is_cancelled)(stream as *const NemoRelayNativeAsyncStream) } - { - return Err("Relay caller cancelled the output stream".into()); - } - match unsafe { - (host.async_stream_push_json)( - stream as *const NemoRelayNativeAsyncStream, - value.as_ptr(), - ) - } { - NemoRelayStatus::Ok => return Ok(()), - // Native API v1 reports its bounded queue's WouldBlock state as Internal. - NemoRelayStatus::Internal => tokio::time::sleep(BACKPRESSURE_POLL).await, - status => return Err(format!("Relay rejected output stream event: {status:?}")), - } - } -} - -pub(crate) fn finish_stream( - host: &NemoRelayNativeHostApiV3, - stream: *const NemoRelayNativeAsyncStream, -) -> NemoRelayStatus { - unsafe { (host.async_stream_finish)(stream) } -} - -pub(crate) async fn reject_stream( - host: &NemoRelayNativeHostApiV3, - stream: usize, - message: &str, -) -> NemoRelayStatus { - let Ok(message) = HostString::text(&host.v1, message) else { - return NemoRelayStatus::Internal; - }; - loop { - if unsafe { (host.async_stream_is_cancelled)(stream as *const NemoRelayNativeAsyncStream) } - { - return NemoRelayStatus::InvalidArg; - } - match unsafe { - (host.async_stream_reject)( - stream as *const NemoRelayNativeAsyncStream, - message.as_ptr(), - ) - } { - NemoRelayStatus::Internal => tokio::time::sleep(BACKPRESSURE_POLL).await, - status => return status, - } - } -} - -pub(crate) unsafe fn release_completion( - host: &NemoRelayNativeHostApiV3, - completion: *const NemoRelayNativeAsyncCompletion, -) { - unsafe { (host.async_completion_release)(completion) }; -} - -pub(crate) unsafe fn release_next( - host: &NemoRelayNativeHostApiV3, - next: *const NemoRelayNativeAsyncNext, -) { - unsafe { (host.async_next_release)(next) }; -} - -pub(crate) unsafe fn release_stream( - host: &NemoRelayNativeHostApiV3, - stream: *const NemoRelayNativeAsyncStream, -) { - unsafe { (host.async_stream_release)(stream) }; -} diff --git a/crates/switchyard-nemo-relay-plugin/src/lib.rs b/crates/switchyard-nemo-relay-plugin/src/lib.rs index f3a2fd7c6..f32ed9af0 100644 --- a/crates/switchyard-nemo-relay-plugin/src/lib.rs +++ b/crates/switchyard-nemo-relay-plugin/src/lib.rs @@ -3,42 +3,23 @@ mod client; mod config; -mod executor; -mod ffi; mod runtime; mod translation; -use std::ffi::c_void; -use std::mem; -use std::panic::AssertUnwindSafe; +use std::future::Future; +use std::pin::Pin; use std::sync::Arc; +use std::task::{Context, Poll}; -use futures_util::FutureExt; use nemo_relay_plugin::{ - ConfigDiagnostic, DiagnosticLevel, Json, NEMO_RELAY_NATIVE_ABI_VERSION_ASYNC_MIDDLEWARE, - NativePlugin, NemoRelayNativeAsyncCallbackState, NemoRelayNativeAsyncCompletion, - NemoRelayNativeAsyncMiddlewareKind, NemoRelayNativeAsyncNext, NemoRelayNativeAsyncStream, - NemoRelayNativeHostApiV3, NemoRelayNativeString, NemoRelayStatus, PluginContext, + ConfigDiagnostic, DiagnosticLevel, Json, LlmJsonAsyncStream, NativePlugin, PluginContext, + PluginRuntime, }; -use serde::Deserialize; use serde_json::Map; use crate::config::SwitchyardConfig; -use crate::executor::PluginExecutor; use crate::runtime::{RoutingMark, StreamMessage, SwitchyardRuntime}; -#[derive(Deserialize)] -struct Invocation { - name: String, - request: nemo_relay_plugin::LlmRequest, -} - -struct CallbackState { - host: NemoRelayNativeHostApiV3, - runtime: Arc, - executor: PluginExecutor, -} - #[derive(Default)] struct SwitchyardPlugin; @@ -69,26 +50,13 @@ impl NativePlugin for SwitchyardPlugin { plugin_config: &Map, ctx: &mut PluginContext<'_>, ) -> nemo_relay_plugin::Result<()> { - let host_v1 = ctx.host_api(); - if host_v1.abi_version < NEMO_RELAY_NATIVE_ABI_VERSION_ASYNC_MIDDLEWARE - || host_v1.struct_size < mem::size_of::() - { - return Err( - "Switchyard requires Relay 0.7 or newer with the generic asynchronous native host table" - .into(), - ); - } - let host = unsafe { *(host_v1 as *const _ as *const NemoRelayNativeHostApiV3) }; let config = parse_config(plugin_config)?; let priority = config.priority; - let state = Arc::new(CallbackState { - host, - runtime: Arc::new(SwitchyardRuntime::new(config)?), - executor: PluginExecutor::new()?, - }); + let runtime = Arc::new(SwitchyardRuntime::new(config)?); + let plugin_runtime = ctx.runtime(); - register_buffered(ctx, priority, Arc::clone(&state))?; - register_stream(ctx, priority, state)?; + register_buffered(ctx, priority, Arc::clone(&runtime), plugin_runtime.clone())?; + register_stream(ctx, priority, runtime, plugin_runtime)?; Ok(()) } } @@ -96,51 +64,57 @@ impl NativePlugin for SwitchyardPlugin { fn register_buffered( ctx: &mut PluginContext<'_>, priority: i32, - state: Arc, + runtime: Arc, + plugin_runtime: PluginRuntime, ) -> Result<(), String> { - let user_data = Box::into_raw(Box::new(state)).cast::(); - let status = unsafe { - ctx.register_async_middleware_raw( - NemoRelayNativeAsyncMiddlewareKind::LlmExecutionIntercept, - "switchyard.run_stream.buffered", - priority, - false, - buffered_callback, - user_data, - Some(free_callback_state), - ) - }; - if status == NemoRelayStatus::Ok { - Ok(()) - } else { - Err(format!( - "failed to register Switchyard buffered execution: {status:?}" - )) - } + ctx.register_llm_execution_intercept( + "switchyard.run_stream.buffered", + priority, + move |name, request, next| { + let runtime = Arc::clone(&runtime); + let plugin_runtime = plugin_runtime.clone(); + async move { + let Some(inbound) = runtime.managed_protocol(&name) else { + return next.call(request).await; + }; + let request = runtime.decode_request(inbound, &request, false)?; + let mut marks = Vec::new(); + let response = runtime + .execute_buffered(inbound, request, &mut marks) + .await?; + emit_marks(&plugin_runtime, marks); + Ok(response) + } + }, + ) } fn register_stream( ctx: &mut PluginContext<'_>, priority: i32, - state: Arc, + runtime: Arc, + plugin_runtime: PluginRuntime, ) -> Result<(), String> { - let user_data = Box::into_raw(Box::new(state)).cast::(); - let status = unsafe { - ctx.register_async_stream_middleware_raw( - "switchyard.run_stream.streaming", - priority, - stream_callback, - user_data, - Some(free_callback_state), - ) - }; - if status == NemoRelayStatus::Ok { - Ok(()) - } else { - Err(format!( - "failed to register Switchyard streaming execution: {status:?}" - )) - } + ctx.register_llm_stream_execution_intercept( + "switchyard.run_stream.streaming", + priority, + move |name, request, next| { + let runtime = Arc::clone(&runtime); + let plugin_runtime = plugin_runtime.clone(); + async move { + let Some(inbound) = runtime.managed_protocol(&name) else { + return next.call(request).await; + }; + let request = runtime.decode_request(inbound, &request, true)?; + Ok(Box::pin(ManagedStream::new( + runtime, + plugin_runtime, + inbound, + request, + )) as LlmJsonAsyncStream) + } + }, + ) } fn parse_config(plugin_config: &Map) -> Result { @@ -159,246 +133,74 @@ fn parse_config(plugin_config: &Map) -> Result, marks: Vec) { - let Some(parent) = parent else { - return; - }; +fn emit_marks(runtime: &PluginRuntime, marks: Vec) { for mark in marks { - if let Err(error) = parent.emit_mark(&mark.name, &mark.data, &mark.metadata) { - eprintln!( - "Switchyard could not emit routing mark {:?}: {error}", - mark.name - ); - } + emit_mark(runtime, mark); } } -async fn execute_managed_stream( - state: &CallbackState, - output: usize, - inbound: switchyard_protocol::WireFormat, - request: switchyard_protocol::Request, - parent: Option<&ffi::ParentScope>, -) -> Result<(), String> { - let (sender, receiver) = async_channel::bounded(32); - let runtime = Arc::clone(&state.runtime); - let execution = async move { runtime.execute_stream(inbound, request, &sender).await }; - let forwarding = async { - while let Ok(message) = receiver.recv().await { - match message { - StreamMessage::Mark(mark) => emit_marks(parent, vec![mark]), - StreamMessage::Event(event) => { - ffi::push_stream(&state.host, output, &event).await? - } - } - } - Ok(()) - }; - tokio::try_join!(execution, forwarding)?; - Ok(()) +fn emit_mark(runtime: &PluginRuntime, mark: RoutingMark) { + if let Err(error) = runtime.emit_mark(&mark.name, Some(&mark.data), Some(&mark.metadata)) { + eprintln!( + "Switchyard could not emit routing mark {:?}: {error}", + mark.name + ); + } } -unsafe extern "C" fn free_callback_state(user_data: *mut c_void) { - if !user_data.is_null() { - unsafe { drop(Box::from_raw(user_data.cast::>())) }; - } +type StreamExecution = Pin> + Send>>; + +struct ManagedStream { + execution: Option, + messages: Pin>>, + plugin_runtime: PluginRuntime, } -unsafe extern "C" fn buffered_callback( - user_data: *mut c_void, - invocation_json: *const NemoRelayNativeString, - next: *const NemoRelayNativeAsyncNext, - completion: *const NemoRelayNativeAsyncCompletion, -) -> u32 { - if user_data.is_null() || completion.is_null() || next.is_null() { - return NemoRelayNativeAsyncCallbackState::Complete as u32; - } - let state = unsafe { &*user_data.cast::>() }.clone(); - let invocation = ffi::read_json(&state.host.v1, invocation_json).and_then(|value| { - serde_json::from_value::(value).map_err(|error| error.to_string()) - }); - let next = next as usize; - let completion = completion as usize; - let invocation = match invocation { - Ok(invocation) => invocation, - Err(error) => { - let _ = ffi::reject_completion( - &state.host, - completion as *const NemoRelayNativeAsyncCompletion, - &format!("invalid Relay LLM invocation: {error}"), - ); - unsafe { - ffi::release_next(&state.host, next as *const NemoRelayNativeAsyncNext); - ffi::release_completion( - &state.host, - completion as *const NemoRelayNativeAsyncCompletion, - ); - } - return NemoRelayNativeAsyncCallbackState::Pending as u32; - } - }; - let Some(inbound) = state.runtime.managed_protocol(&invocation.name) else { - if let Err(error) = - ffi::invoke_next_buffered(&state.host, next, completion, &invocation.request) - { - let _ = ffi::reject_completion( - &state.host, - completion as *const NemoRelayNativeAsyncCompletion, - &error, - ); +impl ManagedStream { + fn new( + runtime: Arc, + plugin_runtime: PluginRuntime, + inbound: switchyard_protocol::WireFormat, + request: switchyard_protocol::Request, + ) -> Self { + let (sender, messages) = async_channel::bounded(32); + let execution = async move { runtime.execute_stream(inbound, request, &sender).await }; + Self { + execution: Some(Box::pin(execution)), + messages: Box::pin(messages), + plugin_runtime, } - unsafe { - ffi::release_next(&state.host, next as *const NemoRelayNativeAsyncNext); - ffi::release_completion( - &state.host, - completion as *const NemoRelayNativeAsyncCompletion, - ); - } - return NemoRelayNativeAsyncCallbackState::Pending as u32; - }; - let request = match state - .runtime - .decode_request(inbound, &invocation.request, false) - { - Ok(request) => request, - Err(error) => { - let _ = ffi::reject_completion( - &state.host, - completion as *const NemoRelayNativeAsyncCompletion, - &error, - ); - unsafe { - ffi::release_next(&state.host, next as *const NemoRelayNativeAsyncNext); - ffi::release_completion( - &state.host, - completion as *const NemoRelayNativeAsyncCompletion, - ); - } - return NemoRelayNativeAsyncCallbackState::Pending as u32; - } - }; - let parent = ffi::ParentScope::capture(&state.host.v1); - let task_state = Arc::clone(&state); - state.executor.spawn(async move { - let execution = AssertUnwindSafe(async { - let mut marks = Vec::new(); - let result = task_state - .runtime - .execute_buffered(inbound, request, &mut marks) - .await; - emit_marks(parent.as_ref(), marks); - result - }) - .catch_unwind(); - tokio::pin!(execution); - let result = tokio::select! { - biased; - () = ffi::wait_for_completion_cancellation(&task_state.host, completion) => None, - result = &mut execution => Some( - result.unwrap_or_else(|_| Err("Switchyard buffered execution panicked".into())) - ), - }; + } +} - let completion_ptr = completion as *const NemoRelayNativeAsyncCompletion; - if let Some(result) = result { - match result { - Ok(response) => { - let _ = ffi::resolve_completion(&task_state.host, completion_ptr, &response); - } - Err(error) => { - let _ = ffi::reject_completion(&task_state.host, completion_ptr, &error); +impl futures_util::Stream for ManagedStream { + type Item = Result; + + fn poll_next(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { + if let Some(execution) = self.execution.as_mut() { + match execution.as_mut().poll(cx) { + Poll::Ready(Ok(())) => self.execution = None, + Poll::Ready(Err(error)) => { + self.execution = None; + return Poll::Ready(Some(Err(error))); } + Poll::Pending => {} } } - unsafe { - ffi::release_next(&task_state.host, next as *const NemoRelayNativeAsyncNext); - ffi::release_completion(&task_state.host, completion_ptr); - } - }); - NemoRelayNativeAsyncCallbackState::Pending as u32 -} -unsafe extern "C" fn stream_callback( - user_data: *mut c_void, - invocation_json: *const NemoRelayNativeString, - next: *const NemoRelayNativeAsyncNext, - output: *const NemoRelayNativeAsyncStream, -) -> u32 { - if user_data.is_null() || output.is_null() || next.is_null() { - return NemoRelayNativeAsyncCallbackState::Complete as u32; - } - let state = unsafe { &*user_data.cast::>() }.clone(); - let invocation = ffi::read_json(&state.host.v1, invocation_json).and_then(|value| { - serde_json::from_value::(value).map_err(|error| error.to_string()) - }); - let managed_protocol = invocation - .as_ref() - .ok() - .and_then(|invocation| state.runtime.managed_protocol(&invocation.name)); - let parent = managed_protocol.and_then(|_| ffi::ParentScope::capture(&state.host.v1)); - let next = next as usize; - let output = output as usize; - let task_state = Arc::clone(&state); - state.executor.spawn(async move { - let execution = AssertUnwindSafe(async { - match invocation { - Ok(invocation) => { - if let Some(inbound) = managed_protocol { - match task_state - .runtime - .decode_request(inbound, &invocation.request, true) - { - Ok(request) => { - execute_managed_stream( - &task_state, - output, - inbound, - request, - parent.as_ref(), - ) - .await - } - Err(error) => Err(error), - } - } else { - ffi::invoke_next_stream(&task_state.host, next, output, &invocation.request) - .await - } + loop { + match self.messages.as_mut().poll_next(cx) { + Poll::Ready(Some(StreamMessage::Mark(mark))) => { + emit_mark(&self.plugin_runtime, mark) } - Err(error) => Err(format!("invalid Relay LLM stream invocation: {error}")), - } - }) - .catch_unwind(); - tokio::pin!(execution); - let result = tokio::select! { - biased; - () = ffi::wait_for_stream_cancellation(&task_state.host, output) => None, - result = &mut execution => Some( - result.unwrap_or_else(|_| Err("Switchyard streaming execution panicked".into())) - ), - }; - - match result { - Some(Ok(())) => { - let _ = ffi::finish_stream( - &task_state.host, - output as *const NemoRelayNativeAsyncStream, - ); - } - Some(Err(error)) => { - let _ = ffi::reject_stream(&task_state.host, output, &error).await; + Poll::Ready(Some(StreamMessage::Event(event))) => { + return Poll::Ready(Some(Ok(event))); + } + Poll::Ready(None) => return Poll::Ready(None), + Poll::Pending => return Poll::Pending, } - None => {} } - unsafe { - ffi::release_next(&task_state.host, next as *const NemoRelayNativeAsyncNext); - ffi::release_stream( - &task_state.host, - output as *const NemoRelayNativeAsyncStream, - ); - } - }); - NemoRelayNativeAsyncCallbackState::Pending as u32 + } } nemo_relay_plugin::nemo_relay_plugin!(nemo_relay_register_plugin, SwitchyardPlugin::default); From 1980009a16968cc1662daaff86ed7fb55851752d Mon Sep 17 00:00:00 2001 From: Bryan Bednarski Date: Wed, 19 Aug 2026 19:18:20 -0700 Subject: [PATCH 4/4] refactor(relay): delegate HTTP retries to client Signed-off-by: Bryan Bednarski --- crates/switchyard-nemo-relay-plugin/README.md | 22 +- .../config.schema.json | 7 - .../src/client.rs | 67 +++- .../src/config.rs | 11 - .../src/config/tests.rs | 23 +- .../src/runtime.rs | 299 ++++++------------ .../src/runtime/tests.rs | 64 ++-- 7 files changed, 236 insertions(+), 257 deletions(-) diff --git a/crates/switchyard-nemo-relay-plugin/README.md b/crates/switchyard-nemo-relay-plugin/README.md index d5bc92be5..20abab1d7 100644 --- a/crates/switchyard-nemo-relay-plugin/README.md +++ b/crates/switchyard-nemo-relay-plugin/README.md @@ -115,7 +115,7 @@ algorithm. | Streaming responses | Supported | Supported after the judge selects a target | Conditional: an unlatched weak stream is aggregated before the judge runs | Supported after the signal cascade selects a target | | Retained routing state | No selection affinity; context-overflow eviction can use session identity | Optional session affinity and message-hash fallback | Confirmation streak and strong latch require stable session identity | No classifier affinity; context-overflow eviction can use session identity | | Router-specific prompts | Not applicable | Optional judge prompt | Optional escalation-judge prompt | Optional tier prompts, handoff notes, and classifier prompt | -| Relay decision marks | Algorithm, attempt, selected target, reasoning, answer-call status, and identity | Algorithm, attempt, selected target, reasoning, answer-call status, and identity | Algorithm, attempt, selected target, reasoning, answer-call status, and identity | Algorithm, attempt, selected target, reasoning, answer-call status, and identity | +| Relay decision marks | Algorithm, attempt, selected target, and identity | Algorithm, attempt, selected target, and identity | Algorithm, attempt, selected target, and identity | Algorithm, attempt, selected target, and identity | | ATOF routing-LLM usage | Not applicable unless a failed candidate is replaced | Judge calls, plus failed candidates | Judge calls and discarded weak candidates | Optional classifier judge calls, plus failed candidates | Anthropic Messages is supported for callers and serving targets, but not for a @@ -155,17 +155,13 @@ provider response that omits usage, has `usage = null`. Consumers can therefore add these marks to the outer LLM usage to measure total request compute without double-counting the serving model. -The plugin owns the outer routing retry loop. Each retry starts a fresh libsy -run. Random routing draws again; an algorithm configured with persistent state, -such as classifier session affinity, may intentionally retain its assignment. -Each target's built-in HTTP retry count is set to zero to avoid retrying a -failed target invisibly before reselection. A random target with `weight = 0` -is fallback-only and is not considered by the algorithm. Trusted fallback is -attempted at most once and, for streaming responses, only before the first -caller event is emitted. Outer routing retries use exponential backoff starting -at 250 milliseconds and capped at 2 seconds. They do not currently honor -provider `Retry-After` headers because the client error contract does not expose -that metadata to the routing loop. +`switchyard-llm-client` owns provider retries. Every target, including the +trusted fallback target, uses its default of two additional attempts for +transient provider failures and honors capped `Retry-After` delays. A retry +stays on the selected target and does not rerun the routing algorithm. A random +target with `weight = 0` is fallback-only and is not considered by the +algorithm. Trusted fallback is attempted at most once and, for streaming +responses, only before the first caller event is emitted. ## Translation and stream fidelity @@ -203,8 +199,6 @@ manifest = "/opt/switchyard-relay-plugin/relay-plugin.toml" [plugins.dynamic.config] version = 2 priority = 0 -max_retries = 3 - [plugins.dynamic.config.algorithm] kind = "random" seed = 42 diff --git a/crates/switchyard-nemo-relay-plugin/config.schema.json b/crates/switchyard-nemo-relay-plugin/config.schema.json index 63db7b40c..8ea12cece 100644 --- a/crates/switchyard-nemo-relay-plugin/config.schema.json +++ b/crates/switchyard-nemo-relay-plugin/config.schema.json @@ -14,13 +14,6 @@ "type": "integer", "default": 0 }, - "max_retries": { - "type": "integer", - "minimum": 0, - "maximum": 10, - "default": 3, - "description": "Routing retries after the initial libsy run. Every retry starts a fresh run." - }, "algorithm": { "description": "In-process random, capability, escalation, or stage-router configuration.", "oneOf": [ diff --git a/crates/switchyard-nemo-relay-plugin/src/client.rs b/crates/switchyard-nemo-relay-plugin/src/client.rs index c3d345dad..2db0c933b 100644 --- a/crates/switchyard-nemo-relay-plugin/src/client.rs +++ b/crates/switchyard-nemo-relay-plugin/src/client.rs @@ -7,7 +7,9 @@ use std::collections::BTreeMap; use async_trait::async_trait; use serde_json::Value as Json; -use switchyard_llm_client::{Backend, HttpBackendConfig, ModelConfig, TranslatingLlmClient}; +use switchyard_llm_client::{ + Backend, DEFAULT_MAX_RETRIES, HttpBackendConfig, ModelConfig, TranslatingLlmClient, +}; use switchyard_protocol::{ LlmClientError, ModelId, Request, Response, RoutedLlmClient, WireFormat, }; @@ -47,9 +49,7 @@ impl TargetClient { forward_auth: false, extra_headers: headers, extra_body, - // Routing retries belong to the plugin: every retry must start a - // fresh libsy run and obtain a fresh decision. - max_retries: 0, + max_retries: DEFAULT_MAX_RETRIES, }; let backend = match target_format { WireFormat::OpenAiChat => Backend::OpenAiChat(backend_config), @@ -106,9 +106,15 @@ impl RoutedLlmClient for TargetClient { #[cfg(test)] mod tests { + use std::io::{Read, Write}; + use std::net::TcpListener; + use std::thread; + use super::*; use serde_json::json; - use switchyard_protocol::{LlmRequest, Metadata, PreservationMetadata, ProviderExtensions}; + use switchyard_protocol::{ + LlmRequest, LlmResponse, Metadata, PreservationMetadata, ProviderExtensions, text_request, + }; fn client(format: WireFormat) -> TargetClient { TargetClient::new( @@ -220,4 +226,55 @@ mod tests { .all(|body| body.get("extra_body").is_none()) ); } + + #[tokio::test] + async fn target_client_retries_transient_provider_failures() { + let listener = TcpListener::bind("127.0.0.1:0").expect("bind provider stub"); + let address = listener.local_addr().expect("read provider stub address"); + let server = thread::spawn(move || -> std::io::Result<()> { + let response_body = "{\"id\":\"chatcmpl-1\",\"model\":\"provider/model\",\"choices\":[{\"index\":0,\"message\":{\"role\":\"assistant\",\"content\":\"recovered\"},\"finish_reason\":\"stop\"}],\"usage\":{}}"; + let responses = [ + "HTTP/1.1 503 Service Unavailable\r\nContent-Length: 11\r\nConnection: close\r\n\r\nunavailable".to_string(), + "HTTP/1.1 503 Service Unavailable\r\nContent-Length: 11\r\nConnection: close\r\n\r\nunavailable".to_string(), + format!( + "HTTP/1.1 200 OK\r\nContent-Type: application/json\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{response_body}", + response_body.len() + ), + ]; + for response in responses { + let (mut stream, _) = listener.accept()?; + let mut request = [0_u8; 1024]; + if stream.read(&mut request)? == 0 { + return Err(std::io::Error::new( + std::io::ErrorKind::UnexpectedEof, + "client closed before sending a request", + )); + } + stream.write_all(response.as_bytes())?; + } + Ok(()) + }); + let client = TargetClient::new( + "provider/model".into(), + WireFormat::OpenAiChat, + format!("http://{address}/v1"), + BTreeMap::new(), + BTreeMap::new(), + false, + ) + .expect("build target client"); + let response = client + .call(Request { + llm_request: text_request(Some("route".into()), "hello"), + ..Request::default() + }) + .await + .expect("third provider attempt succeeds"); + + assert!(matches!(response.llm_response, LlmResponse::Agg(_))); + server + .join() + .expect("provider stub panicked") + .expect("provider stub failed"); + } } diff --git a/crates/switchyard-nemo-relay-plugin/src/config.rs b/crates/switchyard-nemo-relay-plugin/src/config.rs index 8e986c888..23dbe2d43 100644 --- a/crates/switchyard-nemo-relay-plugin/src/config.rs +++ b/crates/switchyard-nemo-relay-plugin/src/config.rs @@ -292,15 +292,12 @@ pub(crate) struct SwitchyardConfig { version: u32, #[serde(default)] pub(crate) priority: i32, - #[serde(default = "default_max_retries")] - max_retries: u32, algorithm: AlgorithmConfig, targets: BTreeMap, default_targets: BTreeMap, } pub(crate) struct PreparedConfig { - pub(crate) max_retries: u32, pub(crate) algorithm: Arc, pub(crate) targets: BTreeMap, pub(crate) default_targets: BTreeMap, @@ -319,9 +316,6 @@ impl SwitchyardConfig { self.version )); } - if self.max_retries > 10 { - return Err("max_retries must not exceed 10".into()); - } if self.targets.is_empty() { return Err("targets must not be empty".into()); } @@ -358,7 +352,6 @@ impl SwitchyardConfig { .collect::, _>>()?; let algorithm = self.build_algorithm(Some(&targets))?; Ok(PreparedConfig { - max_retries: self.max_retries, algorithm, targets, default_targets: self.default_targets, @@ -567,10 +560,6 @@ fn is_forbidden_target_header(name: &str) -> bool { ) || name.starts_with("x-nemo-relay-internal-") } -const fn default_max_retries() -> u32 { - 3 -} - const fn default_weight() -> f64 { 1.0 } diff --git a/crates/switchyard-nemo-relay-plugin/src/config/tests.rs b/crates/switchyard-nemo-relay-plugin/src/config/tests.rs index 1a1153e9f..06b5d6d31 100644 --- a/crates/switchyard-nemo-relay-plugin/src/config/tests.rs +++ b/crates/switchyard-nemo-relay-plugin/src/config/tests.rs @@ -21,7 +21,6 @@ fn config() -> SwitchyardConfig { SwitchyardConfig { version: 2, priority: 0, - max_retries: 3, algorithm: AlgorithmConfig::Random { seed: Some(42) }, targets: BTreeMap::from([ ( @@ -171,6 +170,28 @@ fn schema_required_contract_fields_do_not_default_during_deserialization() { } } +#[test] +fn plugin_retry_budget_is_not_configurable() { + let value = json!({ + "version": 2, + "max_retries": 3, + "algorithm": {"kind": "random"}, + "targets": { + "chat": { + "model": "provider/chat", + "protocol": "openai_chat", + "base_url": "https://provider.example/v1" + } + }, + "default_targets": {"openai_chat": "chat"} + }); + + let error = serde_json::from_value::(value) + .err() + .expect("plugin retry budget must be rejected"); + assert!(error.to_string().contains("max_retries")); +} + #[test] fn unknown_target_fields_are_rejected() { let value = json!({ diff --git a/crates/switchyard-nemo-relay-plugin/src/runtime.rs b/crates/switchyard-nemo-relay-plugin/src/runtime.rs index cb15180d3..48914eab3 100644 --- a/crates/switchyard-nemo-relay-plugin/src/runtime.rs +++ b/crates/switchyard-nemo-relay-plugin/src/runtime.rs @@ -3,7 +3,6 @@ use std::collections::{BTreeMap, HashMap}; use std::sync::{Arc, Mutex}; -use std::time::Duration; use futures_util::{StreamExt, stream}; use nemo_relay_plugin::{Json, LlmRequest as RelayRequest}; @@ -18,9 +17,6 @@ use switchyard_translation::{TranslationEngine, encode_stream}; use crate::config::{PreparedTargetBinding, SwitchyardConfig, protocol_from_call}; use crate::translation; -const INITIAL_RETRY_BACKOFF: Duration = Duration::from_millis(250); -const MAX_RETRY_BACKOFF: Duration = Duration::from_secs(2); - #[derive(Debug)] pub(crate) struct RoutingMark { pub(crate) name: String, @@ -35,7 +31,6 @@ pub(crate) enum StreamMessage { } pub(crate) struct SwitchyardRuntime { - max_retries: u32, algorithm: Arc, targets: BTreeMap, default_targets: BTreeMap, @@ -46,7 +41,6 @@ impl SwitchyardRuntime { pub(crate) fn new(config: SwitchyardConfig) -> Result { let prepared = config.prepare()?; Ok(Self { - max_retries: prepared.max_retries, algorithm: prepared.algorithm, targets: prepared.targets, default_targets: prepared.default_targets, @@ -95,49 +89,33 @@ impl SwitchyardRuntime { marks: &mut Vec, ) -> Result { let metadata = identity_metadata(request.metadata.as_ref()); - let max_attempts = self.max_retries + 1; - let mut attempt = 1; - loop { - self.mark( - marks, - "switchyard.routing.requested", - json!({"algorithm": self.algorithm.name(), "attempt": attempt}), - &metadata, - ); - let result = self - .drive(request.clone(), attempt, marks, &metadata) - .await - .and_then(|response| { - finalize_buffered_response(&self.translation, inbound, response) - .map_err(|source| LibsyError::client_call("return_to_agent", source)) - }); - match result { - Ok(response) => return Ok(response), - Err(failure) if libsy_error_retryable(&failure) && attempt < max_attempts => { - self.mark( - marks, - "switchyard.routing.retry", - failure_mark_data(attempt, &failure), - &metadata, - ); - sleep_before_retry(attempt).await; - attempt += 1; - } - Err(failure) => { - self.mark( - marks, - "switchyard.routing.error", - failure_mark_data(attempt, &failure), - &metadata, - ); - let response = self - .fallback_response(inbound, request, marks, &metadata) - .await?; - return finalize_buffered_response(&self.translation, inbound, response) - .map_err(|error| { - public_response_failure("trusted fallback response", &error) - }); - } + self.mark( + marks, + "switchyard.routing.requested", + json!({"algorithm": self.algorithm.name(), "attempt": 1}), + &metadata, + ); + let result = self + .drive(request.clone(), 1, marks, &metadata) + .await + .and_then(|response| { + finalize_buffered_response(&self.translation, inbound, response) + .map_err(|source| LibsyError::client_call("return_to_agent", source)) + }); + match result { + Ok(response) => Ok(response), + Err(failure) => { + self.mark( + marks, + "switchyard.routing.error", + failure_mark_data(1, &failure), + &metadata, + ); + let response = self + .fallback_response(inbound, request, marks, &metadata) + .await?; + finalize_buffered_response(&self.translation, inbound, response) + .map_err(|error| public_response_failure("trusted fallback response", &error)) } } } @@ -149,38 +127,21 @@ impl SwitchyardRuntime { output: &async_channel::Sender, ) -> Result<(), String> { let metadata = identity_metadata(request.metadata.as_ref()); - let max_attempts = self.max_retries + 1; - let mut attempt = 1; let mut marks = Vec::new(); - 'attempts: loop { - self.mark( - &mut marks, - "switchyard.routing.requested", - json!({"algorithm": self.algorithm.name(), "attempt": attempt}), - &metadata, - ); - let (response, mut fallback_used) = match self - .drive(request.clone(), attempt, &mut marks, &metadata) - .await - { + self.mark( + &mut marks, + "switchyard.routing.requested", + json!({"algorithm": self.algorithm.name(), "attempt": 1}), + &metadata, + ); + let (response, mut fallback_used) = + match self.drive(request.clone(), 1, &mut marks, &metadata).await { Ok(response) => (response, false), - Err(failure) if libsy_error_retryable(&failure) && attempt < max_attempts => { - self.mark( - &mut marks, - "switchyard.routing.retry", - failure_mark_data(attempt, &failure), - &metadata, - ); - send_marks(output, &mut marks).await?; - sleep_before_retry(attempt).await; - attempt += 1; - continue; - } Err(failure) => { self.mark( &mut marks, "switchyard.routing.error", - failure_mark_data(attempt, &failure), + failure_mark_data(1, &failure), &metadata, ); let fallback = self @@ -190,118 +151,84 @@ impl SwitchyardRuntime { (fallback?, true) } }; - send_marks(output, &mut marks).await?; - - let mut events = match returned_events(response, inbound).await { - Ok(events) => events, - Err(failure) - if !fallback_used - && libsy_error_retryable(&failure) - && attempt < max_attempts => - { - self.mark( - &mut marks, - "switchyard.routing.retry", - failure_mark_data(attempt, &failure), - &metadata, - ); - send_marks(output, &mut marks).await?; - sleep_before_retry(attempt).await; - attempt += 1; - continue; + send_marks(output, &mut marks).await?; + + let mut events = match returned_events(response, inbound).await { + Ok(events) => events, + Err(failure) if !fallback_used => { + self.mark( + &mut marks, + "switchyard.routing.error", + failure_mark_data(1, &failure), + &metadata, + ); + fallback_used = true; + let fallback = self + .fallback_response(inbound, request.clone(), &mut marks, &metadata) + .await; + send_marks(output, &mut marks).await?; + let fallback = fallback?; + returned_events(fallback, inbound) + .await + .map_err(|error| public_libsy_failure("trusted fallback stream", &error))? + } + Err(failure) => { + return Err(public_libsy_failure("trusted fallback stream", &failure)); + } + }; + + let mut committed = false; + while let Some(item) = events.next().await { + match item { + Ok(event) => { + send_event(output, event).await?; + committed = true; } - Err(failure) if !fallback_used => { + Err(failure) if !fallback_used && !committed => { self.mark( &mut marks, "switchyard.routing.error", - failure_mark_data(attempt, &failure), + failure_mark_data(1, &failure), &metadata, ); - fallback_used = true; let fallback = self .fallback_response(inbound, request.clone(), &mut marks, &metadata) .await; send_marks(output, &mut marks).await?; let fallback = fallback?; - returned_events(fallback, inbound) + let mut fallback = returned_events(fallback, inbound) .await - .map_err(|error| public_libsy_failure("trusted fallback stream", &error))? + .map_err(|error| public_libsy_failure("trusted fallback stream", &error))?; + while let Some(item) = fallback.next().await { + let event = item.map_err(|error| { + public_libsy_failure("trusted fallback stream", &error) + })?; + send_event(output, event).await?; + } + return Ok(()); } - Err(failure) => { + Err(failure) if !committed => { return Err(public_libsy_failure("trusted fallback stream", &failure)); } - }; - - let mut committed = false; - while let Some(item) = events.next().await { - match item { - Ok(event) => { - send_event(output, event).await?; - committed = true; - } - Err(failure) - if !fallback_used - && !committed - && libsy_error_retryable(&failure) - && attempt < max_attempts => - { - self.mark( - &mut marks, - "switchyard.routing.retry", - failure_mark_data(attempt, &failure), - &metadata, - ); - send_marks(output, &mut marks).await?; - sleep_before_retry(attempt).await; - attempt += 1; - continue 'attempts; - } - Err(failure) if !fallback_used && !committed => { - self.mark( - &mut marks, - "switchyard.routing.error", - failure_mark_data(attempt, &failure), - &metadata, - ); - let fallback = self - .fallback_response(inbound, request.clone(), &mut marks, &metadata) - .await; - send_marks(output, &mut marks).await?; - let fallback = fallback?; - let mut fallback = - returned_events(fallback, inbound).await.map_err(|error| { - public_libsy_failure("trusted fallback stream", &error) - })?; - while let Some(item) = fallback.next().await { - let event = item.map_err(|error| { - public_libsy_failure("trusted fallback stream", &error) - })?; - send_event(output, event).await?; - } - return Ok(()); - } - Err(failure) if !committed => { - return Err(public_libsy_failure("trusted fallback stream", &failure)); - } - Err(failure) => { - self.mark( - &mut marks, - "switchyard.routing.error", - failure_mark_data(attempt, &failure), - &metadata, - ); - send_marks(output, &mut marks).await?; - return Err(public_libsy_failure( - "Switchyard stream failed after response commitment", - &failure, - )); - } + Err(failure) => { + self.mark( + &mut marks, + "switchyard.routing.error", + failure_mark_data(1, &failure), + &metadata, + ); + send_marks(output, &mut marks).await?; + return Err(public_libsy_failure( + "Switchyard stream failed after response commitment", + &failure, + )); } } - if committed { - return Ok(()); - } - return Err("Switchyard response stream produced no caller events".into()); + } + if committed { + Ok(()) + } else { + Err("Switchyard response stream produced no caller events".into()) } } @@ -525,38 +452,8 @@ async fn returned_events( }))) } -fn libsy_error_retryable(error: &LibsyError) -> bool { - let LibsyError::ClientCall { source, .. } = error else { - return false; - }; - match source { - LlmClientError::UpstreamHttp { status, .. } => { - matches!(status.as_u16(), 408 | 425 | 429 | 500 | 502 | 503 | 504) - } - LlmClientError::Transport { .. } | LlmClientError::Timeout { .. } => true, - _ => false, - } -} - -fn retry_backoff(attempt: u32) -> Duration { - let exponent = attempt.saturating_sub(1).min(3); - INITIAL_RETRY_BACKOFF - .saturating_mul(1_u32 << exponent) - .min(MAX_RETRY_BACKOFF) -} - -async fn sleep_before_retry(attempt: u32) { - tokio::time::sleep(retry_backoff(attempt)).await; -} - fn failure_mark_data(attempt: u32, failure: &LibsyError) -> Json { - let mut data = Map::from_iter([ - ("attempt".into(), Json::from(attempt)), - ( - "retryable".into(), - Json::from(libsy_error_retryable(failure)), - ), - ]); + let mut data = Map::from_iter([("attempt".into(), Json::from(attempt))]); match failure { LibsyError::ClientCall { source: LlmClientError::UpstreamHttp { status, .. }, diff --git a/crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs b/crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs index a4266764f..b0dcaa39c 100644 --- a/crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs +++ b/crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs @@ -9,7 +9,8 @@ use switchyard_libsy::{ TaskClassifierConfig, }; use switchyard_protocol::{ - LlmResponseStream, ModelId, RoutedLlmClient, Usage, text_request, text_response, + LlmResponseChunk, LlmResponseStream, LlmResponseStreamEvent, ModelId, RoutedLlmClient, Usage, + text_request, text_response, }; use super::*; @@ -19,6 +20,7 @@ enum ScriptedBehavior { EmptyBuffered, EmptyStream, FailingStream, + PartialThenFailure, TransportFailure(&'static str), } @@ -71,6 +73,21 @@ impl RoutedLlmClient for ScriptedClient { metadata: None, }) } + ScriptedBehavior::PartialThenFailure => { + let stream: LlmResponseStream = Box::pin(stream::iter(vec![ + Ok(LlmResponseStreamEvent::from(LlmResponseChunk::TextDelta { + index: 0, + text: "partial".into(), + })), + Err(LlmClientError::Transport { + source: Box::new(std::io::Error::other("stream failed after a chunk")), + }), + ])); + Ok(Response { + llm_response: LlmResponse::Stream(stream), + metadata: None, + }) + } ScriptedBehavior::TransportFailure(message) => Err(LlmClientError::Transport { source: Box::new(std::io::Error::other(message)), }), @@ -111,7 +128,6 @@ fn runtime_with_algorithm_clients( ); } SwitchyardRuntime { - max_retries: 0, algorithm, targets, default_targets: BTreeMap::from([(protocol, "fallback".into())]), @@ -271,7 +287,6 @@ async fn buffered_finalization_failure_uses_fallback_once() { let selected = scripted(ScriptedBehavior::EmptyStream); let fallback = scripted(ScriptedBehavior::EmptyBuffered); let runtime = SwitchyardRuntime { - max_retries: 1, algorithm: Arc::new(Passthrough::new(ModelId::from("selected"))), targets: BTreeMap::from([ ( @@ -300,16 +315,10 @@ async fn buffered_finalization_failure_uses_fallback_once() { assert!(response.is_object()); assert_eq!(selected.calls.load(Ordering::Relaxed), 1); assert_eq!(fallback.calls.load(Ordering::Relaxed), 1); - assert!( - !marks - .iter() - .any(|mark| mark.name == "switchyard.routing.retry") - ); let error = marks .iter() .find(|mark| mark.name == "switchyard.routing.error") .expect("finalization failure should emit an error mark"); - assert_eq!(error.data["retryable"], false); assert_eq!(error.data["non_http_kind"], "invalid_response"); assert_eq!( marks @@ -371,7 +380,6 @@ async fn invalid_selected_stream_does_not_invoke_failing_fallback_twice() { let selected = scripted(ScriptedBehavior::EmptyStream); let fallback = scripted(ScriptedBehavior::FailingStream); let runtime = SwitchyardRuntime { - max_retries: 0, algorithm: Arc::new(Passthrough::new(ModelId::from("selected"))), targets: BTreeMap::from([ ( @@ -407,7 +415,6 @@ async fn failing_fallback_call_flushes_error_and_fallback_marks() { let selected = scripted(ScriptedBehavior::EmptyStream); let fallback = scripted(ScriptedBehavior::TransportFailure("fallback call failed")); let runtime = SwitchyardRuntime { - max_retries: 0, algorithm: Arc::new(Passthrough::new(ModelId::from("selected"))), targets: BTreeMap::from([ ( @@ -453,13 +460,34 @@ async fn failing_fallback_call_flushes_error_and_fallback_marks() { ); } -#[test] -fn retry_backoff_increases_exponentially_and_is_capped() { - assert_eq!(retry_backoff(1), Duration::from_millis(250)); - assert_eq!(retry_backoff(2), Duration::from_millis(500)); - assert_eq!(retry_backoff(3), Duration::from_secs(1)); - assert_eq!(retry_backoff(4), Duration::from_secs(2)); - assert_eq!(retry_backoff(u32::MAX), Duration::from_secs(2)); +#[tokio::test] +async fn committed_stream_failure_does_not_fallback() { + let selected = scripted(ScriptedBehavior::PartialThenFailure); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let runtime = runtime_with_algorithm_clients( + Arc::new(Passthrough::new(ModelId::from("selected"))), + fallback.clone(), + WireFormat::OpenAiChat, + vec![("selected", selected.clone())], + ); + let (output, messages) = async_channel::bounded(32); + + let error = runtime + .execute_stream(WireFormat::OpenAiChat, Request::default(), &output) + .await + .expect_err("committed stream failures must reject the stream"); + + assert_eq!( + error, + "Switchyard stream failed after response commitment: provider transport failure" + ); + assert_eq!(selected.calls.load(Ordering::Relaxed), 1); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 0); + let mut emitted_event = false; + while let Ok(message) = messages.try_recv() { + emitted_event |= matches!(message, StreamMessage::Event(_)); + } + assert!(emitted_event); } #[tokio::test]