From 1a44602a66ac76a7faee3f8bef2c0f7886700b89 Mon Sep 17 00:00:00 2001 From: Owen McGirr Date: Sat, 3 Oct 2026 20:13:22 +0100 Subject: [PATCH 1/8] Add isolated SmolLM2 companion and freeze evaluation protocol (#17) --- .gitignore | 1 + neural/Cargo.lock | 1853 ++++++++++++++++++++++++++ neural/Cargo.toml | 36 + neural/MODEL_LICENSE.txt | 202 +++ neural/evaluation-protocol.json | 12 + neural/fixtures/general-writing.json | 58 + neural/fixtures/regression.json | 106 ++ neural/model-bundle.json | 35 + neural/source-manifest.json | 26 + neural/src/bundle.rs | 49 + neural/src/lib.rs | 282 ++++ neural/src/main.rs | 198 +++ neural/src/process.rs | 196 +++ neural/src/protocol.rs | 80 ++ neural/tests/fake_worker.rs | 64 + neural/tests/lifecycle.rs | 181 +++ neural/worker/Cargo.toml | 19 + neural/worker/src/bin/quantize.rs | 143 ++ neural/worker/src/main.rs | 74 + neural/worker/src/model.rs | 144 ++ scripts/neural_bundle.py | 54 + scripts/neural_evaluate.py | 228 ++++ 22 files changed, 4041 insertions(+) create mode 100644 neural/Cargo.lock create mode 100644 neural/Cargo.toml create mode 100644 neural/MODEL_LICENSE.txt create mode 100644 neural/evaluation-protocol.json create mode 100644 neural/fixtures/general-writing.json create mode 100644 neural/fixtures/regression.json create mode 100644 neural/model-bundle.json create mode 100644 neural/source-manifest.json create mode 100644 neural/src/bundle.rs create mode 100644 neural/src/lib.rs create mode 100644 neural/src/main.rs create mode 100644 neural/src/process.rs create mode 100644 neural/src/protocol.rs create mode 100644 neural/tests/fake_worker.rs create mode 100644 neural/tests/lifecycle.rs create mode 100644 neural/worker/Cargo.toml create mode 100644 neural/worker/src/bin/quantize.rs create mode 100644 neural/worker/src/main.rs create mode 100644 neural/worker/src/model.rs create mode 100644 scripts/neural_bundle.py create mode 100644 scripts/neural_evaluate.py diff --git a/.gitignore b/.gitignore index 0ea5c82..1a0c44e 100644 --- a/.gitignore +++ b/.gitignore @@ -1,4 +1,5 @@ /target/ +/neural/target/ /data/ /artifacts/ *.sqlite* diff --git a/neural/Cargo.lock b/neural/Cargo.lock new file mode 100644 index 0000000..14aecdd --- /dev/null +++ b/neural/Cargo.lock @@ -0,0 +1,1853 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "ahash" +version = "0.8.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5a15f179cd60c4584b8a8c596927aadc462e27f2ca70c04e0071964a73ba7a75" +dependencies = [ + "cfg-if", + "getrandom 0.3.4", + "once_cell", + "serde", + "version_check", + "zerocopy", +] + +[[package]] +name = "aho-corasick" +version = "1.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c982642fa9e8606056828ee9a8505737230110bb1099153c79efe865c59d12ba" +dependencies = [ + "memchr", +] + +[[package]] +name = "allocator-api2" +version = "0.2.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "683d7910e743518b0e34f1186f92494becacb047c7b6bf616c96772180fef923" + +[[package]] +name = "anstream" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "824a212faf96e9acacdbd09febd34438f8f711fb84e09a8916013cd7815ca28d" +dependencies = [ + "anstyle", + "anstyle-parse", + "anstyle-query", + "anstyle-wincon", + "colorchoice", + "is_terminal_polyfill", + "utf8parse", +] + +[[package]] +name = "anstyle" +version = "1.0.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "940b3a0ca603d1eade50a4846a2afffd5ef57a9feac2c0e2ec2e14f9ead76000" + +[[package]] +name = "anstyle-parse" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "52ce7f38b242319f7cabaa6813055467063ecdc9d355bbb4ce0c68908cd8130e" +dependencies = [ + "utf8parse", +] + +[[package]] +name = "anstyle-query" +version = "1.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "40c48f72fd53cd289104fc64099abca73db4166ad86ea0b4341abe65af83dadc" +dependencies = [ + "windows-sys", +] + +[[package]] +name = "anstyle-wincon" +version = "3.0.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "291e6a250ff86cd4a820112fb8898808a366d8f9f58ce16d1f538353ad55747d" +dependencies = [ + "anstyle", + "once_cell_polyfill", + "windows-sys", +] + +[[package]] +name = "anyhow" +version = "1.0.104" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" + +[[package]] +name = "autocfg" +version = "1.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" + +[[package]] +name = "base64" +version = "0.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9e1b586273c5702936fe7b7d6896644d8be71e6314cfe09d3167c95f712589e8" + +[[package]] +name = "bit-set" +version = "0.8.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "08807e080ed7f9d5433fa9b275196cfc35414f66a0c79d864dc51a0d825231a3" +dependencies = [ + "bit-vec", +] + +[[package]] +name = "bit-vec" +version = "0.8.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5e764a1d40d510daf35e07be9eb06e75770908c27d411ee6c92109c9840eaaf7" + +[[package]] +name = "bitflags" +version = "2.13.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3ded4057c258ba199e2d26386d3af3780957ecaee6c4ef4041c6b4b8b97c0b06" + +[[package]] +name = "block-buffer" +version = "0.10.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" +dependencies = [ + "generic-array", +] + +[[package]] +name = "bumpalo" +version = "3.20.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" + +[[package]] +name = "bytemuck" +version = "1.25.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "95832e849adfb21180ccb6826a99da14e5d266ae5c2e668e1602cf234f153797" +dependencies = [ + "bytemuck_derive", +] + +[[package]] +name = "bytemuck_derive" +version = "1.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6a1f896587b6f2c069c73d2f0913e2d590c3990285cd2f0b6aa02b786b4c679c" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "byteorder" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b" + +[[package]] +name = "candle-core" +version = "0.11.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5ecb245093b0f791b89d3420c3df9c6d49c60ab63ba54db896bf8a3baf486706" +dependencies = [ + "byteorder", + "float8", + "gemm", + "half", + "libc", + "libm", + "memmap2", + "num-traits", + "num_cpus", + "rand", + "rand_distr", + "rayon", + "safetensors", + "thiserror 2.0.21", + "tokenizers", + "yoke", + "zerocopy", + "zip", +] + +[[package]] +name = "candle-nn" +version = "0.11.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "eaa10b6ccc365b33210ce404fbf45e60d3e0bdac1004463cf1052e6ee1c1739a" +dependencies = [ + "candle-core", + "half", + "libc", + "num-traits", + "rayon", + "safetensors", + "serde", + "thiserror 2.0.21", +] + +[[package]] +name = "candle-transformers" +version = "0.11.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3bcbbf7ff00ff6fe2af22b93600195917fe90e90ff48424a140d1a926c44b1c1" +dependencies = [ + "byteorder", + "candle-core", + "candle-nn", + "fancy-regex 0.18.0", + "num-traits", + "rand", + "rayon", + "serde", + "serde_json", + "serde_plain", + "tracing", +] + +[[package]] +name = "castaway" +version = "0.2.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dec551ab6e7578819132c713a93c022a05d60159dc86e7a7050223577484c55a" +dependencies = [ + "rustversion", +] + +[[package]] +name = "cc" +version = "1.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f74872d07caf508b30a21f6836e7d7016a2eaf7d9ff4f48deaa58cd8a0407630" +dependencies = [ + "find-msvc-tools", + "shlex", +] + +[[package]] +name = "cfg-if" +version = "1.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600" + +[[package]] +name = "clap" +version = "4.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "aa8876b300ab35ba921adea3dfd70157a46249b33f95c9084ae5709785478946" +dependencies = [ + "clap_builder", + "clap_derive", +] + +[[package]] +name = "clap_builder" +version = "4.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ec0797fb7aeb1406c84efac526901f7ec3ead2124f946b494e72879d4b54704d" +dependencies = [ + "anstream", + "anstyle", + "clap_lex", + "strsim", +] + +[[package]] +name = "clap_derive" +version = "4.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9c751b79415d4e559e3d1fcf128e09e720eb673a06d26cf6f392d37d75b66e0" +dependencies = [ + "heck", + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "clap_lex" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1c133bc6a41be0d194c306b5506d15e6feeea7b1d6604bd3f8310dfb2ca96486" + +[[package]] +name = "colorchoice" +version = "1.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1d07550c9036bf2ae0c684c4297d503f838287c83c53686d05370d0e139ae570" + +[[package]] +name = "compact_str" +version = "0.9.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9dfdd1c2274d9aa354115b09dc9a901d6c5576818cdf70d14cae2bdb47df00ab" +dependencies = [ + "castaway", + "cfg-if", + "itoa", + "rustversion", + "ryu", + "serde", + "static_assertions", +] + +[[package]] +name = "cpufeatures" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280" +dependencies = [ + "libc", +] + +[[package]] +name = "crc32fast" +version = "1.5.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78" +dependencies = [ + "cfg-if", +] + +[[package]] +name = "crossbeam-deque" +version = "0.8.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "622f3fc73690be383c7214310406f28a90e6edeadc3cea882f9d71e495b9711a" +dependencies = [ + "crossbeam-epoch", + "crossbeam-utils", +] + +[[package]] +name = "crossbeam-epoch" +version = "0.9.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dc74980687109a3b14c72fd458107bf0baa1da1a1a805e178d15501ba9b86d9d" +dependencies = [ + "crossbeam-utils", +] + +[[package]] +name = "crossbeam-utils" +version = "0.8.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a31eee39dddec8330830986fcd7625edb5a24ec90ea038215273bbc3adb08ac6" + +[[package]] +name = "crunchy" +version = "0.2.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5" + +[[package]] +name = "crypto-common" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" +dependencies = [ + "generic-array", + "typenum", +] + +[[package]] +name = "darling" +version = "0.20.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc7f46116c46ff9ab3eb1597a45688b6715c6e628b5c133e288e709a29bcb4ee" +dependencies = [ + "darling_core", + "darling_macro", +] + +[[package]] +name = "darling_core" +version = "0.20.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0d00b9596d185e565c2207a0b01f8bd1a135483d02d9b7b0a54b11da8d53412e" +dependencies = [ + "fnv", + "ident_case", + "proc-macro2", + "quote", + "strsim", + "syn 2.0.119", +] + +[[package]] +name = "darling_macro" +version = "0.20.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc34b93ccb385b40dc71c6fceac4b2ad23662c7eeb248cf10d529b7e055b6ead" +dependencies = [ + "darling_core", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "dary_heap" +version = "0.3.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8b1e3a325bc115f096c8b77bbf027a7c2592230e70be2d985be950d3d5e60ebe" +dependencies = [ + "serde", +] + +[[package]] +name = "derive_builder" +version = "0.20.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "507dfb09ea8b7fa618fcf76e953f4f5e192547945816d5358edffe39f6f94947" +dependencies = [ + "derive_builder_macro", +] + +[[package]] +name = "derive_builder_core" +version = "0.20.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2d5bcf7b024d6835cfb3d473887cd966994907effbe9227e8c8219824d06c4e8" +dependencies = [ + "darling", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "derive_builder_macro" +version = "0.20.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ab63b0e2bf4d5928aff72e83a7dace85d7bba5fe12dcc3c5a572d78caffd3f3c" +dependencies = [ + "derive_builder_core", + "syn 2.0.119", +] + +[[package]] +name = "digest" +version = "0.10.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" +dependencies = [ + "block-buffer", + "crypto-common", +] + +[[package]] +name = "dyn-stack" +version = "0.13.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1c4713e43e2886ba72b8271aa66c93d722116acf7a75555cce11dcde84388fe8" +dependencies = [ + "bytemuck", + "dyn-stack-macros", +] + +[[package]] +name = "dyn-stack-macros" +version = "0.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e1d926b4d407d372f141f93bb444696142c29d32962ccbd3531117cf3aa0bfa9" + +[[package]] +name = "either" +version = "1.18.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" + +[[package]] +name = "enum-as-inner" +version = "0.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a1e6a265c649f3f5979b601d26f1d05ada116434c87741c9493cb56218f76cbc" +dependencies = [ + "heck", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "equivalent" +version = "1.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" + +[[package]] +name = "errno" +version = "0.3.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" +dependencies = [ + "libc", + "windows-sys", +] + +[[package]] +name = "esaxx-rs" +version = "0.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d817e038c30374a4bcb22f94d0a8a0e216958d4c3dcde369b1439fec4bdda6e6" + +[[package]] +name = "fallible-iterator" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2acce4a10f12dc2fb14a218589d4f1f62ef011b2d0cc4b3cb1bba8e94da14649" + +[[package]] +name = "fallible-streaming-iterator" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a" + +[[package]] +name = "fancy-regex" +version = "0.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6e24cb5a94bcae1e5408b0effca5cd7172ea3c5755049c5f3af4cd283a165298" +dependencies = [ + "bit-set", + "regex-automata", + "regex-syntax", +] + +[[package]] +name = "fancy-regex" +version = "0.18.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e1e1dacd0d2082dfcf1351c4bdd566bbe89a2b263235a2b50058f1e130a47277" +dependencies = [ + "bit-set", + "regex-automata", + "regex-syntax", +] + +[[package]] +name = "fastrand" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" + +[[package]] +name = "find-msvc-tools" +version = "0.1.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "aedcfb3409746eddb02b9e19ebda1c3394f759a152e48ee875a0844d1b955484" + +[[package]] +name = "float8" +version = "0.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c2d1f04709a8ac06e8e8042875a3c466cc4832d3c1a18dbcb9dba3c6e83046bc" +dependencies = [ + "half", + "num-traits", + "rand", + "rand_distr", +] + +[[package]] +name = "fnv" +version = "1.0.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3f9eec918d3f24069decb9af1554cad7c880e2da24a9afd88aca000531ab82c1" + +[[package]] +name = "foldhash" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "77ce24cb58228fbb8aa041425bb1050850ac19177686ea6e0f41a70416f56fdb" + +[[package]] +name = "gemm" +version = "0.19.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "aa0673db364b12263d103b68337a68fbecc541d6f6b61ba72fe438654709eacb" +dependencies = [ + "dyn-stack", + "gemm-c32", + "gemm-c64", + "gemm-common", + "gemm-f16", + "gemm-f32", + "gemm-f64", + "num-complex", + "num-traits", + "paste", + "raw-cpuid", + "seq-macro", +] + +[[package]] +name = "gemm-c32" +version = "0.19.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "086936dbdcb99e37aad81d320f98f670e53c1e55a98bee70573e83f95beb128c" +dependencies = [ + "dyn-stack", + "gemm-common", + "num-complex", + "num-traits", + "paste", + "raw-cpuid", + "seq-macro", +] + +[[package]] +name = "gemm-c64" +version = "0.19.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "20c8aeeeec425959bda4d9827664029ba1501a90a0d1e6228e48bef741db3a3f" +dependencies = [ + "dyn-stack", + "gemm-common", + "num-complex", + "num-traits", + "paste", + "raw-cpuid", + "seq-macro", +] + +[[package]] +name = "gemm-common" +version = "0.19.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "88027625910cc9b1085aaaa1c4bc46bb3a36aad323452b33c25b5e4e7c8e2a3e" +dependencies = [ + "bytemuck", + "dyn-stack", + "half", + "libm", + "num-complex", + "num-traits", + "once_cell", + "paste", + "pulp", + "raw-cpuid", + "rayon", + "seq-macro", + "sysctl", +] + +[[package]] +name = "gemm-f16" +version = "0.19.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e3df7a55202e6cd6739d82ae3399c8e0c7e1402859b30e4cb780e61525d9486e" +dependencies = [ + "dyn-stack", + "gemm-common", + "gemm-f32", + "half", + "num-complex", + "num-traits", + "paste", + "raw-cpuid", + "rayon", + "seq-macro", +] + +[[package]] +name = "gemm-f32" +version = "0.19.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "02e0b8c9da1fbec6e3e3ab2ce6bc259ef18eb5f6f0d3e4edf54b75f9fd41a81c" +dependencies = [ + "dyn-stack", + "gemm-common", + "num-complex", + "num-traits", + "paste", + "raw-cpuid", + "seq-macro", +] + +[[package]] +name = "gemm-f64" +version = "0.19.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "056131e8f2a521bfab322f804ccd652520c79700d81209e9d9275bbdecaadc6a" +dependencies = [ + "dyn-stack", + "gemm-common", + "num-complex", + "num-traits", + "paste", + "raw-cpuid", + "seq-macro", +] + +[[package]] +name = "generic-array" +version = "0.14.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" +dependencies = [ + "typenum", + "version_check", +] + +[[package]] +name = "getrandom" +version = "0.3.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" +dependencies = [ + "cfg-if", + "libc", + "r-efi 5.3.0", + "wasip2", +] + +[[package]] +name = "getrandom" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" +dependencies = [ + "cfg-if", + "libc", + "r-efi 6.0.0", +] + +[[package]] +name = "half" +version = "2.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ea2d84b969582b4b1864a92dc5d27cd2b77b622a8d79306834f1be5ba20d84b" +dependencies = [ + "bytemuck", + "cfg-if", + "crunchy", + "num-traits", + "rand", + "rand_distr", + "zerocopy", +] + +[[package]] +name = "hashbrown" +version = "0.16.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "841d1cc9bed7f9236f321df977030373f4a4163ae1a7dbfe1a51a2c1a51d9100" +dependencies = [ + "allocator-api2", + "equivalent", + "foldhash", + "serde", + "serde_core", +] + +[[package]] +name = "hashbrown" +version = "0.17.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" +dependencies = [ + "foldhash", +] + +[[package]] +name = "hashlink" +version = "0.12.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a596f1b20ed2cc5ecac41a164aaebc7258057060f06c0cf7a2ba3991ee7990fb" +dependencies = [ + "hashbrown 0.17.1", +] + +[[package]] +name = "heck" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" + +[[package]] +name = "hermit-abi" +version = "0.5.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e17592d60ebacc7d5e169f4663c5f84f9161cc90328abcfe8456f41e4dfcb284" + +[[package]] +name = "ident_case" +version = "1.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9e0384b61958566e926dc50660321d12159025e767c18e043daf26b70104c39" + +[[package]] +name = "indexmap" +version = "2.14.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cc4e190f5d26ca7051642629da2c52fc03bde85a03197c99408dcd291734c855" +dependencies = [ + "equivalent", + "hashbrown 0.17.1", +] + +[[package]] +name = "is_terminal_polyfill" +version = "1.70.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a6cb138bb79a146c1bd460005623e142ef0181e3d0219cb493e02f7d08a35695" + +[[package]] +name = "itertools" +version = "0.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2b192c782037fadd9cfa75548310488aabdbf3d2da73885b31bd0abd03351285" +dependencies = [ + "either", +] + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "js-sys" +version = "0.3.106" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7883d941dae510fb2d978fc3fe018c71c9e2892fd38854de3e8b92c2e5ad9cc5" +dependencies = [ + "cfg-if", + "wasm-bindgen", +] + +[[package]] +name = "libc" +version = "0.2.190" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ce5d3ddc6d3fa000eb1536d85e147bfe31aacaba692ed6a876f95cb7c855be78" + +[[package]] +name = "libm" +version = "0.2.16" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6d2cec3eae94f9f509c767b45932f1ada8350c4bdb85af2fcab4a3c14807981" + +[[package]] +name = "libsqlite3-sys" +version = "0.38.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f1d20bef17f513b9b3004532233187769cd072d790971f4e4da0e346eb6401e8" +dependencies = [ + "cc", + "pkg-config", + "vcpkg", +] + +[[package]] +name = "linux-raw-sys" +version = "0.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" + +[[package]] +name = "log" +version = "0.4.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" + +[[package]] +name = "macro_rules_attribute" +version = "0.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b3ae8f6d608c795738406608304d30a2dfbdc8e58e44f7ba43236da5208ded3c" +dependencies = [ + "macro_rules_attribute-proc_macro", + "pastey", +] + +[[package]] +name = "macro_rules_attribute-proc_macro" +version = "0.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc04a4c58212d57930a24bf47d3fa87485264a3a054e9c10e042eb373573ad3c" + +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + +[[package]] +name = "memmap2" +version = "0.9.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d1219ed1b7f229ee7104d281dd01d6802fe28bb6e95d292942c4daacdeb798c0" +dependencies = [ + "libc", + "stable_deref_trait", +] + +[[package]] +name = "minimal-lexical" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "68354c5c6bd36d73ff3feceb05efa59b6acb7626617f4962be322a825e61f79a" + +[[package]] +name = "monostate" +version = "0.1.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3341a273f6c9d5bef1908f17b7267bbab0e95c9bf69a0d4dcf8e9e1b2c76ef67" +dependencies = [ + "monostate-impl", + "serde", + "serde_core", +] + +[[package]] +name = "monostate-impl" +version = "0.1.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e4db6d5580af57bf992f59068d4ea26fd518574ff48d7639b255a36f9de6e7e9" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "nom" +version = "7.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d273983c5a657a70a3e8f2a01329822f3b8c8172b73826411a55751e404a0a4a" +dependencies = [ + "memchr", + "minimal-lexical", +] + +[[package]] +name = "num-complex" +version = "0.4.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "73f88a1307638156682bada9d7604135552957b7818057dcef22705b4d509495" +dependencies = [ + "bytemuck", + "num-traits", +] + +[[package]] +name = "num-traits" +version = "0.2.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "071dfc062690e90b734c0b2273ce72ad0ffa95f0c74596bc250dcfd960262841" +dependencies = [ + "autocfg", + "libm", +] + +[[package]] +name = "num_cpus" +version = "1.17.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "91df4bbde75afed763b708b7eee1e8e7651e02d97f6d5dd763e89367e957b23b" +dependencies = [ + "hermit-abi", + "libc", +] + +[[package]] +name = "once_cell" +version = "1.21.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" + +[[package]] +name = "once_cell_polyfill" +version = "1.70.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe" + +[[package]] +name = "onig" +version = "6.5.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0cc3cbf698f9438986c11a880c90a6d04b9de27575afd28bbf45b154b6c709e2" +dependencies = [ + "bitflags", + "libc", + "once_cell", + "onig_sys", +] + +[[package]] +name = "onig_sys" +version = "69.9.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1e68317604e77e53b85896388e1a803c1d21b74c899ec9e5e1112db90735edd7" +dependencies = [ + "cc", + "pkg-config", +] + +[[package]] +name = "paste" +version = "1.0.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "57c0d7b74b563b49d38dae00a0c37d4d6de9b432382b2892f0574ddcae73fd0a" + +[[package]] +name = "pastey" +version = "0.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2ee67f1008b1ba2321834326597b8e186293b049a023cdef258527550b9935b4" + +[[package]] +name = "pin-project-lite" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" + +[[package]] +name = "pkg-config" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" + +[[package]] +name = "ppv-lite86" +version = "0.2.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85eae3c4ed2f50dcfe72643da4befc30deadb458a9b590d720cde2f2b1e97da9" +dependencies = [ + "zerocopy", +] + +[[package]] +name = "proc-macro2" +version = "1.0.107" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "pulp" +version = "0.22.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "046aa45b989642ec2e4717c8e72d677b13edd831a4d3b6cf37d9a3e54912496a" +dependencies = [ + "bytemuck", + "cfg-if", + "libm", + "num-complex", + "paste", + "pulp-wasm-simd-flag", + "raw-cpuid", + "reborrow", + "version_check", +] + +[[package]] +name = "pulp-wasm-simd-flag" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1d8f70e07b9c3962945a74e59ca1c511bba65b6419468acc217c457d93f3c740" + +[[package]] +name = "quote" +version = "1.0.47" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" +dependencies = [ + "proc-macro2", +] + +[[package]] +name = "r-efi" +version = "5.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "69cdb34c158ceb288df11e18b4bd39de994f6657d83847bdffdbd7f346754b0f" + +[[package]] +name = "r-efi" +version = "6.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" + +[[package]] +name = "rand" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41" +dependencies = [ + "rand_chacha", + "rand_core", +] + +[[package]] +name = "rand_chacha" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d3022b5f1df60f26e1ffddd6c66e8aa15de382ae63b3a0c1bfc0e4d3e3f325cb" +dependencies = [ + "ppv-lite86", + "rand_core", +] + +[[package]] +name = "rand_core" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "76afc826de14238e6e8c374ddcc1fa19e374fd8dd986b0d2af0d02377261d83c" +dependencies = [ + "getrandom 0.3.4", +] + +[[package]] +name = "rand_distr" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6a8615d50dcf34fa31f7ab52692afec947c4dd0ab803cc87cb3b0b4570ff7463" +dependencies = [ + "num-traits", + "rand", +] + +[[package]] +name = "raw-cpuid" +version = "11.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "498cd0dc59d73224351ee52a95fee0f1a617a2eae0e7d9d720cc622c73a54186" +dependencies = [ + "bitflags", +] + +[[package]] +name = "rayon" +version = "1.12.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fb39b166781f92d482534ef4b4b1b2568f42613b53e5b6c160e24cfbfa30926d" +dependencies = [ + "either", + "rayon-core", +] + +[[package]] +name = "rayon-cond" +version = "0.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2964d0cf57a3e7a06e8183d14a8b527195c706b7983549cd5462d5aa3747438f" +dependencies = [ + "either", + "itertools", + "rayon", +] + +[[package]] +name = "rayon-core" +version = "1.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "22e18b0f0062d30d4230b2e85ff77fdfe4326feb054b9783a3460d8435c8ab91" +dependencies = [ + "crossbeam-deque", + "crossbeam-utils", +] + +[[package]] +name = "reborrow" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "03251193000f4bd3b042892be858ee50e8b3719f2b08e5833ac4353724632430" + +[[package]] +name = "regex" +version = "1.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f020237b6c8eed93db2e2cb53c00c60a8e1bc73da7d073199a1180401450218d" +dependencies = [ + "aho-corasick", + "memchr", + "regex-automata", + "regex-syntax", +] + +[[package]] +name = "regex-automata" +version = "0.4.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" +dependencies = [ + "aho-corasick", + "memchr", + "regex-syntax", +] + +[[package]] +name = "regex-syntax" +version = "0.8.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4" + +[[package]] +name = "rsqlite-vfs" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c51c9ae4df8a7fba42103df5c621fa3c37eccf3a3c650879e90fc48b11cc192c" +dependencies = [ + "hashbrown 0.16.1", + "thiserror 2.0.21", +] + +[[package]] +name = "rusqlite" +version = "0.40.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "23f2a97da3e3873c73cb2a2e71b35c40ff95e0b1eefa8d72d8499a6928c3b5b3" +dependencies = [ + "bitflags", + "fallible-iterator", + "fallible-streaming-iterator", + "hashlink", + "libsqlite3-sys", + "smallvec", + "sqlite-wasm-rs", +] + +[[package]] +name = "rustix" +version = "1.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "891efababe418670775f199f0d233d84843c227a0949a883ce15b37c78d6629d" +dependencies = [ + "bitflags", + "errno", + "libc", + "linux-raw-sys", + "windows-sys", +] + +[[package]] +name = "rustversion" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" + +[[package]] +name = "ryu" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f" + +[[package]] +name = "safetensors" +version = "0.8.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "79b079b829cb27a1c3c374341345ed2e8b2c0c839034522cee576c140bd7f846" +dependencies = [ + "hashbrown 0.16.1", + "libc", + "serde", + "serde_json", + "tempfile", +] + +[[package]] +name = "same-file" +version = "1.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "93fc1dc3aaa9bfed95e02e6eadabb4baf7e3078b0bd1b4d7b6b0b68378900502" +dependencies = [ + "winapi-util", +] + +[[package]] +name = "seq-macro" +version = "0.3.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1bc711410fbe7399f390ca1c3b60ad0f53f80e95c5eb935e52268a0e2cd49acc" + +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", + "serde_derive", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "serde_plain" +version = "1.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9ce1fc6db65a611022b23a0dec6975d63fb80a302cb3388835ff02c097258d50" +dependencies = [ + "serde", +] + +[[package]] +name = "sha2" +version = "0.10.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283" +dependencies = [ + "cfg-if", + "cpufeatures", + "digest", +] + +[[package]] +name = "shlex" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" + +[[package]] +name = "smallvec" +version = "1.16.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9395f0f0eee849a9b707b2f06bb92a6a422090e2123bb2ef8e87a0e61892a8e" + +[[package]] +name = "spm_precompiled" +version = "0.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5851699c4033c63636f7ea4cf7b7c1f1bf06d0cc03cfb42e711de5a5c46cf326" +dependencies = [ + "base64", + "nom", + "serde", + "unicode-segmentation", +] + +[[package]] +name = "sqlite-wasm-rs" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dc3efc0da82635d7e1ced0053bbbfa8c7ab9645d0bf36ceb4f7127bb85315d75" +dependencies = [ + "cc", + "js-sys", + "rsqlite-vfs", + "wasm-bindgen", +] + +[[package]] +name = "stable_deref_trait" +version = "1.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" + +[[package]] +name = "static_assertions" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a2eb9349b6444b326872e140eb1cf5e7c522154d69e7a0ffb0fb81c06b37543f" + +[[package]] +name = "strsim" +version = "0.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" + +[[package]] +name = "switchify-prediction" +version = "0.1.0" +dependencies = [ + "clap", + "rusqlite", + "serde", + "serde_json", + "sha2", + "tempfile", + "thiserror 2.0.21", + "unicode-normalization", + "unicode-segmentation", +] + +[[package]] +name = "switchify-prediction-neural" +version = "0.1.0" +dependencies = [ + "clap", + "serde", + "serde_json", + "sha2", + "switchify-prediction", + "tempfile", + "thiserror 2.0.21", +] + +[[package]] +name = "switchify-smol-worker" +version = "0.1.0" +dependencies = [ + "anyhow", + "candle-core", + "candle-transformers", + "serde_json", + "switchify-prediction-neural", + "tokenizers", +] + +[[package]] +name = "syn" +version = "2.0.119" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "syn" +version = "3.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "synstructure" +version = "0.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "901704edd0dfe137f1987838ee4f259e4e063c31371bdb423f7ae38ec6f77f02" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "sysctl" +version = "0.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "01198a2debb237c62b6826ec7081082d951f46dbb64b0e8c7649a452230d1dfc" +dependencies = [ + "bitflags", + "byteorder", + "enum-as-inner", + "libc", + "thiserror 1.0.69", + "walkdir", +] + +[[package]] +name = "tempfile" +version = "3.27.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" +dependencies = [ + "fastrand", + "getrandom 0.4.3", + "once_cell", + "rustix", + "windows-sys", +] + +[[package]] +name = "thiserror" +version = "1.0.69" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6aaf5339b578ea85b50e080feb250a3e8ae8cfcdff9a461c9ec2904bc923f52" +dependencies = [ + "thiserror-impl 1.0.69", +] + +[[package]] +name = "thiserror" +version = "2.0.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "09e52cb86a36cede5cb101bf8908837b3e4c6e5e59fe7fd85c23fb56200d189e" +dependencies = [ + "thiserror-impl 2.0.21", +] + +[[package]] +name = "thiserror-impl" +version = "1.0.69" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4fee6c4efc90059e10f81e6d42c60a18f76588c3d74cb83a0b242a2b6c7504c1" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "thiserror-impl" +version = "2.0.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fe5197923287db20a58125f0bc85c062f7f2c892de97b18c356f9efb14b28524" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "tinyvec" +version = "1.13.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fd3ca314f692efd6c868f8408f53fe444634a845f96c028b97d35f6a1f79f0ee" + +[[package]] +name = "tokenizers" +version = "0.22.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b238e22d44a15349529690fb07bd645cf58149a1b1e44d6cb5bd1641ff1a6223" +dependencies = [ + "ahash", + "aho-corasick", + "compact_str", + "dary_heap", + "derive_builder", + "esaxx-rs", + "fancy-regex 0.14.0", + "getrandom 0.3.4", + "itertools", + "log", + "macro_rules_attribute", + "monostate", + "onig", + "paste", + "rand", + "rayon", + "rayon-cond", + "regex", + "regex-syntax", + "serde", + "serde_json", + "spm_precompiled", + "thiserror 2.0.21", + "unicode-normalization-alignments", + "unicode-segmentation", + "unicode_categories", +] + +[[package]] +name = "tracing" +version = "0.1.44" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "63e71662fa4b2a2c3a26f570f037eb95bb1f85397f3cd8076caed2f026a6d100" +dependencies = [ + "pin-project-lite", + "tracing-attributes", + "tracing-core", +] + +[[package]] +name = "tracing-attributes" +version = "0.1.31" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "tracing-core" +version = "0.1.36" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "db97caf9d906fbde555dd62fa95ddba9eecfd14cb388e4f491a66d74cd5fb79a" +dependencies = [ + "once_cell", +] + +[[package]] +name = "typed-path" +version = "0.12.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e28f89b80c87b8fb0cf04ab448d5dd0dd0ade2f8891bae878de66a75a28600e" + +[[package]] +name = "typenum" +version = "1.20.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" + +[[package]] +name = "unicode-ident" +version = "1.0.26" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954" + +[[package]] +name = "unicode-normalization" +version = "0.1.25" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5fd4f6878c9cb28d874b009da9e8d183b5abc80117c40bbd187a1fde336be6e8" +dependencies = [ + "tinyvec", +] + +[[package]] +name = "unicode-normalization-alignments" +version = "0.1.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "43f613e4fa046e69818dd287fdc4bc78175ff20331479dab6e1b0f98d57062de" +dependencies = [ + "smallvec", +] + +[[package]] +name = "unicode-segmentation" +version = "1.13.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c6f5d3c3b1bf09027a88a6bc961fc00497d651009560b5463668dc81b0fa87a8" + +[[package]] +name = "unicode_categories" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "39ec24b3121d976906ece63c9daad25b85969647682eee313cb5779fdd69e14e" + +[[package]] +name = "utf8parse" +version = "0.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" + +[[package]] +name = "vcpkg" +version = "0.2.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" + +[[package]] +name = "version_check" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" + +[[package]] +name = "walkdir" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29790946404f91d9c5d06f9874efddea1dc06c5efe94541a7d6863108e3a5e4b" +dependencies = [ + "same-file", + "winapi-util", +] + +[[package]] +name = "wasip2" +version = "1.0.4+wasi-0.2.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" +dependencies = [ + "wit-bindgen", +] + +[[package]] +name = "wasm-bindgen" +version = "0.2.129" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9bb54f33acc68fd454578d9820b0bde1a1a3d17aa17bb7b6595806d02886d409" +dependencies = [ + "cfg-if", + "once_cell", + "rustversion", + "wasm-bindgen-macro", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-macro" +version = "0.2.129" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2e29d0c35b16e224a7eeb5cd2d25e3e1968fbd65604117b44d3b789d00ee8535" +dependencies = [ + "quote", + "wasm-bindgen-macro-support", +] + +[[package]] +name = "wasm-bindgen-macro-support" +version = "0.2.129" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6f501a8bc3719dba86ef8ae4728879c08001bea749eb1333ac5b91e040e2a6b7" +dependencies = [ + "bumpalo", + "proc-macro2", + "quote", + "syn 3.0.6", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-shared" +version = "0.2.129" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "23f0c9c52aa7cd7d77769a4cfe2a9adb1b331f489a41d912ce14513d5ab995c6" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "winapi-util" +version = "0.1.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c2a7b1c03c876122aa43f3020e6c3c3ee5c05081c9a00739faf7503aeba10d22" +dependencies = [ + "windows-sys", +] + +[[package]] +name = "windows-link" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" + +[[package]] +name = "windows-sys" +version = "0.61.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" +dependencies = [ + "windows-link", +] + +[[package]] +name = "wit-bindgen" +version = "0.57.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" + +[[package]] +name = "yoke" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "709fe23a0424b6a435d82152b1bd3fdfb0833487d5fa90d05d42762a9891fef5" +dependencies = [ + "stable_deref_trait", + "yoke-derive", + "zerofrom", +] + +[[package]] +name = "yoke-derive" +version = "0.8.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ec8ebde2db3681e8c9980cc27822030e68752690ddfa9473e739aeb4dbde6d71" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", + "synstructure", +] + +[[package]] +name = "zerocopy" +version = "0.8.59" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6df92bf3d9227be3d53173901ddbffac2babc27ae50f397776ffd6dc33f800cb" +dependencies = [ + "zerocopy-derive", +] + +[[package]] +name = "zerocopy-derive" +version = "0.8.59" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ac4f328cf2f05d084e496c3e9c3f33ed0a183656a16e1fcec4d464d8373aec82" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "zerofrom" +version = "0.1.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ec05a11813ea801ff6d75110ad09cd0824ddba17dfe17128ea0d5f68e6c5272" +dependencies = [ + "zerofrom-derive", +] + +[[package]] +name = "zerofrom-derive" +version = "0.1.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f75b4683f6c7f45248d4d64056a24298c6281e0993356d7d1b4a1a962ef10d4a" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", + "synstructure", +] + +[[package]] +name = "zip" +version = "8.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2d04a6b5381502aa6087c94c669499eb1602eb9c5e8198e534de571f7154809b" +dependencies = [ + "crc32fast", + "indexmap", + "memchr", + "typed-path", +] + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/neural/Cargo.toml b/neural/Cargo.toml new file mode 100644 index 0000000..f9290cd --- /dev/null +++ b/neural/Cargo.toml @@ -0,0 +1,36 @@ +[package] +name = "switchify-prediction-neural" +version = "0.1.0" +edition = "2024" +rust-version = "1.97.1" +license = "MIT" +description = "Opt-in offline SmolLM2 refinement for Switchify Prediction" +autotests = false + +[features] +test-support = [] + +[[bin]] +name = "fake-worker" +path = "tests/fake_worker.rs" +required-features = ["test-support"] + +[[test]] +name = "lifecycle" +path = "tests/lifecycle.rs" +required-features = ["test-support"] + +[workspace] +members = ["worker"] +resolver = "3" + +[dependencies] +switchify-prediction = { path = ".." } +serde = { version = "1", features = ["derive"] } +serde_json = "1" +sha2 = "0.10" +thiserror = "2" +clap = { version = "4", features = ["derive"] } + +[dev-dependencies] +tempfile = "3" diff --git a/neural/MODEL_LICENSE.txt b/neural/MODEL_LICENSE.txt new file mode 100644 index 0000000..d645695 --- /dev/null +++ b/neural/MODEL_LICENSE.txt @@ -0,0 +1,202 @@ + + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright [yyyy] [name of copyright owner] + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/neural/evaluation-protocol.json b/neural/evaluation-protocol.json new file mode 100644 index 0000000..e1db06e --- /dev/null +++ b/neural/evaluation-protocol.json @@ -0,0 +1,12 @@ +{ + "version": 1, + "frozen_before_scoring": true, + "policy": "SmolLM2-135M Q8, top 8 statistical candidates, last sentence/64 tokens, whole-word plus boundary probability, sequential, return 5", + "regression": "Existing 96 authored sentences; 64 hash-selected queries per domain/prefix using the previous experiment selection rule", + "additional": "48 new general-writing sentences; first 6 per domain development, remaining 6 test. Select 8 dev and 16 test target positions per domain by SHA256, words longer than four graphemes. Run each position at 0 through 4 typed graphemes in order.", + "personal_learning": false, + "warmup_requests": 20, + "latency": "At least 1000 successful warmed queries/configuration; CLI immediate and refinement end-to-end, exact-context hits and misses; cold start to ready; 10ms sampled sum of parent and descendant RSS", + "gates": {"overall_top5_non_decreasing": true, "early_cell_max_decline_percentage_points": 1.0, "immediate_p95_ms": 20, "refinement_p95_ms_reference_windows": 150}, + "interpretation": "Regression comparisons, not unseen-data accuracy. Unknown pretraining overlap. Freeze before scoring; do not tune on test failures. No automatic promotion. Other platforms require their own measurements." +} diff --git a/neural/fixtures/general-writing.json b/neural/fixtures/general-writing.json new file mode 100644 index 0000000..565ba56 --- /dev/null +++ b/neural/fixtures/general-writing.json @@ -0,0 +1,58 @@ +{ + "messages": [ + "Could you collect the parcel from reception this afternoon", + "The train has stopped outside the station again", + "I left your charger beside the kitchen window", + "We should book a table before everyone arrives", + "Please bring another towel for the swimming lesson", + "My appointment was moved to Thursday morning", + "The delivery driver called while I was sleeping", + "Can someone check whether the garden gate is locked", + "I will send the photographs after dinner tonight", + "There is fresh bread on the counter for lunch", + "The concert finishes too late for the last bus", + "Please remind me to return your book tomorrow" + ], + "email": [ + "Please confirm which address we should use for the invoice", + "The attached spreadsheet includes the revised delivery dates", + "We have reserved a meeting room for next Wednesday", + "Thank you for sending the updated contract yesterday", + "Could you explain the difference between these two estimates", + "I am writing to request a replacement for the damaged item", + "Your application has been received by our recruitment team", + "The project schedule needs another review before approval", + "Please remove my previous address from your mailing list", + "I would appreciate a response before the end of next week", + "We need permission to include these images in the report", + "The maintenance team will inspect the equipment on Friday" + ], + "documents": [ + "The survey found that most residents preferred the revised proposal", + "Each participant received written instructions before the session began", + "The library provides quiet study spaces on the upper floor", + "Water samples were collected from several locations along the river", + "The committee recommended further consultation with local businesses", + "These measurements should be repeated under controlled conditions", + "The final chapter examines changes in public transport funding", + "A separate entrance allows visitors to reach the exhibition directly", + "The procedure requires careful inspection of every completed component", + "This section describes the methods used to estimate annual demand", + "The building was restored using materials from the original structure", + "Several independent studies reached similar conclusions about the results" + ], + "search": [ + "How to replace the battery in a wireless keyboard", + "Opening hours for the public swimming pool on Sunday", + "Best way to remove coffee stains from a cotton shirt", + "Where to find the serial number on a washing machine", + "Train tickets from Dublin to Cork tomorrow morning", + "Simple vegetarian recipes with potatoes and frozen spinach", + "How long does bread dough need to rise in winter", + "Weather forecast for the west coast this weekend", + "Instructions for changing the default browser on a laptop", + "Local recycling centre rules for old electrical appliances", + "Difference between a fixed rate and a variable rate", + "How to keep indoor plants healthy during a holiday" + ] +} diff --git a/neural/fixtures/regression.json b/neural/fixtures/regression.json new file mode 100644 index 0000000..fcf5924 --- /dev/null +++ b/neural/fixtures/regression.json @@ -0,0 +1,106 @@ +{ + "messages": [ + "I'm running a little late because the train stopped outside the station.", + "Could you bring my blue jacket when you come over this evening?", + "We should book a table somewhere near the cinema before the weekend.", + "The delivery arrived this morning and I left your parcel beside the stairs.", + "Let me know when you finish work and we can arrange dinner.", + "I thought the meeting was tomorrow but the calendar says Thursday afternoon.", + "My phone battery is nearly empty so I'll call you after lunch.", + "Thanks for helping me move the furniture into the spare bedroom yesterday.", + "Do you remember which supermarket sells the coffee we bought last month?", + "I'll pick up some vegetables and bread on the way back home.", + "The weather looks better tomorrow so we could walk along the beach.", + "Your photographs from the holiday look amazing especially the mountain views.", + "Please send me the address again because I can't find the message.", + "We have enough chairs but somebody needs to bring another folding table.", + "I haven't watched the final episode yet so please don't spoil anything.", + "The tickets are cheaper if we travel before the morning rush hour.", + "Can you check whether the front door is locked before you leave?", + "I found your headphones underneath the sofa while I was cleaning today.", + "Let's choose another restaurant because the place we wanted is fully booked.", + "The children are staying with their grandparents until late on Sunday evening.", + "I've ordered a replacement charger and it should arrive early next week.", + "Would you prefer to meet outside the library or beside the station?", + "There is some leftover pasta in the fridge if you want lunch.", + "We enjoyed the concert although finding somewhere to park was quite difficult." + ], + "email": [ + "Thank you for sending the revised proposal and the updated delivery schedule.", + "Please find the requested documents attached for your review before our meeting.", + "Could you confirm whether the replacement parts will arrive by Friday afternoon?", + "I would appreciate a breakdown of the additional charges on this invoice.", + "Following our discussion yesterday we have adjusted the project milestones accordingly.", + "The finance team needs your approval before processing the outstanding supplier payment.", + "Unfortunately I am unavailable on Tuesday but Wednesday morning would suit me.", + "Please include the purchase reference when replying so we can locate your order.", + "We are reviewing the applications and expect to contact shortlisted candidates shortly.", + "Your subscription renewal has been processed and the receipt is attached below.", + "I noticed that the contact details in the shared document are outdated.", + "Would it be possible to extend the deadline until the following Monday?", + "The customer reported that the replacement device still displays the same error.", + "We have reserved the conference room for the training session next week.", + "Please let us know if you require any further information about registration.", + "I am writing to request a quotation for the equipment listed below.", + "The updated contract reflects the changes agreed during our last telephone conversation.", + "Thank you for your patience while our technical team investigates this issue.", + "We will send the final agenda once all speakers have confirmed attendance.", + "Please ensure that each expense claim includes a copy of the original receipt.", + "I have forwarded your enquiry to the colleague responsible for international deliveries.", + "The maintenance work is scheduled outside normal office hours to minimise disruption.", + "Could everyone review the attached spreadsheet and correct any missing information?", + "We look forward to welcoming your team to our offices next month." + ], + "documents": [ + "The experiment compared three methods under identical conditions to reduce measurement bias.", + "Participants completed a short questionnaire before beginning the practical assessment session.", + "The proposed timetable allows additional preparation time between the morning presentations.", + "A reliable backup process should preserve recent changes without interrupting normal work.", + "The building includes a shared kitchen and several meeting rooms on each floor.", + "This section explains how to replace the filter and clean the surrounding housing.", + "The committee recommended further consultation before making a decision about the redevelopment.", + "Changes in customer demand affected both production planning and warehouse storage requirements.", + "The first chapter describes the historical background and introduces the main research questions.", + "Employees should report damaged equipment to their supervisor before attempting any repairs.", + "The new route connects the residential neighbourhood with the hospital and central station.", + "Each observation was recorded separately and checked against the original laboratory notes.", + "The installation process creates a shortcut and copies the required files automatically.", + "A small increase in temperature caused the material to expand along its length.", + "The report identifies several opportunities to improve communication between regional offices.", + "Successful applicants will receive written confirmation together with instructions for the next stage.", + "The museum collection includes paintings and photographs donated by local residents.", + "Regular maintenance helps prevent unexpected failures and extends the useful life of equipment.", + "The survey results suggest that visitors value clear information and convenient opening hours.", + "A detailed inventory should include the location and condition of every stored item.", + "The software displays a warning when the available storage falls below the recommended level.", + "The final design includes wider entrances and improved lighting throughout the public areas.", + "Researchers compared the original measurements with results collected during the second trial.", + "This guide provides practical examples of common problems and explains possible solutions." + ], + "search": [ + "best lightweight waterproof jackets for walking during rainy weather", + "how to replace a damaged charging cable on a laptop", + "restaurants serving vegetarian meals near the central railway station", + "compare electricity tariffs for apartments with electric heating systems", + "train timetables between the airport and the city centre", + "instructions for cleaning reusable coffee filters without harsh chemicals", + "comfortable office chairs with adjustable armrests and lumbar support", + "weekend activities for families visiting the natural history museum", + "troubleshooting wireless headphones that disconnect during telephone conversations", + "simple recipes using roasted vegetables and leftover cooked chicken", + "opening hours for the public swimming pool during school holidays", + "how to recover an accidentally deleted folder from a backup", + "affordable accommodation within walking distance of the conference centre", + "cycling routes suitable for beginners around the coastal villages", + "portable monitors compatible with laptops using standard display connections", + "average rainfall during autumn in popular European holiday destinations", + "where to recycle broken household appliances and electronic equipment", + "beginner photography courses covering lighting composition and camera settings", + "quiet washing machines with efficient programmes for delicate clothing", + "requirements for renewing an expired passport before travelling abroad", + "comparison of cloud storage services offering automatic photograph backups", + "how to transfer calendar appointments between different email accounts", + "replacement cushions for outdoor furniture exposed to winter weather", + "local evening classes teaching conversational Spanish to complete beginners" + ] +} diff --git a/neural/model-bundle.json b/neural/model-bundle.json new file mode 100644 index 0000000..fd4497c --- /dev/null +++ b/neural/model-bundle.json @@ -0,0 +1,35 @@ +{ + "format_version": 1, + "model_id": "smollm2-135m-q8-v1", + "source": "HuggingFaceTB/SmolLM2-135M", + "revision": "93efa2f097d58c2a74874c7e644dbc9b0cee75a2", + "conversion": "Candle 0.11.0 Q8_0, F32 norms, adjacent-pair RoPE, tied embeddings; neural/worker/src/bin/quantize.rs", + "policy": { + "context_tokens": 64, + "shortlist": 8, + "max_results": 5, + "scoring": "whole-word log probability plus boundary probability, sequential" + }, + "files": { + "model.gguf": { + "bytes": 143041952, + "sha256": "8d75e9b96c4b64e8a1180224cbaefbbc7744f21ca2e0be2319a429f2c589d342" + }, + "config.json": { + "bytes": 704, + "sha256": "1d556eab73b69c7f11f64c557a2f9c6f440bd4c6b89bb2584a6b498c92603843" + }, + "tokenizer.json": { + "bytes": 2104556, + "sha256": "9ca9acddb6525a194ec8ac7a87f24fbba7232a9a15ffa1af0c1224fcd888e47c" + }, + "MODEL_CARD.md": { + "bytes": 6340, + "sha256": "d1ba68cae64a89b6b434b11526e6e2271ee5ffd2c914ec35ed515f9d84c6085c" + }, + "MODEL_LICENSE.txt": { + "bytes": 11358, + "sha256": "cfc7749b96f63bd31c3c42b5c471bf756814053e847c10f3eb003417bc523d30" + } + } +} diff --git a/neural/source-manifest.json b/neural/source-manifest.json new file mode 100644 index 0000000..c6c54d6 --- /dev/null +++ b/neural/source-manifest.json @@ -0,0 +1,26 @@ +{ + "model": "HuggingFaceTB/SmolLM2-135M", + "revision": "93efa2f097d58c2a74874c7e644dbc9b0cee75a2", + "files": { + "config.json": { + "url": "https://huggingface.co/HuggingFaceTB/SmolLM2-135M/resolve/93efa2f097d58c2a74874c7e644dbc9b0cee75a2/config.json", + "sha256": "1d556eab73b69c7f11f64c557a2f9c6f440bd4c6b89bb2584a6b498c92603843", + "bytes": 704 + }, + "tokenizer.json": { + "url": "https://huggingface.co/HuggingFaceTB/SmolLM2-135M/resolve/93efa2f097d58c2a74874c7e644dbc9b0cee75a2/tokenizer.json", + "sha256": "9ca9acddb6525a194ec8ac7a87f24fbba7232a9a15ffa1af0c1224fcd888e47c", + "bytes": 2104556 + }, + "model.safetensors": { + "url": "https://huggingface.co/HuggingFaceTB/SmolLM2-135M/resolve/93efa2f097d58c2a74874c7e644dbc9b0cee75a2/model.safetensors", + "sha256": "80521b40281d6ce74e35c9282c22539e75aa0ac8578892b2a59955ef78d55da1", + "bytes": 269060552 + }, + "README.md": { + "url": "https://huggingface.co/HuggingFaceTB/SmolLM2-135M/resolve/93efa2f097d58c2a74874c7e644dbc9b0cee75a2/README.md", + "sha256": "d1ba68cae64a89b6b434b11526e6e2271ee5ffd2c914ec35ed515f9d84c6085c", + "bytes": 6340 + } + } +} diff --git a/neural/src/bundle.rs b/neural/src/bundle.rs new file mode 100644 index 0000000..40211ea --- /dev/null +++ b/neural/src/bundle.rs @@ -0,0 +1,49 @@ +//! Compiled compatibility pins, independent of claims in a downloaded manifest. +use crate::{Error, Result}; +use sha2::{Digest, Sha256}; +use std::{fs, path::Path}; + +pub const MODEL_ID: &str = "smollm2-135m-q8-v1"; +pub const MANIFEST: &str = include_str!("../model-bundle.json"); + +pub struct Bundle { + pub weights: Vec, + pub tokenizer: Vec, +} + +pub fn load(path: &Path) -> Result { + let manifest = fs::read(path.join("model-bundle.json")).map_err(|_| Error::Bundle)?; + if manifest.len() > 16_384 { + return Err(Error::Bundle); + } + let actual: serde_json::Value = serde_json::from_slice(&manifest).map_err(|_| Error::Bundle)?; + let expected: serde_json::Value = serde_json::from_str(MANIFEST).map_err(|_| Error::Bundle)?; + if actual != expected { + return Err(Error::Bundle); + } + let mut weights = None; + let mut tokenizer = None; + for (name, pin) in expected["files"].as_object().ok_or(Error::Bundle)? { + let file = path.join(name); + let size = pin["bytes"].as_u64().ok_or(Error::Bundle)?; + if fs::metadata(&file).map_err(|_| Error::Bundle)?.len() != size { + return Err(Error::Bundle); + } + let bytes = fs::read(file).map_err(|_| Error::Bundle)?; + if bytes.len() as u64 != size + || format!("{:x}", Sha256::digest(&bytes)) + != pin["sha256"].as_str().ok_or(Error::Bundle)? + { + return Err(Error::Bundle); + } + match name.as_str() { + "model.gguf" => weights = Some(bytes), + "tokenizer.json" => tokenizer = Some(bytes), + _ => {} + } + } + Ok(Bundle { + weights: weights.ok_or(Error::Bundle)?, + tokenizer: tokenizer.ok_or(Error::Bundle)?, + }) +} diff --git a/neural/src/lib.rs b/neural/src/lib.rs new file mode 100644 index 0000000..c937a3b --- /dev/null +++ b/neural/src/lib.rs @@ -0,0 +1,282 @@ +//! Optional asynchronous refinement. The caller owns the statistical predictor. +//! Text is held in bounded memory and sent only through private child-process pipes. +pub mod bundle; +mod process; +pub mod protocol; + +use protocol::Query; +use serde::Serialize; +use std::{ + path::PathBuf, + sync::{Arc, Condvar, Mutex}, + thread::{self, JoinHandle}, + time::Duration, +}; +use switchify_prediction::{Options, Predictor, normalize, sentences}; + +pub type Result = std::result::Result; +#[derive(Debug, thiserror::Error)] +pub enum Error { + #[error("invalid neural configuration")] + Config, + #[error("model bundle is missing, corrupt or incompatible")] + Bundle, + #[error("input exceeds neural limits")] + Input, +} + +#[derive(Clone)] +pub struct Config { + pub bundle: PathBuf, + pub portable_worker: PathBuf, + pub accelerated_worker: Option, + pub threads: usize, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +pub enum Failure { + Load, + Timeout, + Worker, + Protocol, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +pub enum Status { + Loading, + Ready, + Unavailable(Failure), + Stopped, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +pub struct Capabilities { + pub accelerated: bool, + pub threads: usize, + pub context_tokens: usize, + pub shortlist: usize, + pub deadline_ms: u64, +} +#[derive(Serialize)] +pub struct Immediate { + pub request_id: u64, + pub words: Vec, + pub status: Status, + pub refinement_requested: bool, +} +#[derive(Serialize)] +pub struct Refined { + pub request_id: u64, + pub words: Vec, + pub cache_hit: bool, +} + +struct Shared { + latest: u64, + session: u64, + pending: Option, + result: Option, + status: Status, + reset: bool, + retry: bool, + stop: bool, +} +type State = Arc<(Mutex, Condvar)>; + +/// Single controller, one in-flight query and one replaceable pending query. +/// Poll and submit on the application's controller thread. No callback races. +pub struct Refiner { + state: State, + actor: Option>, + capabilities: Capabilities, +} + +pub fn accelerated_supported() -> bool { + #[cfg(target_arch = "x86_64")] + { + std::is_x86_feature_detected!("avx2") + && std::is_x86_feature_detected!("fma") + && std::is_x86_feature_detected!("f16c") + } + #[cfg(not(target_arch = "x86_64"))] + { + false + } +} + +/// Normalize only the current sentence, matching the established word tokenizer. +pub fn effective_context(before: &str) -> String { + sentences( + before + .rsplit(['.', '!', '?', '\n', '\r']) + .next() + .unwrap_or_default(), + ) + .into_iter() + .flatten() + .collect::>() + .join(" ") +} + +impl Refiner { + pub fn new(config: Config) -> Result { + if !(1..=4).contains(&config.threads) + || !config.bundle.is_dir() + || !config.portable_worker.is_file() + || config + .accelerated_worker + .as_ref() + .is_some_and(|p| !p.is_file()) + { + return Err(Error::Config); + } + let accelerated = accelerated_supported() && config.accelerated_worker.is_some(); + let capabilities = Capabilities { + accelerated, + threads: config.threads, + context_tokens: 64, + shortlist: 8, + deadline_ms: 500, + }; + let state = Arc::new(( + Mutex::new(Shared { + latest: 0, + session: 0, + pending: None, + result: None, + status: Status::Loading, + reset: false, + retry: true, + stop: false, + }), + Condvar::new(), + )); + let cloned = Arc::clone(&state); + let actor = thread::Builder::new() + .name("prediction-refiner".into()) + .spawn(move || process::run(config, accelerated, cloned)) + .map_err(|_| Error::Config)?; + Ok(Self { + state, + actor: Some(actor), + capabilities, + }) + } + + pub fn capabilities(&self) -> Capabilities { + self.capabilities + } + pub fn status(&self) -> Status { + self.state.0.lock().unwrap().status + } + + /// Capture both immediate results and the shortlist from the same caller snapshot. + /// Limits above five are rejected. Context is capped at 16 KiB, prefix at 256 bytes. + pub fn submit( + &mut self, + predictor: &Predictor, + before: &str, + prefix: &str, + options: Options, + session: u64, + ) -> Result { + if before.len() > 16_384 || prefix.len() > 256 || options.limit > 5 { + return Err(Error::Input); + } + let candidates: Vec = predictor + .predict( + before, + prefix, + Options { + limit: 8, + ..options + }, + ) + .into_iter() + .map(|s| s.word) + .collect(); + let words: Vec = candidates.iter().take(options.limit).cloned().collect(); + let valid = candidates + .iter() + .all(|w| w.len() <= 128 && normalize(w) == *w && sentences(w) == vec![vec![w.clone()]]); + let mut shared = self.state.0.lock().unwrap(); + shared.latest = shared.latest.checked_add(1).ok_or(Error::Input)?; + if shared.session != session { + shared.reset = true; + shared.session = session; + } + shared.result = None; + shared.pending = None; + let refinement_requested = valid + && options.limit > 0 + && !options.unigram_only + && !candidates.is_empty() + && matches!(shared.status, Status::Loading | Status::Ready); + if refinement_requested { + shared.pending = Some(Query { + id: shared.latest, + session, + before: effective_context(before), + candidates, + limit: options.limit, + }); + } + let result = Immediate { + request_id: shared.latest, + words, + status: shared.status, + refinement_requested, + }; + self.state.1.notify_one(); + Ok(result) + } + + /// A subsequent submit/reset invalidates any previously unconsumed result. + pub fn poll(&mut self) -> Option { + self.state.0.lock().unwrap().result.take() + } + + /// Clear queued text and worker context. In-flight replies are discarded. + pub fn reset(&mut self) { + let mut shared = self.state.0.lock().unwrap(); + shared.pending = None; + shared.result = None; + shared.reset = true; + shared.latest = shared.latest.saturating_add(1); + self.state.1.notify_one(); + } + + /// Explicitly reload after failure. Never retries automatically in the background. + pub fn retry(&mut self) { + let mut shared = self.state.0.lock().unwrap(); + if matches!(shared.status, Status::Unavailable(_)) { + shared.status = Status::Loading; + shared.retry = true; + self.state.1.notify_one(); + } + } + + pub fn shutdown(&mut self) { + { + let mut shared = self.state.0.lock().unwrap(); + shared.stop = true; + shared.pending = None; + shared.result = None; + self.state.1.notify_one(); + } + if let Some(actor) = self.actor.take() { + let _ = actor.join(); + } + } +} +impl Drop for Refiner { + fn drop(&mut self) { + self.shutdown(); + } +} + +fn pause(state: &State) { + let shared = state.0.lock().unwrap(); + let _ = state + .1 + .wait_timeout(shared, Duration::from_millis(5)) + .unwrap(); +} diff --git a/neural/src/main.rs b/neural/src/main.rs new file mode 100644 index 0000000..c0b0bc6 --- /dev/null +++ b/neural/src/main.rs @@ -0,0 +1,198 @@ +use clap::{Parser, Subcommand}; +use serde::Deserialize; +use serde_json::json; +use std::{ + io::{self, BufRead, Read, Write}, + path::PathBuf, + sync::mpsc, + thread, + time::{Duration, Instant}, +}; +use switchify_prediction::{Options, Predictor}; +use switchify_prediction_neural::{Config, Refiner, Status, bundle}; + +#[derive(Parser)] +#[command( + version, + about = "Offline SmolLM2 companion. User text is accepted only on stdin." +)] +struct Args { + #[command(subcommand)] + command: Commands, +} +#[derive(Subcommand)] +enum Commands { + Validate { + #[arg(long)] + bundle: PathBuf, + }, + Once(Run), + Stream(Run), +} +#[derive(clap::Args)] +struct Run { + #[arg(long)] + baseline: PathBuf, + #[arg(long)] + personal: Option, + #[arg(long)] + bundle: PathBuf, + #[arg(long)] + worker: PathBuf, + #[arg(long)] + accelerated_worker: Option, + #[arg(long, default_value_t = 4)] + threads: usize, +} +#[derive(Deserialize)] +#[serde(tag = "command", rename_all = "snake_case", deny_unknown_fields)] +enum Input { + Predict { + before: String, + prefix: String, + session: u64, + #[serde(default = "five")] + limit: usize, + #[serde(default = "two")] + min_chars: usize, + #[serde(default)] + unigram_only: bool, + }, + Reset, + Retry, +} +fn five() -> usize { + 5 +} +fn two() -> usize { + 2 +} +fn emit(value: serde_json::Value) -> Result<(), ()> { + let mut out = io::stdout().lock(); + serde_json::to_writer(&mut out, &value).map_err(|_| ())?; + writeln!(out).and_then(|_| out.flush()).map_err(|_| ()) +} +fn input() -> mpsc::Receiver> { + let (tx, rx) = mpsc::sync_channel(1); + thread::spawn(move || { + let mut stdin = io::stdin().lock(); + loop { + let mut bytes = Vec::new(); + let read = (&mut stdin).take(65_537).read_until(b'\n', &mut bytes); + if matches!(read, Ok(0)) { + break; + } + let item = if read.is_err() || bytes.len() > 65_536 { + Err(()) + } else { + serde_json::from_slice(&bytes).map_err(|_| ()) + }; + let failed = item.is_err(); + if tx.send(item).is_err() || failed { + break; + } + } + }); + rx +} +fn run(args: Run, once: bool) -> Result<(), ()> { + let predictor = Predictor::open(&args.baseline, args.personal.as_deref()).map_err(|_| ())?; + let mut engine = Refiner::new(Config { + bundle: args.bundle, + portable_worker: args.worker, + accelerated_worker: args.accelerated_worker, + threads: args.threads, + }) + .map_err(|_| ())?; + while engine.status() == Status::Loading { + thread::sleep(Duration::from_millis(5)); + } + if engine.status() != Status::Ready { + return Err(()); + } + emit(json!({"type":"ready", "capabilities":engine.capabilities()}))?; + let rx = input(); + let mut outstanding = false; + let mut eof = false; + let mut started = Instant::now(); + let mut last_status = Status::Ready; + loop { + if let Some(result) = engine.poll() { + emit( + json!({"type":"refined", "result":result, "elapsed_ms":started.elapsed().as_secs_f64()*1000.}), + )?; + outstanding = false; + } + let status = engine.status(); + if status != last_status { + emit(json!({"type":"status", "status":status}))?; + last_status = status; + if matches!(status, Status::Unavailable(_)) { + outstanding = false; + } + } + if eof && !outstanding { + return Ok(()); + } + if eof { + thread::sleep(Duration::from_millis(1)); + continue; + } + match rx.recv_timeout(Duration::from_millis(1)) { + Ok(Ok(Input::Predict { + before, + prefix, + session, + limit, + min_chars, + unigram_only, + })) => { + started = Instant::now(); + let result = engine + .submit( + &predictor, + &before, + &prefix, + Options { + limit, + min_chars, + unigram_only, + }, + session, + ) + .map_err(|_| ())?; + outstanding = result.refinement_requested; + emit( + json!({"type":"immediate", "result":result, "elapsed_ms":started.elapsed().as_secs_f64()*1000.}), + )?; + if once { + eof = true; + } + } + Ok(Ok(Input::Reset)) => { + engine.reset(); + outstanding = false; + emit(json!({"type":"reset"}))?; + } + Ok(Ok(Input::Retry)) => engine.retry(), + Ok(Err(())) => return Err(()), + Err(mpsc::RecvTimeoutError::Disconnected) => eof = true, + Err(mpsc::RecvTimeoutError::Timeout) => {} + } + } +} +fn main() { + std::panic::set_hook(Box::new(|_| {})); + let result = match Args::parse().command { + Commands::Validate { bundle: path } => bundle::load(&path) + .map(|_| ()) + .map_err(|_| ()) + .and_then(|_| emit(json!({"valid":true,"model_id":bundle::MODEL_ID}))), + Commands::Once(args) => run(args, true), + Commands::Stream(args) => run(args, false), + }; + if result.is_err() { + eprintln!("neural command failed; check configuration, bundle and input schema"); + std::process::exit(1); + } +} diff --git a/neural/src/process.rs b/neural/src/process.rs new file mode 100644 index 0000000..952b348 --- /dev/null +++ b/neural/src/process.rs @@ -0,0 +1,196 @@ +use crate::{ + Config, Failure, Refined, State, Status, pause, + protocol::{self, Command, Reply}, +}; +use std::{ + io, + process::{Child, Command as Spawn, Stdio}, + sync::mpsc::{self, Receiver, SyncSender}, + thread::{self, JoinHandle}, + time::{Duration, Instant}, +}; + +struct Worker { + child: Child, + commands: Option>, + writer: Option>, + replies: Receiver>, + reader: Option>, +} +impl Worker { + fn start(config: &Config, accelerated: bool) -> Result { + let path = if accelerated { + config.accelerated_worker.as_ref().unwrap() + } else { + &config.portable_worker + }; + let mut command = Spawn::new(path); + command + .arg(&config.bundle) + .env("RAYON_NUM_THREADS", config.threads.to_string()) + .env("TOKENIZERS_PARALLELISM", "false") + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::null()); + #[cfg(windows)] + { + use std::os::windows::process::CommandExt; + command.creation_flags(0x08000000); + } + let mut child = command.spawn().map_err(|_| Failure::Load)?; + let mut input = child.stdin.take().unwrap(); + let mut output = child.stdout.take().unwrap(); + // The reader never buffers a stream of unsolicited replies. + let (tx, replies) = mpsc::sync_channel(1); + let reader = thread::spawn(move || { + loop { + let result = protocol::read_frame(&mut output); + let failed = result.is_err(); + if tx.try_send(result).is_err() || failed { + break; + } + } + }); + let (commands, rx) = mpsc::sync_channel(1); + let writer = thread::spawn(move || { + while let Ok(command) = rx.recv() { + if protocol::write_frame(&mut input, &command).is_err() { + break; + } + } + }); + Ok(Self { + child, + commands: Some(commands), + writer: Some(writer), + replies, + reader: Some(reader), + }) + } + fn receive(&self, state: &State, timeout: Duration) -> Result { + let start = Instant::now(); + loop { + if state.0.lock().unwrap().stop { + return Err(Failure::Worker); + } + match self.replies.recv_timeout(Duration::from_millis(5)) { + Ok(Ok(reply)) => return Ok(reply), + Ok(Err(_)) => return Err(Failure::Protocol), + Err(mpsc::RecvTimeoutError::Disconnected) => return Err(Failure::Worker), + Err(mpsc::RecvTimeoutError::Timeout) => {} + } + if start.elapsed() >= timeout { + return Err(Failure::Timeout); + } + } + } + fn send(&mut self, command: Command) -> Result<(), Failure> { + self.commands + .as_ref() + .unwrap() + .try_send(command) + .map_err(|_| Failure::Worker) + } +} +impl Drop for Worker { + fn drop(&mut self) { + let _ = self.child.kill(); + let _ = self.child.wait(); + self.commands.take(); + if let Some(writer) = self.writer.take() { + let _ = writer.join(); + } + if let Some(reader) = self.reader.take() { + let _ = reader.join(); + } + } +} + +pub(super) fn run(config: Config, accelerated: bool, state: State) { + let mut worker: Option = None; + loop { + let (stop, retry, reset, query) = { + let mut s = state.0.lock().unwrap(); + ( + s.stop, + std::mem::take(&mut s.retry), + std::mem::take(&mut s.reset), + s.pending.take(), + ) + }; + if stop { + break; + } + let result = (|| { + if retry { + worker = Some(Worker::start(&config, accelerated)?); + match worker + .as_ref() + .unwrap() + .receive(&state, Duration::from_secs(30))? + { + Reply::Ready { + version: protocol::VERSION, + accelerated: actual, + } if actual == accelerated => { + state.0.lock().unwrap().status = Status::Ready; + } + _ => return Err(Failure::Load), + } + } + if let Some(w) = &mut worker { + if reset { + w.send(Command::Reset)?; + if !matches!(w.receive(&state, Duration::from_millis(500))?, Reply::Reset) { + return Err(Failure::Protocol); + } + } + if let Some(query) = query { + let id = query.id; + if state.0.lock().unwrap().latest != id { + return Ok(()); + } + let limit = query.limit.min(query.candidates.len()); + let candidates = query.candidates.clone(); + w.send(Command::Predict(query))?; + match w.receive(&state, Duration::from_millis(500))? { + Reply::Ranked { + id: actual, + words, + cache_hit, + } if actual == id + && words.len() == limit + && words.iter().all(|w| candidates.contains(w)) + && words + .iter() + .collect::>() + .len() + == words.len() => + { + let mut s = state.0.lock().unwrap(); + if s.latest == id && !s.reset && !s.stop { + s.result = Some(Refined { + request_id: id, + words, + cache_hit, + }); + } + } + _ => return Err(Failure::Protocol), + } + } + } + Ok(()) + })(); + if let Err(error) = result { + worker = None; + let mut s = state.0.lock().unwrap(); + s.status = Status::Unavailable(error); + s.pending = None; + s.result = None; + } + pause(&state); + } + drop(worker); + state.0.lock().unwrap().status = Status::Stopped; +} diff --git a/neural/src/protocol.rs b/neural/src/protocol.rs new file mode 100644 index 0000000..44a08d7 --- /dev/null +++ b/neural/src/protocol.rs @@ -0,0 +1,80 @@ +//! Bounded private child-process transport. No sockets and no text diagnostics. +use serde::{Deserialize, Serialize, de::DeserializeOwned}; +use std::io::{self, Read, Write}; + +pub const VERSION: u32 = 1; +pub const MAX_FRAME: usize = 65_536; + +#[derive(Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct Query { + pub id: u64, + pub session: u64, + pub before: String, + pub candidates: Vec, + pub limit: usize, +} + +#[derive(Serialize, Deserialize)] +pub enum Command { + Predict(Query), + Reset, +} + +#[derive(Serialize, Deserialize)] +pub enum Reply { + Ready { + version: u32, + accelerated: bool, + }, + Ranked { + id: u64, + words: Vec, + cache_hit: bool, + }, + Reset, + Failed, +} + +fn invalid() -> io::Error { + io::Error::new(io::ErrorKind::InvalidData, "invalid worker frame") +} + +pub fn write_frame(mut out: impl Write, value: &impl Serialize) -> io::Result<()> { + let bytes = serde_json::to_vec(value).map_err(|_| invalid())?; + if bytes.is_empty() || bytes.len() > MAX_FRAME { + return Err(invalid()); + } + out.write_all(&(bytes.len() as u32).to_le_bytes())?; + out.write_all(&bytes)?; + out.flush() +} + +pub fn read_frame(mut input: impl Read) -> io::Result { + let mut header = [0; 4]; + input.read_exact(&mut header)?; + let length = u32::from_le_bytes(header) as usize; + if length == 0 || length > MAX_FRAME { + return Err(invalid()); + } + let mut bytes = vec![0; length]; + input.read_exact(&mut bytes)?; + serde_json::from_slice(&bytes).map_err(|_| invalid()) +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn rejects_bad_frames_without_echoing_input() { + for bytes in [vec![], vec![255; 4], vec![0; 4], vec![1, 0, 0, 0, b'!']] { + assert!(read_frame::(&bytes[..]).is_err()); + } + let mut bytes = Vec::new(); + write_frame(&mut bytes, &Command::Reset).unwrap(); + assert!(matches!( + read_frame::(&bytes[..]).unwrap(), + Command::Reset + )); + } +} diff --git a/neural/tests/fake_worker.rs b/neural/tests/fake_worker.rs new file mode 100644 index 0000000..fc92ee1 --- /dev/null +++ b/neural/tests/fake_worker.rs @@ -0,0 +1,64 @@ +use std::{ + io::{self, Write}, + path::PathBuf, + thread, + time::Duration, +}; +use switchify_prediction_neural::protocol::{self, Command, Reply}; +fn main() { + let mode = + std::fs::read_to_string(PathBuf::from(std::env::args_os().nth(1).unwrap()).join("mode")) + .unwrap(); + if mode == "load-stall" { + thread::sleep(Duration::from_secs(30)); + return; + } + let mut out = io::stdout().lock(); + protocol::write_frame( + &mut out, + &Reply::Ready { + version: protocol::VERSION, + accelerated: false, + }, + ) + .unwrap(); + if mode == "read-stall" { + thread::sleep(Duration::from_secs(30)); + return; + } + let mut input = io::stdin().lock(); + while let Ok(command) = protocol::read_frame(&mut input) { + let reply = match command { + Command::Reset => Reply::Reset, + Command::Predict(mut query) => { + match mode.as_str() { + "crash" => return, + "stall" => thread::sleep(Duration::from_secs(30)), + "delay" => thread::sleep(Duration::from_millis(100)), + "malformed" => { + out.write_all(&[1, 0, 0, 0, b'!']).unwrap(); + out.flush().unwrap(); + return; + } + "oversized" => { + out.write_all(&[255; 4]).unwrap(); + out.flush().unwrap(); + return; + } + "id" => query.id += 1, + "foreign" => query.candidates = vec!["not-in-shortlist".into()], + _ => {} + } + query.candidates.reverse(); + Reply::Ranked { + id: query.id, + words: query.candidates.into_iter().take(query.limit).collect(), + cache_hit: false, + } + } + }; + if protocol::write_frame(&mut out, &reply).is_err() { + break; + } + } +} diff --git a/neural/tests/lifecycle.rs b/neural/tests/lifecycle.rs new file mode 100644 index 0000000..b8f4740 --- /dev/null +++ b/neural/tests/lifecycle.rs @@ -0,0 +1,181 @@ +use std::{ + thread, + time::{Duration, Instant}, +}; +use switchify_prediction::{Options, Predictor, build}; +use switchify_prediction_neural::{Config, Refiner, Status, bundle, effective_context}; +use tempfile::TempDir; +fn fixture(mode: &str) -> (TempDir, Predictor, Refiner) { + let temp = tempfile::tempdir().unwrap(); + std::fs::write(temp.path().join("mode"), mode).unwrap(); + let path = temp.path().join("baseline.sqlite"); + build( + &path, + "alpha beta gamma delta epsilon zeta eta theta iota café can't", + "test", + ) + .unwrap(); + let predictor = Predictor::open(&path, Some(&temp.path().join("personal.sqlite"))).unwrap(); + let refiner = Refiner::new(Config { + bundle: temp.path().to_owned(), + portable_worker: env!("CARGO_BIN_EXE_fake-worker").into(), + accelerated_worker: None, + threads: 1, + }) + .unwrap(); + (temp, predictor, refiner) +} +fn until(mut check: impl FnMut() -> bool) { + let start = Instant::now(); + while !check() { + assert!( + start.elapsed() < Duration::from_secs(5), + "condition timed out" + ); + thread::sleep(Duration::from_millis(2)); + } +} +fn options() -> Options { + Options { + min_chars: 0, + ..Options::default() + } +} +#[test] +fn immediate_snapshot_options_and_latest_only() { + let (_temp, mut predictor, mut engine) = fixture("delay"); + until(|| engine.status() == Status::Ready); + let expected: Vec<_> = predictor + .predict("", "", options()) + .into_iter() + .map(|s| s.word) + .collect(); + let first = engine.submit(&predictor, "", "", options(), 1).unwrap(); + assert_eq!(first.words, expected); + thread::sleep(Duration::from_millis(20)); + for _ in 0..30 { + engine.submit(&predictor, "", "", options(), 1).unwrap(); + } + predictor.learn("zeta zeta zeta zeta zeta").unwrap(); + let latest = engine.submit(&predictor, "", "ze", options(), 2).unwrap(); + assert_eq!(latest.words, ["zeta"]); + let mut refined = None; + until(|| { + refined = engine.poll(); + refined.is_some() + }); + let refined = refined.unwrap(); + assert_eq!(refined.request_id, latest.request_id); + assert_eq!(refined.words, ["zeta"]); + engine.submit(&predictor, "", "", options(), 2).unwrap(); + engine.reset(); + thread::sleep(Duration::from_millis(150)); + assert!(engine.poll().is_none()); + for opts in [ + Options { + limit: 0, + ..options() + }, + Options { + min_chars: 2, + ..options() + }, + Options { + unigram_only: true, + ..options() + }, + ] { + assert!( + !engine + .submit(&predictor, "", "", opts, 3) + .unwrap() + .refinement_requested + ); + } + assert!( + engine + .submit( + &predictor, + "", + "", + Options { + limit: 6, + ..options() + }, + 3 + ) + .is_err() + ); + assert_eq!( + engine + .submit(&predictor, "", "CAFE\u{301}", options(), 3) + .unwrap() + .words, + ["café"] + ); + assert_eq!( + engine + .submit(&predictor, "", "CAN’T", options(), 3) + .unwrap() + .words, + ["can't"] + ); +} +#[test] +fn failures_leave_immediate_available_and_require_explicit_retry() { + for mode in [ + "stall", + "read-stall", + "crash", + "malformed", + "oversized", + "id", + "foreign", + ] { + let (temp, predictor, mut engine) = fixture(mode); + until(|| engine.status() == Status::Ready); + // This also exceeds the OS pipe buffer for a worker that never reads. + let before = "word ".repeat(3000); + let first = engine + .submit(&predictor, &before, "", options(), 1) + .unwrap(); + assert_eq!(first.words.len(), 5); + until(|| matches!(engine.status(), Status::Unavailable(_))); + assert!(engine.poll().is_none()); + assert!( + !engine + .submit(&predictor, "", "", options(), 1) + .unwrap() + .refinement_requested + ); + std::fs::write(temp.path().join("mode"), "ok").unwrap(); + engine.retry(); + until(|| engine.status() == Status::Ready); + engine.submit(&predictor, "", "", options(), 1).unwrap(); + until(|| engine.poll().is_some()); + } +} +#[test] +fn shutdown_interrupts_model_loading() { + let (_temp, _predictor, mut engine) = fixture("load-stall"); + thread::sleep(Duration::from_millis(50)); + let start = Instant::now(); + engine.shutdown(); + assert!(start.elapsed() < Duration::from_secs(1)); + assert_eq!(engine.status(), Status::Stopped); +} +#[test] +fn context_and_bundle_validation() { + assert_eq!( + effective_context("Ignore me! CAFE\u{301} CAN’T"), + "café can't" + ); + assert_eq!(effective_context("last sentence."), ""); + assert_eq!(effective_context("first\nsecond line"), "second line"); + let temp = tempfile::tempdir().unwrap(); + assert!(bundle::load(temp.path()).is_err()); + std::fs::write(temp.path().join("model-bundle.json"), "{}").unwrap(); + assert!(bundle::load(temp.path()).is_err()); + std::fs::write(temp.path().join("model-bundle.json"), bundle::MANIFEST).unwrap(); + assert!(bundle::load(temp.path()).is_err()); +} diff --git a/neural/worker/Cargo.toml b/neural/worker/Cargo.toml new file mode 100644 index 0000000..cb46d92 --- /dev/null +++ b/neural/worker/Cargo.toml @@ -0,0 +1,19 @@ +[package] +name = "switchify-smol-worker" +version = "0.1.0" +edition = "2024" +rust-version = "1.97.1" +license = "MIT" +publish = false + +[features] +accelerated = [] + +[dependencies] +switchify-prediction-neural = { path = ".." } +anyhow = "1" +candle-core = "=0.11.0" +candle-transformers = "=0.11.0" +serde_json = "1" +tokenizers = { version = "0.22", default-features = false, features = ["fancy-regex"] } + diff --git a/neural/worker/src/bin/quantize.rs b/neural/worker/src/bin/quantize.rs new file mode 100644 index 0000000..f0764b7 --- /dev/null +++ b/neural/worker/src/bin/quantize.rs @@ -0,0 +1,143 @@ +//! Local conversion of the pinned SmolLM2 weights. No downloaded GGUF is trusted. +use anyhow::{Result, ensure}; +use candle_core::{ + DType, Device, Tensor, + quantized::{ + GgmlDType, QTensor, + gguf_file::{self, Value}, + }, +}; +use candle_transformers::models::llama::LlamaConfig; +use std::{collections::BTreeMap, fs, path::PathBuf}; + +fn interleave_rope(tensor: Tensor, heads: usize) -> Result { + let (rows, cols) = tensor.dims2()?; + ensure!(heads > 0 && rows % (heads * 2) == 0, "Invalid rotary shape"); + Ok(tensor + .reshape((heads, 2, rows / heads / 2, cols))? + .transpose(1, 2)? + .contiguous()? + .reshape((rows, cols))?) +} + +fn main() -> Result<()> { + let args: Vec<_> = std::env::args_os().skip(1).collect(); + ensure!( + args.len() == 3, + "Usage: quantize MODEL_DIR OUTPUT.gguf f32|q8" + ); + let root = PathBuf::from(&args[0]); + let output = PathBuf::from(&args[1]); + ensure!(!output.exists(), "Output already exists"); + let dtype = match args[2].to_str() { + Some("f32") => GgmlDType::F32, + Some("q8") => GgmlDType::Q8_0, + _ => anyhow::bail!("Expected f32 or q8"), + }; + let cfg: LlamaConfig = serde_json::from_slice(&fs::read(root.join("config.json"))?)?; + ensure!( + cfg.hidden_size == 576 + && cfg.num_hidden_layers == 30 + && cfg.tie_word_embeddings == Some(true) + && cfg.rope_scaling.is_none(), + "Expected pinned SmolLM2 configuration" + ); + let source = candle_core::safetensors::load(root.join("model.safetensors"), &Device::Cpu)?; + let mut tensors = BTreeMap::new(); + let mut add = |target: String, name: &str, heads: Option| -> Result<()> { + let mut tensor: Tensor = source + .get(name) + .ok_or_else(|| anyhow::anyhow!("Missing {name}"))? + .to_dtype(DType::F32)?; + // HF RoPE pairs the two halves; GGUF llama pairs adjacent coordinates. + if let Some(heads) = heads { + tensor = interleave_rope(tensor, heads)?; + } + let kind = if tensor.rank() == 2 { + dtype + } else { + GgmlDType::F32 + }; + tensors.insert(target, QTensor::quantize(&tensor, kind)?); + Ok(()) + }; + add( + "token_embd.weight".into(), + "model.embed_tokens.weight", + None, + )?; + add("output_norm.weight".into(), "model.norm.weight", None)?; + for i in 0..cfg.num_hidden_layers { + for (target, name, heads) in [ + ("attn_q", "self_attn.q_proj", Some(cfg.num_attention_heads)), + ( + "attn_k", + "self_attn.k_proj", + Some(cfg.num_key_value_heads()), + ), + ("attn_v", "self_attn.v_proj", None), + ("attn_output", "self_attn.o_proj", None), + ("ffn_gate", "mlp.gate_proj", None), + ("ffn_down", "mlp.down_proj", None), + ("ffn_up", "mlp.up_proj", None), + ("attn_norm", "input_layernorm", None), + ("ffn_norm", "post_attention_layernorm", None), + ] { + add( + format!("blk.{i}.{target}.weight"), + &format!("model.layers.{i}.{name}.weight"), + heads, + )?; + } + } + let metadata = [ + ("general.architecture", Value::String("llama".into())), + ( + "llama.attention.head_count", + Value::U32(cfg.num_attention_heads as u32), + ), + ( + "llama.attention.head_count_kv", + Value::U32(cfg.num_key_value_heads() as u32), + ), + ( + "llama.block_count", + Value::U32(cfg.num_hidden_layers as u32), + ), + ("llama.embedding_length", Value::U32(cfg.hidden_size as u32)), + ( + "llama.rope.dimension_count", + Value::U32((cfg.hidden_size / cfg.num_attention_heads) as u32), + ), + ( + "llama.attention.layer_norm_rms_epsilon", + Value::F32(cfg.rms_norm_eps as f32), + ), + ("llama.rope.freq_base", Value::F32(cfg.rope_theta)), + ]; + let metadata: Vec<_> = metadata.iter().map(|(k, v)| (*k, v)).collect(); + let tensors: Vec<_> = tensors.iter().map(|(k, v)| (k.as_str(), v)).collect(); + gguf_file::write(&mut fs::File::create_new(output)?, &metadata, &tensors)?; + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn rotary_permutation_stays_within_each_attention_head() { + let tensor = Tensor::arange(0_f32, 16_f32, &Device::Cpu) + .unwrap() + .reshape((8, 2)) + .unwrap(); + let result = interleave_rope(tensor.clone(), 2) + .unwrap() + .to_vec2::() + .unwrap(); + assert_eq!( + result.iter().map(|row| row[0]).collect::>(), + [0., 4., 2., 6., 8., 12., 10., 14.] + ); + assert!(interleave_rope(tensor, 3).is_err()); + } +} diff --git a/neural/worker/src/main.rs b/neural/worker/src/main.rs new file mode 100644 index 0000000..8d55eb5 --- /dev/null +++ b/neural/worker/src/main.rs @@ -0,0 +1,74 @@ +//! Dedicated offline inference process. All errors are deliberately text-free. +mod model; +use std::{io, path::PathBuf}; +use switchify_prediction_neural::{ + bundle, + protocol::{self, Command, Reply}, +}; + +#[cfg(all( + feature = "accelerated", + not(all( + target_arch = "x86_64", + target_feature = "avx2", + target_feature = "fma", + target_feature = "f16c" + )) +))] +compile_error!("accelerated workers require x86_64 +avx2,+fma,+f16c"); + +fn run() -> anyhow::Result<()> { + let args: Vec<_> = std::env::args_os().skip(1).collect(); + anyhow::ensure!(args.len() == 1, "configuration"); + let bundle = bundle::load(&PathBuf::from(&args[0]))?; + let mut model = model::Model::load(bundle)?; + let mut input = io::stdin().lock(); + let mut output = io::stdout().lock(); + protocol::write_frame( + &mut output, + &Reply::Ready { + version: protocol::VERSION, + accelerated: cfg!(feature = "accelerated"), + }, + )?; + loop { + let command = match protocol::read_frame(&mut input) { + Ok(command) => command, + Err(e) if e.kind() == io::ErrorKind::UnexpectedEof => return Ok(()), + Err(_) => anyhow::bail!("protocol"), + }; + let reply = match command { + Command::Reset => { + model.reset(); + Reply::Reset + } + Command::Predict(query) => { + anyhow::ensure!( + query.before.len() <= 16_384 + && query.candidates.len() <= 8 + && query.limit <= 5 + && query + .candidates + .iter() + .all(|s| !s.is_empty() && s.len() <= 128), + "query" + ); + let (words, cache_hit) = + model.rank(query.session, &query.before, &query.candidates, query.limit)?; + Reply::Ranked { + id: query.id, + words, + cache_hit, + } + } + }; + protocol::write_frame(&mut output, &reply)?; + } +} +fn main() { + // Panic messages may include third-party input values. Never send them to logs. + std::panic::set_hook(Box::new(|_| {})); + if run().is_err() { + std::process::exit(1); + } +} diff --git a/neural/worker/src/model.rs b/neural/worker/src/model.rs new file mode 100644 index 0000000..387b68b --- /dev/null +++ b/neural/worker/src/model.rs @@ -0,0 +1,144 @@ +use anyhow::{Result, ensure}; +use candle_core::{Device, Tensor}; +use candle_transformers::models::quantized_llama::ModelWeights; +use switchify_prediction_neural::bundle::Bundle; +use tokenizers::Tokenizer; + +struct Prepared { + session: u64, + ids: Vec, + state: ModelWeights, + logits: Vec, +} +pub struct Model { + empty: ModelWeights, + tokenizer: Tokenizer, + boundaries: Vec, + prepared: Option, +} +fn forward(state: &mut ModelWeights, ids: &[u32], position: usize) -> Result> { + Ok(state + .forward(&Tensor::new(ids, &Device::Cpu)?.unsqueeze(0)?, position)? + .squeeze(0)? + .to_vec1()?) +} +fn log_prob(logits: &[f32], token: usize) -> f64 { + let max = logits.iter().copied().fold(f32::NEG_INFINITY, f32::max) as f64; + logits[token] as f64 + - max + - logits + .iter() + .map(|x| (*x as f64 - max).exp()) + .sum::() + .ln() +} +fn log_boundary(logits: &[f32], ids: &[usize]) -> f64 { + let max = logits.iter().copied().fold(f32::NEG_INFINITY, f32::max) as f64; + let all = logits.iter().map(|x| (*x as f64 - max).exp()).sum::(); + (ids.iter() + .map(|&i| (logits[i] as f64 - max).exp()) + .sum::() + / all) + .ln() +} +impl Model { + pub fn load(bundle: Bundle) -> Result { + let tokenizer = Tokenizer::from_bytes(&bundle.tokenizer).map_err(anyhow::Error::msg)?; + ensure!( + tokenizer.get_vocab_size(true) == 49152 + && tokenizer.token_to_id("<|endoftext|>") == Some(0), + "tokenizer" + ); + let mut reader = std::io::Cursor::new(bundle.weights); + let content = candle_core::quantized::gguf_file::Content::read(&mut reader)?; + let empty = ModelWeights::from_gguf(content, &mut reader, &Device::Cpu)?; + let mut boundaries = vec![0]; + for id in 1..49152 { + let text = tokenizer.decode(&[id], false).map_err(anyhow::Error::msg)?; + if text.chars().next().is_some_and(|c| { + c.is_whitespace() || c.is_numeric() || ".!?;:,()[]{}\"-/".contains(c) + }) { + boundaries.push(id as usize); + } + } + Ok(Self { + empty, + tokenizer, + boundaries, + prepared: None, + }) + } + pub fn reset(&mut self) { + self.prepared = None; + } + pub fn rank( + &mut self, + session: u64, + before: &str, + candidates: &[String], + limit: usize, + ) -> Result<(Vec, bool)> { + let encoded = self + .tokenizer + .encode(before, false) + .map_err(anyhow::Error::msg)?; + let tokens = encoded.get_ids(); + let mut ids = vec![0]; + ids.extend_from_slice(&tokens[tokens.len().saturating_sub(63)..]); + let cache_hit = self + .prepared + .as_ref() + .is_some_and(|p| p.session == session && p.ids == ids); + if !cache_hit { + self.reset(); + let mut state = self.empty.clone(); + let logits = forward(&mut state, &ids, 0)?; + self.prepared = Some(Prepared { + session, + ids, + state, + logits, + }); + } + let prepared = self.prepared.as_ref().unwrap(); + let mut scored = Vec::new(); + for (index, word) in candidates.iter().enumerate() { + let text = if before.is_empty() { + word.clone() + } else { + format!(" {word}") + }; + let encoded = self + .tokenizer + .encode(text, false) + .map_err(anyhow::Error::msg)?; + ensure!(!encoded.get_ids().is_empty(), "candidate"); + let mut state = prepared.state.clone(); + let mut logits = prepared.logits.clone(); + let mut score = 0.; + for (offset, &token) in encoded.get_ids().iter().enumerate() { + score += log_prob(&logits, token as usize); + logits = forward(&mut state, &[token], prepared.ids.len() + offset)?; + } + score += log_boundary(&logits, &self.boundaries); + ensure!(score.is_finite(), "score"); + scored.push((score, index, word.clone())); + } + scored.sort_by(|a, b| b.0.total_cmp(&a.0).then(a.1.cmp(&b.1))); + Ok(( + scored.into_iter().take(limit).map(|x| x.2).collect(), + cache_hit, + )) + } +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn boundary_probability_is_part_of_whole_word_score() { + assert!((log_prob(&[0., 0.], 0) + 2_f64.ln()).abs() < 1e-12); + assert!((log_boundary(&[0., 0.], &[1]) + 2_f64.ln()).abs() < 1e-12); + assert_eq!(log_boundary(&[0., 0.], &[0, 1]), 0.); + } +} diff --git a/scripts/neural_bundle.py b/scripts/neural_bundle.py new file mode 100644 index 0000000..9aa8da0 --- /dev/null +++ b/scripts/neural_bundle.py @@ -0,0 +1,54 @@ +#!/usr/bin/env python3 +"""Assemble a local pinned model bundle. Never uploads or downloads assets.""" +import argparse +import hashlib +import json +from pathlib import Path +import shutil +import subprocess +import tempfile + +ROOT = Path(__file__).resolve().parent.parent + + +def verify(path, pin): + if path.stat().st_size != pin['bytes'] or hashlib.file_digest(path.open('rb'), 'sha256').hexdigest() != pin['sha256']: + raise ValueError(f'Pinned file mismatch: {path.name}') + + +def assemble(source, gguf, output): + manifest = json.loads((ROOT / 'neural/model-bundle.json').read_bytes()) + pins = json.loads((ROOT / 'neural/source-manifest.json').read_bytes()) + for name, pin in pins['files'].items(): + verify(source / name, pin) + paths = {'model.gguf': gguf, 'MODEL_CARD.md': source / 'README.md', + 'MODEL_LICENSE.txt': ROOT / 'neural/MODEL_LICENSE.txt', + 'config.json': source / 'config.json', 'tokenizer.json': source / 'tokenizer.json'} + for name, path in paths.items(): + verify(path, manifest['files'][name]) + if output.exists(): + raise ValueError('Output already exists') + output.parent.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory(dir=output.parent) as temp: + stage = Path(temp) / 'bundle' + stage.mkdir() + for name, path in paths.items(): + shutil.copyfile(path, stage / name) + verify(stage / name, manifest['files'][name]) + shutil.copyfile(ROOT / 'neural/model-bundle.json', stage / 'model-bundle.json') + shutil.copyfile(ROOT / 'neural/source-manifest.json', stage / 'source-manifest.json') + shutil.copyfile(ROOT / 'neural/worker/src/bin/quantize.rs', stage / 'quantize.rs') + stage.rename(output) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--source', type=Path, required=True) + parser.add_argument('--gguf', type=Path, required=True) + parser.add_argument('--output', type=Path, required=True) + args = parser.parse_args() + assemble(args.source, args.gguf, args.output) + + +if __name__ == '__main__': + main() diff --git a/scripts/neural_evaluate.py b/scripts/neural_evaluate.py new file mode 100644 index 0000000..43b4758 --- /dev/null +++ b/scripts/neural_evaluate.py @@ -0,0 +1,228 @@ +#!/usr/bin/env python3 +"""Reproduce the frozen SmolLM2 regression and progressive-prefix comparison. + +Requires psutil and regex. Synthetic fixture text travels through stdin only. +Reports contain counts, hashes and timing, never captured application text. +""" +import argparse +import collections +import hashlib +import json +import math +from pathlib import Path +import platform +import queue +import subprocess +import threading +import time +import unicodedata + +import psutil +import regex + +ROOT = Path(__file__).resolve().parent.parent + + +def sha(path): + with path.open('rb') as file: + return hashlib.file_digest(file, 'sha256').hexdigest() + + +def words(text): + text = unicodedata.normalize('NFC', text.replace('’', "'").lower()) + return regex.findall(r"\p{L}[\p{L}\p{M}]*(?:'\p{L}[\p{L}\p{M}]*)*", text) + + +def key(text): + return hashlib.sha256(text.encode()).hexdigest() + + +def workload(): + result = [] + regression = json.loads((ROOT / 'neural/fixtures/regression.json').read_bytes()) + for domain, texts in sorted(regression.items()): + for n in range(5): + cell, seen = [], set() + for sentence in sorted({' '.join(words(text)) for text in texts}): + tokens = sentence.split() + for i, target in enumerate(tokens): + graphemes = regex.findall(r'\X', target) + if len(graphemes) <= n: + continue + before, prefix = ' '.join(tokens[:i]), ''.join(graphemes[:n]) + if (before, prefix, target) in seen: + continue + seen.add((before, prefix, target)) + cell.append((key(f'{domain}\n{sentence}\n{i}\n{n}'), 'regression', domain, n, before, prefix, target)) + assert len(cell) >= 64 + result.extend(sorted(cell)[:64]) + extra = json.loads((ROOT / 'neural/fixtures/general-writing.json').read_bytes()) + for part, start, end, count in [('development', 0, 6, 8), ('test', 6, 12, 16)]: + for domain, texts in sorted(extra.items()): + positions = [] + for text in texts[start:end]: + tokens = words(text) + sentence = ' '.join(tokens) + for i, target in enumerate(tokens): + if len(regex.findall(r'\X', target)) > 4: + positions.append((key(f'{domain}\n{sentence}\n{i}'), tokens, i)) + assert len(positions) >= count + for identifier, tokens, i in sorted(positions)[:count]: + graphemes = regex.findall(r'\X', tokens[i]) + for n in range(5): + result.append((identifier + str(n), part, domain, n, ' '.join(tokens[:i]), ''.join(graphemes[:n]), tokens[i])) + return result + + +def timing(values): + values = sorted(values) + if not values: + return None + return {'samples': len(values), 'median_ms': values[len(values)//2], + 'p95_ms': values[math.ceil(len(values)*.95)-1], 'max_ms': values[-1]} + + +class Client: + def __init__(self, command): + self.started = time.perf_counter() + self.process = subprocess.Popen(command, stdin=subprocess.PIPE, stdout=subprocess.PIPE, + stderr=subprocess.PIPE, text=True, encoding='utf-8') + self.messages = queue.Queue() + self.stop = threading.Event() + self.peak = 0 + def read(): + try: + for line in self.process.stdout: + self.messages.put(json.loads(line)) + finally: + self.messages.put(None) + def memory(): + parent = psutil.Process(self.process.pid) + while not self.stop.is_set(): + try: + total = sum(p.memory_info().rss for p in [parent] + parent.children(recursive=True)) + self.peak = max(self.peak, total) + except psutil.Error: + pass + self.stop.wait(.01) + self.reader = threading.Thread(target=read, daemon=True) + self.monitor = threading.Thread(target=memory, daemon=True) + self.reader.start() + self.monitor.start() + self.ready = self.receive() + if self.ready is None or self.ready['type'] != 'ready': + self.close() + raise RuntimeError('Worker did not become ready') + self.cold_ms = (time.perf_counter()-self.started)*1000 + + def receive(self): + return self.messages.get(timeout=40) + + def query(self, before, prefix): + start = time.perf_counter() + self.process.stdin.write(json.dumps({'command':'predict', 'before':before, 'prefix':prefix, + 'session':1, 'min_chars':0})+'\n') + self.process.stdin.flush() + immediate = self.receive() + assert immediate['type'] == 'immediate' + ipc_immediate = (time.perf_counter()-start)*1000 + if not immediate['result']['refinement_requested']: + return immediate, None, ipc_immediate, None + refined = self.receive() + elapsed = (time.perf_counter()-start)*1000 + if refined['type'] != 'refined': + return immediate, None, ipc_immediate, elapsed + assert refined['result']['request_id'] == immediate['result']['request_id'] + return immediate, refined, ipc_immediate, elapsed + + def close(self): + self.process.stdin.close() + try: + self.process.wait(timeout=5) + except subprocess.TimeoutExpired: + self.process.kill() + self.process.wait() + self.stop.set() + self.monitor.join() + self.reader.join() + self.process.stdout.close() + self.process.stderr.close() + + +def evaluate(args): + queries = workload() + training = {' '.join(words(line)) for line in args.training.read_text(encoding='utf-8').splitlines()} + for name in ['regression', 'general-writing']: + fixtures = json.loads((ROOT / f'neural/fixtures/{name}.json').read_bytes()) + assert not training.intersection(' '.join(words(text)) for texts in fixtures.values() for text in texts) + command = [str(args.cli.resolve()), 'stream', '--baseline', str(args.baseline.resolve()), + '--bundle', str(args.bundle.resolve()), '--worker', str(args.worker.resolve())] + if args.accelerated_worker: + command += ['--accelerated-worker', str(args.accelerated_worker.resolve())] + client = Client(command) + times = collections.defaultdict(list) + cells = collections.defaultdict(lambda: {'queries':0, 'baseline_top1':0, 'baseline_top5':0, + 'refined_top1':0, 'refined_top5':0, 'failures':0}) + digest = hashlib.sha256() + failures = 0 + try: + for query in queries[:20]: + client.query(query[4], query[5]) + for index, (_, part, domain, n, before, prefix, target) in enumerate(queries): + immediate, refined, ipc_immediate, elapsed = client.query(before, prefix) + baseline = immediate['result']['words'] + final = refined['result']['words'] if refined else baseline + failed = refined is None and immediate['result']['refinement_requested'] + failures += int(failed) + times['immediate'].append(immediate['elapsed_ms']) + times['immediate_with_ipc'].append(ipc_immediate) + if refined: + times['refinement_with_ipc'].append(elapsed) + times['context_hit' if refined['result']['cache_hit'] else 'context_miss'].append(elapsed) + cell = cells[f'{part}/{domain}/{n}'] + cell['queries'] += 1 + cell['baseline_top1'] += int(bool(baseline) and baseline[0] == target) + cell['baseline_top5'] += int(target in baseline) + cell['refined_top1'] += int(bool(final) and final[0] == target) + cell['refined_top5'] += int(target in final) + cell['failures'] += int(failed) + digest.update(json.dumps(final, ensure_ascii=False, separators=(',', ':')).encode()+b'\n') + if index % 200 == 0: + print(f'{index}/{len(queries)} completed', flush=True) + finally: + client.close() + early_failures = [name for name, cell in cells.items() if int(name.rsplit('/',1)[1]) < 3 + and (cell['baseline_top5']-cell['refined_top5'])*100/cell['queries'] > 1] + total_baseline = sum(c['baseline_top5'] for c in cells.values()) + total_refined = sum(c['refined_top5'] for c in cells.values()) + report = {'platform':platform.platform(), 'processor':platform.processor(), 'logical_cpus':psutil.cpu_count(), + 'capabilities':client.ready['capabilities'], 'cold_load_ms':client.cold_ms, + 'process_tree_peak_rss_bytes':client.peak, 'memory_method':'10ms sum of parent and descendants RSS; shared pages may count twice', + 'queries':len(queries), 'failures':failures, 'timings':{k:timing(v) for k,v in times.items()}, + 'cells':dict(cells), 'prediction_sha256':digest.hexdigest(), + 'inputs':{str(p.relative_to(ROOT)) if p.is_relative_to(ROOT) else p.name:sha(p) for p in + [ROOT/'neural/evaluation-protocol.json', ROOT/'neural/fixtures/regression.json', ROOT/'neural/fixtures/general-writing.json', + args.baseline.resolve(), args.training.resolve(), args.cli.resolve(), args.worker.resolve()]}, + 'gates':{'overall_top5':total_refined >= total_baseline, 'early_cell_regressions':early_failures, + 'immediate_latency':timing(times['immediate'])['p95_ms'] < 20, + 'refinement_latency':bool(times['refinement_with_ipc']) and timing(times['refinement_with_ipc'])['p95_ms'] < 150, + 'minimum_successful_samples':len(times['refinement_with_ipc']) >= 1000, 'no_failures':failures == 0}, + 'interpretation':'Frozen regression comparison, unknown pretraining overlap; no automatic promotion.'} + if args.accelerated_worker: + report['accelerated_worker_sha256'] = sha(args.accelerated_worker) + args.output.write_text(json.dumps(report, indent=2)+'\n', encoding='utf-8') + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + for name in ['cli', 'worker', 'baseline', 'bundle', 'training', 'output']: + parser.add_argument('--'+name, type=Path, required=True) + parser.add_argument('--accelerated-worker', type=Path) + args = parser.parse_args() + if args.output.exists(): + raise ValueError('Output already exists') + evaluate(args) + + +if __name__ == '__main__': + main() From 2bcec2395f351e3b6582979f3a14f3efa855cac8 Mon Sep 17 00:00:00 2001 From: Owen McGirr Date: Sat, 3 Oct 2026 20:17:45 +0100 Subject: [PATCH 2/8] Add neural packaging, platform CI and deployment guidance --- .github/workflows/neural.yml | 62 +++++++++ neural/README.md | 64 +++++++++ neural/SECURITY.md | 11 ++ neural/evaluation-requirements.txt | 2 + neural/examples/refine.rs | 41 ++++++ neural/notices/candle-LICENSE-APACHE | 201 +++++++++++++++++++++++++++ neural/notices/candle-LICENSE-MIT | 23 +++ neural/notices/pulp-LICENSE | 21 +++ neural/notices/sources.json | 14 ++ neural/src/bundle.rs | 19 ++- neural/tests/lifecycle.rs | 28 ++++ scripts/neural_bundle.py | 6 +- scripts/package_neural.py | 99 +++++++++++++ scripts/test_neural.py | 42 ++++++ 14 files changed, 625 insertions(+), 8 deletions(-) create mode 100644 .github/workflows/neural.yml create mode 100644 neural/README.md create mode 100644 neural/SECURITY.md create mode 100644 neural/evaluation-requirements.txt create mode 100644 neural/examples/refine.rs create mode 100644 neural/notices/candle-LICENSE-APACHE create mode 100644 neural/notices/candle-LICENSE-MIT create mode 100644 neural/notices/pulp-LICENSE create mode 100644 neural/notices/sources.json create mode 100644 scripts/package_neural.py create mode 100644 scripts/test_neural.py diff --git a/.github/workflows/neural.yml b/.github/workflows/neural.yml new file mode 100644 index 0000000..faf6c63 --- /dev/null +++ b/.github/workflows/neural.yml @@ -0,0 +1,62 @@ +name: Neural companion +on: + push: + branches: [main] + pull_request: + workflow_dispatch: +permissions: + contents: read +concurrency: + group: neural-${{ github.ref }} + cancel-in-progress: true +jobs: + neural: + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, windows-latest, macos-latest, macos-15-intel] + runs-on: ${{ matrix.os }} + env: + RUSTFLAGS: ${{ matrix.os == 'windows-latest' && '-C target-feature=+crt-static' || '' }} + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + - uses: dtolnay/rust-toolchain@4716b85f2fac3e324e64fa2810f6b5c3905760a5 + with: + components: rustfmt, clippy + - uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 + with: + workspaces: | + neural -> ../target/smol-portable + - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 + with: + python-version: '3.13' + - run: cargo fmt --manifest-path neural/Cargo.toml --all -- --check + - run: cargo clippy --manifest-path neural/Cargo.toml --workspace --all-targets --features test-support --locked -- -D warnings + - run: cargo test --manifest-path neural/Cargo.toml --workspace --features test-support --locked + - run: python -m unittest discover -s scripts -p "test_neural*.py" + - run: cargo build --manifest-path neural/Cargo.toml --workspace --release --locked --target-dir target/smol-portable + - name: Build explicit x64 ISA worker + if: runner.arch == 'X64' + shell: bash + run: | + RUSTFLAGS="${RUSTFLAGS} -C target-feature=+avx2,+fma,+f16c" cargo build --manifest-path neural/Cargo.toml -p switchify-smol-worker --features accelerated --release --locked --target-dir target/smol-avx2 + - name: Package x64 + if: runner.arch == 'X64' + run: python scripts/package_neural.py --portable target/smol-portable/release --accelerated target/smol-avx2/release + - name: Package ARM64 + if: runner.arch == 'ARM64' + run: python scripts/package_neural.py --portable target/smol-portable/release + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: neural-${{ matrix.os }} + path: artifacts/neural-cli/* + if-no-files-found: error + neural-audit: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + - uses: dtolnay/rust-toolchain@4716b85f2fac3e324e64fa2810f6b5c3905760a5 + - uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 + - run: cargo install cargo-audit --locked --version 0.22.0 + # Scoped, documented maintenance-only exception. Root CI remains unchanged. + - run: cargo audit --file neural/Cargo.lock --deny warnings --ignore RUSTSEC-2024-0436 diff --git a/neural/README.md b/neural/README.md new file mode 100644 index 0000000..c5d4cfd --- /dev/null +++ b/neural/README.md @@ -0,0 +1,64 @@ +# SmolLM2 companion + +An optional library and CLI for immediate statistical suggestions followed by offline SmolLM2 refinement. The root predictor, learning APIs and SQLite formats are unchanged. This package has its own workspace and lockfile. It is not integrated with Switchify PC. + +The fixed policy reranks eight statistical candidates using SmolLM2-135M Q8. It scores every token of a word plus the probability of a following word boundary, uses the current sentence capped at 64 tokens and returns at most five ordered words. Scores from different models are never combined or exposed as shared probabilities. Limits above five are errors. Minimum grapheme settings are honored; unigram-only requests stay statistical. The same caller-owned predictor supplies immediate suggestions and the shortlist, including its current personal snapshot. + +The controller returns immediate words synchronously. A persistent child loads the model once and refines through private bounded pipes. There is one active request and one replaceable pending request. `poll()` yields only the latest request; callers should also match its request ID before displaying it. Session changes and `reset()` invalidate old results and clear the worker's context. Loading is separate from the 500 ms inference deadline. Startup has a 30 second bound. Failure stops the worker and leaves immediate suggestions available. Recovery requires `retry()`. `shutdown()` kills and joins the worker and clears pending text. + +## Build + +Use Rust 1.97.1. From the repository root: + +```sh +cargo build --manifest-path neural/Cargo.toml --workspace --release --locked --target-dir target/smol-portable +cargo test --manifest-path neural/Cargo.toml --workspace --features test-support --locked +cargo clippy --manifest-path neural/Cargo.toml --workspace --all-targets --features test-support --locked -- -D warnings +cargo fmt --manifest-path neural/Cargo.toml --all -- --check +``` + +The companion library depends on the root crate by path. Add `switchify-prediction-neural = { path = "path/to/switchify-prediction/neural" }` to an embedding application. See `examples/refine.rs` for the complete lifecycle. No crates have been published. + +On x64 only, build a separate optimized worker with `RUSTFLAGS="-C target-feature=+avx2,+fma,+f16c"` and `cargo build --manifest-path neural/Cargo.toml -p switchify-smol-worker --features accelerated --release --locked --target-dir target/smol-avx2`. PowerShell uses `$env:RUSTFLAGS` and should clear it after that command. Windows packaging adds `+crt-static` to both builds. Never use `target-cpu=native`. Pass the optimized worker as an optional path to the portable parent; do not start it directly on an unknown CPU. + +## Model bundle + +The runtime is offline and never downloads assets. `source-manifest.json` pins the upstream revision and source file hashes. `model-bundle.json` pins the converted Q8 bytes, tokenizer, config, model card, license and scoring policy. Missing, changed or unsupported bundles fail explicitly. The 143,041,952-byte GGUF is separate from the existing statistical database and from binary packages. + +Acquire the files in `source-manifest.json` into a local source directory. Run the local converter and bundle assembler: + +```sh +target/smol-portable/release/quantize SOURCE_DIR MODEL.gguf q8 +python scripts/neural_bundle.py --source SOURCE_DIR --gguf MODEL.gguf --output MODEL_BUNDLE +target/smol-portable/release/switchify-prediction-neural validate --bundle MODEL_BUNDLE +``` + +Append `.exe` on Windows. The assembler verifies every source and converted hash and refuses to overwrite an existing destination. It includes source provenance, converter source, model card and Apache 2.0 license. Do not substitute a third-party GGUF with an edited manifest. + +## CLI + +`once` accepts one JSON request on stdin. `stream` accepts JSONL and keeps the worker loaded. Supply `--baseline`, `--bundle` and `--worker` explicitly. Optional flags are `--personal`, `--accelerated-worker` and `--threads 1..4`. Existing personal files are read through the established predictor; these commands never learn. + +```sh +switchify-prediction-neural stream --baseline english.sqlite --bundle MODEL_BUNDLE --worker switchify-smol-worker +``` + +Example stdin line: + +```json +{"command":"predict","before":"please send the","prefix":"","session":1,"limit":5,"min_chars":0} +``` + +The default minimum is two graphemes. Output starts with `ready` and capabilities, then `immediate` with a request ID, and later `refined` with the same ID if still current. `status` reports Loading, Ready or Unavailable. Send `{"command":"reset"}` on focus/session cleanup and `{"command":"retry"}` to recover explicitly. Invalid startup configuration or malformed/oversized input exits nonzero. EOF drains the last request; closing the process ends the worker. Text is accepted only on stdin, never command-line arguments. No application fields are read automatically. + +## Validation and packaging + +The frozen protocol is `evaluation-protocol.json`. Install the benchmark-only dependencies with `python -m pip install -r neural/evaluation-requirements.txt` and run: + +```sh +python scripts/neural_evaluate.py --cli target/smol-portable/release/switchify-prediction-neural --worker target/smol-portable/release/switchify-smol-worker --baseline english.sqlite --bundle MODEL_BUNDLE --training data/aac-oanc/prepared/candidate.txt --output results.json +``` + +Add `--accelerated-worker target/smol-avx2/release/switchify-smol-worker` for the optimized comparison. Each run contains 1,760 warmed queries with personal learning disabled. Reports include quality cells, immediate and IPC-inclusive refinement timings, cache hits/misses, cold load and sampled process-tree RSS. Training overlap is checked; unknown neural pretraining overlap remains possible. These are regression comparisons and do not prove unseen-data accuracy. A failed quality or latency gate prevents a production-quality claim; it does not cause test-set tuning or automatic promotion. + +`python scripts/package_neural.py --portable target/smol-portable/release --accelerated target/smol-avx2/release` packages binaries, hashes and dependency notices without model files. Omit the accelerated path for ARM. CI builds/tests Windows x64, Linux x64, macOS ARM64 and macOS x64. Build/test success is not a claim of measured model latency on those platforms. See `SECURITY.md` for the scoped dependency advisory exception and deployment boundaries. diff --git a/neural/SECURITY.md b/neural/SECURITY.md new file mode 100644 index 0000000..e253574 --- /dev/null +++ b/neural/SECURITY.md @@ -0,0 +1,11 @@ +# Dependency and deployment notes + +This companion is opt-in. It does not change the root crate's dependencies or audit policy. + +The neural dependency audit uses `cargo audit --file neural/Cargo.lock --deny warnings --ignore RUSTSEC-2024-0436`. This single exception covers `paste` 1.0.15, an unmaintained build-time macro dependency of Candle. The advisory reports maintenance status, not a known memory-safety exploit. Replacement requires an upstream-compatible Candle update or an independently reviewed patch. Review this exception at each dependency update. No other advisory is ignored. The root audit still denies all warnings. + +Worker files are executable code supplied by the embedding application. Install them in a directory writable only by trusted users. The model manifest is not a signature; compatibility comes from hashes compiled into the library. Runtime loading hashes the exact owned bytes used by inference. The worker has no network API, file writes, clipboard access or input-injection code. Process isolation bounds hangs and crashes; it is not an operating-system security sandbox. + +Context, prefixes and suggestions are kept in process memory and private pipes. Error messages deliberately omit their contents. The CLI writes requested suggestions to stdout, so callers must avoid logging or redirecting that stream when handling private text. Reset drops context buffers and invalidates pending results; it does not promise cryptographic memory erasure or protection against OS paging, crash dumps or a privileged debugger. + +The default four-thread cap applies to the worker's Rayon pool. Accelerated binaries require AVX2, FMA and F16C and must be launched through the portable parent, which checks CPU support. Portable builds must not inherit `target-cpu=native` or additional ISA flags. Packaged executables are unsigned and are not desktop installers. Model redistribution is separate and must retain the Apache 2.0 license and model card. diff --git a/neural/evaluation-requirements.txt b/neural/evaluation-requirements.txt new file mode 100644 index 0000000..2e9007d --- /dev/null +++ b/neural/evaluation-requirements.txt @@ -0,0 +1,2 @@ +psutil==7.0.0 +regex==2024.11.6 diff --git a/neural/examples/refine.rs b/neural/examples/refine.rs new file mode 100644 index 0000000..7cd4ef8 --- /dev/null +++ b/neural/examples/refine.rs @@ -0,0 +1,41 @@ +use std::{path::PathBuf, thread, time::Duration}; +use switchify_prediction::{Options, Predictor}; +use switchify_prediction_neural::{Config, Refiner, Status}; + +fn main() -> Result<(), Box> { + let args: Vec = std::env::args_os().skip(1).map(PathBuf::from).collect(); + if args.len() != 3 { + return Err("usage: refine BASELINE BUNDLE WORKER".into()); + } + let predictor = Predictor::open(&args[0], None)?; + let mut engine = Refiner::new(Config { + bundle: args[1].clone(), + portable_worker: args[2].clone(), + accelerated_worker: None, + threads: 4, + })?; + // Synthetic example only. A UI can submit while Loading and still gets immediate results. + let immediate = engine.submit(&predictor, "please send", "th", Options::default(), 1)?; + println!( + "request {}, {} immediate words", + immediate.request_id, + immediate.words.len() + ); + loop { + if let Some(refined) = engine.poll() { + println!( + "request {}, {} refined words", + refined.request_id, + refined.words.len() + ); + break; + } + if !immediate.refinement_requested || matches!(engine.status(), Status::Unavailable(_)) { + break; + } + thread::sleep(Duration::from_millis(5)); + } + engine.reset(); + engine.shutdown(); + Ok(()) +} diff --git a/neural/notices/candle-LICENSE-APACHE b/neural/notices/candle-LICENSE-APACHE new file mode 100644 index 0000000..261eeb9 --- /dev/null +++ b/neural/notices/candle-LICENSE-APACHE @@ -0,0 +1,201 @@ + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright [yyyy] [name of copyright owner] + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/neural/notices/candle-LICENSE-MIT b/neural/notices/candle-LICENSE-MIT new file mode 100644 index 0000000..31aa793 --- /dev/null +++ b/neural/notices/candle-LICENSE-MIT @@ -0,0 +1,23 @@ +Permission is hereby granted, free of charge, to any +person obtaining a copy of this software and associated +documentation files (the "Software"), to deal in the +Software without restriction, including without +limitation the rights to use, copy, modify, merge, +publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software +is furnished to do so, subject to the following +conditions: + +The above copyright notice and this permission notice +shall be included in all copies or substantial portions +of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF +ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED +TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A +PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT +SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY +CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION +OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR +IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER +DEALINGS IN THE SOFTWARE. diff --git a/neural/notices/pulp-LICENSE b/neural/notices/pulp-LICENSE new file mode 100644 index 0000000..84e3f44 --- /dev/null +++ b/neural/notices/pulp-LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2021 sarah + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/neural/notices/sources.json b/neural/notices/sources.json new file mode 100644 index 0000000..3826b1e --- /dev/null +++ b/neural/notices/sources.json @@ -0,0 +1,14 @@ +{ + "candle-LICENSE-MIT": { + "url": "https://raw.githubusercontent.com/huggingface/candle/31f35b147389700ed2a178ee66a91c3cc25cc80d/LICENSE-MIT", + "sha256": "23f18e03dc49df91622fe2a76176497404e46ced8a715d9d2b67a7446571cca3" + }, + "candle-LICENSE-APACHE": { + "url": "https://raw.githubusercontent.com/huggingface/candle/31f35b147389700ed2a178ee66a91c3cc25cc80d/LICENSE-APACHE", + "sha256": "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4" + }, + "pulp-LICENSE": { + "url": "https://raw.githubusercontent.com/sarah-quinones/pulp/5eb07fd7b68edf0a5e19f71737d315f72a510295/LICENSE", + "sha256": "d64f878c89bd5f1e5ada7e4aad57690b122727cf5229b7b87ea1980c188da1d9" + } +} diff --git a/neural/src/bundle.rs b/neural/src/bundle.rs index 40211ea..2596140 100644 --- a/neural/src/bundle.rs +++ b/neural/src/bundle.rs @@ -1,7 +1,7 @@ //! Compiled compatibility pins, independent of claims in a downloaded manifest. use crate::{Error, Result}; use sha2::{Digest, Sha256}; -use std::{fs, path::Path}; +use std::{fs, io::Read, path::Path}; pub const MODEL_ID: &str = "smollm2-135m-q8-v1"; pub const MANIFEST: &str = include_str!("../model-bundle.json"); @@ -11,11 +11,20 @@ pub struct Bundle { pub tokenizer: Vec, } -pub fn load(path: &Path) -> Result { - let manifest = fs::read(path.join("model-bundle.json")).map_err(|_| Error::Bundle)?; - if manifest.len() > 16_384 { +fn bounded_read(path: &Path, limit: u64) -> Result> { + let file = fs::File::open(path).map_err(|_| Error::Bundle)?; + let mut bytes = Vec::new(); + file.take(limit + 1) + .read_to_end(&mut bytes) + .map_err(|_| Error::Bundle)?; + if bytes.len() as u64 > limit { return Err(Error::Bundle); } + Ok(bytes) +} + +pub fn load(path: &Path) -> Result { + let manifest = bounded_read(&path.join("model-bundle.json"), 16_384)?; let actual: serde_json::Value = serde_json::from_slice(&manifest).map_err(|_| Error::Bundle)?; let expected: serde_json::Value = serde_json::from_str(MANIFEST).map_err(|_| Error::Bundle)?; if actual != expected { @@ -29,7 +38,7 @@ pub fn load(path: &Path) -> Result { if fs::metadata(&file).map_err(|_| Error::Bundle)?.len() != size { return Err(Error::Bundle); } - let bytes = fs::read(file).map_err(|_| Error::Bundle)?; + let bytes = bounded_read(&file, size)?; if bytes.len() as u64 != size || format!("{:x}", Sha256::digest(&bytes)) != pin["sha256"].as_str().ok_or(Error::Bundle)? diff --git a/neural/tests/lifecycle.rs b/neural/tests/lifecycle.rs index b8f4740..4a33b09 100644 --- a/neural/tests/lifecycle.rs +++ b/neural/tests/lifecycle.rs @@ -164,6 +164,34 @@ fn shutdown_interrupts_model_loading() { assert!(start.elapsed() < Duration::from_secs(1)); assert_eq!(engine.status(), Status::Stopped); } + +#[test] +fn learning_after_submission_does_not_change_captured_shortlist() { + let (_temp, mut predictor, mut engine) = fixture("delay"); + until(|| engine.status() == Status::Ready); + let mut expected: Vec<_> = predictor + .predict( + "", + "", + Options { + limit: 8, + ..options() + }, + ) + .into_iter() + .map(|s| s.word) + .collect(); + expected.reverse(); + expected.truncate(5); + engine.submit(&predictor, "", "", options(), 1).unwrap(); + predictor.learn("newword newword newword newword").unwrap(); + let mut result = None; + until(|| { + result = engine.poll(); + result.is_some() + }); + assert_eq!(result.unwrap().words, expected); +} #[test] fn context_and_bundle_validation() { assert_eq!( diff --git a/scripts/neural_bundle.py b/scripts/neural_bundle.py index 9aa8da0..32b2b72 100644 --- a/scripts/neural_bundle.py +++ b/scripts/neural_bundle.py @@ -5,15 +5,15 @@ import json from pathlib import Path import shutil -import subprocess import tempfile ROOT = Path(__file__).resolve().parent.parent def verify(path, pin): - if path.stat().st_size != pin['bytes'] or hashlib.file_digest(path.open('rb'), 'sha256').hexdigest() != pin['sha256']: - raise ValueError(f'Pinned file mismatch: {path.name}') + with path.open('rb') as file: + if path.stat().st_size != pin['bytes'] or hashlib.file_digest(file, 'sha256').hexdigest() != pin['sha256']: + raise ValueError(f'Pinned file mismatch: {path.name}') def assemble(source, gguf, output): diff --git a/scripts/package_neural.py b/scripts/package_neural.py new file mode 100644 index 0000000..26a9108 --- /dev/null +++ b/scripts/package_neural.py @@ -0,0 +1,99 @@ +#!/usr/bin/env python3 +"""Package explicitly built workers and notices, without model assets.""" +import argparse +import hashlib +import json +import os +from pathlib import Path +import platform +import shutil +import subprocess +import tempfile + +import dependency_notices + +ROOT = Path(__file__).resolve().parent.parent +# All additions have a permissive redistribution choice. Keep original notices. +NEURAL_LICENSES = {'Apache-2.0', 'Apache-2.0 / MIT', 'Apache-2.0/MIT', + 'Apache-2.0 OR BSL-1.0', 'Apache-2.0 OR MIT OR Zlib', + 'BSD-2-Clause OR Apache-2.0 OR MIT', 'Unicode-3.0', + 'Unlicense/MIT', 'MIT OR Apache-2.0 OR LGPL-2.1-or-later'} + + +def notices(target): + previous = dependency_notices.PERMITTED + dependency_notices.PERMITTED = previous | NEURAL_LICENSES + try: + metadata = json.loads(subprocess.check_output(['cargo', 'metadata', '--locked', '--format-version', '1', + '--filter-platform', target], cwd=ROOT / 'neural')) + sections = ['# Neural third-party notices\n\nExact lockfile dependencies, including build tools.\n'] + sources = json.loads((ROOT / 'neural/notices/sources.json').read_bytes()) + for package in sorted(metadata['packages'], key=lambda p: (p['name'], p['version'])): + if not package['source']: + continue + extra = {('candle-nn','0.11.0'):['candle-LICENSE-MIT','candle-LICENSE-APACHE'], + ('candle-transformers','0.11.0'):['candle-LICENSE-MIT','candle-LICENSE-APACHE'], + ('pulp-wasm-simd-flag','0.1.1'):['pulp-LICENSE']}.get((package['name'],package['version'])) + if extra: + sections.append(f"## {package['name']} {package['version']}\n\nDeclared license: {package['license']}\n") + for name in extra: + path = ROOT / 'neural/notices' / name + if sha(path) != sources[name]['sha256']: + raise ValueError('Notice checksum mismatch') + sections.append(sources[name]['url']+'\n\n'+path.read_text(encoding='utf-8')) + else: + sections.append(dependency_notices.crate_notices(package)) + if package['name'] == 'onig_sys': + path = Path(package['manifest_path']).parent / 'oniguruma/COPYING' + sections.append('### Bundled Oniguruma\n\n'+path.read_text(encoding='utf-8')) + return '\n'.join(sections) + finally: + dependency_notices.PERMITTED = previous + + +def sha(path): + with path.open('rb') as file: + return hashlib.file_digest(file, 'sha256').hexdigest() + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--portable', type=Path, required=True, help='Release binary directory') + parser.add_argument('--accelerated', type=Path, help='AVX2 release binary directory') + parser.add_argument('--output', type=Path, default=ROOT / 'artifacts/neural-cli') + args = parser.parse_args() + rustc = subprocess.check_output(['rustc', '-vV'], text=True) + target = next(line.split(': ',1)[1] for line in rustc.splitlines() if line.startswith('host: ')) + ext = '.exe' if os.name == 'nt' else '' + name = 'switchify-prediction-neural-0.1.0-' + target + args.output.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory() as temp: + stage = Path(temp) / name + stage.mkdir() + for exe in ['switchify-prediction-neural', 'switchify-smol-worker']: + shutil.copy2(args.portable / (exe+ext), stage / (exe+ext)) + if args.accelerated: + if not target.startswith('x86_64-'): + raise ValueError('Accelerated worker requires x86_64') + shutil.copy2(args.accelerated / ('switchify-smol-worker'+ext), stage / ('switchify-smol-worker-avx2'+ext)) + for source, destination in [('LICENSE','LICENSE'), ('neural/Cargo.lock','Cargo.lock'), + ('neural/README.md','README.md'), ('neural/SECURITY.md','SECURITY.md'), + ('scripts/verify_bundle.py','verify_bundle.py')]: + shutil.copyfile(ROOT / source, stage / destination) + (stage / 'THIRD_PARTY_NOTICES.md').write_text(notices(target), encoding='utf-8') + sysroot = Path(subprocess.check_output(['rustc','--print','sysroot'], text=True).strip()) + shutil.copyfile(sysroot / 'share/doc/rust/COPYRIGHT-library.html', stage / 'RUST_LIBRARY_COPYRIGHT.html') + (stage / 'BUILD.json').write_text(json.dumps({'version':'0.1.0','target':target, 'rustc':rustc, + 'platform':platform.platform(), 'libc':platform.libc_ver(), 'signed':False, + 'commit':subprocess.check_output(['git','rev-parse','HEAD'],cwd=ROOT,text=True).strip(), + 'model_id':'smollm2-135m-q8-v1', 'accelerated_requires':['avx2','fma','f16c'] if args.accelerated else [], + 'model_assets_included':False}, indent=2)+'\n',encoding='utf-8') + (stage / 'SHA256SUMS').write_text(''.join(f'{sha(p)} {p.name}\n' for p in sorted(stage.iterdir())), encoding='utf-8') + subprocess.run([str(stage / ('switchify-prediction-neural'+ext)), '--version'], check=True) + archive = Path(shutil.make_archive(str(args.output / name), 'zip', temp, name)) + (args.output / (name+'.sha256')).write_text(f'{sha(archive)} {archive.name}\n',encoding='utf-8') + print(archive) + + +if __name__ == '__main__': + main() diff --git a/scripts/test_neural.py b/scripts/test_neural.py new file mode 100644 index 0000000..6c7f348 --- /dev/null +++ b/scripts/test_neural.py @@ -0,0 +1,42 @@ +import hashlib +import json +from pathlib import Path +import tempfile +import unittest + +from neural_bundle import verify + +ROOT = Path(__file__).resolve().parent.parent + + +class NeuralBundleTests(unittest.TestCase): + def test_corrupt_file_and_wrong_length_are_rejected(self): + with tempfile.TemporaryDirectory() as temp: + path = Path(temp) / 'test' + path.write_bytes(b'good') + pin = {'bytes':4, 'sha256':hashlib.sha256(b'good').hexdigest()} + verify(path, pin) + for value in [b'evil', b'', b'longer']: + path.write_bytes(value) + with self.assertRaises(ValueError): + verify(path, pin) + + def test_pins_and_preserved_model_license(self): + manifest = json.loads((ROOT / 'neural/model-bundle.json').read_bytes()) + source = json.loads((ROOT / 'neural/source-manifest.json').read_bytes()) + self.assertEqual(manifest['revision'], source['revision']) + verify(ROOT / 'neural/MODEL_LICENSE.txt', manifest['files']['MODEL_LICENSE.txt']) + self.assertEqual(manifest['files']['tokenizer.json']['sha256'], source['files']['tokenizer.json']['sha256']) + self.assertEqual(manifest['policy']['max_results'], 5) + + def test_frozen_fixture_partitions_are_disjoint(self): + extra = json.loads((ROOT / 'neural/fixtures/general-writing.json').read_bytes()) + old = json.loads((ROOT / 'neural/fixtures/regression.json').read_bytes()) + texts = [s for group in extra.values() for s in group] + self.assertEqual(len(texts), len(set(texts))) + self.assertFalse(set(texts).intersection(s for group in old.values() for s in group)) + self.assertTrue(all(len(group) == 12 for group in extra.values())) + + +if __name__ == '__main__': + unittest.main() From 6850614b2cd8658127c2531c8383b810c083f6ea Mon Sep 17 00:00:00 2001 From: Owen McGirr Date: Sat, 3 Oct 2026 20:23:37 +0100 Subject: [PATCH 3/8] Invalidate rejected requests and resolve worker identity before spawn --- .gitattributes | 2 + neural/README.md | 2 +- neural/src/lib.rs | 13 ++++- neural/src/main.rs | 4 +- neural/tests/lifecycle.rs | 103 +++++++++++++++++++++++++++++++++++++ neural/worker/src/model.rs | 32 ++++++++++++ scripts/neural_evaluate.py | 18 +++++++ 7 files changed, 171 insertions(+), 3 deletions(-) create mode 100644 .gitattributes diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..1809afa --- /dev/null +++ b/.gitattributes @@ -0,0 +1,2 @@ +# These are byte-pinned model notices, fixtures and compatibility manifests. +neural/** text eol=lf diff --git a/neural/README.md b/neural/README.md index c5d4cfd..83ba1d7 100644 --- a/neural/README.md +++ b/neural/README.md @@ -37,7 +37,7 @@ Append `.exe` on Windows. The assembler verifies every source and converted hash ## CLI -`once` accepts one JSON request on stdin. `stream` accepts JSONL and keeps the worker loaded. Supply `--baseline`, `--bundle` and `--worker` explicitly. Optional flags are `--personal`, `--accelerated-worker` and `--threads 1..4`. Existing personal files are read through the established predictor; these commands never learn. +`once` accepts one JSON request on stdin. `stream` accepts JSONL and keeps the worker loaded. The CLI waits for a validated model before accepting requests, and exits nonzero if startup fails. The library can return immediate suggestions during Loading. After successful CLI startup, an inference failure leaves immediate predictions available and allows explicit retry. Supply `--baseline`, `--bundle` and `--worker` explicitly. Optional flags are `--personal`, `--accelerated-worker` and `--threads 1..4`. Existing personal files are read through the established predictor; these commands never learn. ```sh switchify-prediction-neural stream --baseline english.sqlite --bundle MODEL_BUNDLE --worker switchify-smol-worker diff --git a/neural/src/lib.rs b/neural/src/lib.rs index c937a3b..61fcfb7 100644 --- a/neural/src/lib.rs +++ b/neural/src/lib.rs @@ -117,7 +117,17 @@ pub fn effective_context(before: &str) -> String { } impl Refiner { - pub fn new(config: Config) -> Result { + pub fn new(mut config: Config) -> Result { + config.bundle = config.bundle.canonicalize().map_err(|_| Error::Config)?; + config.portable_worker = config + .portable_worker + .canonicalize() + .map_err(|_| Error::Config)?; + config.accelerated_worker = config + .accelerated_worker + .map(|p| p.canonicalize()) + .transpose() + .map_err(|_| Error::Config)?; if !(1..=4).contains(&config.threads) || !config.bundle.is_dir() || !config.portable_worker.is_file() @@ -179,6 +189,7 @@ impl Refiner { session: u64, ) -> Result { if before.len() > 16_384 || prefix.len() > 256 || options.limit > 5 { + self.reset(); return Err(Error::Input); } let candidates: Vec = predictor diff --git a/neural/src/main.rs b/neural/src/main.rs index c0b0bc6..326af5a 100644 --- a/neural/src/main.rs +++ b/neural/src/main.rs @@ -114,6 +114,7 @@ fn run(args: Run, once: bool) -> Result<(), ()> { let rx = input(); let mut outstanding = false; let mut eof = false; + let mut predicted = false; let mut started = Instant::now(); let mut last_status = Status::Ready; loop { @@ -132,7 +133,7 @@ fn run(args: Run, once: bool) -> Result<(), ()> { } } if eof && !outstanding { - return Ok(()); + return if once && !predicted { Err(()) } else { Ok(()) }; } if eof { thread::sleep(Duration::from_millis(1)); @@ -148,6 +149,7 @@ fn run(args: Run, once: bool) -> Result<(), ()> { unigram_only, })) => { started = Instant::now(); + predicted = true; let result = engine .submit( &predictor, diff --git a/neural/tests/lifecycle.rs b/neural/tests/lifecycle.rs index 4a33b09..b9257c2 100644 --- a/neural/tests/lifecycle.rs +++ b/neural/tests/lifecycle.rs @@ -192,6 +192,38 @@ fn learning_after_submission_does_not_change_captured_shortlist() { }); assert_eq!(result.unwrap().words, expected); } + +#[test] +fn rejected_request_also_invalidates_previous_work() { + let (_temp, predictor, mut engine) = fixture("delay"); + until(|| engine.status() == Status::Ready); + engine.submit(&predictor, "", "", options(), 1).unwrap(); + thread::sleep(Duration::from_millis(20)); + assert!( + engine + .submit(&predictor, &"x".repeat(16_385), "", options(), 2) + .is_err() + ); + thread::sleep(Duration::from_millis(200)); + assert!(engine.poll().is_none()); + engine.submit(&predictor, "", "", options(), 2).unwrap(); + thread::sleep(Duration::from_millis(200)); + assert!( + engine + .submit( + &predictor, + "", + "", + Options { + limit: 6, + ..options() + }, + 2 + ) + .is_err() + ); + assert!(engine.poll().is_none()); +} #[test] fn context_and_bundle_validation() { assert_eq!( @@ -207,3 +239,74 @@ fn context_and_bundle_validation() { std::fs::write(temp.path().join("model-bundle.json"), bundle::MANIFEST).unwrap(); assert!(bundle::load(temp.path()).is_err()); } + +#[test] +fn cli_errors_do_not_echo_input_and_once_requires_request() { + use std::io::Write; + use std::process::{Command, Stdio}; + let (temp, _predictor, mut engine) = fixture("ok"); + engine.shutdown(); + for input in [ + "private-malformed-text".to_string(), + "private".repeat(10_000), + String::new(), + ] { + let mut child = Command::new(env!("CARGO_BIN_EXE_switchify-prediction-neural")) + .arg("once") + .arg("--baseline") + .arg(temp.path().join("baseline.sqlite")) + .arg("--bundle") + .arg(temp.path()) + .arg("--worker") + .arg(env!("CARGO_BIN_EXE_fake-worker")) + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .spawn() + .unwrap(); + let _ = child.stdin.take().unwrap().write_all(input.as_bytes()); + let output = child.wait_with_output().unwrap(); + assert!(!output.status.success()); + assert!(!String::from_utf8_lossy(&output.stderr).contains("private")); + assert!(!String::from_utf8_lossy(&output.stdout).contains("private")); + } +} + +#[test] +fn bare_worker_filename_resolves_in_callers_directory() { + use std::io::Write; + use std::process::{Command, Stdio}; + let (temp, _predictor, mut engine) = fixture("ok"); + engine.shutdown(); + let filename = std::path::Path::new(env!("CARGO_BIN_EXE_fake-worker")) + .file_name() + .unwrap(); + std::fs::copy( + env!("CARGO_BIN_EXE_fake-worker"), + temp.path().join(filename), + ) + .unwrap(); + let mut child = Command::new(env!("CARGO_BIN_EXE_switchify-prediction-neural")) + .current_dir(temp.path()) + .arg("once") + .arg("--baseline") + .arg("baseline.sqlite") + .arg("--bundle") + .arg(".") + .arg("--worker") + .arg(filename) + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .spawn() + .unwrap(); + child + .stdin + .take() + .unwrap() + .write_all(br#"{"command":"predict","before":"","prefix":"","session":1,"min_chars":0}"#) + .unwrap(); + let output = child.wait_with_output().unwrap(); + assert!(output.status.success()); + assert!(String::from_utf8_lossy(&output.stdout).contains("refined")); +} diff --git a/neural/worker/src/model.rs b/neural/worker/src/model.rs index 387b68b..e729cb9 100644 --- a/neural/worker/src/model.rs +++ b/neural/worker/src/model.rs @@ -141,4 +141,36 @@ mod tests { assert!((log_boundary(&[0., 0.], &[1]) + 2_f64.ln()).abs() < 1e-12); assert_eq!(log_boundary(&[0., 0.], &[0, 1]), 0.); } + + #[test] + #[ignore = "requires explicit SWITCHIFY_SMOL_BUNDLE; never downloads model assets"] + fn pinned_model_order_and_session_cache() { + let path = + std::env::var_os("SWITCHIFY_SMOL_BUNDLE").expect("explicit model bundle required"); + let bundle = + switchify_prediction_neural::bundle::load(std::path::Path::new(&path)).unwrap(); + let mut model = Model::load(bundle).unwrap(); + let candidates: Vec<_> = [ + "receipt", + "order", + "tickets", + "confirmation", + "car", + "details", + "directions", + "address", + ] + .into_iter() + .map(String::from) + .collect(); + let (first, hit) = model.rank(1, "please send the", &candidates, 5).unwrap(); + assert!(!hit); + assert_eq!(first, ["address", "details", "order", "directions", "car"]); + let (second, hit) = model.rank(1, "please send the", &candidates, 5).unwrap(); + assert!(hit); + assert_eq!(first, second); + assert!(!model.rank(2, "please send the", &candidates, 5).unwrap().1); + model.reset(); + assert!(!model.rank(2, "please send the", &candidates, 5).unwrap().1); + } } diff --git a/scripts/neural_evaluate.py b/scripts/neural_evaluate.py index 43b4758..13f17a9 100644 --- a/scripts/neural_evaluate.py +++ b/scripts/neural_evaluate.py @@ -135,6 +135,19 @@ def query(self, before, prefix): assert refined['result']['request_id'] == immediate['result']['request_id'] return immediate, refined, ipc_immediate, elapsed + def retry(self): + start = time.perf_counter() + self.process.stdin.write('{"command":"retry"}\n') + self.process.stdin.flush() + while True: + message = self.receive() + if message is None: + raise RuntimeError('Worker exited during explicit benchmark retry') + if message.get('status') == 'Ready': + return (time.perf_counter()-start)*1000 + if isinstance(message.get('status'), dict): + raise RuntimeError('Worker retry failed') + def close(self): self.process.stdin.close() try: @@ -165,6 +178,7 @@ def evaluate(args): 'refined_top1':0, 'refined_top5':0, 'failures':0}) digest = hashlib.sha256() failures = 0 + reload_ms = [] try: for query in queries[:20]: client.query(query[4], query[5]) @@ -187,6 +201,9 @@ def evaluate(args): cell['refined_top5'] += int(target in final) cell['failures'] += int(failed) digest.update(json.dumps(final, ensure_ascii=False, separators=(',', ':')).encode()+b'\n') + if failed: + # The benchmark caller explicitly retries; the library never auto-reloads. + reload_ms.append(client.retry()) if index % 200 == 0: print(f'{index}/{len(queries)} completed', flush=True) finally: @@ -199,6 +216,7 @@ def evaluate(args): 'capabilities':client.ready['capabilities'], 'cold_load_ms':client.cold_ms, 'process_tree_peak_rss_bytes':client.peak, 'memory_method':'10ms sum of parent and descendants RSS; shared pages may count twice', 'queries':len(queries), 'failures':failures, 'timings':{k:timing(v) for k,v in times.items()}, + 'explicit_retry_load_ms':reload_ms, 'cells':dict(cells), 'prediction_sha256':digest.hexdigest(), 'inputs':{str(p.relative_to(ROOT)) if p.is_relative_to(ROOT) else p.name:sha(p) for p in [ROOT/'neural/evaluation-protocol.json', ROOT/'neural/fixtures/regression.json', ROOT/'neural/fixtures/general-writing.json', From 047461e1ed1ce1974bbe50a047abf9239d4fa293 Mon Sep 17 00:00:00 2001 From: Owen McGirr Date: Sat, 3 Oct 2026 20:37:31 +0100 Subject: [PATCH 4/8] Record qualification limits and harden benchmark and deadline handling --- .github/workflows/neural.yml | 6 +- README.md | 6 + docs/smol-avx2-results.json | 555 ++++++++++++++++++++++++++++++++ docs/smol-production-results.md | 46 +++ neural/README.md | 2 + neural/qualification.json | 13 + neural/src/process.rs | 10 +- scripts/neural_evaluate.py | 20 +- scripts/package_neural.py | 3 + scripts/test_neural.py | 14 + 10 files changed, 666 insertions(+), 9 deletions(-) create mode 100644 docs/smol-avx2-results.json create mode 100644 docs/smol-production-results.md create mode 100644 neural/qualification.json diff --git a/.github/workflows/neural.yml b/.github/workflows/neural.yml index faf6c63..9be3c77 100644 --- a/.github/workflows/neural.yml +++ b/.github/workflows/neural.yml @@ -37,9 +37,9 @@ jobs: - run: cargo build --manifest-path neural/Cargo.toml --workspace --release --locked --target-dir target/smol-portable - name: Build explicit x64 ISA worker if: runner.arch == 'X64' - shell: bash - run: | - RUSTFLAGS="${RUSTFLAGS} -C target-feature=+avx2,+fma,+f16c" cargo build --manifest-path neural/Cargo.toml -p switchify-smol-worker --features accelerated --release --locked --target-dir target/smol-avx2 + env: + RUSTFLAGS: ${{ runner.os == 'Windows' && '-C target-feature=+crt-static,+avx2,+fma,+f16c' || '-C target-feature=+avx2,+fma,+f16c' }} + run: cargo build --manifest-path neural/Cargo.toml -p switchify-smol-worker --features accelerated --release --locked --target-dir target/smol-avx2 - name: Package x64 if: runner.arch == 'X64' run: python scripts/package_neural.py --portable target/smol-portable/release --accelerated target/smol-avx2/release diff --git a/README.md b/README.md index 9c0633f..eb6252f 100644 --- a/README.md +++ b/README.md @@ -180,3 +180,9 @@ New code is MIT licensed. Corpus licences are separate: retain distributions. [Production guidance](docs/production.md) covers the recorded licensing evidence, OANC's differing historical/current notices, private-data handling, backup, upgrades, rollback and releases. + +The optional [SmolLM2 companion](neural/README.md) adds a separate library, worker +and CLI for immediate statistical results followed by offline neural refinement. +It has its own dependencies and model bundle. It remains opt-in and is not yet +production-qualified: the frozen comparison found two development-cell quality +regressions. See the [qualification report](docs/smol-production-results.md). diff --git a/docs/smol-avx2-results.json b/docs/smol-avx2-results.json new file mode 100644 index 0000000..4a099af --- /dev/null +++ b/docs/smol-avx2-results.json @@ -0,0 +1,555 @@ +{ + "platform": "Windows-11-10.0.26200-SP0", + "processor": "AMD64 Family 26 Model 36 Stepping 0, AuthenticAMD", + "logical_cpus": 24, + "capabilities": { + "accelerated": true, + "context_tokens": 64, + "deadline_ms": 500, + "shortlist": 8, + "threads": 4 + }, + "cold_load_ms": 3294.658499988145, + "process_tree_peak_rss_bytes": 669810688, + "memory_method": "10ms sum of parent and descendants RSS; shared pages may count twice", + "queries": 1760, + "failures": 0, + "timings": { + "immediate": { + "samples": 1760, + "median_ms": 0.5691999999999999, + "p95_ms": 7.4966, + "max_ms": 9.9528 + }, + "immediate_with_ipc": { + "samples": 1760, + "median_ms": 0.8190999942598864, + "p95_ms": 7.820099999662489, + "max_ms": 17.916700002388097 + }, + "refinement_with_ipc": { + "samples": 1759, + "median_ms": 91.55710000777617, + "p95_ms": 123.78069999977015, + "max_ms": 210.26760000677314 + }, + "context_miss": { + "samples": 1368, + "median_ms": 92.28149999398738, + "p95_ms": 125.7591999892611, + "max_ms": 210.26760000677314 + }, + "context_hit": { + "samples": 391, + "median_ms": 67.62349999917205, + "p95_ms": 115.72970000270288, + "max_ms": 152.74410000711214 + } + }, + "explicit_retry_load_ms": [], + "cells": { + "regression/documents/0": { + "queries": 64, + "baseline_top1": 2, + "baseline_top5": 6, + "refined_top1": 7, + "refined_top5": 10, + "failures": 0 + }, + "regression/documents/1": { + "queries": 64, + "baseline_top1": 7, + "baseline_top5": 9, + "refined_top1": 9, + "refined_top5": 10, + "failures": 0 + }, + "regression/documents/2": { + "queries": 64, + "baseline_top1": 10, + "baseline_top5": 19, + "refined_top1": 21, + "refined_top5": 23, + "failures": 0 + }, + "regression/documents/3": { + "queries": 64, + "baseline_top1": 12, + "baseline_top5": 34, + "refined_top1": 38, + "refined_top5": 41, + "failures": 0 + }, + "regression/documents/4": { + "queries": 64, + "baseline_top1": 22, + "baseline_top5": 46, + "refined_top1": 47, + "refined_top5": 53, + "failures": 0 + }, + "regression/email/0": { + "queries": 64, + "baseline_top1": 3, + "baseline_top5": 12, + "refined_top1": 10, + "refined_top5": 18, + "failures": 0 + }, + "regression/email/1": { + "queries": 64, + "baseline_top1": 16, + "baseline_top5": 29, + "refined_top1": 24, + "refined_top5": 30, + "failures": 0 + }, + "regression/email/2": { + "queries": 64, + "baseline_top1": 27, + "baseline_top5": 42, + "refined_top1": 34, + "refined_top5": 44, + "failures": 0 + }, + "regression/email/3": { + "queries": 64, + "baseline_top1": 26, + "baseline_top5": 42, + "refined_top1": 38, + "refined_top5": 47, + "failures": 0 + }, + "regression/email/4": { + "queries": 64, + "baseline_top1": 26, + "baseline_top5": 47, + "refined_top1": 43, + "refined_top5": 52, + "failures": 0 + }, + "regression/messages/0": { + "queries": 64, + "baseline_top1": 10, + "baseline_top5": 13, + "refined_top1": 12, + "refined_top5": 18, + "failures": 0 + }, + "regression/messages/1": { + "queries": 64, + "baseline_top1": 23, + "baseline_top5": 34, + "refined_top1": 26, + "refined_top5": 38, + "failures": 0 + }, + "regression/messages/2": { + "queries": 64, + "baseline_top1": 32, + "baseline_top5": 55, + "refined_top1": 41, + "refined_top5": 56, + "failures": 0 + }, + "regression/messages/3": { + "queries": 64, + "baseline_top1": 38, + "baseline_top5": 59, + "refined_top1": 50, + "refined_top5": 59, + "failures": 0 + }, + "regression/messages/4": { + "queries": 64, + "baseline_top1": 42, + "baseline_top5": 62, + "refined_top1": 50, + "refined_top5": 63, + "failures": 0 + }, + "regression/search/0": { + "queries": 64, + "baseline_top1": 2, + "baseline_top5": 5, + "refined_top1": 5, + "refined_top5": 7, + "failures": 0 + }, + "regression/search/1": { + "queries": 64, + "baseline_top1": 6, + "baseline_top5": 10, + "refined_top1": 11, + "refined_top5": 12, + "failures": 0 + }, + "regression/search/2": { + "queries": 64, + "baseline_top1": 10, + "baseline_top5": 19, + "refined_top1": 19, + "refined_top5": 22, + "failures": 0 + }, + "regression/search/3": { + "queries": 64, + "baseline_top1": 15, + "baseline_top5": 36, + "refined_top1": 32, + "refined_top5": 41, + "failures": 0 + }, + "regression/search/4": { + "queries": 64, + "baseline_top1": 24, + "baseline_top5": 49, + "refined_top1": 35, + "refined_top5": 53, + "failures": 0 + }, + "development/documents/0": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "development/documents/1": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 1, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "development/documents/2": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 2, + "refined_top1": 1, + "refined_top5": 2, + "failures": 0 + }, + "development/documents/3": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 2, + "refined_top5": 3, + "failures": 0 + }, + "development/documents/4": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 5, + "refined_top1": 6, + "refined_top5": 6, + "failures": 0 + }, + "development/email/0": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/email/1": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 2, + "refined_top5": 2, + "failures": 0 + }, + "development/email/2": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 3, + "refined_top5": 4, + "failures": 0 + }, + "development/email/3": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 6, + "refined_top1": 7, + "refined_top5": 7, + "failures": 0 + }, + "development/email/4": { + "queries": 8, + "baseline_top1": 4, + "baseline_top5": 8, + "refined_top1": 7, + "refined_top5": 8, + "failures": 0 + }, + "development/messages/0": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 2, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/messages/1": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 1, + "refined_top5": 3, + "failures": 0 + }, + "development/messages/2": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 4, + "refined_top1": 3, + "refined_top5": 4, + "failures": 0 + }, + "development/messages/3": { + "queries": 8, + "baseline_top1": 4, + "baseline_top5": 6, + "refined_top1": 6, + "refined_top5": 6, + "failures": 0 + }, + "development/messages/4": { + "queries": 8, + "baseline_top1": 6, + "baseline_top5": 8, + "refined_top1": 7, + "refined_top5": 8, + "failures": 0 + }, + "development/search/0": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/search/1": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/search/2": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 3, + "refined_top5": 3, + "failures": 0 + }, + "development/search/3": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 6, + "refined_top1": 5, + "refined_top5": 6, + "failures": 0 + }, + "development/search/4": { + "queries": 8, + "baseline_top1": 5, + "baseline_top5": 7, + "refined_top1": 7, + "refined_top5": 8, + "failures": 0 + }, + "test/documents/0": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "test/documents/1": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 1, + "refined_top5": 2, + "failures": 0 + }, + "test/documents/2": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 2, + "refined_top5": 3, + "failures": 0 + }, + "test/documents/3": { + "queries": 16, + "baseline_top1": 3, + "baseline_top5": 7, + "refined_top1": 10, + "refined_top5": 10, + "failures": 0 + }, + "test/documents/4": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 14, + "refined_top1": 13, + "refined_top5": 14, + "failures": 0 + }, + "test/email/0": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "test/email/1": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 5, + "refined_top1": 1, + "refined_top5": 5, + "failures": 0 + }, + "test/email/2": { + "queries": 16, + "baseline_top1": 3, + "baseline_top5": 8, + "refined_top1": 4, + "refined_top5": 9, + "failures": 0 + }, + "test/email/3": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 11, + "refined_top1": 8, + "refined_top5": 11, + "failures": 0 + }, + "test/email/4": { + "queries": 16, + "baseline_top1": 9, + "baseline_top5": 14, + "refined_top1": 14, + "refined_top5": 15, + "failures": 0 + }, + "test/messages/0": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 1, + "refined_top1": 0, + "refined_top5": 1, + "failures": 0 + }, + "test/messages/1": { + "queries": 16, + "baseline_top1": 3, + "baseline_top5": 4, + "refined_top1": 3, + "refined_top5": 6, + "failures": 0 + }, + "test/messages/2": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 9, + "refined_top1": 4, + "refined_top5": 10, + "failures": 0 + }, + "test/messages/3": { + "queries": 16, + "baseline_top1": 8, + "baseline_top5": 14, + "refined_top1": 12, + "refined_top5": 15, + "failures": 0 + }, + "test/messages/4": { + "queries": 16, + "baseline_top1": 14, + "baseline_top5": 16, + "refined_top1": 15, + "refined_top5": 16, + "failures": 0 + }, + "test/search/0": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "test/search/1": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 2, + "refined_top1": 1, + "refined_top5": 2, + "failures": 0 + }, + "test/search/2": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 4, + "refined_top1": 5, + "refined_top5": 6, + "failures": 0 + }, + "test/search/3": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 10, + "refined_top1": 9, + "refined_top5": 10, + "failures": 0 + }, + "test/search/4": { + "queries": 16, + "baseline_top1": 6, + "baseline_top5": 12, + "refined_top1": 13, + "refined_top5": 14, + "failures": 0 + } + }, + "prediction_sha256": "a1521815e0742be25878e609da9a930a0a3e3a2ce504bbe81daab0854b159b70", + "inputs": { + "neural\\evaluation-protocol.json": "2a2f93a39de9a8a9ab7d18350aef51539b05600781ea972a666ea626b97023ca", + "neural\\fixtures\\regression.json": "5e7fe93fc126862da5895d6c7304414c2a49a48303688648349afb3db956fe76", + "neural\\fixtures\\general-writing.json": "315cb0dacf6026d752c60482915d2dae17ff8842a77290325c6b5f780769fee9", + "english.sqlite": "222253417d0a7a705823ffb7e599a3bcf5d5d3daf4a9d76161ac6b3e555aeaad", + "data\\aac-oanc\\prepared\\candidate.txt": "ff724de5ab3b6e609ed77b699c51287fc82f79d026fdecfa4892939abfbb2834", + "target\\smol-portable\\release\\switchify-prediction-neural.exe": "19791cf8287ffcfcf1ae3e351ed5157c08311e149df66b4fc29f82dc4f421e6a", + "target\\smol-portable\\release\\switchify-smol-worker.exe": "7827ccaf9a05dff03ac59bdd04ab01dd7bf64e90734c23e68f95a52c86b2fca4" + }, + "gates": { + "overall_top5": true, + "early_cell_regressions": [ + "development/documents/1", + "development/messages/0" + ], + "immediate_latency": true, + "refinement_latency": true, + "minimum_successful_samples": true, + "no_failures": true + }, + "interpretation": "Frozen regression comparison, unknown pretraining overlap; no automatic promotion.", + "accelerated_worker_sha256": "97fc8d5665ed06ded104536c560bfa2a1d73ce7219b4a5a1bc84d5553729dc32" +} diff --git a/docs/smol-production-results.md b/docs/smol-production-results.md new file mode 100644 index 0000000..5241785 --- /dev/null +++ b/docs/smol-production-results.md @@ -0,0 +1,46 @@ +# SmolLM2 companion qualification + +The implementation provides the separate library, CLI and isolated worker. The model policy is **not production-qualified**. Keep it opt-in. The optimized Windows worker meets the speed target and improves aggregate top-five accuracy, but two development cells fail the frozen regression limit. No desktop integration, model publication or release has been performed. + +The policy and fixtures were committed in `1a44602` before scoring. Both workers use SmolLM2-135M Q8, eight statistical candidates, the last sentence capped at 64 tokens, whole-word plus boundary probability and sequential scoring. Personal learning is disabled. The immediate list comes from the same predictor snapshot as the shortlist. The existing statistical APIs, model files and counts are unchanged. + +## Quality + +The optimized worker completed 1,760 queries, with 1,759 neural refinements and one legitimate empty shortlist. There were no failed requests. Each target in the new partitions is tested with zero through four typed Unicode graphemes, in order. Existing regression queries retain their earlier hash-selection rule. The new development and test sentences were frozen together before evaluation. + +| Partition | Queries | Statistical top one | Refined top one | Statistical top five | Refined top five | +| --- | ---: | ---: | ---: | ---: | ---: | +| Existing regression | 1,280 | 27.58% | 43.12% | 49.06% | 54.45% | +| New development | 160 | 24.38% | 40.00% | 44.38% | 46.25% | +| New test | 320 | 22.19% | 36.25% | 42.19% | 46.88% | +| All | 1,760 | | | 47.39% | 52.33% | + +At zero through two graphemes, existing-regression top five rises from 32.94% to 37.50%, development is unchanged at 22.92%, and the new test partition rises from 19.27% to 23.44%. + +Two development cells fail the maximum one-percentage-point decline rule. Documents at one grapheme loses one hit out of eight, from 12.5% to 0%. Messages at zero graphemes loses one hit out of eight, from 25% to 12.5%. All existing-regression and new-test early cells pass. The small cells are noisy, but the agreed gate still fails. The policy was not tuned after seeing these results. A new policy or qualification protocol needs a separately frozen comparison, not a rewritten pass threshold. + +The corpus is synthetic general writing covering messages, email, documents and search. Exact normalized overlap with the statistical training file is checked. These results are regression evidence only; overlap with the neural model's unknown pretraining examples cannot be excluded. The older statistical baseline's AAC/spoken-data provenance is retained as a control, not treated as the intended product use case. + +## Windows performance + +Reference machine: Windows 11, AMD Ryzen AI 9 HX 370, 24 logical CPUs, four inference threads. Timings include private IPC and controller polling unless marked immediate. The optimized binary uses explicit AVX2, FMA and F16C flags. It does not use `target-cpu=native`. + +| Optimized measurement | Median | p95 | Maximum | +| --- | ---: | ---: | ---: | +| Immediate predictor call | 0.57 ms | 7.50 ms | 9.95 ms | +| Immediate including CLI transport | 0.82 ms | 7.82 ms | 17.92 ms | +| Refinement including IPC | 91.56 ms | 123.78 ms | 210.27 ms | +| Context miss, 1,368 samples | 92.28 ms | 125.76 ms | 210.27 ms | +| Context hit, 391 samples | 67.62 ms | 115.73 ms | 152.74 ms | + +Cold startup to CLI ready was 3.29 seconds, including the statistical database and model verification/load. Peak sampled parent-plus-worker RSS was 638.8 MiB. RSS is summed every 10 ms; shared pages can be counted twice, and brief peaks can be missed. This measures the whole process tree, not just the controller. Weights occupy 143,041,952 bytes; tokenizer and notices are separate bundle files. + +The initial portable run recorded a 335.50 ms refinement p95 and one inference timeout after 924 successful refinements. The worker stopped and statistical results remained available. That initial diagnostic did not qualify as the required 1,000 successful neural samples. The benchmark now explicitly requests retry after a measured failure, counts the failed query and reports reload time separately. A warm-up failure aborts clearly instead of silently benchmarking statistical fallback. + +## Reproduction and limits + +Run the command in [the companion README](../neural/README.md) once with only `--worker`, then once with `--accelerated-worker`. Both use the same pinned model bundle and baseline file. Reports contain binary/input hashes, per-cell counts, timing distributions and process-tree memory. The benchmark source and environment are committed; model and database bytes remain local. The existing baseline logical fingerprint remains `e7d83562681951d4b1a1c12a598c3f9934b211bc697d210cb787672e18c7ec64`. + +Windows x64 has actual model runtime measurements. Linux x64 and both macOS architectures have CI compilation, lifecycle tests and packaging, not measured model performance. Do not infer macOS latency from Windows results. No keyboard or pointer input was injected. Tests use synthetic text, fake processes and an explicit local-model parity test. + +The remaining qualification work is the failed quality gate and actual model runtime/performance validation on the other target platforms. The portable worker also misses the reference latency target. Passing software checks does not override these limits. diff --git a/neural/README.md b/neural/README.md index 83ba1d7..6a1e36b 100644 --- a/neural/README.md +++ b/neural/README.md @@ -2,6 +2,8 @@ An optional library and CLI for immediate statistical suggestions followed by offline SmolLM2 refinement. The root predictor, learning APIs and SQLite formats are unchanged. This package has its own workspace and lockfile. It is not integrated with Switchify PC. +This candidate is not yet production-qualified. The optimized Windows worker passed the latency target and improved overall top-five accuracy, but two development cells failed the frozen quality gate. See `docs/smol-production-results.md` in the repository for the full comparison and platform limits. `Ready` means the worker is available, not that the quality gate has passed. + The fixed policy reranks eight statistical candidates using SmolLM2-135M Q8. It scores every token of a word plus the probability of a following word boundary, uses the current sentence capped at 64 tokens and returns at most five ordered words. Scores from different models are never combined or exposed as shared probabilities. Limits above five are errors. Minimum grapheme settings are honored; unigram-only requests stay statistical. The same caller-owned predictor supplies immediate suggestions and the shortlist, including its current personal snapshot. The controller returns immediate words synchronously. A persistent child loads the model once and refines through private bounded pipes. There is one active request and one replaceable pending request. `poll()` yields only the latest request; callers should also match its request ID before displaying it. Session changes and `reset()` invalidate old results and clear the worker's context. Loading is separate from the 500 ms inference deadline. Startup has a 30 second bound. Failure stops the worker and leaves immediate suggestions available. Recovery requires `retry()`. `shutdown()` kills and joins the worker and clears pending text. diff --git a/neural/qualification.json b/neural/qualification.json new file mode 100644 index 0000000..99d660d --- /dev/null +++ b/neural/qualification.json @@ -0,0 +1,13 @@ +{ + "production_qualified": false, + "model_id": "smollm2-135m-q8-v1", + "policy_frozen_commit": "1a44602", + "reasons": [ + "Two early development cells exceed the one-percentage-point top-five regression limit", + "Portable Windows refinement misses the 150 ms p95 target", + "Actual model runtime performance has not been measured on Linux or macOS" + ], + "optimized_windows_latency_passed": true, + "desktop_integration": false, + "automatic_promotion": false +} diff --git a/neural/src/process.rs b/neural/src/process.rs index 952b348..5f8b476 100644 --- a/neural/src/process.rs +++ b/neural/src/process.rs @@ -74,7 +74,15 @@ impl Worker { return Err(Failure::Worker); } match self.replies.recv_timeout(Duration::from_millis(5)) { - Ok(Ok(reply)) => return Ok(reply), + // A descheduled controller may wake with a reply already queued. + // Do not accept it after the request's deadline has elapsed. + Ok(Ok(reply)) => { + return if start.elapsed() < timeout { + Ok(reply) + } else { + Err(Failure::Timeout) + }; + } Ok(Err(_)) => return Err(Failure::Protocol), Err(mpsc::RecvTimeoutError::Disconnected) => return Err(Failure::Worker), Err(mpsc::RecvTimeoutError::Timeout) => {} diff --git a/scripts/neural_evaluate.py b/scripts/neural_evaluate.py index 13f17a9..13d70df 100644 --- a/scripts/neural_evaluate.py +++ b/scripts/neural_evaluate.py @@ -17,9 +17,6 @@ import time import unicodedata -import psutil -import regex - ROOT = Path(__file__).resolve().parent.parent @@ -29,6 +26,7 @@ def sha(path): def words(text): + import regex text = unicodedata.normalize('NFC', text.replace('’', "'").lower()) return regex.findall(r"\p{L}[\p{L}\p{M}]*(?:'\p{L}[\p{L}\p{M}]*)*", text) @@ -38,6 +36,7 @@ def key(text): def workload(): + import regex result = [] regression = json.loads((ROOT / 'neural/fixtures/regression.json').read_bytes()) for domain, texts in sorted(regression.items()): @@ -84,6 +83,7 @@ def timing(values): class Client: def __init__(self, command): + import psutil self.started = time.perf_counter() self.process = subprocess.Popen(command, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, encoding='utf-8') @@ -162,7 +162,16 @@ def close(self): self.process.stderr.close() +def warmup(client, queries): + for query in queries[:20]: + immediate, refined, _, _ = client.query(query[4], query[5]) + result = immediate['result'] + if result['status'] != 'Ready' or (result['refinement_requested'] and refined is None): + raise RuntimeError('Neural warm-up failed; no measured run was produced') + + def evaluate(args): + import psutil queries = workload() training = {' '.join(words(line)) for line in args.training.read_text(encoding='utf-8').splitlines()} for name in ['regression', 'general-writing']: @@ -180,8 +189,7 @@ def evaluate(args): failures = 0 reload_ms = [] try: - for query in queries[:20]: - client.query(query[4], query[5]) + warmup(client, queries) for index, (_, part, domain, n, before, prefix, target) in enumerate(queries): immediate, refined, ipc_immediate, elapsed = client.query(before, prefix) baseline = immediate['result']['words'] @@ -217,6 +225,8 @@ def evaluate(args): 'process_tree_peak_rss_bytes':client.peak, 'memory_method':'10ms sum of parent and descendants RSS; shared pages may count twice', 'queries':len(queries), 'failures':failures, 'timings':{k:timing(v) for k,v in times.items()}, 'explicit_retry_load_ms':reload_ms, + 'model_file_bytes':(args.bundle / 'model.gguf').stat().st_size, + 'baseline_file_bytes':args.baseline.stat().st_size, 'cells':dict(cells), 'prediction_sha256':digest.hexdigest(), 'inputs':{str(p.relative_to(ROOT)) if p.is_relative_to(ROOT) else p.name:sha(p) for p in [ROOT/'neural/evaluation-protocol.json', ROOT/'neural/fixtures/regression.json', ROOT/'neural/fixtures/general-writing.json', diff --git a/scripts/package_neural.py b/scripts/package_neural.py index 26a9108..aaeec62 100644 --- a/scripts/package_neural.py +++ b/scripts/package_neural.py @@ -78,6 +78,8 @@ def main(): shutil.copy2(args.accelerated / ('switchify-smol-worker'+ext), stage / ('switchify-smol-worker-avx2'+ext)) for source, destination in [('LICENSE','LICENSE'), ('neural/Cargo.lock','Cargo.lock'), ('neural/README.md','README.md'), ('neural/SECURITY.md','SECURITY.md'), + ('neural/qualification.json','QUALIFICATION.json'), + ('docs/smol-production-results.md','QUALIFICATION.md'), ('scripts/verify_bundle.py','verify_bundle.py')]: shutil.copyfile(ROOT / source, stage / destination) (stage / 'THIRD_PARTY_NOTICES.md').write_text(notices(target), encoding='utf-8') @@ -87,6 +89,7 @@ def main(): 'platform':platform.platform(), 'libc':platform.libc_ver(), 'signed':False, 'commit':subprocess.check_output(['git','rev-parse','HEAD'],cwd=ROOT,text=True).strip(), 'model_id':'smollm2-135m-q8-v1', 'accelerated_requires':['avx2','fma','f16c'] if args.accelerated else [], + 'production_qualified':json.loads((ROOT / 'neural/qualification.json').read_bytes())['production_qualified'], 'model_assets_included':False}, indent=2)+'\n',encoding='utf-8') (stage / 'SHA256SUMS').write_text(''.join(f'{sha(p)} {p.name}\n' for p in sorted(stage.iterdir())), encoding='utf-8') subprocess.run([str(stage / ('switchify-prediction-neural'+ext)), '--version'], check=True) diff --git a/scripts/test_neural.py b/scripts/test_neural.py index 6c7f348..2b5a194 100644 --- a/scripts/test_neural.py +++ b/scripts/test_neural.py @@ -5,11 +5,25 @@ import unittest from neural_bundle import verify +from neural_evaluate import warmup ROOT = Path(__file__).resolve().parent.parent class NeuralBundleTests(unittest.TestCase): + def test_warmup_cannot_silently_turn_into_statistical_only_measurement(self): + class FakeClient: + def __init__(self, status, requested, refined): + self.status, self.requested, self.refined = status, requested, refined + def query(self, before, prefix): + return {'result':{'status':self.status, 'refinement_requested':self.requested}}, self.refined, 0, 0 + query = [None, None, None, None, '', ''] + for client in [FakeClient('Ready', True, None), FakeClient({'Unavailable':'Timeout'}, False, None)]: + with self.assertRaisesRegex(RuntimeError, 'warm-up failed'): + warmup(client, [query]) + warmup(FakeClient('Ready', True, {'result':{}}), [query]) + warmup(FakeClient('Ready', False, None), [query]) + def test_corrupt_file_and_wrong_length_are_rejected(self): with tempfile.TemporaryDirectory() as temp: path = Path(temp) / 'test' From bf3dc529fa638c0a66a20c56fb7fd09a0e2c7b11 Mon Sep 17 00:00:00 2001 From: Owen McGirr Date: Sat, 3 Oct 2026 20:49:31 +0100 Subject: [PATCH 5/8] Finalize measured worker comparisons and per-kernel regression fixtures --- docs/smol-avx2-results.json | 42 +-- docs/smol-portable-results.json | 559 ++++++++++++++++++++++++++++++++ docs/smol-production-results.md | 27 +- neural/worker/src/model.rs | 9 +- 4 files changed, 610 insertions(+), 27 deletions(-) create mode 100644 docs/smol-portable-results.json diff --git a/docs/smol-avx2-results.json b/docs/smol-avx2-results.json index 4a099af..01039a6 100644 --- a/docs/smol-avx2-results.json +++ b/docs/smol-avx2-results.json @@ -9,44 +9,46 @@ "shortlist": 8, "threads": 4 }, - "cold_load_ms": 3294.658499988145, - "process_tree_peak_rss_bytes": 669810688, + "cold_load_ms": 3235.6024000036996, + "process_tree_peak_rss_bytes": 674332672, "memory_method": "10ms sum of parent and descendants RSS; shared pages may count twice", "queries": 1760, "failures": 0, "timings": { "immediate": { "samples": 1760, - "median_ms": 0.5691999999999999, - "p95_ms": 7.4966, - "max_ms": 9.9528 + "median_ms": 0.48700000000000004, + "p95_ms": 7.95, + "max_ms": 10.4084 }, "immediate_with_ipc": { "samples": 1760, - "median_ms": 0.8190999942598864, - "p95_ms": 7.820099999662489, - "max_ms": 17.916700002388097 + "median_ms": 0.7729999924777076, + "p95_ms": 8.307900003273971, + "max_ms": 16.008799997507595 }, "refinement_with_ipc": { "samples": 1759, - "median_ms": 91.55710000777617, - "p95_ms": 123.78069999977015, - "max_ms": 210.26760000677314 + "median_ms": 91.58360000583343, + "p95_ms": 122.44340000324883, + "max_ms": 242.16829999932088 }, "context_miss": { "samples": 1368, - "median_ms": 92.28149999398738, - "p95_ms": 125.7591999892611, - "max_ms": 210.26760000677314 + "median_ms": 92.3454000003403, + "p95_ms": 122.94800000381656, + "max_ms": 242.16829999932088 }, "context_hit": { "samples": 391, - "median_ms": 67.62349999917205, - "p95_ms": 115.72970000270288, - "max_ms": 152.74410000711214 + "median_ms": 66.0391000128584, + "p95_ms": 106.13019998709206, + "max_ms": 125.21689999266528 } }, "explicit_retry_load_ms": [], + "model_file_bytes": 143041952, + "baseline_file_bytes": 29802496, "cells": { "regression/documents/0": { "queries": 64, @@ -536,8 +538,8 @@ "neural\\fixtures\\general-writing.json": "315cb0dacf6026d752c60482915d2dae17ff8842a77290325c6b5f780769fee9", "english.sqlite": "222253417d0a7a705823ffb7e599a3bcf5d5d3daf4a9d76161ac6b3e555aeaad", "data\\aac-oanc\\prepared\\candidate.txt": "ff724de5ab3b6e609ed77b699c51287fc82f79d026fdecfa4892939abfbb2834", - "target\\smol-portable\\release\\switchify-prediction-neural.exe": "19791cf8287ffcfcf1ae3e351ed5157c08311e149df66b4fc29f82dc4f421e6a", - "target\\smol-portable\\release\\switchify-smol-worker.exe": "7827ccaf9a05dff03ac59bdd04ab01dd7bf64e90734c23e68f95a52c86b2fca4" + "target\\smol-portable\\release\\switchify-prediction-neural.exe": "ee4d74eed3650de251bfe02f4ac752e2bd5f17527653510ea731a28036d7a27d", + "target\\smol-portable\\release\\switchify-smol-worker.exe": "431ded20627f571d89045983520a877feb40060446910359b1c72feecd93cb92" }, "gates": { "overall_top5": true, @@ -551,5 +553,5 @@ "no_failures": true }, "interpretation": "Frozen regression comparison, unknown pretraining overlap; no automatic promotion.", - "accelerated_worker_sha256": "97fc8d5665ed06ded104536c560bfa2a1d73ce7219b4a5a1bc84d5553729dc32" + "accelerated_worker_sha256": "c501222092380219354153eebd9d58211cb94e17e316fd2a9662f18cdcd65afe" } diff --git a/docs/smol-portable-results.json b/docs/smol-portable-results.json new file mode 100644 index 0000000..08c20e5 --- /dev/null +++ b/docs/smol-portable-results.json @@ -0,0 +1,559 @@ +{ + "platform": "Windows-11-10.0.26200-SP0", + "processor": "AMD64 Family 26 Model 36 Stepping 0, AuthenticAMD", + "logical_cpus": 24, + "capabilities": { + "accelerated": false, + "context_tokens": 64, + "deadline_ms": 500, + "shortlist": 8, + "threads": 4 + }, + "cold_load_ms": 3529.6957999962615, + "process_tree_peak_rss_bytes": 687869952, + "memory_method": "10ms sum of parent and descendants RSS; shared pages may count twice", + "queries": 1760, + "failures": 2, + "timings": { + "immediate": { + "samples": 1760, + "median_ms": 0.6013999999999999, + "p95_ms": 7.892899999999999, + "max_ms": 21.4904 + }, + "immediate_with_ipc": { + "samples": 1760, + "median_ms": 0.8744999940972775, + "p95_ms": 8.31330000073649, + "max_ms": 34.94259998842608 + }, + "refinement_with_ipc": { + "samples": 1757, + "median_ms": 213.817800002289, + "p95_ms": 327.69039999402594, + "max_ms": 428.9647999976296 + }, + "context_miss": { + "samples": 1366, + "median_ms": 239.4914000033168, + "p95_ms": 332.7148000098532, + "max_ms": 428.9647999976296 + }, + "context_hit": { + "samples": 391, + "median_ms": 152.550999991945, + "p95_ms": 217.16149999701884, + "max_ms": 302.0225000072969 + } + }, + "explicit_retry_load_ms": [ + 368.2270999997854, + 426.7621000035433 + ], + "model_file_bytes": 143041952, + "baseline_file_bytes": 29802496, + "cells": { + "regression/documents/0": { + "queries": 64, + "baseline_top1": 2, + "baseline_top5": 6, + "refined_top1": 7, + "refined_top5": 10, + "failures": 0 + }, + "regression/documents/1": { + "queries": 64, + "baseline_top1": 7, + "baseline_top5": 9, + "refined_top1": 9, + "refined_top5": 10, + "failures": 0 + }, + "regression/documents/2": { + "queries": 64, + "baseline_top1": 10, + "baseline_top5": 19, + "refined_top1": 22, + "refined_top5": 23, + "failures": 0 + }, + "regression/documents/3": { + "queries": 64, + "baseline_top1": 12, + "baseline_top5": 34, + "refined_top1": 37, + "refined_top5": 41, + "failures": 1 + }, + "regression/documents/4": { + "queries": 64, + "baseline_top1": 22, + "baseline_top5": 46, + "refined_top1": 47, + "refined_top5": 53, + "failures": 0 + }, + "regression/email/0": { + "queries": 64, + "baseline_top1": 3, + "baseline_top5": 12, + "refined_top1": 11, + "refined_top5": 18, + "failures": 0 + }, + "regression/email/1": { + "queries": 64, + "baseline_top1": 16, + "baseline_top5": 29, + "refined_top1": 25, + "refined_top5": 30, + "failures": 0 + }, + "regression/email/2": { + "queries": 64, + "baseline_top1": 27, + "baseline_top5": 42, + "refined_top1": 34, + "refined_top5": 44, + "failures": 0 + }, + "regression/email/3": { + "queries": 64, + "baseline_top1": 26, + "baseline_top5": 42, + "refined_top1": 38, + "refined_top5": 47, + "failures": 0 + }, + "regression/email/4": { + "queries": 64, + "baseline_top1": 26, + "baseline_top5": 47, + "refined_top1": 43, + "refined_top5": 52, + "failures": 0 + }, + "regression/messages/0": { + "queries": 64, + "baseline_top1": 10, + "baseline_top5": 13, + "refined_top1": 12, + "refined_top5": 17, + "failures": 0 + }, + "regression/messages/1": { + "queries": 64, + "baseline_top1": 23, + "baseline_top5": 34, + "refined_top1": 26, + "refined_top5": 38, + "failures": 0 + }, + "regression/messages/2": { + "queries": 64, + "baseline_top1": 32, + "baseline_top5": 55, + "refined_top1": 40, + "refined_top5": 56, + "failures": 0 + }, + "regression/messages/3": { + "queries": 64, + "baseline_top1": 38, + "baseline_top5": 59, + "refined_top1": 50, + "refined_top5": 59, + "failures": 0 + }, + "regression/messages/4": { + "queries": 64, + "baseline_top1": 42, + "baseline_top5": 62, + "refined_top1": 48, + "refined_top5": 63, + "failures": 1 + }, + "regression/search/0": { + "queries": 64, + "baseline_top1": 2, + "baseline_top5": 5, + "refined_top1": 5, + "refined_top5": 7, + "failures": 0 + }, + "regression/search/1": { + "queries": 64, + "baseline_top1": 6, + "baseline_top5": 10, + "refined_top1": 11, + "refined_top5": 12, + "failures": 0 + }, + "regression/search/2": { + "queries": 64, + "baseline_top1": 10, + "baseline_top5": 19, + "refined_top1": 19, + "refined_top5": 22, + "failures": 0 + }, + "regression/search/3": { + "queries": 64, + "baseline_top1": 15, + "baseline_top5": 36, + "refined_top1": 32, + "refined_top5": 41, + "failures": 0 + }, + "regression/search/4": { + "queries": 64, + "baseline_top1": 24, + "baseline_top5": 49, + "refined_top1": 35, + "refined_top5": 53, + "failures": 0 + }, + "development/documents/0": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "development/documents/1": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 1, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "development/documents/2": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 2, + "refined_top1": 1, + "refined_top5": 2, + "failures": 0 + }, + "development/documents/3": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 2, + "refined_top5": 3, + "failures": 0 + }, + "development/documents/4": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 5, + "refined_top1": 6, + "refined_top5": 6, + "failures": 0 + }, + "development/email/0": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/email/1": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 2, + "refined_top5": 2, + "failures": 0 + }, + "development/email/2": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 3, + "refined_top5": 4, + "failures": 0 + }, + "development/email/3": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 6, + "refined_top1": 7, + "refined_top5": 7, + "failures": 0 + }, + "development/email/4": { + "queries": 8, + "baseline_top1": 4, + "baseline_top5": 8, + "refined_top1": 7, + "refined_top5": 8, + "failures": 0 + }, + "development/messages/0": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 2, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/messages/1": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 1, + "refined_top5": 3, + "failures": 0 + }, + "development/messages/2": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 4, + "refined_top1": 3, + "refined_top5": 4, + "failures": 0 + }, + "development/messages/3": { + "queries": 8, + "baseline_top1": 4, + "baseline_top5": 6, + "refined_top1": 6, + "refined_top5": 6, + "failures": 0 + }, + "development/messages/4": { + "queries": 8, + "baseline_top1": 6, + "baseline_top5": 8, + "refined_top1": 7, + "refined_top5": 8, + "failures": 0 + }, + "development/search/0": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/search/1": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/search/2": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 3, + "refined_top5": 3, + "failures": 0 + }, + "development/search/3": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 6, + "refined_top1": 5, + "refined_top5": 6, + "failures": 0 + }, + "development/search/4": { + "queries": 8, + "baseline_top1": 5, + "baseline_top5": 7, + "refined_top1": 7, + "refined_top5": 8, + "failures": 0 + }, + "test/documents/0": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "test/documents/1": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 1, + "refined_top5": 2, + "failures": 0 + }, + "test/documents/2": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 2, + "refined_top5": 3, + "failures": 0 + }, + "test/documents/3": { + "queries": 16, + "baseline_top1": 3, + "baseline_top5": 7, + "refined_top1": 10, + "refined_top5": 10, + "failures": 0 + }, + "test/documents/4": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 14, + "refined_top1": 13, + "refined_top5": 14, + "failures": 0 + }, + "test/email/0": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "test/email/1": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 5, + "refined_top1": 1, + "refined_top5": 5, + "failures": 0 + }, + "test/email/2": { + "queries": 16, + "baseline_top1": 3, + "baseline_top5": 8, + "refined_top1": 4, + "refined_top5": 8, + "failures": 0 + }, + "test/email/3": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 11, + "refined_top1": 8, + "refined_top5": 11, + "failures": 0 + }, + "test/email/4": { + "queries": 16, + "baseline_top1": 9, + "baseline_top5": 14, + "refined_top1": 14, + "refined_top5": 15, + "failures": 0 + }, + "test/messages/0": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 1, + "refined_top1": 0, + "refined_top5": 1, + "failures": 0 + }, + "test/messages/1": { + "queries": 16, + "baseline_top1": 3, + "baseline_top5": 4, + "refined_top1": 3, + "refined_top5": 6, + "failures": 0 + }, + "test/messages/2": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 9, + "refined_top1": 4, + "refined_top5": 10, + "failures": 0 + }, + "test/messages/3": { + "queries": 16, + "baseline_top1": 8, + "baseline_top5": 14, + "refined_top1": 12, + "refined_top5": 15, + "failures": 0 + }, + "test/messages/4": { + "queries": 16, + "baseline_top1": 14, + "baseline_top5": 16, + "refined_top1": 15, + "refined_top5": 16, + "failures": 0 + }, + "test/search/0": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "test/search/1": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 2, + "refined_top1": 1, + "refined_top5": 2, + "failures": 0 + }, + "test/search/2": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 4, + "refined_top1": 5, + "refined_top5": 6, + "failures": 0 + }, + "test/search/3": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 10, + "refined_top1": 9, + "refined_top5": 10, + "failures": 0 + }, + "test/search/4": { + "queries": 16, + "baseline_top1": 6, + "baseline_top5": 12, + "refined_top1": 13, + "refined_top5": 14, + "failures": 0 + } + }, + "prediction_sha256": "3deef3a6c68355c0326b610fe19acf7005028c6a143fce782d225c7754c59df1", + "inputs": { + "neural\\evaluation-protocol.json": "2a2f93a39de9a8a9ab7d18350aef51539b05600781ea972a666ea626b97023ca", + "neural\\fixtures\\regression.json": "5e7fe93fc126862da5895d6c7304414c2a49a48303688648349afb3db956fe76", + "neural\\fixtures\\general-writing.json": "315cb0dacf6026d752c60482915d2dae17ff8842a77290325c6b5f780769fee9", + "english.sqlite": "222253417d0a7a705823ffb7e599a3bcf5d5d3daf4a9d76161ac6b3e555aeaad", + "data\\aac-oanc\\prepared\\candidate.txt": "ff724de5ab3b6e609ed77b699c51287fc82f79d026fdecfa4892939abfbb2834", + "target\\smol-portable\\release\\switchify-prediction-neural.exe": "ee4d74eed3650de251bfe02f4ac752e2bd5f17527653510ea731a28036d7a27d", + "target\\smol-portable\\release\\switchify-smol-worker.exe": "431ded20627f571d89045983520a877feb40060446910359b1c72feecd93cb92" + }, + "gates": { + "overall_top5": true, + "early_cell_regressions": [ + "development/documents/1", + "development/messages/0" + ], + "immediate_latency": true, + "refinement_latency": false, + "minimum_successful_samples": true, + "no_failures": false + }, + "interpretation": "Frozen regression comparison, unknown pretraining overlap; no automatic promotion." +} diff --git a/docs/smol-production-results.md b/docs/smol-production-results.md index 5241785..41516be 100644 --- a/docs/smol-production-results.md +++ b/docs/smol-production-results.md @@ -27,16 +27,27 @@ Reference machine: Windows 11, AMD Ryzen AI 9 HX 370, 24 logical CPUs, four infe | Optimized measurement | Median | p95 | Maximum | | --- | ---: | ---: | ---: | -| Immediate predictor call | 0.57 ms | 7.50 ms | 9.95 ms | -| Immediate including CLI transport | 0.82 ms | 7.82 ms | 17.92 ms | -| Refinement including IPC | 91.56 ms | 123.78 ms | 210.27 ms | -| Context miss, 1,368 samples | 92.28 ms | 125.76 ms | 210.27 ms | -| Context hit, 391 samples | 67.62 ms | 115.73 ms | 152.74 ms | +| Immediate predictor call | 0.49 ms | 7.95 ms | 10.41 ms | +| Immediate including CLI transport | 0.77 ms | 8.31 ms | 16.01 ms | +| Refinement including IPC | 91.58 ms | 122.44 ms | 242.17 ms | +| Context miss, 1,368 samples | 92.35 ms | 122.95 ms | 242.17 ms | +| Context hit, 391 samples | 66.04 ms | 106.13 ms | 125.22 ms | -Cold startup to CLI ready was 3.29 seconds, including the statistical database and model verification/load. Peak sampled parent-plus-worker RSS was 638.8 MiB. RSS is summed every 10 ms; shared pages can be counted twice, and brief peaks can be missed. This measures the whole process tree, not just the controller. Weights occupy 143,041,952 bytes; tokenizer and notices are separate bundle files. +Cold startup to CLI ready was 3.24 seconds, including the statistical database and model verification/load. Peak sampled parent-plus-worker RSS was 643.1 MiB. RSS is summed every 10 ms; shared pages can be counted twice, and brief peaks can be missed. This measures the whole process tree, not just the controller. Weights occupy 143,041,952 bytes; tokenizer and notices are separate bundle files. The initial portable run recorded a 335.50 ms refinement p95 and one inference timeout after 924 successful refinements. The worker stopped and statistical results remained available. That initial diagnostic did not qualify as the required 1,000 successful neural samples. The benchmark now explicitly requests retry after a measured failure, counts the failed query and reports reload time separately. A warm-up failure aborts clearly instead of silently benchmarking statistical fallback. +The final portable run completed 1,760 queries with 1,757 successful refinements and 2 failed requests. Cold startup was 3.53 seconds and sampled process-tree peak RSS was 656.0 MiB. It meets the successful-sample minimum but misses the 150 ms refinement target. + +| Portable measurement | Samples | Median | p95 | Maximum | +| --- | ---: | ---: | ---: | ---: | +| Immediate predictor call | 1760 | 0.60 ms | 7.89 ms | 21.49 ms | +| Refinement including IPC | 1757 | 213.82 ms | 327.69 ms | 428.96 ms | +| Context miss | 1366 | 239.49 ms | 332.71 ms | 428.96 ms | +| Context hit | 391 | 152.55 ms | 217.16 ms | 302.02 ms | + +Machine-readable evidence: [optimized](smol-avx2-results.json) and [portable](smol-portable-results.json). End-to-end timings include immediate prediction, dispatch and result polling in addition to the separately enforced 500 ms worker deadline. These measurements ran in a normal desktop session, without real-time scheduling isolation. + ## Reproduction and limits Run the command in [the companion README](../neural/README.md) once with only `--worker`, then once with `--accelerated-worker`. Both use the same pinned model bundle and baseline file. Reports contain binary/input hashes, per-cell counts, timing distributions and process-tree memory. The benchmark source and environment are committed; model and database bytes remain local. The existing baseline logical fingerprint remains `e7d83562681951d4b1a1c12a598c3f9934b211bc697d210cb787672e18c7ec64`. @@ -44,3 +55,7 @@ Run the command in [the companion README](../neural/README.md) once with only `- Windows x64 has actual model runtime measurements. Linux x64 and both macOS architectures have CI compilation, lifecycle tests and packaging, not measured model performance. Do not infer macOS latency from Windows results. No keyboard or pointer input was injected. Tests use synthetic text, fake processes and an explicit local-model parity test. The remaining qualification work is the failed quality gate and actual model runtime/performance validation on the other target platforms. The portable worker also misses the reference latency target. Passing software checks does not override these limits. + +Q8 kernels can produce different candidate orders between the portable and explicit ISA builds. The fixed order fixture records each Windows build separately and checks repeatability, session invalidation and reset. It does not assert bitwise parity across CPU kernels. Aggregate comparisons must therefore use the report for the selected worker, not assume identical outputs from the two builds. + +The local converter reproduced the compiled Q8 SHA-256 exactly with the new lockfile. Both source assets and the existing statistical database retained their recorded hashes. The final runtime measurements use implementation commit `047461e`; subsequent changes are test fixtures and documentation. diff --git a/neural/worker/src/model.rs b/neural/worker/src/model.rs index e729cb9..ae551a6 100644 --- a/neural/worker/src/model.rs +++ b/neural/worker/src/model.rs @@ -165,7 +165,14 @@ mod tests { .collect(); let (first, hit) = model.rank(1, "please send the", &candidates, 5).unwrap(); assert!(!hit); - assert_eq!(first, ["address", "details", "order", "directions", "car"]); + // Q8 kernels have ISA-dependent rounding; freeze each supported reference + // build separately rather than asserting cross-kernel bitwise parity. + let expected = if cfg!(feature = "accelerated") { + ["address", "details", "order", "receipt", "directions"] + } else { + ["address", "details", "order", "directions", "car"] + }; + assert_eq!(first, expected); let (second, hit) = model.rank(1, "please send the", &candidates, 5).unwrap(); assert!(hit); assert_eq!(first, second); From c08f41cba79d753fdf0ab322a3bc3821e9aaad3f Mon Sep 17 00:00:00 2001 From: Owen McGirr Date: Sat, 3 Oct 2026 20:51:27 +0100 Subject: [PATCH 6/8] Keep benchmark evidence byte-stable across platforms --- .gitattributes | 1 + docs/smol-avx2-results.json | 1114 +++++++++++++++--------------- docs/smol-portable-results.json | 1118 +++++++++++++++---------------- docs/smol-production-results.md | 2 +- scripts/neural_evaluate.py | 2 +- 5 files changed, 1119 insertions(+), 1118 deletions(-) diff --git a/.gitattributes b/.gitattributes index 1809afa..620c5fc 100644 --- a/.gitattributes +++ b/.gitattributes @@ -1,2 +1,3 @@ # These are byte-pinned model notices, fixtures and compatibility manifests. neural/** text eol=lf +docs/smol-*.json text eol=lf diff --git a/docs/smol-avx2-results.json b/docs/smol-avx2-results.json index 01039a6..3d60b42 100644 --- a/docs/smol-avx2-results.json +++ b/docs/smol-avx2-results.json @@ -1,557 +1,557 @@ -{ - "platform": "Windows-11-10.0.26200-SP0", - "processor": "AMD64 Family 26 Model 36 Stepping 0, AuthenticAMD", - "logical_cpus": 24, - "capabilities": { - "accelerated": true, - "context_tokens": 64, - "deadline_ms": 500, - "shortlist": 8, - "threads": 4 - }, - "cold_load_ms": 3235.6024000036996, - "process_tree_peak_rss_bytes": 674332672, - "memory_method": "10ms sum of parent and descendants RSS; shared pages may count twice", - "queries": 1760, - "failures": 0, - "timings": { - "immediate": { - "samples": 1760, - "median_ms": 0.48700000000000004, - "p95_ms": 7.95, - "max_ms": 10.4084 - }, - "immediate_with_ipc": { - "samples": 1760, - "median_ms": 0.7729999924777076, - "p95_ms": 8.307900003273971, - "max_ms": 16.008799997507595 - }, - "refinement_with_ipc": { - "samples": 1759, - "median_ms": 91.58360000583343, - "p95_ms": 122.44340000324883, - "max_ms": 242.16829999932088 - }, - "context_miss": { - "samples": 1368, - "median_ms": 92.3454000003403, - "p95_ms": 122.94800000381656, - "max_ms": 242.16829999932088 - }, - "context_hit": { - "samples": 391, - "median_ms": 66.0391000128584, - "p95_ms": 106.13019998709206, - "max_ms": 125.21689999266528 - } - }, - "explicit_retry_load_ms": [], - "model_file_bytes": 143041952, - "baseline_file_bytes": 29802496, - "cells": { - "regression/documents/0": { - "queries": 64, - "baseline_top1": 2, - "baseline_top5": 6, - "refined_top1": 7, - "refined_top5": 10, - "failures": 0 - }, - "regression/documents/1": { - "queries": 64, - "baseline_top1": 7, - "baseline_top5": 9, - "refined_top1": 9, - "refined_top5": 10, - "failures": 0 - }, - "regression/documents/2": { - "queries": 64, - "baseline_top1": 10, - "baseline_top5": 19, - "refined_top1": 21, - "refined_top5": 23, - "failures": 0 - }, - "regression/documents/3": { - "queries": 64, - "baseline_top1": 12, - "baseline_top5": 34, - "refined_top1": 38, - "refined_top5": 41, - "failures": 0 - }, - "regression/documents/4": { - "queries": 64, - "baseline_top1": 22, - "baseline_top5": 46, - "refined_top1": 47, - "refined_top5": 53, - "failures": 0 - }, - "regression/email/0": { - "queries": 64, - "baseline_top1": 3, - "baseline_top5": 12, - "refined_top1": 10, - "refined_top5": 18, - "failures": 0 - }, - "regression/email/1": { - "queries": 64, - "baseline_top1": 16, - "baseline_top5": 29, - "refined_top1": 24, - "refined_top5": 30, - "failures": 0 - }, - "regression/email/2": { - "queries": 64, - "baseline_top1": 27, - "baseline_top5": 42, - "refined_top1": 34, - "refined_top5": 44, - "failures": 0 - }, - "regression/email/3": { - "queries": 64, - "baseline_top1": 26, - "baseline_top5": 42, - "refined_top1": 38, - "refined_top5": 47, - "failures": 0 - }, - "regression/email/4": { - "queries": 64, - "baseline_top1": 26, - "baseline_top5": 47, - "refined_top1": 43, - "refined_top5": 52, - "failures": 0 - }, - "regression/messages/0": { - "queries": 64, - "baseline_top1": 10, - "baseline_top5": 13, - "refined_top1": 12, - "refined_top5": 18, - "failures": 0 - }, - "regression/messages/1": { - "queries": 64, - "baseline_top1": 23, - "baseline_top5": 34, - "refined_top1": 26, - "refined_top5": 38, - "failures": 0 - }, - "regression/messages/2": { - "queries": 64, - "baseline_top1": 32, - "baseline_top5": 55, - "refined_top1": 41, - "refined_top5": 56, - "failures": 0 - }, - "regression/messages/3": { - "queries": 64, - "baseline_top1": 38, - "baseline_top5": 59, - "refined_top1": 50, - "refined_top5": 59, - "failures": 0 - }, - "regression/messages/4": { - "queries": 64, - "baseline_top1": 42, - "baseline_top5": 62, - "refined_top1": 50, - "refined_top5": 63, - "failures": 0 - }, - "regression/search/0": { - "queries": 64, - "baseline_top1": 2, - "baseline_top5": 5, - "refined_top1": 5, - "refined_top5": 7, - "failures": 0 - }, - "regression/search/1": { - "queries": 64, - "baseline_top1": 6, - "baseline_top5": 10, - "refined_top1": 11, - "refined_top5": 12, - "failures": 0 - }, - "regression/search/2": { - "queries": 64, - "baseline_top1": 10, - "baseline_top5": 19, - "refined_top1": 19, - "refined_top5": 22, - "failures": 0 - }, - "regression/search/3": { - "queries": 64, - "baseline_top1": 15, - "baseline_top5": 36, - "refined_top1": 32, - "refined_top5": 41, - "failures": 0 - }, - "regression/search/4": { - "queries": 64, - "baseline_top1": 24, - "baseline_top5": 49, - "refined_top1": 35, - "refined_top5": 53, - "failures": 0 - }, - "development/documents/0": { - "queries": 8, - "baseline_top1": 0, - "baseline_top5": 0, - "refined_top1": 0, - "refined_top5": 0, - "failures": 0 - }, - "development/documents/1": { - "queries": 8, - "baseline_top1": 0, - "baseline_top5": 1, - "refined_top1": 0, - "refined_top5": 0, - "failures": 0 - }, - "development/documents/2": { - "queries": 8, - "baseline_top1": 0, - "baseline_top5": 2, - "refined_top1": 1, - "refined_top5": 2, - "failures": 0 - }, - "development/documents/3": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 3, - "refined_top1": 2, - "refined_top5": 3, - "failures": 0 - }, - "development/documents/4": { - "queries": 8, - "baseline_top1": 3, - "baseline_top5": 5, - "refined_top1": 6, - "refined_top5": 6, - "failures": 0 - }, - "development/email/0": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 1, - "refined_top1": 1, - "refined_top5": 1, - "failures": 0 - }, - "development/email/1": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 1, - "refined_top1": 2, - "refined_top5": 2, - "failures": 0 - }, - "development/email/2": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 3, - "refined_top1": 3, - "refined_top5": 4, - "failures": 0 - }, - "development/email/3": { - "queries": 8, - "baseline_top1": 3, - "baseline_top5": 6, - "refined_top1": 7, - "refined_top5": 7, - "failures": 0 - }, - "development/email/4": { - "queries": 8, - "baseline_top1": 4, - "baseline_top5": 8, - "refined_top1": 7, - "refined_top5": 8, - "failures": 0 - }, - "development/messages/0": { - "queries": 8, - "baseline_top1": 0, - "baseline_top5": 2, - "refined_top1": 1, - "refined_top5": 1, - "failures": 0 - }, - "development/messages/1": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 3, - "refined_top1": 1, - "refined_top5": 3, - "failures": 0 - }, - "development/messages/2": { - "queries": 8, - "baseline_top1": 3, - "baseline_top5": 4, - "refined_top1": 3, - "refined_top5": 4, - "failures": 0 - }, - "development/messages/3": { - "queries": 8, - "baseline_top1": 4, - "baseline_top5": 6, - "refined_top1": 6, - "refined_top5": 6, - "failures": 0 - }, - "development/messages/4": { - "queries": 8, - "baseline_top1": 6, - "baseline_top5": 8, - "refined_top1": 7, - "refined_top5": 8, - "failures": 0 - }, - "development/search/0": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 1, - "refined_top1": 1, - "refined_top5": 1, - "failures": 0 - }, - "development/search/1": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 1, - "refined_top1": 1, - "refined_top5": 1, - "failures": 0 - }, - "development/search/2": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 3, - "refined_top1": 3, - "refined_top5": 3, - "failures": 0 - }, - "development/search/3": { - "queries": 8, - "baseline_top1": 3, - "baseline_top5": 6, - "refined_top1": 5, - "refined_top5": 6, - "failures": 0 - }, - "development/search/4": { - "queries": 8, - "baseline_top1": 5, - "baseline_top5": 7, - "refined_top1": 7, - "refined_top5": 8, - "failures": 0 - }, - "test/documents/0": { - "queries": 16, - "baseline_top1": 0, - "baseline_top5": 0, - "refined_top1": 0, - "refined_top5": 0, - "failures": 0 - }, - "test/documents/1": { - "queries": 16, - "baseline_top1": 0, - "baseline_top5": 0, - "refined_top1": 1, - "refined_top5": 2, - "failures": 0 - }, - "test/documents/2": { - "queries": 16, - "baseline_top1": 1, - "baseline_top5": 3, - "refined_top1": 2, - "refined_top5": 3, - "failures": 0 - }, - "test/documents/3": { - "queries": 16, - "baseline_top1": 3, - "baseline_top5": 7, - "refined_top1": 10, - "refined_top5": 10, - "failures": 0 - }, - "test/documents/4": { - "queries": 16, - "baseline_top1": 5, - "baseline_top5": 14, - "refined_top1": 13, - "refined_top5": 14, - "failures": 0 - }, - "test/email/0": { - "queries": 16, - "baseline_top1": 0, - "baseline_top5": 0, - "refined_top1": 0, - "refined_top5": 0, - "failures": 0 - }, - "test/email/1": { - "queries": 16, - "baseline_top1": 1, - "baseline_top5": 5, - "refined_top1": 1, - "refined_top5": 5, - "failures": 0 - }, - "test/email/2": { - "queries": 16, - "baseline_top1": 3, - "baseline_top5": 8, - "refined_top1": 4, - "refined_top5": 9, - "failures": 0 - }, - "test/email/3": { - "queries": 16, - "baseline_top1": 5, - "baseline_top5": 11, - "refined_top1": 8, - "refined_top5": 11, - "failures": 0 - }, - "test/email/4": { - "queries": 16, - "baseline_top1": 9, - "baseline_top5": 14, - "refined_top1": 14, - "refined_top5": 15, - "failures": 0 - }, - "test/messages/0": { - "queries": 16, - "baseline_top1": 0, - "baseline_top5": 1, - "refined_top1": 0, - "refined_top5": 1, - "failures": 0 - }, - "test/messages/1": { - "queries": 16, - "baseline_top1": 3, - "baseline_top5": 4, - "refined_top1": 3, - "refined_top5": 6, - "failures": 0 - }, - "test/messages/2": { - "queries": 16, - "baseline_top1": 5, - "baseline_top5": 9, - "refined_top1": 4, - "refined_top5": 10, - "failures": 0 - }, - "test/messages/3": { - "queries": 16, - "baseline_top1": 8, - "baseline_top5": 14, - "refined_top1": 12, - "refined_top5": 15, - "failures": 0 - }, - "test/messages/4": { - "queries": 16, - "baseline_top1": 14, - "baseline_top5": 16, - "refined_top1": 15, - "refined_top5": 16, - "failures": 0 - }, - "test/search/0": { - "queries": 16, - "baseline_top1": 1, - "baseline_top5": 1, - "refined_top1": 1, - "refined_top5": 1, - "failures": 0 - }, - "test/search/1": { - "queries": 16, - "baseline_top1": 1, - "baseline_top5": 2, - "refined_top1": 1, - "refined_top5": 2, - "failures": 0 - }, - "test/search/2": { - "queries": 16, - "baseline_top1": 1, - "baseline_top5": 4, - "refined_top1": 5, - "refined_top5": 6, - "failures": 0 - }, - "test/search/3": { - "queries": 16, - "baseline_top1": 5, - "baseline_top5": 10, - "refined_top1": 9, - "refined_top5": 10, - "failures": 0 - }, - "test/search/4": { - "queries": 16, - "baseline_top1": 6, - "baseline_top5": 12, - "refined_top1": 13, - "refined_top5": 14, - "failures": 0 - } - }, - "prediction_sha256": "a1521815e0742be25878e609da9a930a0a3e3a2ce504bbe81daab0854b159b70", - "inputs": { - "neural\\evaluation-protocol.json": "2a2f93a39de9a8a9ab7d18350aef51539b05600781ea972a666ea626b97023ca", - "neural\\fixtures\\regression.json": "5e7fe93fc126862da5895d6c7304414c2a49a48303688648349afb3db956fe76", - "neural\\fixtures\\general-writing.json": "315cb0dacf6026d752c60482915d2dae17ff8842a77290325c6b5f780769fee9", - "english.sqlite": "222253417d0a7a705823ffb7e599a3bcf5d5d3daf4a9d76161ac6b3e555aeaad", - "data\\aac-oanc\\prepared\\candidate.txt": "ff724de5ab3b6e609ed77b699c51287fc82f79d026fdecfa4892939abfbb2834", - "target\\smol-portable\\release\\switchify-prediction-neural.exe": "ee4d74eed3650de251bfe02f4ac752e2bd5f17527653510ea731a28036d7a27d", - "target\\smol-portable\\release\\switchify-smol-worker.exe": "431ded20627f571d89045983520a877feb40060446910359b1c72feecd93cb92" - }, - "gates": { - "overall_top5": true, - "early_cell_regressions": [ - "development/documents/1", - "development/messages/0" - ], - "immediate_latency": true, - "refinement_latency": true, - "minimum_successful_samples": true, - "no_failures": true - }, - "interpretation": "Frozen regression comparison, unknown pretraining overlap; no automatic promotion.", - "accelerated_worker_sha256": "c501222092380219354153eebd9d58211cb94e17e316fd2a9662f18cdcd65afe" -} +{ + "platform": "Windows-11-10.0.26200-SP0", + "processor": "AMD64 Family 26 Model 36 Stepping 0, AuthenticAMD", + "logical_cpus": 24, + "capabilities": { + "accelerated": true, + "context_tokens": 64, + "deadline_ms": 500, + "shortlist": 8, + "threads": 4 + }, + "cold_load_ms": 3235.6024000036996, + "process_tree_peak_rss_bytes": 674332672, + "memory_method": "10ms sum of parent and descendants RSS; shared pages may count twice", + "queries": 1760, + "failures": 0, + "timings": { + "immediate": { + "samples": 1760, + "median_ms": 0.48700000000000004, + "p95_ms": 7.95, + "max_ms": 10.4084 + }, + "immediate_with_ipc": { + "samples": 1760, + "median_ms": 0.7729999924777076, + "p95_ms": 8.307900003273971, + "max_ms": 16.008799997507595 + }, + "refinement_with_ipc": { + "samples": 1759, + "median_ms": 91.58360000583343, + "p95_ms": 122.44340000324883, + "max_ms": 242.16829999932088 + }, + "context_miss": { + "samples": 1368, + "median_ms": 92.3454000003403, + "p95_ms": 122.94800000381656, + "max_ms": 242.16829999932088 + }, + "context_hit": { + "samples": 391, + "median_ms": 66.0391000128584, + "p95_ms": 106.13019998709206, + "max_ms": 125.21689999266528 + } + }, + "explicit_retry_load_ms": [], + "model_file_bytes": 143041952, + "baseline_file_bytes": 29802496, + "cells": { + "regression/documents/0": { + "queries": 64, + "baseline_top1": 2, + "baseline_top5": 6, + "refined_top1": 7, + "refined_top5": 10, + "failures": 0 + }, + "regression/documents/1": { + "queries": 64, + "baseline_top1": 7, + "baseline_top5": 9, + "refined_top1": 9, + "refined_top5": 10, + "failures": 0 + }, + "regression/documents/2": { + "queries": 64, + "baseline_top1": 10, + "baseline_top5": 19, + "refined_top1": 21, + "refined_top5": 23, + "failures": 0 + }, + "regression/documents/3": { + "queries": 64, + "baseline_top1": 12, + "baseline_top5": 34, + "refined_top1": 38, + "refined_top5": 41, + "failures": 0 + }, + "regression/documents/4": { + "queries": 64, + "baseline_top1": 22, + "baseline_top5": 46, + "refined_top1": 47, + "refined_top5": 53, + "failures": 0 + }, + "regression/email/0": { + "queries": 64, + "baseline_top1": 3, + "baseline_top5": 12, + "refined_top1": 10, + "refined_top5": 18, + "failures": 0 + }, + "regression/email/1": { + "queries": 64, + "baseline_top1": 16, + "baseline_top5": 29, + "refined_top1": 24, + "refined_top5": 30, + "failures": 0 + }, + "regression/email/2": { + "queries": 64, + "baseline_top1": 27, + "baseline_top5": 42, + "refined_top1": 34, + "refined_top5": 44, + "failures": 0 + }, + "regression/email/3": { + "queries": 64, + "baseline_top1": 26, + "baseline_top5": 42, + "refined_top1": 38, + "refined_top5": 47, + "failures": 0 + }, + "regression/email/4": { + "queries": 64, + "baseline_top1": 26, + "baseline_top5": 47, + "refined_top1": 43, + "refined_top5": 52, + "failures": 0 + }, + "regression/messages/0": { + "queries": 64, + "baseline_top1": 10, + "baseline_top5": 13, + "refined_top1": 12, + "refined_top5": 18, + "failures": 0 + }, + "regression/messages/1": { + "queries": 64, + "baseline_top1": 23, + "baseline_top5": 34, + "refined_top1": 26, + "refined_top5": 38, + "failures": 0 + }, + "regression/messages/2": { + "queries": 64, + "baseline_top1": 32, + "baseline_top5": 55, + "refined_top1": 41, + "refined_top5": 56, + "failures": 0 + }, + "regression/messages/3": { + "queries": 64, + "baseline_top1": 38, + "baseline_top5": 59, + "refined_top1": 50, + "refined_top5": 59, + "failures": 0 + }, + "regression/messages/4": { + "queries": 64, + "baseline_top1": 42, + "baseline_top5": 62, + "refined_top1": 50, + "refined_top5": 63, + "failures": 0 + }, + "regression/search/0": { + "queries": 64, + "baseline_top1": 2, + "baseline_top5": 5, + "refined_top1": 5, + "refined_top5": 7, + "failures": 0 + }, + "regression/search/1": { + "queries": 64, + "baseline_top1": 6, + "baseline_top5": 10, + "refined_top1": 11, + "refined_top5": 12, + "failures": 0 + }, + "regression/search/2": { + "queries": 64, + "baseline_top1": 10, + "baseline_top5": 19, + "refined_top1": 19, + "refined_top5": 22, + "failures": 0 + }, + "regression/search/3": { + "queries": 64, + "baseline_top1": 15, + "baseline_top5": 36, + "refined_top1": 32, + "refined_top5": 41, + "failures": 0 + }, + "regression/search/4": { + "queries": 64, + "baseline_top1": 24, + "baseline_top5": 49, + "refined_top1": 35, + "refined_top5": 53, + "failures": 0 + }, + "development/documents/0": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "development/documents/1": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 1, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "development/documents/2": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 2, + "refined_top1": 1, + "refined_top5": 2, + "failures": 0 + }, + "development/documents/3": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 2, + "refined_top5": 3, + "failures": 0 + }, + "development/documents/4": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 5, + "refined_top1": 6, + "refined_top5": 6, + "failures": 0 + }, + "development/email/0": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/email/1": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 2, + "refined_top5": 2, + "failures": 0 + }, + "development/email/2": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 3, + "refined_top5": 4, + "failures": 0 + }, + "development/email/3": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 6, + "refined_top1": 7, + "refined_top5": 7, + "failures": 0 + }, + "development/email/4": { + "queries": 8, + "baseline_top1": 4, + "baseline_top5": 8, + "refined_top1": 7, + "refined_top5": 8, + "failures": 0 + }, + "development/messages/0": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 2, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/messages/1": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 1, + "refined_top5": 3, + "failures": 0 + }, + "development/messages/2": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 4, + "refined_top1": 3, + "refined_top5": 4, + "failures": 0 + }, + "development/messages/3": { + "queries": 8, + "baseline_top1": 4, + "baseline_top5": 6, + "refined_top1": 6, + "refined_top5": 6, + "failures": 0 + }, + "development/messages/4": { + "queries": 8, + "baseline_top1": 6, + "baseline_top5": 8, + "refined_top1": 7, + "refined_top5": 8, + "failures": 0 + }, + "development/search/0": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/search/1": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/search/2": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 3, + "refined_top5": 3, + "failures": 0 + }, + "development/search/3": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 6, + "refined_top1": 5, + "refined_top5": 6, + "failures": 0 + }, + "development/search/4": { + "queries": 8, + "baseline_top1": 5, + "baseline_top5": 7, + "refined_top1": 7, + "refined_top5": 8, + "failures": 0 + }, + "test/documents/0": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "test/documents/1": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 1, + "refined_top5": 2, + "failures": 0 + }, + "test/documents/2": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 2, + "refined_top5": 3, + "failures": 0 + }, + "test/documents/3": { + "queries": 16, + "baseline_top1": 3, + "baseline_top5": 7, + "refined_top1": 10, + "refined_top5": 10, + "failures": 0 + }, + "test/documents/4": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 14, + "refined_top1": 13, + "refined_top5": 14, + "failures": 0 + }, + "test/email/0": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "test/email/1": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 5, + "refined_top1": 1, + "refined_top5": 5, + "failures": 0 + }, + "test/email/2": { + "queries": 16, + "baseline_top1": 3, + "baseline_top5": 8, + "refined_top1": 4, + "refined_top5": 9, + "failures": 0 + }, + "test/email/3": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 11, + "refined_top1": 8, + "refined_top5": 11, + "failures": 0 + }, + "test/email/4": { + "queries": 16, + "baseline_top1": 9, + "baseline_top5": 14, + "refined_top1": 14, + "refined_top5": 15, + "failures": 0 + }, + "test/messages/0": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 1, + "refined_top1": 0, + "refined_top5": 1, + "failures": 0 + }, + "test/messages/1": { + "queries": 16, + "baseline_top1": 3, + "baseline_top5": 4, + "refined_top1": 3, + "refined_top5": 6, + "failures": 0 + }, + "test/messages/2": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 9, + "refined_top1": 4, + "refined_top5": 10, + "failures": 0 + }, + "test/messages/3": { + "queries": 16, + "baseline_top1": 8, + "baseline_top5": 14, + "refined_top1": 12, + "refined_top5": 15, + "failures": 0 + }, + "test/messages/4": { + "queries": 16, + "baseline_top1": 14, + "baseline_top5": 16, + "refined_top1": 15, + "refined_top5": 16, + "failures": 0 + }, + "test/search/0": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "test/search/1": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 2, + "refined_top1": 1, + "refined_top5": 2, + "failures": 0 + }, + "test/search/2": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 4, + "refined_top1": 5, + "refined_top5": 6, + "failures": 0 + }, + "test/search/3": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 10, + "refined_top1": 9, + "refined_top5": 10, + "failures": 0 + }, + "test/search/4": { + "queries": 16, + "baseline_top1": 6, + "baseline_top5": 12, + "refined_top1": 13, + "refined_top5": 14, + "failures": 0 + } + }, + "prediction_sha256": "a1521815e0742be25878e609da9a930a0a3e3a2ce504bbe81daab0854b159b70", + "inputs": { + "neural\\evaluation-protocol.json": "2a2f93a39de9a8a9ab7d18350aef51539b05600781ea972a666ea626b97023ca", + "neural\\fixtures\\regression.json": "5e7fe93fc126862da5895d6c7304414c2a49a48303688648349afb3db956fe76", + "neural\\fixtures\\general-writing.json": "315cb0dacf6026d752c60482915d2dae17ff8842a77290325c6b5f780769fee9", + "english.sqlite": "222253417d0a7a705823ffb7e599a3bcf5d5d3daf4a9d76161ac6b3e555aeaad", + "data\\aac-oanc\\prepared\\candidate.txt": "ff724de5ab3b6e609ed77b699c51287fc82f79d026fdecfa4892939abfbb2834", + "target\\smol-portable\\release\\switchify-prediction-neural.exe": "ee4d74eed3650de251bfe02f4ac752e2bd5f17527653510ea731a28036d7a27d", + "target\\smol-portable\\release\\switchify-smol-worker.exe": "431ded20627f571d89045983520a877feb40060446910359b1c72feecd93cb92" + }, + "gates": { + "overall_top5": true, + "early_cell_regressions": [ + "development/documents/1", + "development/messages/0" + ], + "immediate_latency": true, + "refinement_latency": true, + "minimum_successful_samples": true, + "no_failures": true + }, + "interpretation": "Frozen regression comparison, unknown pretraining overlap; no automatic promotion.", + "accelerated_worker_sha256": "c501222092380219354153eebd9d58211cb94e17e316fd2a9662f18cdcd65afe" +} diff --git a/docs/smol-portable-results.json b/docs/smol-portable-results.json index 08c20e5..b77b160 100644 --- a/docs/smol-portable-results.json +++ b/docs/smol-portable-results.json @@ -1,559 +1,559 @@ -{ - "platform": "Windows-11-10.0.26200-SP0", - "processor": "AMD64 Family 26 Model 36 Stepping 0, AuthenticAMD", - "logical_cpus": 24, - "capabilities": { - "accelerated": false, - "context_tokens": 64, - "deadline_ms": 500, - "shortlist": 8, - "threads": 4 - }, - "cold_load_ms": 3529.6957999962615, - "process_tree_peak_rss_bytes": 687869952, - "memory_method": "10ms sum of parent and descendants RSS; shared pages may count twice", - "queries": 1760, - "failures": 2, - "timings": { - "immediate": { - "samples": 1760, - "median_ms": 0.6013999999999999, - "p95_ms": 7.892899999999999, - "max_ms": 21.4904 - }, - "immediate_with_ipc": { - "samples": 1760, - "median_ms": 0.8744999940972775, - "p95_ms": 8.31330000073649, - "max_ms": 34.94259998842608 - }, - "refinement_with_ipc": { - "samples": 1757, - "median_ms": 213.817800002289, - "p95_ms": 327.69039999402594, - "max_ms": 428.9647999976296 - }, - "context_miss": { - "samples": 1366, - "median_ms": 239.4914000033168, - "p95_ms": 332.7148000098532, - "max_ms": 428.9647999976296 - }, - "context_hit": { - "samples": 391, - "median_ms": 152.550999991945, - "p95_ms": 217.16149999701884, - "max_ms": 302.0225000072969 - } - }, - "explicit_retry_load_ms": [ - 368.2270999997854, - 426.7621000035433 - ], - "model_file_bytes": 143041952, - "baseline_file_bytes": 29802496, - "cells": { - "regression/documents/0": { - "queries": 64, - "baseline_top1": 2, - "baseline_top5": 6, - "refined_top1": 7, - "refined_top5": 10, - "failures": 0 - }, - "regression/documents/1": { - "queries": 64, - "baseline_top1": 7, - "baseline_top5": 9, - "refined_top1": 9, - "refined_top5": 10, - "failures": 0 - }, - "regression/documents/2": { - "queries": 64, - "baseline_top1": 10, - "baseline_top5": 19, - "refined_top1": 22, - "refined_top5": 23, - "failures": 0 - }, - "regression/documents/3": { - "queries": 64, - "baseline_top1": 12, - "baseline_top5": 34, - "refined_top1": 37, - "refined_top5": 41, - "failures": 1 - }, - "regression/documents/4": { - "queries": 64, - "baseline_top1": 22, - "baseline_top5": 46, - "refined_top1": 47, - "refined_top5": 53, - "failures": 0 - }, - "regression/email/0": { - "queries": 64, - "baseline_top1": 3, - "baseline_top5": 12, - "refined_top1": 11, - "refined_top5": 18, - "failures": 0 - }, - "regression/email/1": { - "queries": 64, - "baseline_top1": 16, - "baseline_top5": 29, - "refined_top1": 25, - "refined_top5": 30, - "failures": 0 - }, - "regression/email/2": { - "queries": 64, - "baseline_top1": 27, - "baseline_top5": 42, - "refined_top1": 34, - "refined_top5": 44, - "failures": 0 - }, - "regression/email/3": { - "queries": 64, - "baseline_top1": 26, - "baseline_top5": 42, - "refined_top1": 38, - "refined_top5": 47, - "failures": 0 - }, - "regression/email/4": { - "queries": 64, - "baseline_top1": 26, - "baseline_top5": 47, - "refined_top1": 43, - "refined_top5": 52, - "failures": 0 - }, - "regression/messages/0": { - "queries": 64, - "baseline_top1": 10, - "baseline_top5": 13, - "refined_top1": 12, - "refined_top5": 17, - "failures": 0 - }, - "regression/messages/1": { - "queries": 64, - "baseline_top1": 23, - "baseline_top5": 34, - "refined_top1": 26, - "refined_top5": 38, - "failures": 0 - }, - "regression/messages/2": { - "queries": 64, - "baseline_top1": 32, - "baseline_top5": 55, - "refined_top1": 40, - "refined_top5": 56, - "failures": 0 - }, - "regression/messages/3": { - "queries": 64, - "baseline_top1": 38, - "baseline_top5": 59, - "refined_top1": 50, - "refined_top5": 59, - "failures": 0 - }, - "regression/messages/4": { - "queries": 64, - "baseline_top1": 42, - "baseline_top5": 62, - "refined_top1": 48, - "refined_top5": 63, - "failures": 1 - }, - "regression/search/0": { - "queries": 64, - "baseline_top1": 2, - "baseline_top5": 5, - "refined_top1": 5, - "refined_top5": 7, - "failures": 0 - }, - "regression/search/1": { - "queries": 64, - "baseline_top1": 6, - "baseline_top5": 10, - "refined_top1": 11, - "refined_top5": 12, - "failures": 0 - }, - "regression/search/2": { - "queries": 64, - "baseline_top1": 10, - "baseline_top5": 19, - "refined_top1": 19, - "refined_top5": 22, - "failures": 0 - }, - "regression/search/3": { - "queries": 64, - "baseline_top1": 15, - "baseline_top5": 36, - "refined_top1": 32, - "refined_top5": 41, - "failures": 0 - }, - "regression/search/4": { - "queries": 64, - "baseline_top1": 24, - "baseline_top5": 49, - "refined_top1": 35, - "refined_top5": 53, - "failures": 0 - }, - "development/documents/0": { - "queries": 8, - "baseline_top1": 0, - "baseline_top5": 0, - "refined_top1": 0, - "refined_top5": 0, - "failures": 0 - }, - "development/documents/1": { - "queries": 8, - "baseline_top1": 0, - "baseline_top5": 1, - "refined_top1": 0, - "refined_top5": 0, - "failures": 0 - }, - "development/documents/2": { - "queries": 8, - "baseline_top1": 0, - "baseline_top5": 2, - "refined_top1": 1, - "refined_top5": 2, - "failures": 0 - }, - "development/documents/3": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 3, - "refined_top1": 2, - "refined_top5": 3, - "failures": 0 - }, - "development/documents/4": { - "queries": 8, - "baseline_top1": 3, - "baseline_top5": 5, - "refined_top1": 6, - "refined_top5": 6, - "failures": 0 - }, - "development/email/0": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 1, - "refined_top1": 1, - "refined_top5": 1, - "failures": 0 - }, - "development/email/1": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 1, - "refined_top1": 2, - "refined_top5": 2, - "failures": 0 - }, - "development/email/2": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 3, - "refined_top1": 3, - "refined_top5": 4, - "failures": 0 - }, - "development/email/3": { - "queries": 8, - "baseline_top1": 3, - "baseline_top5": 6, - "refined_top1": 7, - "refined_top5": 7, - "failures": 0 - }, - "development/email/4": { - "queries": 8, - "baseline_top1": 4, - "baseline_top5": 8, - "refined_top1": 7, - "refined_top5": 8, - "failures": 0 - }, - "development/messages/0": { - "queries": 8, - "baseline_top1": 0, - "baseline_top5": 2, - "refined_top1": 1, - "refined_top5": 1, - "failures": 0 - }, - "development/messages/1": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 3, - "refined_top1": 1, - "refined_top5": 3, - "failures": 0 - }, - "development/messages/2": { - "queries": 8, - "baseline_top1": 3, - "baseline_top5": 4, - "refined_top1": 3, - "refined_top5": 4, - "failures": 0 - }, - "development/messages/3": { - "queries": 8, - "baseline_top1": 4, - "baseline_top5": 6, - "refined_top1": 6, - "refined_top5": 6, - "failures": 0 - }, - "development/messages/4": { - "queries": 8, - "baseline_top1": 6, - "baseline_top5": 8, - "refined_top1": 7, - "refined_top5": 8, - "failures": 0 - }, - "development/search/0": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 1, - "refined_top1": 1, - "refined_top5": 1, - "failures": 0 - }, - "development/search/1": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 1, - "refined_top1": 1, - "refined_top5": 1, - "failures": 0 - }, - "development/search/2": { - "queries": 8, - "baseline_top1": 1, - "baseline_top5": 3, - "refined_top1": 3, - "refined_top5": 3, - "failures": 0 - }, - "development/search/3": { - "queries": 8, - "baseline_top1": 3, - "baseline_top5": 6, - "refined_top1": 5, - "refined_top5": 6, - "failures": 0 - }, - "development/search/4": { - "queries": 8, - "baseline_top1": 5, - "baseline_top5": 7, - "refined_top1": 7, - "refined_top5": 8, - "failures": 0 - }, - "test/documents/0": { - "queries": 16, - "baseline_top1": 0, - "baseline_top5": 0, - "refined_top1": 0, - "refined_top5": 0, - "failures": 0 - }, - "test/documents/1": { - "queries": 16, - "baseline_top1": 0, - "baseline_top5": 0, - "refined_top1": 1, - "refined_top5": 2, - "failures": 0 - }, - "test/documents/2": { - "queries": 16, - "baseline_top1": 1, - "baseline_top5": 3, - "refined_top1": 2, - "refined_top5": 3, - "failures": 0 - }, - "test/documents/3": { - "queries": 16, - "baseline_top1": 3, - "baseline_top5": 7, - "refined_top1": 10, - "refined_top5": 10, - "failures": 0 - }, - "test/documents/4": { - "queries": 16, - "baseline_top1": 5, - "baseline_top5": 14, - "refined_top1": 13, - "refined_top5": 14, - "failures": 0 - }, - "test/email/0": { - "queries": 16, - "baseline_top1": 0, - "baseline_top5": 0, - "refined_top1": 0, - "refined_top5": 0, - "failures": 0 - }, - "test/email/1": { - "queries": 16, - "baseline_top1": 1, - "baseline_top5": 5, - "refined_top1": 1, - "refined_top5": 5, - "failures": 0 - }, - "test/email/2": { - "queries": 16, - "baseline_top1": 3, - "baseline_top5": 8, - "refined_top1": 4, - "refined_top5": 8, - "failures": 0 - }, - "test/email/3": { - "queries": 16, - "baseline_top1": 5, - "baseline_top5": 11, - "refined_top1": 8, - "refined_top5": 11, - "failures": 0 - }, - "test/email/4": { - "queries": 16, - "baseline_top1": 9, - "baseline_top5": 14, - "refined_top1": 14, - "refined_top5": 15, - "failures": 0 - }, - "test/messages/0": { - "queries": 16, - "baseline_top1": 0, - "baseline_top5": 1, - "refined_top1": 0, - "refined_top5": 1, - "failures": 0 - }, - "test/messages/1": { - "queries": 16, - "baseline_top1": 3, - "baseline_top5": 4, - "refined_top1": 3, - "refined_top5": 6, - "failures": 0 - }, - "test/messages/2": { - "queries": 16, - "baseline_top1": 5, - "baseline_top5": 9, - "refined_top1": 4, - "refined_top5": 10, - "failures": 0 - }, - "test/messages/3": { - "queries": 16, - "baseline_top1": 8, - "baseline_top5": 14, - "refined_top1": 12, - "refined_top5": 15, - "failures": 0 - }, - "test/messages/4": { - "queries": 16, - "baseline_top1": 14, - "baseline_top5": 16, - "refined_top1": 15, - "refined_top5": 16, - "failures": 0 - }, - "test/search/0": { - "queries": 16, - "baseline_top1": 1, - "baseline_top5": 1, - "refined_top1": 1, - "refined_top5": 1, - "failures": 0 - }, - "test/search/1": { - "queries": 16, - "baseline_top1": 1, - "baseline_top5": 2, - "refined_top1": 1, - "refined_top5": 2, - "failures": 0 - }, - "test/search/2": { - "queries": 16, - "baseline_top1": 1, - "baseline_top5": 4, - "refined_top1": 5, - "refined_top5": 6, - "failures": 0 - }, - "test/search/3": { - "queries": 16, - "baseline_top1": 5, - "baseline_top5": 10, - "refined_top1": 9, - "refined_top5": 10, - "failures": 0 - }, - "test/search/4": { - "queries": 16, - "baseline_top1": 6, - "baseline_top5": 12, - "refined_top1": 13, - "refined_top5": 14, - "failures": 0 - } - }, - "prediction_sha256": "3deef3a6c68355c0326b610fe19acf7005028c6a143fce782d225c7754c59df1", - "inputs": { - "neural\\evaluation-protocol.json": "2a2f93a39de9a8a9ab7d18350aef51539b05600781ea972a666ea626b97023ca", - "neural\\fixtures\\regression.json": "5e7fe93fc126862da5895d6c7304414c2a49a48303688648349afb3db956fe76", - "neural\\fixtures\\general-writing.json": "315cb0dacf6026d752c60482915d2dae17ff8842a77290325c6b5f780769fee9", - "english.sqlite": "222253417d0a7a705823ffb7e599a3bcf5d5d3daf4a9d76161ac6b3e555aeaad", - "data\\aac-oanc\\prepared\\candidate.txt": "ff724de5ab3b6e609ed77b699c51287fc82f79d026fdecfa4892939abfbb2834", - "target\\smol-portable\\release\\switchify-prediction-neural.exe": "ee4d74eed3650de251bfe02f4ac752e2bd5f17527653510ea731a28036d7a27d", - "target\\smol-portable\\release\\switchify-smol-worker.exe": "431ded20627f571d89045983520a877feb40060446910359b1c72feecd93cb92" - }, - "gates": { - "overall_top5": true, - "early_cell_regressions": [ - "development/documents/1", - "development/messages/0" - ], - "immediate_latency": true, - "refinement_latency": false, - "minimum_successful_samples": true, - "no_failures": false - }, - "interpretation": "Frozen regression comparison, unknown pretraining overlap; no automatic promotion." -} +{ + "platform": "Windows-11-10.0.26200-SP0", + "processor": "AMD64 Family 26 Model 36 Stepping 0, AuthenticAMD", + "logical_cpus": 24, + "capabilities": { + "accelerated": false, + "context_tokens": 64, + "deadline_ms": 500, + "shortlist": 8, + "threads": 4 + }, + "cold_load_ms": 3529.6957999962615, + "process_tree_peak_rss_bytes": 687869952, + "memory_method": "10ms sum of parent and descendants RSS; shared pages may count twice", + "queries": 1760, + "failures": 2, + "timings": { + "immediate": { + "samples": 1760, + "median_ms": 0.6013999999999999, + "p95_ms": 7.892899999999999, + "max_ms": 21.4904 + }, + "immediate_with_ipc": { + "samples": 1760, + "median_ms": 0.8744999940972775, + "p95_ms": 8.31330000073649, + "max_ms": 34.94259998842608 + }, + "refinement_with_ipc": { + "samples": 1757, + "median_ms": 213.817800002289, + "p95_ms": 327.69039999402594, + "max_ms": 428.9647999976296 + }, + "context_miss": { + "samples": 1366, + "median_ms": 239.4914000033168, + "p95_ms": 332.7148000098532, + "max_ms": 428.9647999976296 + }, + "context_hit": { + "samples": 391, + "median_ms": 152.550999991945, + "p95_ms": 217.16149999701884, + "max_ms": 302.0225000072969 + } + }, + "explicit_retry_load_ms": [ + 368.2270999997854, + 426.7621000035433 + ], + "model_file_bytes": 143041952, + "baseline_file_bytes": 29802496, + "cells": { + "regression/documents/0": { + "queries": 64, + "baseline_top1": 2, + "baseline_top5": 6, + "refined_top1": 7, + "refined_top5": 10, + "failures": 0 + }, + "regression/documents/1": { + "queries": 64, + "baseline_top1": 7, + "baseline_top5": 9, + "refined_top1": 9, + "refined_top5": 10, + "failures": 0 + }, + "regression/documents/2": { + "queries": 64, + "baseline_top1": 10, + "baseline_top5": 19, + "refined_top1": 22, + "refined_top5": 23, + "failures": 0 + }, + "regression/documents/3": { + "queries": 64, + "baseline_top1": 12, + "baseline_top5": 34, + "refined_top1": 37, + "refined_top5": 41, + "failures": 1 + }, + "regression/documents/4": { + "queries": 64, + "baseline_top1": 22, + "baseline_top5": 46, + "refined_top1": 47, + "refined_top5": 53, + "failures": 0 + }, + "regression/email/0": { + "queries": 64, + "baseline_top1": 3, + "baseline_top5": 12, + "refined_top1": 11, + "refined_top5": 18, + "failures": 0 + }, + "regression/email/1": { + "queries": 64, + "baseline_top1": 16, + "baseline_top5": 29, + "refined_top1": 25, + "refined_top5": 30, + "failures": 0 + }, + "regression/email/2": { + "queries": 64, + "baseline_top1": 27, + "baseline_top5": 42, + "refined_top1": 34, + "refined_top5": 44, + "failures": 0 + }, + "regression/email/3": { + "queries": 64, + "baseline_top1": 26, + "baseline_top5": 42, + "refined_top1": 38, + "refined_top5": 47, + "failures": 0 + }, + "regression/email/4": { + "queries": 64, + "baseline_top1": 26, + "baseline_top5": 47, + "refined_top1": 43, + "refined_top5": 52, + "failures": 0 + }, + "regression/messages/0": { + "queries": 64, + "baseline_top1": 10, + "baseline_top5": 13, + "refined_top1": 12, + "refined_top5": 17, + "failures": 0 + }, + "regression/messages/1": { + "queries": 64, + "baseline_top1": 23, + "baseline_top5": 34, + "refined_top1": 26, + "refined_top5": 38, + "failures": 0 + }, + "regression/messages/2": { + "queries": 64, + "baseline_top1": 32, + "baseline_top5": 55, + "refined_top1": 40, + "refined_top5": 56, + "failures": 0 + }, + "regression/messages/3": { + "queries": 64, + "baseline_top1": 38, + "baseline_top5": 59, + "refined_top1": 50, + "refined_top5": 59, + "failures": 0 + }, + "regression/messages/4": { + "queries": 64, + "baseline_top1": 42, + "baseline_top5": 62, + "refined_top1": 48, + "refined_top5": 63, + "failures": 1 + }, + "regression/search/0": { + "queries": 64, + "baseline_top1": 2, + "baseline_top5": 5, + "refined_top1": 5, + "refined_top5": 7, + "failures": 0 + }, + "regression/search/1": { + "queries": 64, + "baseline_top1": 6, + "baseline_top5": 10, + "refined_top1": 11, + "refined_top5": 12, + "failures": 0 + }, + "regression/search/2": { + "queries": 64, + "baseline_top1": 10, + "baseline_top5": 19, + "refined_top1": 19, + "refined_top5": 22, + "failures": 0 + }, + "regression/search/3": { + "queries": 64, + "baseline_top1": 15, + "baseline_top5": 36, + "refined_top1": 32, + "refined_top5": 41, + "failures": 0 + }, + "regression/search/4": { + "queries": 64, + "baseline_top1": 24, + "baseline_top5": 49, + "refined_top1": 35, + "refined_top5": 53, + "failures": 0 + }, + "development/documents/0": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "development/documents/1": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 1, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "development/documents/2": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 2, + "refined_top1": 1, + "refined_top5": 2, + "failures": 0 + }, + "development/documents/3": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 2, + "refined_top5": 3, + "failures": 0 + }, + "development/documents/4": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 5, + "refined_top1": 6, + "refined_top5": 6, + "failures": 0 + }, + "development/email/0": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/email/1": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 2, + "refined_top5": 2, + "failures": 0 + }, + "development/email/2": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 3, + "refined_top5": 4, + "failures": 0 + }, + "development/email/3": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 6, + "refined_top1": 7, + "refined_top5": 7, + "failures": 0 + }, + "development/email/4": { + "queries": 8, + "baseline_top1": 4, + "baseline_top5": 8, + "refined_top1": 7, + "refined_top5": 8, + "failures": 0 + }, + "development/messages/0": { + "queries": 8, + "baseline_top1": 0, + "baseline_top5": 2, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/messages/1": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 1, + "refined_top5": 3, + "failures": 0 + }, + "development/messages/2": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 4, + "refined_top1": 3, + "refined_top5": 4, + "failures": 0 + }, + "development/messages/3": { + "queries": 8, + "baseline_top1": 4, + "baseline_top5": 6, + "refined_top1": 6, + "refined_top5": 6, + "failures": 0 + }, + "development/messages/4": { + "queries": 8, + "baseline_top1": 6, + "baseline_top5": 8, + "refined_top1": 7, + "refined_top5": 8, + "failures": 0 + }, + "development/search/0": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/search/1": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "development/search/2": { + "queries": 8, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 3, + "refined_top5": 3, + "failures": 0 + }, + "development/search/3": { + "queries": 8, + "baseline_top1": 3, + "baseline_top5": 6, + "refined_top1": 5, + "refined_top5": 6, + "failures": 0 + }, + "development/search/4": { + "queries": 8, + "baseline_top1": 5, + "baseline_top5": 7, + "refined_top1": 7, + "refined_top5": 8, + "failures": 0 + }, + "test/documents/0": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "test/documents/1": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 1, + "refined_top5": 2, + "failures": 0 + }, + "test/documents/2": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 3, + "refined_top1": 2, + "refined_top5": 3, + "failures": 0 + }, + "test/documents/3": { + "queries": 16, + "baseline_top1": 3, + "baseline_top5": 7, + "refined_top1": 10, + "refined_top5": 10, + "failures": 0 + }, + "test/documents/4": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 14, + "refined_top1": 13, + "refined_top5": 14, + "failures": 0 + }, + "test/email/0": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 0, + "refined_top1": 0, + "refined_top5": 0, + "failures": 0 + }, + "test/email/1": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 5, + "refined_top1": 1, + "refined_top5": 5, + "failures": 0 + }, + "test/email/2": { + "queries": 16, + "baseline_top1": 3, + "baseline_top5": 8, + "refined_top1": 4, + "refined_top5": 8, + "failures": 0 + }, + "test/email/3": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 11, + "refined_top1": 8, + "refined_top5": 11, + "failures": 0 + }, + "test/email/4": { + "queries": 16, + "baseline_top1": 9, + "baseline_top5": 14, + "refined_top1": 14, + "refined_top5": 15, + "failures": 0 + }, + "test/messages/0": { + "queries": 16, + "baseline_top1": 0, + "baseline_top5": 1, + "refined_top1": 0, + "refined_top5": 1, + "failures": 0 + }, + "test/messages/1": { + "queries": 16, + "baseline_top1": 3, + "baseline_top5": 4, + "refined_top1": 3, + "refined_top5": 6, + "failures": 0 + }, + "test/messages/2": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 9, + "refined_top1": 4, + "refined_top5": 10, + "failures": 0 + }, + "test/messages/3": { + "queries": 16, + "baseline_top1": 8, + "baseline_top5": 14, + "refined_top1": 12, + "refined_top5": 15, + "failures": 0 + }, + "test/messages/4": { + "queries": 16, + "baseline_top1": 14, + "baseline_top5": 16, + "refined_top1": 15, + "refined_top5": 16, + "failures": 0 + }, + "test/search/0": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 1, + "refined_top1": 1, + "refined_top5": 1, + "failures": 0 + }, + "test/search/1": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 2, + "refined_top1": 1, + "refined_top5": 2, + "failures": 0 + }, + "test/search/2": { + "queries": 16, + "baseline_top1": 1, + "baseline_top5": 4, + "refined_top1": 5, + "refined_top5": 6, + "failures": 0 + }, + "test/search/3": { + "queries": 16, + "baseline_top1": 5, + "baseline_top5": 10, + "refined_top1": 9, + "refined_top5": 10, + "failures": 0 + }, + "test/search/4": { + "queries": 16, + "baseline_top1": 6, + "baseline_top5": 12, + "refined_top1": 13, + "refined_top5": 14, + "failures": 0 + } + }, + "prediction_sha256": "3deef3a6c68355c0326b610fe19acf7005028c6a143fce782d225c7754c59df1", + "inputs": { + "neural\\evaluation-protocol.json": "2a2f93a39de9a8a9ab7d18350aef51539b05600781ea972a666ea626b97023ca", + "neural\\fixtures\\regression.json": "5e7fe93fc126862da5895d6c7304414c2a49a48303688648349afb3db956fe76", + "neural\\fixtures\\general-writing.json": "315cb0dacf6026d752c60482915d2dae17ff8842a77290325c6b5f780769fee9", + "english.sqlite": "222253417d0a7a705823ffb7e599a3bcf5d5d3daf4a9d76161ac6b3e555aeaad", + "data\\aac-oanc\\prepared\\candidate.txt": "ff724de5ab3b6e609ed77b699c51287fc82f79d026fdecfa4892939abfbb2834", + "target\\smol-portable\\release\\switchify-prediction-neural.exe": "ee4d74eed3650de251bfe02f4ac752e2bd5f17527653510ea731a28036d7a27d", + "target\\smol-portable\\release\\switchify-smol-worker.exe": "431ded20627f571d89045983520a877feb40060446910359b1c72feecd93cb92" + }, + "gates": { + "overall_top5": true, + "early_cell_regressions": [ + "development/documents/1", + "development/messages/0" + ], + "immediate_latency": true, + "refinement_latency": false, + "minimum_successful_samples": true, + "no_failures": false + }, + "interpretation": "Frozen regression comparison, unknown pretraining overlap; no automatic promotion." +} diff --git a/docs/smol-production-results.md b/docs/smol-production-results.md index 41516be..0a878f6 100644 --- a/docs/smol-production-results.md +++ b/docs/smol-production-results.md @@ -58,4 +58,4 @@ The remaining qualification work is the failed quality gate and actual model run Q8 kernels can produce different candidate orders between the portable and explicit ISA builds. The fixed order fixture records each Windows build separately and checks repeatability, session invalidation and reset. It does not assert bitwise parity across CPU kernels. Aggregate comparisons must therefore use the report for the selected worker, not assume identical outputs from the two builds. -The local converter reproduced the compiled Q8 SHA-256 exactly with the new lockfile. Both source assets and the existing statistical database retained their recorded hashes. The final runtime measurements use implementation commit `047461e`; subsequent changes are test fixtures and documentation. +The local converter reproduced the compiled Q8 SHA-256 exactly with the new lockfile. Both source assets and the existing statistical database retained their recorded hashes. The final runtime measurements use implementation commit `047461e`; subsequent changes are test fixtures, report formatting and documentation. diff --git a/scripts/neural_evaluate.py b/scripts/neural_evaluate.py index 13d70df..93fb61b 100644 --- a/scripts/neural_evaluate.py +++ b/scripts/neural_evaluate.py @@ -238,7 +238,7 @@ def evaluate(args): 'interpretation':'Frozen regression comparison, unknown pretraining overlap; no automatic promotion.'} if args.accelerated_worker: report['accelerated_worker_sha256'] = sha(args.accelerated_worker) - args.output.write_text(json.dumps(report, indent=2)+'\n', encoding='utf-8') + args.output.write_bytes((json.dumps(report, indent=2)+'\n').encode('utf-8')) def main(): From 4a316613194348b1d19794e3f9da1234f8a12fca Mon Sep 17 00:00:00 2001 From: Owen McGirr Date: Sat, 3 Oct 2026 20:54:34 +0100 Subject: [PATCH 7/8] Remove trailing manifest whitespace --- neural/worker/Cargo.toml | 1 - 1 file changed, 1 deletion(-) diff --git a/neural/worker/Cargo.toml b/neural/worker/Cargo.toml index cb46d92..c5595e6 100644 --- a/neural/worker/Cargo.toml +++ b/neural/worker/Cargo.toml @@ -16,4 +16,3 @@ candle-core = "=0.11.0" candle-transformers = "=0.11.0" serde_json = "1" tokenizers = { version = "0.22", default-features = false, features = ["fancy-regex"] } - From 8b201d342940d2233a6df71f1175d2a457366815 Mon Sep 17 00:00:00 2001 From: Owen McGirr Date: Sat, 3 Oct 2026 21:24:56 +0100 Subject: [PATCH 8/8] Remove unused neural protocol and conversion options --- neural/README.md | 2 +- neural/src/protocol.rs | 1 - neural/worker/src/bin/quantize.rs | 12 ++---------- 3 files changed, 3 insertions(+), 12 deletions(-) diff --git a/neural/README.md b/neural/README.md index 6a1e36b..5c969a4 100644 --- a/neural/README.md +++ b/neural/README.md @@ -30,7 +30,7 @@ The runtime is offline and never downloads assets. `source-manifest.json` pins t Acquire the files in `source-manifest.json` into a local source directory. Run the local converter and bundle assembler: ```sh -target/smol-portable/release/quantize SOURCE_DIR MODEL.gguf q8 +target/smol-portable/release/quantize SOURCE_DIR MODEL.gguf python scripts/neural_bundle.py --source SOURCE_DIR --gguf MODEL.gguf --output MODEL_BUNDLE target/smol-portable/release/switchify-prediction-neural validate --bundle MODEL_BUNDLE ``` diff --git a/neural/src/protocol.rs b/neural/src/protocol.rs index 44a08d7..21978eb 100644 --- a/neural/src/protocol.rs +++ b/neural/src/protocol.rs @@ -33,7 +33,6 @@ pub enum Reply { cache_hit: bool, }, Reset, - Failed, } fn invalid() -> io::Error { diff --git a/neural/worker/src/bin/quantize.rs b/neural/worker/src/bin/quantize.rs index f0764b7..71abbd1 100644 --- a/neural/worker/src/bin/quantize.rs +++ b/neural/worker/src/bin/quantize.rs @@ -22,18 +22,10 @@ fn interleave_rope(tensor: Tensor, heads: usize) -> Result { fn main() -> Result<()> { let args: Vec<_> = std::env::args_os().skip(1).collect(); - ensure!( - args.len() == 3, - "Usage: quantize MODEL_DIR OUTPUT.gguf f32|q8" - ); + ensure!(args.len() == 2, "Usage: quantize MODEL_DIR OUTPUT.gguf"); let root = PathBuf::from(&args[0]); let output = PathBuf::from(&args[1]); ensure!(!output.exists(), "Output already exists"); - let dtype = match args[2].to_str() { - Some("f32") => GgmlDType::F32, - Some("q8") => GgmlDType::Q8_0, - _ => anyhow::bail!("Expected f32 or q8"), - }; let cfg: LlamaConfig = serde_json::from_slice(&fs::read(root.join("config.json"))?)?; ensure!( cfg.hidden_size == 576 @@ -54,7 +46,7 @@ fn main() -> Result<()> { tensor = interleave_rope(tensor, heads)?; } let kind = if tensor.rank() == 2 { - dtype + GgmlDType::Q8_0 } else { GgmlDType::F32 };