From 6fa66de2798d0723c76e91fe473f96a1a7732cc9 Mon Sep 17 00:00:00 2001 From: Owen McGirr Date: Sun, 4 Oct 2026 15:51:11 +0100 Subject: [PATCH 1/5] Add bounded asynchronous whole-word generation --- .github/workflows/neural.yml | 36 +++ docs/six-slot-generation.md | 13 + neural/Cargo.lock | 6 +- neural/Cargo.toml | 2 +- neural/README.md | 28 ++ neural/fixtures/baseline-pin.json | 11 + neural/model-bundle.json | 14 +- neural/src/lib.rs | 50 +++- neural/src/main.rs | 38 +++ neural/src/process.rs | 55 ++-- neural/src/protocol.rs | 60 +++- neural/tests/fake_worker.rs | 27 +- neural/tests/lifecycle.rs | 64 +++++ neural/worker/Cargo.toml | 4 +- neural/worker/src/generation.rs | 302 ++++++++++++++++++++ neural/worker/src/main.rs | 11 + neural/worker/src/model.rs | 41 ++- scripts/generation_evaluate.py | 132 +++++++++ scripts/package_neural.py | 2 +- scripts/prepare_generation_qualification.py | 69 +++++ 20 files changed, 923 insertions(+), 42 deletions(-) create mode 100644 docs/six-slot-generation.md create mode 100644 neural/fixtures/baseline-pin.json create mode 100644 neural/worker/src/generation.rs create mode 100644 scripts/generation_evaluate.py create mode 100644 scripts/prepare_generation_qualification.py diff --git a/.github/workflows/neural.yml b/.github/workflows/neural.yml index 934bd3c..26c563a 100644 --- a/.github/workflows/neural.yml +++ b/.github/workflows/neural.yml @@ -52,6 +52,42 @@ jobs: name: neural-${{ matrix.os }} path: artifacts/neural-cli/* if-no-files-found: error + generation-qualification: + strategy: + fail-fast: false + matrix: + os: [windows-latest, macos-latest] + runs-on: ${{ matrix.os }} + timeout-minutes: 120 + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + - uses: dtolnay/rust-toolchain@4716b85f2fac3e324e64fa2810f6b5c3905760a5 + - uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 + with: + workspaces: neural -> ../target/qualification + - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 + with: + python-version: '3.13' + - run: python -m pip install psutil==7.0.0 regex==2025.11.3 + - run: cargo build --manifest-path neural/Cargo.toml --workspace --release --locked --target-dir target/qualification + - name: Prepare pinned offline bundle + shell: bash + run: | + ext="" + if [ "$RUNNER_OS" = "Windows" ]; then ext=".exe"; fi + python scripts/prepare_generation_qualification.py --quantize "target/qualification/release/quantize$ext" --output target/generation-inputs + - name: Measure 1000 warmed queries and compare frozen fixtures + shell: bash + run: | + ext="" + if [ "$RUNNER_OS" = "Windows" ]; then ext=".exe"; fi + python scripts/generation_evaluate.py --cli "target/qualification/release/switchify-prediction-neural$ext" --worker "target/qualification/release/switchify-smol-worker$ext" --baseline target/generation-inputs/english.sqlite --bundle target/generation-inputs/bundle --samples 1000 --output "artifacts/generation-${{ runner.os }}.json" + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + if: always() + with: + name: generation-${{ matrix.os }} + path: artifacts/generation-*.json + if-no-files-found: error neural-audit: runs-on: ubuntu-latest steps: diff --git a/docs/six-slot-generation.md b/docs/six-slot-generation.md new file mode 100644 index 0000000..af1a2ba --- /dev/null +++ b/docs/six-slot-generation.md @@ -0,0 +1,13 @@ +# Six-slot generation qualification + +Issue #23 adds generation while preserving statistical and reranking APIs. +The first three slots belong to the unchanged statistical predictor. The next +three are generated words, excluding the normalized instant words. Empty positions +stay empty. + +The worker uses protocol 2 and the unchanged pinned Q8 model/tokenizer. See +neural/model-bundle.json for the frozen search policy. Reports are produced by +scripts/generation_evaluate.py on the existing frozen fixtures. Unknown overlap +with model pretraining prevents claims of unseen-data accuracy. + +Validation and platform measurements are in progress. No release is published. diff --git a/neural/Cargo.lock b/neural/Cargo.lock index d27eeaf..8b120e6 100644 --- a/neural/Cargo.lock +++ b/neural/Cargo.lock @@ -1416,7 +1416,7 @@ dependencies = [ [[package]] name = "switchify-prediction-neural" -version = "0.2.0" +version = "0.2.1" dependencies = [ "clap", "serde", @@ -1429,14 +1429,16 @@ dependencies = [ [[package]] name = "switchify-smol-worker" -version = "0.2.0" +version = "0.2.1" dependencies = [ "anyhow", "candle-core", "candle-transformers", "serde_json", + "switchify-prediction", "switchify-prediction-neural", "tokenizers", + "unicode-normalization", ] [[package]] diff --git a/neural/Cargo.toml b/neural/Cargo.toml index 1d6da3d..ef0cd84 100644 --- a/neural/Cargo.toml +++ b/neural/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "switchify-prediction-neural" -version = "0.2.0" +version = "0.2.1" edition = "2024" rust-version = "1.97.1" license = "MIT" diff --git a/neural/README.md b/neural/README.md index 9c6485f..6e34796 100644 --- a/neural/README.md +++ b/neural/README.md @@ -64,3 +64,31 @@ python scripts/neural_evaluate.py --cli target/smol-portable/release/switchify-p Add `--accelerated-worker target/smol-avx2/release/switchify-smol-worker` for the optimized comparison. Each run contains 1,760 warmed queries with personal learning disabled. Reports include quality cells, immediate and IPC-inclusive refinement timings, cache hits/misses, cold load and sampled process-tree RSS. Training overlap is checked; unknown neural pretraining overlap remains possible. These are regression comparisons and do not prove unseen-data accuracy. A failed quality or latency gate prevents a production-quality claim; it does not cause test-set tuning or automatic promotion. `python scripts/package_neural.py --portable target/smol-portable/release --accelerated target/smol-avx2/release` packages binaries, hashes and dependency notices without model files. Omit the accelerated path for ARM. CI builds/tests Windows x64, Linux x64, macOS ARM64 and macOS x64. Build/test success is not a claim of measured model latency on those platforms. See `SECURITY.md` for the scoped dependency advisory exception and deployment boundaries. + + +## Asynchronous generation + +The generation API returns up to three additional normalized whole words without +requiring statistical-vocabulary membership. Call +Refiner::generate(before, prefix, session, instant_words, 3), then poll using the +returned request ID. A subsequent submit, generate or reset invalidates old results. +The existing submit/reranking API is unchanged. The stream CLI accepts +{"command":"generate","before":"please send the","prefix":"","session":1} on stdin. + +Worker protocol 2 is required; protocol 1 workers fail cleanly. Model weights and +tokenizer hashes are unchanged. The manifest records beam width 8, at most 8 tokens +per word, at most 64 forward evaluations including uncached context, and a 1600 ms +search cutoff within the unchanged 2000 ms parent reply deadline. Results are +ranked by whole-word probability including following boundary mass. A boundary +probability of at least 0.5 excludes likely unfinished fragments. Search can return +fewer than three words, including none. This is English-focused and not a spelling +dictionary; plausible but incorrect words remain possible. + +Run scripts/generation_evaluate.py with explicit --cli, --worker, --baseline, +--bundle and --output paths. Optional --samples 1000 selects a reproducible spread +across the existing frozen corpus partitions and zero through four graphemes. +Omit --samples for the full comparison. Install psutil==7.0.0 and regex==2025.11.3. +The report includes top-three/top-six counts, fill, OOV, regressions, latency and +process-tree RSS. This is a regression comparison, not unseen-data qualification. +Existing model-quality failures remain; generation has not been promoted to a +qualified model. CI artifacts are prepared without publishing a release. diff --git a/neural/fixtures/baseline-pin.json b/neural/fixtures/baseline-pin.json new file mode 100644 index 0000000..1a58114 --- /dev/null +++ b/neural/fixtures/baseline-pin.json @@ -0,0 +1,11 @@ +{ + "archive": { + "url": "https://github.com/switchifyapp/switchify-prediction/releases/download/v0.1.0/switchify-english-en-aac-oanc-v1.zip", + "sha256": "4537c70b44f553b31e618378a5cc500cd83ffa2dff940263ecabf88e18945b17", + "bytes": 10393776 + }, + "database": { + "sha256": "222253417d0a7a705823ffb7e599a3bcf5d5d3daf4a9d76161ac6b3e555aeaad", + "bytes": 29802496 + } +} diff --git a/neural/model-bundle.json b/neural/model-bundle.json index fd4497c..b2ac8d6 100644 --- a/neural/model-bundle.json +++ b/neural/model-bundle.json @@ -8,7 +8,19 @@ "context_tokens": 64, "shortlist": 8, "max_results": 5, - "scoring": "whole-word log probability plus boundary probability, sequential" + "scoring": "whole-word log probability plus boundary probability, sequential", + "generation": { + "protocol_version": 2, + "beam_width": 8, + "max_tokens_per_word": 8, + "max_forward_evaluations": 64, + "max_results": 3, + "max_word_bytes": 128, + "reply_deadline_ms": 2000, + "scoring": "whole-word probability including following boundary; normalized duplicates excluded", + "minimum_boundary_probability": 0.5, + "search_cutoff_ms": 1600 + } }, "files": { "model.gguf": { diff --git a/neural/src/lib.rs b/neural/src/lib.rs index 0f66a42..598a410 100644 --- a/neural/src/lib.rs +++ b/neural/src/lib.rs @@ -4,7 +4,7 @@ pub mod bundle; mod process; pub mod protocol; -use protocol::Query; +use protocol::{Command, GenerationQuery, Query}; use serde::Serialize; use std::{ path::PathBuf, @@ -74,7 +74,7 @@ pub struct Refined { struct Shared { latest: u64, session: u64, - pending: Option, + pending: Option, result: Option, status: Status, reset: bool, @@ -224,13 +224,13 @@ impl Refiner { && !candidates.is_empty() && matches!(shared.status, Status::Loading | Status::Ready); if refinement_requested { - shared.pending = Some(Query { + shared.pending = Some(Command::Predict(Query { id: shared.latest, session, before: effective_context(before), candidates, limit: options.limit, - }); + })); } let result = Immediate { request_id: shared.latest, @@ -242,6 +242,48 @@ impl Refiner { Ok(result) } + /// Generate additional whole words without a statistical vocabulary restriction. + /// Results use the same request-ID-based poll method as reranking. + /// None means generation is unavailable until an explicit retry. + pub fn generate( + &mut self, + before: &str, + prefix: &str, + session: u64, + exclude: &[String], + limit: usize, + ) -> Result> { + let mut query = GenerationQuery { + id: 0, + session, + before: before.to_owned(), + prefix: normalize(prefix), + exclude: exclude.iter().map(|s| normalize(s)).collect(), + limit, + }; + if before.len() > 16_384 || prefix.len() > 256 || !query.valid() { + self.reset(); + return Err(Error::Input); + } + query.before = effective_context(before); + let mut shared = self.state.0.lock().unwrap(); + shared.latest = shared.latest.checked_add(1).ok_or(Error::Input)?; + shared.pending = None; + shared.result = None; + if shared.session != session { + shared.session = session; + shared.reset = true; + } + let id = shared.latest; + let requested = limit > 0 && matches!(shared.status, Status::Loading | Status::Ready); + if requested { + query.id = id; + shared.pending = Some(Command::Generate(query)); + } + self.state.1.notify_one(); + Ok(requested.then_some(id)) + } + /// A subsequent submit/reset invalidates any previously unconsumed result. pub fn poll(&mut self) -> Option { self.state.0.lock().unwrap().result.take() diff --git a/neural/src/main.rs b/neural/src/main.rs index 326af5a..7abbce5 100644 --- a/neural/src/main.rs +++ b/neural/src/main.rs @@ -47,6 +47,11 @@ struct Run { #[derive(Deserialize)] #[serde(tag = "command", rename_all = "snake_case", deny_unknown_fields)] enum Input { + Generate { + before: String, + prefix: String, + session: u64, + }, Predict { before: String, prefix: String, @@ -140,6 +145,39 @@ fn run(args: Run, once: bool) -> Result<(), ()> { continue; } match rx.recv_timeout(Duration::from_millis(1)) { + Ok(Ok(Input::Generate { + before, + prefix, + session, + })) => { + started = Instant::now(); + predicted = true; + let statistical: Vec = predictor + .predict( + &before, + &prefix, + Options { + limit: 6, + min_chars: 0, + unigram_only: false, + }, + ) + .into_iter() + .map(|s| s.word) + .collect(); + let words: Vec = statistical.iter().take(3).cloned().collect(); + let id = engine + .generate(&before, &prefix, session, &words, 3) + .map_err(|_| ())?; + outstanding = id.is_some(); + emit(json!({"type":"immediate", "result": { + "request_id":id, "words":words, "statistical_six":statistical, + "status":engine.status(), "refinement_requested":outstanding + }, "elapsed_ms":started.elapsed().as_secs_f64()*1000.}))?; + if once { + eof = true; + } + } Ok(Ok(Input::Predict { before, prefix, diff --git a/neural/src/process.rs b/neural/src/process.rs index 8e8a6fc..8653c5b 100644 --- a/neural/src/process.rs +++ b/neural/src/process.rs @@ -157,37 +157,52 @@ pub(super) fn run(config: Config, accelerated: bool, state: State) { } } if let Some(query) = query { - let id = query.id; + let id = match &query { + Command::Predict(q) => q.id, + Command::Generate(q) => q.id, + Command::Reset => return Err(Failure::Protocol), + }; if state.0.lock().unwrap().latest != id { return Ok(()); } - let limit = query.limit.min(query.candidates.len()); - let candidates = query.candidates.clone(); - w.send(Command::Predict(query))?; - match w.receive(&state, Duration::from_millis(REPLY_DEADLINE_MS))? { - Reply::Ranked { - id: actual, - words, - cache_hit, - } if actual == id - && words.len() == limit - && words.iter().all(|w| candidates.contains(w)) + w.send(query.clone())?; + let reply = w.receive(&state, Duration::from_millis(REPLY_DEADLINE_MS))?; + let (words, cache_hit) = match (query, reply) { + ( + Command::Predict(q), + Reply::Ranked { + id: actual, + words, + cache_hit, + }, + ) if actual == id + && words.len() == q.limit.min(q.candidates.len()) + && words.iter().all(|w| q.candidates.contains(w)) && words .iter() .collect::>() .len() == words.len() => { - let mut s = state.0.lock().unwrap(); - if s.latest == id && !s.reset && !s.stop { - s.result = Some(Refined { - request_id: id, - words, - cache_hit, - }); - } + (words, cache_hit) } + ( + Command::Generate(q), + Reply::Generated { + id: actual, + words, + cache_hit, + }, + ) if actual == id && q.accepts(&words) => (words, cache_hit), _ => return Err(Failure::Protocol), + }; + let mut s = state.0.lock().unwrap(); + if s.latest == id && !s.reset && !s.stop { + s.result = Some(Refined { + request_id: id, + words, + cache_hit, + }); } } } diff --git a/neural/src/protocol.rs b/neural/src/protocol.rs index 21978eb..cecddd4 100644 --- a/neural/src/protocol.rs +++ b/neural/src/protocol.rs @@ -2,10 +2,10 @@ use serde::{Deserialize, Serialize, de::DeserializeOwned}; use std::io::{self, Read, Write}; -pub const VERSION: u32 = 1; +pub const VERSION: u32 = 2; pub const MAX_FRAME: usize = 65_536; -#[derive(Serialize, Deserialize)] +#[derive(Clone, Serialize, Deserialize)] #[serde(deny_unknown_fields)] pub struct Query { pub id: u64, @@ -15,13 +15,60 @@ pub struct Query { pub limit: usize, } -#[derive(Serialize, Deserialize)] +#[derive(Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct GenerationQuery { + pub id: u64, + pub session: u64, + pub before: String, + pub prefix: String, + pub exclude: Vec, + pub limit: usize, +} + +impl GenerationQuery { + pub fn valid(&self) -> bool { + self.before.len() <= 16_384 + && self.prefix.len() <= 256 + && self.limit <= 3 + && self.exclude.len() <= 3 + && self.exclude.iter().all(|w| w.len() <= 128) + } + pub fn accepts(&self, words: &[String]) -> bool { + let prefix = switchify_prediction::normalize(&self.prefix); + words.len() <= self.limit + && words.iter().all(|w| { + valid_word(w) + && w.starts_with(&prefix) + && !self + .exclude + .iter() + .any(|e| switchify_prediction::normalize(e) == *w) + }) + && words + .iter() + .collect::>() + .len() + == words.len() + } +} + +/// A canonical, single whole word. Apostrophes are allowed only internally. +pub fn valid_word(word: &str) -> bool { + !word.is_empty() + && word.len() <= 128 + && switchify_prediction::normalize(word) == word + && switchify_prediction::sentences(word) == vec![vec![word.to_owned()]] +} + +#[derive(Clone, Serialize, Deserialize)] pub enum Command { Predict(Query), + Generate(GenerationQuery), Reset, } -#[derive(Serialize, Deserialize)] +#[derive(Clone, Serialize, Deserialize)] pub enum Reply { Ready { version: u32, @@ -32,6 +79,11 @@ pub enum Reply { words: Vec, cache_hit: bool, }, + Generated { + id: u64, + words: Vec, + cache_hit: bool, + }, Reset, } diff --git a/neural/tests/fake_worker.rs b/neural/tests/fake_worker.rs index 6e37ea1..dc1190d 100644 --- a/neural/tests/fake_worker.rs +++ b/neural/tests/fake_worker.rs @@ -17,7 +17,7 @@ fn main() { protocol::write_frame( &mut out, &Reply::Ready { - version: protocol::VERSION, + version: if mode == "old" { 1 } else { protocol::VERSION }, accelerated: false, }, ) @@ -35,6 +35,31 @@ fn main() { } Reply::Reset } + Command::Generate(query) => { + match mode.as_str() { + "crash" => return, + "stall" => thread::sleep(Duration::from_secs(30)), + "delay" => thread::sleep(Duration::from_millis(100)), + _ => {} + } + let mut words: Vec = ["hello", "help", "helium", "café", "can't"] + .into_iter() + .map(String::from) + .filter(|w| query.accepts(std::slice::from_ref(w))) + .take(query.limit) + .collect(); + if mode == "foreign" { + words = vec!["two words".into()]; + } + if mode == "duplicate" { + words = vec!["hello".into(), "hello".into()]; + } + Reply::Generated { + id: query.id + u64::from(mode == "id"), + words, + cache_hit: false, + } + } Command::Predict(mut query) => { match mode.as_str() { "crash" => return, diff --git a/neural/tests/lifecycle.rs b/neural/tests/lifecycle.rs index 3457648..8880d7e 100644 --- a/neural/tests/lifecycle.rs +++ b/neural/tests/lifecycle.rs @@ -330,3 +330,67 @@ fn bare_worker_filename_resolves_in_callers_directory() { assert!(output.status.success()); assert!(String::from_utf8_lossy(&output.stdout).contains("refined")); } + +#[test] +fn generation_is_async_normalized_and_latest_only() { + let (_temp, _predictor, mut engine) = fixture("delay"); + until(|| engine.status() == Status::Ready); + let start = Instant::now(); + let first = engine + .generate("old. say", "he", 1, &["HELLO".into()], 3) + .unwrap() + .unwrap(); + assert!(start.elapsed() < Duration::from_millis(50)); + let latest = engine + .generate("say", "cafe\u{301}", 2, &[], 3) + .unwrap() + .unwrap(); + assert_ne!(first, latest); + let mut result = None; + until(|| { + result = engine.poll(); + result.is_some() + }); + let result = result.unwrap(); + assert_eq!(result.request_id, latest); + assert_eq!(result.words, ["café"]); + engine.generate("", "", 2, &[], 3).unwrap(); + engine.reset(); + thread::sleep(Duration::from_millis(150)); + assert!(engine.poll().is_none()); + assert!(engine.generate("", "", 2, &[], 4).is_err()); +} + +#[test] +fn generation_rejects_invalid_workers_and_timeout_requires_retry() { + for mode in ["foreign", "duplicate", "id", "stall", "crash"] { + let (temp, _predictor, mut engine) = fixture(mode); + until(|| engine.status() == Status::Ready); + engine.generate("", "", 1, &[], 3).unwrap(); + until(|| matches!(engine.status(), Status::Unavailable(_))); + assert!(engine.poll().is_none()); + assert_eq!(engine.generate("", "", 1, &[], 3).unwrap(), None); + std::fs::write(temp.path().join("mode"), "delay").unwrap(); + engine.retry(); + until(|| engine.status() == Status::Ready); + let id = engine + .generate("", "he", 2, &["HELLO".into()], 3) + .unwrap() + .unwrap(); + let mut result = None; + until(|| { + result = engine.poll(); + result.is_some() + }); + let result = result.unwrap(); + assert_eq!(result.request_id, id); + assert_eq!(result.words, ["help", "helium"]); + } +} + +#[test] +fn generation_rejects_old_protocol_without_retry_loop() { + let (_temp, _predictor, mut engine) = fixture("old"); + until(|| matches!(engine.status(), Status::Unavailable(_))); + assert_eq!(engine.generate("", "", 1, &[], 3).unwrap(), None); +} diff --git a/neural/worker/Cargo.toml b/neural/worker/Cargo.toml index 95790ba..89edded 100644 --- a/neural/worker/Cargo.toml +++ b/neural/worker/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "switchify-smol-worker" -version = "0.2.0" +version = "0.2.1" edition = "2024" rust-version = "1.97.1" license = "MIT" @@ -12,6 +12,8 @@ accelerated = [] [dependencies] switchify-prediction-neural = { path = ".." } anyhow = "1" +switchify-prediction = { path = "../.." } +unicode-normalization = "0.1" candle-core = "=0.11.0" candle-transformers = "=0.11.0" serde_json = "1" diff --git a/neural/worker/src/generation.rs b/neural/worker/src/generation.rs new file mode 100644 index 0000000..890a924 --- /dev/null +++ b/neural/worker/src/generation.rs @@ -0,0 +1,302 @@ +//! Bounded byte-level beam search. Prefixes constrain decoded words, not token IDs. +use anyhow::{Result, ensure}; +use std::{ + collections::BTreeMap, + time::{Duration, Instant}, +}; +use switchify_prediction::normalize; +use switchify_prediction_neural::protocol::{GenerationQuery, valid_word}; +use unicode_normalization::{UnicodeNormalization, char::is_combining_mark}; + +const WIDTH: usize = 8; +const TOKENS: usize = 8; +const EVALUATIONS: usize = 64; + +/// Reverse the tokenizer's GPT-2 byte alphabet without replacing partial UTF-8. +pub fn pieces(tokenizer: &tokenizers::Tokenizer) -> Result>> { + let mut bytes: Vec = (33..=126).chain(161..=172).chain(174..=255).collect(); + let mut alphabet: BTreeMap = bytes.iter().map(|&b| (char::from(b), b)).collect(); + let mut extra = 256; + for b in 0..=255 { + if !bytes.contains(&b) { + alphabet.insert(char::from_u32(extra).unwrap(), b); + bytes.push(b); + extra += 1; + } + } + (0..tokenizer.get_vocab_size(true) as u32) + .map(|id| { + let token = tokenizer + .id_to_token(id) + .ok_or_else(|| anyhow::anyhow!("tokenizer"))?; + if token.starts_with("<|") { + return Ok(Vec::new()); + } + token + .chars() + .map(|c| { + alphabet + .get(&c) + .copied() + .ok_or_else(|| anyhow::anyhow!("tokenizer")) + }) + .collect() + }) + .collect() +} + +fn boundary(piece: &[u8], id: usize) -> bool { + id == 0 + || piece.first().is_some_and(|b| { + b.is_ascii_whitespace() || b".!?;:,()[]{}\"-/".contains(b) || b.is_ascii_digit() + }) +} + +/// A trailing incomplete code point may become valid on the next token. +fn partial(bytes: &[u8], separated: bool, prefix: &str) -> bool { + let bytes = if separated { + if bytes.first() != Some(&b' ') { + return false; + } + &bytes[1..] + } else { + bytes.strip_prefix(b" ").unwrap_or(bytes) + }; + if bytes.len() > 128 { + return false; + } + let text = match std::str::from_utf8(bytes) { + Ok(s) => s, + Err(e) if e.error_len().is_none() => { + std::str::from_utf8(&bytes[..e.valid_up_to()]).unwrap() + } + Err(_) => return false, + }; + let normalized = normalize(text); + let mut letter = false; + let mut apostrophe = false; + for c in normalized.chars() { + if c.is_alphabetic() { + letter = true; + apostrophe = false; + } else if is_combining_mark(c) && letter && !apostrophe { + } else if c == '\'' && letter && !apostrophe { + apostrophe = true; + } else { + return false; + } + } + let text: String = normalized.nfd().collect(); + let prefix: String = prefix.nfd().collect(); + text.starts_with(&prefix) || prefix.starts_with(&text) +} + +fn completed(bytes: &[u8], query: &GenerationQuery) -> Option { + let text = std::str::from_utf8(bytes).ok()?; + let word = normalize(text.strip_prefix(' ').unwrap_or(text)); + (valid_word(&word) && query.accepts(std::slice::from_ref(&word))).then_some(word) +} + +struct Beam { + state: S, + logits: Vec, + bytes: Vec, + score: f64, +} +struct Candidate { + parent: usize, + token: usize, + bytes: Vec, + score: f64, +} + +/// The caller accounts for the context forward pass. Each selected extension +/// costs one evaluation; terminal boundary mass costs none. Scores stay private. +pub fn search( + query: &GenerationQuery, + pieces: &[Vec], + state: S, + logits: Vec, + mut evaluations: usize, + started: Instant, + mut forward: impl FnMut(&mut S, u32, usize) -> Result>, +) -> Result> { + let mut beams = vec![Beam { + state, + logits, + bytes: Vec::new(), + score: 0., + }]; + let mut finished: BTreeMap = BTreeMap::new(); + for depth in 0..TOKENS { + let mut candidates: Vec = Vec::new(); + for (parent, beam) in beams.iter().enumerate() { + ensure!( + beam.logits.len() == pieces.len() && beam.logits.iter().all(|v| v.is_finite()), + "logits" + ); + let max = beam + .logits + .iter() + .copied() + .fold(f32::NEG_INFINITY, f32::max) as f64; + let normalizer = max + + beam + .logits + .iter() + .map(|v| (*v as f64 - max).exp()) + .sum::() + .ln(); + for (token, piece) in pieces.iter().enumerate() { + if piece.is_empty() { + continue; + } + let score = beam.score + beam.logits[token] as f64 - normalizer; + if candidates.len() == WIDTH && score <= candidates.last().unwrap().score { + continue; + } + let mut bytes = beam.bytes.clone(); + bytes.extend_from_slice(piece); + if partial(&bytes, !query.before.is_empty(), &query.prefix) { + candidates.push(Candidate { + parent, + token, + bytes, + score, + }); + candidates.sort_by(|a, b| { + b.score + .total_cmp(&a.score) + .then(a.parent.cmp(&b.parent)) + .then(a.token.cmp(&b.token)) + }); + candidates.truncate(WIDTH); + } + } + } + candidates.sort_by(|a, b| { + b.score + .total_cmp(&a.score) + .then(a.parent.cmp(&b.parent)) + .then(a.token.cmp(&b.token)) + }); + let mut next = Vec::new(); + for candidate in candidates.into_iter().take(WIDTH) { + // Reserve time for the final evaluation and reply transport; the parent still enforces 2 s. + if evaluations >= EVALUATIONS || started.elapsed() >= Duration::from_millis(1600) { + break; + } + let mut state = beams[candidate.parent].state.clone(); + let logits = forward(&mut state, candidate.token as u32, depth)?; + evaluations += 1; + ensure!( + logits.len() == pieces.len() && logits.iter().all(|v| v.is_finite()), + "logits" + ); + if let Some(word) = completed(&candidate.bytes, query) { + let max = logits.iter().copied().fold(f32::NEG_INFINITY, f32::max) as f64; + let all: f64 = logits.iter().map(|v| (*v as f64 - max).exp()).sum(); + let mass: f64 = logits + .iter() + .enumerate() + .filter(|(id, _)| boundary(&pieces[*id], *id)) + .map(|(_, v)| (*v as f64 - max).exp()) + .sum(); + let score = candidate.score + (mass / all).ln(); + // A token fragment is not a completed word merely because EOS + // has nonzero probability. Require a probable word boundary. + if mass / all < 0.5 { + next.push(Beam { + state, + logits, + bytes: candidate.bytes, + score: candidate.score, + }); + continue; + } + // Different tokenizations/case variants represent disjoint paths. + finished + .entry(word) + .and_modify(|old| { + let high = old.max(score); + *old = high + ((*old - high).exp() + (score - high).exp()).ln(); + }) + .or_insert(score); + } + next.push(Beam { + state, + logits, + bytes: candidate.bytes, + score: candidate.score, + }); + } + beams = next; + if beams.is_empty() + || evaluations >= EVALUATIONS + || started.elapsed() >= Duration::from_millis(1600) + { + break; + } + } + let mut words: Vec<_> = finished.into_iter().collect(); + words.sort_by(|a, b| b.1.total_cmp(&a.1).then(a.0.cmp(&b.0))); + Ok(words + .into_iter() + .take(query.limit) + .map(|(word, _)| word) + .collect()) +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn byte_prefix_and_word_shapes() { + assert!(partial(b" caf\xc3", true, "café")); + assert!(partial(" cafe\u{301}".as_bytes(), true, "café")); + assert!(partial(b" can't", true, "can'")); + assert!(!partial(b" two words", true, "")); + assert!(!partial(b" help", true, "z")); + assert!(!partial(b" \xff", true, "")); + assert!(!partial(b" ''", true, "")); + } + #[test] + fn constrained_search_spans_prefix_and_respects_budget() { + let query = GenerationQuery { + id: 1, + session: 1, + before: "say".into(), + prefix: "he".into(), + exclude: vec!["hello".into()], + limit: 3, + }; + let pieces = vec![ + vec![], + b" hello".to_vec(), + b" hel".to_vec(), + b"ium".to_vec(), + b" world".to_vec(), + ]; + let mut calls = 0; + let words = search( + &query, + &pieces, + (), + vec![-10., 5., 4., -10., 0.], + 1, + Instant::now(), + |_, token, _| { + calls += 1; + Ok(if token == 2 { + vec![-20., -20., -20., 10., -20.] + } else { + vec![10., -20., -20., -20., -20.] + }) + }, + ) + .unwrap(); + assert_eq!(words[0], "helium"); + assert!(!words.contains(&"hello".into())); + assert!(calls <= 63); + } +} diff --git a/neural/worker/src/main.rs b/neural/worker/src/main.rs index 8d55eb5..6f2f7f6 100644 --- a/neural/worker/src/main.rs +++ b/neural/worker/src/main.rs @@ -1,4 +1,5 @@ //! Dedicated offline inference process. All errors are deliberately text-free. +mod generation; mod model; use std::{io, path::PathBuf}; use switchify_prediction_neural::{ @@ -42,6 +43,16 @@ fn run() -> anyhow::Result<()> { model.reset(); Reply::Reset } + Command::Generate(query) => { + anyhow::ensure!(query.valid(), "query"); + let (words, cache_hit) = model.generate(&query)?; + anyhow::ensure!(query.accepts(&words), "generation"); + Reply::Generated { + id: query.id, + words, + cache_hit, + } + } Command::Predict(query) => { anyhow::ensure!( query.before.len() <= 16_384 diff --git a/neural/worker/src/model.rs b/neural/worker/src/model.rs index ae551a6..849a15a 100644 --- a/neural/worker/src/model.rs +++ b/neural/worker/src/model.rs @@ -14,6 +14,7 @@ pub struct Model { empty: ModelWeights, tokenizer: Tokenizer, boundaries: Vec, + pieces: Vec>, prepared: Option, } fn forward(state: &mut ModelWeights, ids: &[u32], position: usize) -> Result> { @@ -61,7 +62,9 @@ impl Model { boundaries.push(id as usize); } } + let pieces = crate::generation::pieces(&tokenizer)?; Ok(Self { + pieces, empty, tokenizer, boundaries, @@ -71,13 +74,7 @@ impl Model { pub fn reset(&mut self) { self.prepared = None; } - pub fn rank( - &mut self, - session: u64, - before: &str, - candidates: &[String], - limit: usize, - ) -> Result<(Vec, bool)> { + fn prepare(&mut self, session: u64, before: &str) -> Result { let encoded = self .tokenizer .encode(before, false) @@ -100,6 +97,36 @@ impl Model { logits, }); } + Ok(cache_hit) + } + + pub fn generate( + &mut self, + query: &switchify_prediction_neural::protocol::GenerationQuery, + ) -> Result<(Vec, bool)> { + let started = std::time::Instant::now(); + let hit = self.prepare(query.session, &query.before)?; + let prepared = self.prepared.as_ref().unwrap(); + let words = crate::generation::search( + query, + &self.pieces, + prepared.state.clone(), + prepared.logits.clone(), + usize::from(!hit), + started, + |state, token, depth| forward(state, &[token], prepared.ids.len() + depth), + )?; + Ok((words, hit)) + } + + pub fn rank( + &mut self, + session: u64, + before: &str, + candidates: &[String], + limit: usize, + ) -> Result<(Vec, bool)> { + let cache_hit = self.prepare(session, before)?; let prepared = self.prepared.as_ref().unwrap(); let mut scored = Vec::new(); for (index, word) in candidates.iter().enumerate() { diff --git a/scripts/generation_evaluate.py b/scripts/generation_evaluate.py new file mode 100644 index 0000000..5e4719f --- /dev/null +++ b/scripts/generation_evaluate.py @@ -0,0 +1,132 @@ +#!/usr/bin/env python3 +"""Compare six reserved slots against reranking and six statistical suggestions. + +Only frozen synthetic corpus text is submitted. Reports contain aggregate counts, +timing, hashes and process-tree RSS, never words or prompts. No desktop input. +Requires psutil and regex. No personal database is opened. +""" +import argparse +import collections +import json +from pathlib import Path +import platform +import sqlite3 +import time + +from neural_evaluate import Client, ROOT, sha, timing, workload + + +def generated(client, before, prefix): + started = time.perf_counter() + client.process.stdin.write(json.dumps(dict(command='generate', before=before, + prefix=prefix, session=1)) + '\n') + client.process.stdin.flush() + immediate = client.receive() + assert immediate['type'] == 'immediate' + elapsed = (time.perf_counter() - started) * 1000 + if not immediate['result']['refinement_requested']: + return immediate, None, elapsed, None + reply = client.receive() + total = (time.perf_counter() - started) * 1000 + if reply['type'] != 'refined': + return immediate, None, elapsed, total + assert reply['result']['request_id'] == immediate['result']['request_id'] + return immediate, reply, elapsed, total + + +def evaluate(args): + queries = workload() + # Cover every corpus partition and prefix length, even in a bounded run. + if args.samples: + queries = [queries[i * len(queries) // args.samples % len(queries)] + for i in range(args.samples)] + db = sqlite3.connect(args.baseline.resolve().as_uri() + '?mode=ro', uri=True) + vocabulary = {row[0] for row in db.execute('SELECT word FROM vocabulary')} + db.close() + command = [str(args.cli.resolve()), 'stream', '--baseline', str(args.baseline.resolve()), + '--bundle', str(args.bundle.resolve()), '--worker', str(args.worker.resolve())] + if args.accelerated_worker: + command += ['--accelerated-worker', str(args.accelerated_worker.resolve())] + client = Client(command) + times = collections.defaultdict(list) + cells = collections.defaultdict(collections.Counter) + failures = collections.Counter() + warmup_failures = 0 + try: + for query in queries[:20]: + _, reply, _, _ = generated(client, query[4], query[5]) + if reply is None: + warmup_failures += 1 + client.retry() + for index, (_, part, domain, n, before, prefix, target) in enumerate(queries): + old, ranked, _, _ = client.query(before, prefix) + old_words = ranked['result']['words'] if ranked else old['result']['words'] + if old['result']['refinement_requested'] and ranked is None: + failures['rerank'] += 1 + client.retry() + immediate, reply, latency, total = generated(client, before, prefix) + instant = immediate['result']['words'] + stats = immediate['result']['statistical_six'] + assert instant == stats[:3] + words = reply['result']['words'] if reply else [] + assert len(words) <= 3 and not set(words).intersection(instant) + assert len(set(words)) == len(words) + slots = instant + [None] * (3 - len(instant)) + words + [None] * (3 - len(words)) + times['immediate_with_ipc'].append(latency) + if reply: + times['generation_with_ipc'].append(total) + else: + failures['generation'] += 1 + cell = cells[f'{part}/{domain}/{n}'] + cell['queries'] += 1 + cell['generated_words'] += len(words) + cell['filled_neural_queries'] += bool(words) + cell['generated_oov_words'] += sum(w not in vocabulary for w in words) + cell['oov_targets'] += target not in vocabulary + cell['oov_hits'] += target not in vocabulary and target in words + cell['generation_failures'] += reply is None + for mode, values in [('current', old_words), ('statistical_six', stats), ('six_slots', slots)]: + cell[mode + '_top3_hits'] += target in values[:3] + cell[mode + '_top6_hits'] += target in values[:6] + cell['regressions_vs_statistical_six'] += target in stats and target not in slots + cell['gains_vs_statistical_six'] += target not in stats and target in slots + cell['regressions_vs_current'] += target in old_words and target not in slots + if reply is None: + client.retry() # Explicit benchmark recovery, not production policy. + if (index + 1) % 100 == 0: + print(json.dumps({'completed': index + 1, 'failures': dict(failures)}), flush=True) + finally: + client.close() + for cell in cells.values(): + for mode in ['current', 'statistical_six', 'six_slots']: + for k in [3, 6]: + cell[f'{mode}_top{k}_accuracy'] = cell[f'{mode}_top{k}_hits'] / cell['queries'] + cell['neural_slot_fill_rate'] = cell['generated_words'] / (cell['queries'] * 3) + report = dict(schema=1, policy=json.loads((ROOT / 'neural/model-bundle.json').read_bytes())['policy'], + platform=platform.platform(), queries=len(queries), warmup_queries=min(20, len(queries)), + warmup_failures=warmup_failures, failures=dict(failures), + latency={k: timing(v) for k, v in times.items()}, cold_load_ms=client.cold_ms, + peak_process_tree_rss_bytes=client.peak, capabilities=client.ready['capabilities'], + cells=dict(cells), baseline_sha256=sha(args.baseline), worker_sha256=sha(args.worker), + accelerated_worker_sha256=sha(args.accelerated_worker) if args.accelerated_worker else None, + fixture_hashes={n: sha(ROOT / 'neural/fixtures' / n) for n in ['regression.json', 'general-writing.json']}, + note='Regression comparison only; fixture overlap with model pretraining is unknown. Current mode returns at most five words. Explicit benchmark retries counted; production never retries automatically.', + production_qualified=False) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(report, indent=2) + '\n', encoding='utf-8') + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + for name in ['cli', 'baseline', 'bundle', 'worker', 'output']: + parser.add_argument('--' + name, type=Path, required=True) + parser.add_argument('--accelerated-worker', type=Path) + parser.add_argument('--samples', type=int, default=0, help='0 uses the full frozen workload') + args = parser.parse_args() + if args.samples < 0: + parser.error('samples must be nonnegative') + evaluate(args) + + +if __name__ == '__main__': + main() diff --git a/scripts/package_neural.py b/scripts/package_neural.py index 8af4afc..bf32a6f 100644 --- a/scripts/package_neural.py +++ b/scripts/package_neural.py @@ -90,7 +90,7 @@ def main(): (stage / 'BUILD.json').write_text(json.dumps({'version':version,'target':target, 'rustc':rustc, 'platform':platform.platform(), 'libc':platform.libc_ver(), 'signed':False, 'commit':subprocess.check_output(['git','rev-parse','HEAD'],cwd=ROOT,text=True).strip(), - 'model_id':'smollm2-135m-q8-v1', 'accelerated_requires':['avx2','fma','f16c'] if args.accelerated else [], + 'model_id':'smollm2-135m-q8-v1', 'worker_protocol':2, 'generation_policy':json.loads((ROOT / 'neural/model-bundle.json').read_bytes())['policy']['generation'], 'accelerated_requires':['avx2','fma','f16c'] if args.accelerated else [], 'production_qualified':json.loads((ROOT / 'neural/qualification.json').read_bytes())['production_qualified'], 'model_assets_included':False}, indent=2)+'\n',encoding='utf-8') (stage / 'SHA256SUMS').write_text(''.join(f'{sha(p)} {p.name}\n' for p in sorted(stage.iterdir())), encoding='utf-8') diff --git a/scripts/prepare_generation_qualification.py b/scripts/prepare_generation_qualification.py new file mode 100644 index 0000000..53bdad2 --- /dev/null +++ b/scripts/prepare_generation_qualification.py @@ -0,0 +1,69 @@ +#!/usr/bin/env python3 +"""Fetch checksum-pinned qualification inputs at build time, then convert locally.""" +import argparse +import json +from pathlib import Path +import shutil +import subprocess +import tempfile +import urllib.request +import zipfile + +from neural_bundle import ROOT, assemble, verify + + +def fetch(path, pin): + if path.exists(): + verify(path, pin) + return + path.parent.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory(dir=path.parent) as temp: + downloaded = Path(temp) / 'input' + with urllib.request.urlopen(pin['url'], timeout=300) as response, downloaded.open('wb') as out: + remaining = pin['bytes'] + 1 + while remaining: + chunk = response.read(min(65536, remaining)) + if not chunk: + break + out.write(chunk) + remaining -= len(chunk) + verify(downloaded, pin) + downloaded.replace(path) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--quantize', type=Path, required=True) + parser.add_argument('--output', type=Path, required=True) + args = parser.parse_args() + source = args.output / 'source' + pins = json.loads((ROOT / 'neural/source-manifest.json').read_bytes()) + for name, pin in pins['files'].items(): + fetch(source / name, pin) + model = args.output / 'model.gguf' + if not model.exists(): + subprocess.run([str(args.quantize.resolve()), str(source.resolve()), str(model.resolve())], check=True) + manifest = json.loads((ROOT / 'neural/model-bundle.json').read_bytes()) + verify(model, manifest['files']['model.gguf']) + bundle = args.output / 'bundle' + if not bundle.exists(): + assemble(source, model, bundle) + else: + for name, pin in manifest['files'].items(): + verify(bundle / name, pin) + shutil.copyfile(ROOT / 'neural/model-bundle.json', bundle / 'model-bundle.json') + baseline = json.loads((ROOT / 'neural/fixtures/baseline-pin.json').read_bytes()) + archive = args.output / 'baseline.zip' + fetch(archive, baseline['archive']) + database = args.output / 'english.sqlite' + with zipfile.ZipFile(archive) as zipped: + names = [n for n in zipped.namelist() if n.endswith('/english.sqlite') or n == 'english.sqlite'] + if len(names) != 1 or zipped.getinfo(names[0]).file_size != baseline['database']['bytes']: + raise ValueError('Invalid baseline archive') + with zipped.open(names[0]) as input_file, database.open('wb') as output_file: + shutil.copyfileobj(input_file, output_file) + verify(database, baseline['database']) + + +if __name__ == '__main__': + main() From ddbe1f0affc58c9a52c7270c7940fbf4a4fbaec1 Mon Sep 17 00:00:00 2001 From: Owen McGirr Date: Sun, 4 Oct 2026 15:56:04 +0100 Subject: [PATCH 2/5] Measure generation with fresh context and extend Unicode tests --- neural/worker/src/generation.rs | 71 +++++++++++++++++++++++++++++++++ scripts/generation_evaluate.py | 6 +++ scripts/test_neural.py | 11 +++++ 3 files changed, 88 insertions(+) diff --git a/neural/worker/src/generation.rs b/neural/worker/src/generation.rs index 890a924..8301345 100644 --- a/neural/worker/src/generation.rs +++ b/neural/worker/src/generation.rs @@ -299,4 +299,75 @@ mod tests { assert!(!words.contains(&"hello".into())); assert!(calls <= 63); } + + #[test] + fn incomplete_utf8_and_decomposed_words_complete_without_replacement() { + for (first, second) in [ + (b" caf\xc3".to_vec(), b"\xa9".to_vec()), + (b" cafe".to_vec(), "\u{301}".as_bytes().to_vec()), + ] { + let query = GenerationQuery { + id: 1, + session: 1, + before: "a".into(), + prefix: "café".into(), + exclude: vec![], + limit: 3, + }; + let pieces = vec![vec![], first, second]; + let result = search( + &query, + &pieces, + (), + vec![-20., 10., -20.], + 1, + Instant::now(), + |_, token, _| { + Ok(if token == 1 { + vec![-20., -20., 10.] + } else { + vec![10., -20., -20.] + }) + }, + ) + .unwrap(); + assert_eq!(result[0], "café"); + assert!(result.iter().all(|w| !w.contains('\uFFFD'))); + } + } + + #[test] + fn budget_and_fragment_gate_do_not_fill_from_statistics() { + let query = GenerationQuery { + id: 1, + session: 1, + before: "".into(), + prefix: "".into(), + exclude: vec![], + limit: 3, + }; + let pieces = std::iter::once(vec![]) + .chain((b'a'..=b'h').map(|b| vec![b])) + .collect::>(); + let mut calls = 0; + let result = search( + &query, + &pieces, + (), + vec![0.; 9], + 1, + Instant::now(), + |_, _, _| { + calls += 1; + Ok(vec![0.; 9]) + }, + ) + .unwrap(); + assert_eq!(calls, 63); + assert!( + result.is_empty(), + "continuations are more probable than boundaries" + ); + assert!(!partial(&[b'a'; 129], false, "")); + } } diff --git a/scripts/generation_evaluate.py b/scripts/generation_evaluate.py index 5e4719f..cc7739b 100644 --- a/scripts/generation_evaluate.py +++ b/scripts/generation_evaluate.py @@ -17,6 +17,11 @@ def generated(client, before, prefix): + # Match the desktop's changed-context reset. The comparison mode must not + # prefill generation's context cache before the measured request. + client.process.stdin.write('{"command":"reset"}\n') + client.process.stdin.flush() + assert client.receive()['type'] == 'reset' started = time.perf_counter() client.process.stdin.write(json.dumps(dict(command='generate', before=before, prefix=prefix, session=1)) + '\n') @@ -85,6 +90,7 @@ def evaluate(args): cell['oov_targets'] += target not in vocabulary cell['oov_hits'] += target not in vocabulary and target in words cell['generation_failures'] += reply is None + cell['generation_context_hits'] += bool(reply and reply['result']['cache_hit']) for mode, values in [('current', old_words), ('statistical_six', stats), ('six_slots', slots)]: cell[mode + '_top3_hits'] += target in values[:3] cell[mode + '_top6_hits'] += target in values[:6] diff --git a/scripts/test_neural.py b/scripts/test_neural.py index 2b5a194..c22f977 100644 --- a/scripts/test_neural.py +++ b/scripts/test_neural.py @@ -6,11 +6,22 @@ from neural_bundle import verify from neural_evaluate import warmup +from generation_evaluate import generated ROOT = Path(__file__).resolve().parent.parent class NeuralBundleTests(unittest.TestCase): + def test_generation_comparison_resets_context_before_measuring(self): + import io + from types import SimpleNamespace + replies = iter([{'type':'reset'}, {'type':'immediate', 'result':{'refinement_requested':True, 'request_id':8}}, + {'type':'refined', 'result':{'request_id':8}}]) + client = SimpleNamespace(process=SimpleNamespace(stdin=io.StringIO()), receive=lambda: next(replies)) + generated(client, 'synthetic', 'sy') + commands = [json.loads(line) for line in client.process.stdin.getvalue().splitlines()] + self.assertEqual([c['command'] for c in commands], ['reset', 'generate']) + def test_warmup_cannot_silently_turn_into_statistical_only_measurement(self): class FakeClient: def __init__(self, status, requested, refined): From 70fc4a869ffd22a405527e4d81d25c149aebc418 Mon Sep 17 00:00:00 2001 From: Owen McGirr Date: Sun, 4 Oct 2026 15:56:32 +0100 Subject: [PATCH 3/5] Fix replacement-character test escape --- neural/worker/src/generation.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/neural/worker/src/generation.rs b/neural/worker/src/generation.rs index 8301345..56f8fc7 100644 --- a/neural/worker/src/generation.rs +++ b/neural/worker/src/generation.rs @@ -332,7 +332,7 @@ mod tests { ) .unwrap(); assert_eq!(result[0], "café"); - assert!(result.iter().all(|w| !w.contains('\uFFFD'))); + assert!(result.iter().all(|w| !w.contains('\u{FFFD}'))); } } From bc7ef9421fc0f1b7605d312319bdced9540f7ca7 Mon Sep 17 00:00:00 2001 From: Owen McGirr Date: Sun, 4 Oct 2026 16:18:04 +0100 Subject: [PATCH 4/5] Align v0.2.1 release preparation and qualification scope --- Cargo.lock | 2 +- Cargo.toml | 2 +- docs/releases/v0.2.1.md | 23 +++++++++++++++++++++++ neural/Cargo.lock | 2 +- neural/README.md | 4 ++-- neural/qualification.json | 5 ++++- 6 files changed, 32 insertions(+), 6 deletions(-) create mode 100644 docs/releases/v0.2.1.md diff --git a/Cargo.lock b/Cargo.lock index 651d0b9..8ad3e16 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -481,7 +481,7 @@ checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" [[package]] name = "switchify-prediction" -version = "0.2.0" +version = "0.2.1" dependencies = [ "clap", "rusqlite", diff --git a/Cargo.toml b/Cargo.toml index 78f605f..979af45 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "switchify-prediction" -version = "0.2.0" +version = "0.2.1" edition = "2024" rust-version = "1.97.1" license = "MIT" diff --git a/docs/releases/v0.2.1.md b/docs/releases/v0.2.1.md new file mode 100644 index 0000000..1321f3b --- /dev/null +++ b/docs/releases/v0.2.1.md @@ -0,0 +1,23 @@ +# v0.2.1 + +Adds asynchronous whole-word generation to the offline SmolLM2 companion while +preserving the statistical and reranking APIs and SQLite formats. Generation +uses protocol 2, so the matching v0.2.1 workers are required. Model and tokenizer +bytes are unchanged. + +The generation request carries context, prefix, session identity, up to three +excluded instant words and a result limit of up to three. Search uses eight +beams, up to eight tokens per word and 64 forward evaluations. Results are +normalized whole words with a probable following boundary. Words need not +belong to the statistical vocabulary. Empty results are valid. + +The two-second reply deadline and 30-second startup bound are unchanged. +Failure requires explicit retry. No network access is needed at runtime. + +Generation is not automatically production-qualified. Corpus comparisons are +regression tests with unknown pretraining overlap. Existing reranking quality +regressions remain documented in the historical qualification reports. +Prepared CI bundles are unsigned; platform applications sign their embedded +workers through their own packaging workflows. + +Publication of this release and the Switchify PC RC require separate approval. diff --git a/neural/Cargo.lock b/neural/Cargo.lock index 8b120e6..68c758a 100644 --- a/neural/Cargo.lock +++ b/neural/Cargo.lock @@ -1401,7 +1401,7 @@ checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" [[package]] name = "switchify-prediction" -version = "0.2.0" +version = "0.2.1" dependencies = [ "clap", "rusqlite", diff --git a/neural/README.md b/neural/README.md index 6e34796..c0f8813 100644 --- a/neural/README.md +++ b/neural/README.md @@ -1,10 +1,10 @@ # SmolLM2 companion -An optional library and CLI for immediate statistical suggestions followed by offline SmolLM2 refinement. The root predictor, learning APIs and SQLite formats are unchanged. This package has its own workspace and lockfile. It is not integrated with Switchify PC. +An optional library and CLI for immediate statistical suggestions followed by offline SmolLM2 refinement. The root predictor, learning APIs and SQLite formats are unchanged. This package has its own workspace and lockfile. Switchify PC integration is maintained in its separate repository. This candidate is not yet production-qualified. The optimized Windows worker passed the latency target and improved overall top-five accuracy, but two development cells failed the frozen quality gate. See `docs/smol-production-results.md` in the repository for the full comparison and platform limits. `Ready` means the worker is available, not that the quality gate has passed. -The fixed policy reranks eight statistical candidates using SmolLM2-135M Q8. It scores every token of a word plus the probability of a following word boundary, uses the current sentence capped at 64 tokens and returns at most five ordered words. Scores from different models are never combined or exposed as shared probabilities. Limits above five are errors. Minimum grapheme settings are honored; unigram-only requests stay statistical. The same caller-owned predictor supplies immediate suggestions and the shortlist, including its current personal snapshot. +The preserved reranking policy reranks eight statistical candidates using SmolLM2-135M Q8. It scores every token of a word plus the probability of a following word boundary, uses the current sentence capped at 64 tokens and returns at most five ordered words. Scores from different models are never combined or exposed as shared probabilities. Limits above five are errors. Minimum grapheme settings are honored; unigram-only requests stay statistical. The same caller-owned predictor supplies immediate suggestions and the shortlist, including its current personal snapshot. The controller returns immediate words synchronously. A persistent child loads the model once and refines through private bounded pipes. There is one active request and one replaceable pending request. `poll()` yields only the latest request; callers should also match its request ID before displaying it. Session changes and `reset()` invalidate old results and clear the worker's context. Loading is separate from the 2 second reply deadline for inference and reset. Startup has a 30 second bound. Failure stops the worker and leaves immediate suggestions available. Recovery requires `retry()`. `shutdown()` kills and joins the worker and clears pending text. diff --git a/neural/qualification.json b/neural/qualification.json index 99d660d..8f25c8a 100644 --- a/neural/qualification.json +++ b/neural/qualification.json @@ -9,5 +9,8 @@ ], "optimized_windows_latency_passed": true, "desktop_integration": false, - "automatic_promotion": false + "automatic_promotion": false, + "scope": "Historical v0.2.0 reranking qualification; retained unchanged reasons are not generation qualification claims.", + "generation_report_command": "python scripts/generation_evaluate.py", + "generation_production_qualified": false } From c25f068dacef8fbf65dde6b98c745ad4cd56d555 Mon Sep 17 00:00:00 2001 From: Owen McGirr Date: Sun, 4 Oct 2026 16:19:03 +0100 Subject: [PATCH 5/5] Describe reply deadline change relative to shipped release --- docs/releases/v0.2.1.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/releases/v0.2.1.md b/docs/releases/v0.2.1.md index 1321f3b..da1f93e 100644 --- a/docs/releases/v0.2.1.md +++ b/docs/releases/v0.2.1.md @@ -11,8 +11,8 @@ beams, up to eight tokens per word and 64 forward evaluations. Results are normalized whole words with a probable following boundary. Words need not belong to the statistical vocabulary. Empty results are valid. -The two-second reply deadline and 30-second startup bound are unchanged. -Failure requires explicit retry. No network access is needed at runtime. +The inference/reset reply deadline increases from 500 ms in v0.2.0 to two +seconds. Startup remains bounded to 30 seconds. Failure requires explicit retry. No network access is needed at runtime. Generation is not automatically production-qualified. Corpus comparisons are regression tests with unknown pretraining overlap. Existing reranking quality