diff --git a/crates/tracedecay-code-extraction/Cargo.toml b/crates/tracedecay-code-extraction/Cargo.toml index 3551b391a6..949e7fc8b6 100644 --- a/crates/tracedecay-code-extraction/Cargo.toml +++ b/crates/tracedecay-code-extraction/Cargo.toml @@ -80,7 +80,7 @@ lang-lean = ["large-grammars"] [dependencies] hotpath.workspace = true -serde = { version = "1", features = ["derive"] } +serde = { version = "1", features = ["derive", "rc"] } tracedecay-domain = { path = "../tracedecay-domain", version = "0.1.0" } tracedecay-large-treesitters = { package = "tokensave-large-treesitters", version = "0.5.0", optional = true } tracedecay-medium-treesitters = { package = "tokensave-medium-treesitters", version = "0.2.0", optional = true } diff --git a/crates/tracedecay-code-extraction/src/clone_body.rs b/crates/tracedecay-code-extraction/src/clone_body.rs index c579363da9..29bb91be05 100644 --- a/crates/tracedecay-code-extraction/src/clone_body.rs +++ b/crates/tracedecay-code-extraction/src/clone_body.rs @@ -1,4 +1,5 @@ use std::collections::HashMap; +use std::sync::Arc; use serde::{Deserialize, Serialize}; use tracedecay_domain::{NodeKind, SourceSpan}; @@ -71,11 +72,15 @@ pub struct ExtractedCloneBodyV1 { pub eligibility: CloneBodyEligibilityV1, pub tokenization_status: CloneBodyTokenizationStatusV1, pub tokenization_issues: Vec, - pub conservative_tokens: Vec, + /// Shared with every payload built from this body: a token stream is + /// read, hashed, and persisted, never edited, and copying it per body + /// (one `String` per token) was the dominant allocation of the index + /// workers. + pub conservative_tokens: Arc<[ConservativeCloneTokenV1]>, pub rename_normalization_revision: Option, pub rename_status: CloneBodyRenameStatusV1, pub rename_issues: Vec, - pub rename_tokens: Option>, + pub rename_tokens: Option>, } impl ExtractedCloneBodyV1 { @@ -186,7 +191,7 @@ fn extract_clone_body( } struct ConservativeFields { - tokens: Vec, + tokens: Arc<[ConservativeCloneTokenV1]>, issues: Vec, token_count: u32, status: CloneBodyTokenizationStatusV1, @@ -234,7 +239,7 @@ fn conservative_fields( CloneBodyEligibilityV1::Eligible }; ConservativeFields { - tokens: emitter.tokens, + tokens: emitter.tokens.into(), issues: emitter.issues, token_count: emitter.token_count, status: tokenization_status, @@ -246,7 +251,7 @@ struct RenameFields { revision: Option, status: CloneBodyRenameStatusV1, issues: Vec, - tokens: Option>, + tokens: Option>, } fn rename_fields(syntax: CallableSyntax<'_>, source: &str, language: &str) -> RenameFields { @@ -269,7 +274,7 @@ fn rename_token_stream( source: &str, language: &str, replacements: &HashMap<(usize, usize), String>, -) -> Vec { +) -> Arc<[ConservativeCloneTokenV1]> { let mut emitter = TokenEmitter { source: source.as_bytes(), language, @@ -279,7 +284,7 @@ fn rename_token_stream( token_count: 0, }; emitter.emit(body); - emitter.tokens + emitter.tokens.into() } struct TokenEmitter<'a> { diff --git a/crates/tracedecay-code-extraction/tests/main/clone_body_tokens.rs b/crates/tracedecay-code-extraction/tests/main/clone_body_tokens.rs index 8152da06bc..081da3429f 100644 --- a/crates/tracedecay-code-extraction/tests/main/clone_body_tokens.rs +++ b/crates/tracedecay-code-extraction/tests/main/clone_body_tokens.rs @@ -26,7 +26,7 @@ fn tokens( "{:?}", artifact.result.nodes ); - artifact.clone_bodies[0].conservative_tokens.clone() + artifact.clone_bodies[0].conservative_tokens.to_vec() } #[test] diff --git a/crates/tracedecay-code-index/src/clones.rs b/crates/tracedecay-code-index/src/clones.rs index f40a881199..a1f94c8c53 100644 --- a/crates/tracedecay-code-index/src/clones.rs +++ b/crates/tracedecay-code-index/src/clones.rs @@ -120,12 +120,12 @@ pub struct CloneBodyPayloadV1 { pub token_count: u32, pub conservative_normalization_revision: u16, pub conservative_digest: ManifestDigest, - pub conservative_tokens: Vec, + pub conservative_tokens: Arc<[ConservativeCloneTokenV1]>, pub tokenization_status: CloneBodyTokenizationStatusV1, pub tokenization_issues: Vec, pub rename_normalization_revision: Option, pub rename_digest: Option, - pub rename_tokens: Option>, + pub rename_tokens: Option>, pub rename_coverage: CloneBodyRenameStatusV1, pub rename_issues: Vec, } @@ -432,7 +432,7 @@ impl CloneBodyPayloadV1 { token_count: body.non_trivia_token_count, conservative_normalization_revision: body.normalization_revision, conservative_digest: digests.conservative, - conservative_tokens: body.conservative_tokens.clone(), + conservative_tokens: Arc::clone(&body.conservative_tokens), tokenization_status: body.tokenization_status, tokenization_issues: body.tokenization_issues.clone(), rename_normalization_revision: body.rename_normalization_revision, @@ -668,7 +668,7 @@ pub fn verify_exact_clone_payload( if payload.conservative_normalization_revision == key.normalization_revision && payload.conservative_digest == key.digest => { - Some(&payload.conservative_tokens) + Some(&payload.conservative_tokens[..]) } CloneNormalizationClassV1::Rename if payload.rename_normalization_revision == Some(key.normalization_revision) @@ -1304,10 +1304,12 @@ mod fingerprint_tests { .validate() .expect("extracted tokens match their digests"); let mut colliding = payload; - colliding.conservative_tokens[0] = ConservativeCloneTokenV1::Syntax { + let mut tokens = colliding.conservative_tokens.to_vec(); + tokens[0] = ConservativeCloneTokenV1::Syntax { syntax_kind: "identifier".to_owned(), text: "colliding-but-different".to_owned(), }; + colliding.conservative_tokens = tokens.into(); assert_eq!( colliding.validate(), Err("clone payload digests do not match their canonical tokens".to_owned()) diff --git a/crates/tracedecay-privacy/src/rules.rs b/crates/tracedecay-privacy/src/rules.rs index 0e131c990b..00185ebe61 100644 --- a/crates/tracedecay-privacy/src/rules.rs +++ b/crates/tracedecay-privacy/src/rules.rs @@ -657,6 +657,12 @@ fn compile_regex( /// Expanding `\w` to its RE2 meaning fixes the semantics and the size at once /// — every rule in the catalogue then compiles under the default limit, with /// no memory headroom bought and no rule dropped. +////// * **`\b` / `\B`.** RE2's word boundary is ASCII. Rust's is Unicode-aware, +/// and a Unicode boundary is the one construct the lazy DFA gives up on the +/// moment the haystack holds a non-ASCII byte: every file with an em-dash or +/// an emoji in a comment was then scanned by the PikeVM, the slowest engine, +/// once per rule. `(?-u:\b)` is both the upstream meaning and a DFA-eligible +/// pattern. /// /// `\W`, `\D` and `\S` would need the same treatment but appear nowhere in the /// catalogue; a refresh that introduces one is caught by the compile test, @@ -665,7 +671,11 @@ fn compile_regex( /// Character classes are tracked because `\w` expands differently inside one: /// `[\w-]` has to become `[0-9A-Za-z_-]`, never a nested class. fn re2_compatible_regex(pattern: &str) -> Cow<'_, str> { - if !pattern.contains('{') && !pattern.contains(r"\w") { + if !pattern.contains('{') + && !pattern.contains(r"\w") + && !pattern.contains(r"\b") + && !pattern.contains(r"\B") + { return Cow::Borrowed(pattern); } let bytes = pattern.as_bytes(); @@ -683,6 +693,15 @@ fn re2_compatible_regex(pattern: &str) -> Cow<'_, str> { index += 2; continue; } + if !in_class && matches!(bytes[index + 1], b'b' | b'B') { + rewritten.push_str(if bytes[index + 1] == b'b' { + "(?-u:\\b)" + } else { + "(?-u:\\B)" + }); + index += 2; + continue; + } // Any other escape pair is copied whole: its second byte is never // structural, so it must not be re-examined. let end = next_boundary(pattern, index + 1); @@ -1374,10 +1393,15 @@ mod tests { /// meaning both inside and outside a character class. #[test] fn re2_translation_preserves_meaning() { - for untouched in [r"[A-Z]{16}", r"sk-[A-Za-z0-9_-]{20,}", r"\bplain\b"] { + for untouched in [r"[A-Z]{16}", r"sk-[A-Za-z0-9_-]{20,}", r"plain"] { assert_eq!(re2_compatible_regex(untouched), untouched); } + // Word boundaries take RE2's ASCII meaning; inside a class `\b` is a + // backspace and is left alone. + assert_eq!(re2_compatible_regex(r"\bplain\B"), r"(?-u:\b)plain(?-u:\B)"); + assert_eq!(re2_compatible_regex(r"[\b]"), r"[\b]"); + assert_eq!( re2_compatible_regex(r"^\$(?:\d+|{\d+})$"), r"^\$(?:\d+|\{\d+})$"