Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion crates/tracedecay-code-extraction/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -80,7 +80,7 @@ lang-lean = ["large-grammars"]

[dependencies]
hotpath.workspace = true
serde = { version = "1", features = ["derive"] }
serde = { version = "1", features = ["derive", "rc"] }
tracedecay-domain = { path = "../tracedecay-domain", version = "0.1.0" }
tracedecay-large-treesitters = { package = "tokensave-large-treesitters", version = "0.5.0", optional = true }
tracedecay-medium-treesitters = { package = "tokensave-medium-treesitters", version = "0.2.0", optional = true }
Expand Down
19 changes: 12 additions & 7 deletions crates/tracedecay-code-extraction/src/clone_body.rs
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
use std::collections::HashMap;
use std::sync::Arc;

use serde::{Deserialize, Serialize};
use tracedecay_domain::{NodeKind, SourceSpan};
Expand Down Expand Up @@ -71,11 +72,15 @@ pub struct ExtractedCloneBodyV1 {
pub eligibility: CloneBodyEligibilityV1,
pub tokenization_status: CloneBodyTokenizationStatusV1,
pub tokenization_issues: Vec<CloneBodyTokenizationIssueV1>,
pub conservative_tokens: Vec<ConservativeCloneTokenV1>,
/// Shared with every payload built from this body: a token stream is
/// read, hashed, and persisted, never edited, and copying it per body
/// (one `String` per token) was the dominant allocation of the index
/// workers.
pub conservative_tokens: Arc<[ConservativeCloneTokenV1]>,
pub rename_normalization_revision: Option<u16>,
pub rename_status: CloneBodyRenameStatusV1,
pub rename_issues: Vec<CloneBodyRenameIssueV1>,
pub rename_tokens: Option<Vec<ConservativeCloneTokenV1>>,
pub rename_tokens: Option<Arc<[ConservativeCloneTokenV1]>>,
}

impl ExtractedCloneBodyV1 {
Expand Down Expand Up @@ -186,7 +191,7 @@ fn extract_clone_body(
}

struct ConservativeFields {
tokens: Vec<ConservativeCloneTokenV1>,
tokens: Arc<[ConservativeCloneTokenV1]>,
issues: Vec<CloneBodyTokenizationIssueV1>,
token_count: u32,
status: CloneBodyTokenizationStatusV1,
Expand Down Expand Up @@ -234,7 +239,7 @@ fn conservative_fields(
CloneBodyEligibilityV1::Eligible
};
ConservativeFields {
tokens: emitter.tokens,
tokens: emitter.tokens.into(),
issues: emitter.issues,
token_count: emitter.token_count,
status: tokenization_status,
Expand All @@ -246,7 +251,7 @@ struct RenameFields {
revision: Option<u16>,
status: CloneBodyRenameStatusV1,
issues: Vec<CloneBodyRenameIssueV1>,
tokens: Option<Vec<ConservativeCloneTokenV1>>,
tokens: Option<Arc<[ConservativeCloneTokenV1]>>,
}

fn rename_fields(syntax: CallableSyntax<'_>, source: &str, language: &str) -> RenameFields {
Expand All @@ -269,7 +274,7 @@ fn rename_token_stream(
source: &str,
language: &str,
replacements: &HashMap<(usize, usize), String>,
) -> Vec<ConservativeCloneTokenV1> {
) -> Arc<[ConservativeCloneTokenV1]> {
let mut emitter = TokenEmitter {
source: source.as_bytes(),
language,
Expand All @@ -279,7 +284,7 @@ fn rename_token_stream(
token_count: 0,
};
emitter.emit(body);
emitter.tokens
emitter.tokens.into()
}

struct TokenEmitter<'a> {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -26,7 +26,7 @@ fn tokens(
"{:?}",
artifact.result.nodes
);
artifact.clone_bodies[0].conservative_tokens.clone()
artifact.clone_bodies[0].conservative_tokens.to_vec()
}

#[test]
Expand Down
12 changes: 7 additions & 5 deletions crates/tracedecay-code-index/src/clones.rs
Original file line number Diff line number Diff line change
Expand Up @@ -120,12 +120,12 @@ pub struct CloneBodyPayloadV1 {
pub token_count: u32,
pub conservative_normalization_revision: u16,
pub conservative_digest: ManifestDigest,
pub conservative_tokens: Vec<ConservativeCloneTokenV1>,
pub conservative_tokens: Arc<[ConservativeCloneTokenV1]>,
pub tokenization_status: CloneBodyTokenizationStatusV1,
pub tokenization_issues: Vec<CloneBodyTokenizationIssueV1>,
pub rename_normalization_revision: Option<u16>,
pub rename_digest: Option<ManifestDigest>,
pub rename_tokens: Option<Vec<ConservativeCloneTokenV1>>,
pub rename_tokens: Option<Arc<[ConservativeCloneTokenV1]>>,
pub rename_coverage: CloneBodyRenameStatusV1,
pub rename_issues: Vec<CloneBodyRenameIssueV1>,
}
Expand Down Expand Up @@ -432,7 +432,7 @@ impl CloneBodyPayloadV1 {
token_count: body.non_trivia_token_count,
conservative_normalization_revision: body.normalization_revision,
conservative_digest: digests.conservative,
conservative_tokens: body.conservative_tokens.clone(),
conservative_tokens: Arc::clone(&body.conservative_tokens),
tokenization_status: body.tokenization_status,
tokenization_issues: body.tokenization_issues.clone(),
rename_normalization_revision: body.rename_normalization_revision,
Expand Down Expand Up @@ -668,7 +668,7 @@ pub fn verify_exact_clone_payload(
if payload.conservative_normalization_revision == key.normalization_revision
&& payload.conservative_digest == key.digest =>
{
Some(&payload.conservative_tokens)
Some(&payload.conservative_tokens[..])
}
CloneNormalizationClassV1::Rename
if payload.rename_normalization_revision == Some(key.normalization_revision)
Expand Down Expand Up @@ -1304,10 +1304,12 @@ mod fingerprint_tests {
.validate()
.expect("extracted tokens match their digests");
let mut colliding = payload;
colliding.conservative_tokens[0] = ConservativeCloneTokenV1::Syntax {
let mut tokens = colliding.conservative_tokens.to_vec();
tokens[0] = ConservativeCloneTokenV1::Syntax {
syntax_kind: "identifier".to_owned(),
text: "colliding-but-different".to_owned(),
};
colliding.conservative_tokens = tokens.into();
assert_eq!(
colliding.validate(),
Err("clone payload digests do not match their canonical tokens".to_owned())
Expand Down
28 changes: 26 additions & 2 deletions crates/tracedecay-privacy/src/rules.rs
Original file line number Diff line number Diff line change
Expand Up @@ -657,6 +657,12 @@ fn compile_regex(
/// Expanding `\w` to its RE2 meaning fixes the semantics and the size at once
/// — every rule in the catalogue then compiles under the default limit, with
/// no memory headroom bought and no rule dropped.
////// * **`\b` / `\B`.** RE2's word boundary is ASCII. Rust's is Unicode-aware,
/// and a Unicode boundary is the one construct the lazy DFA gives up on the
/// moment the haystack holds a non-ASCII byte: every file with an em-dash or
/// an emoji in a comment was then scanned by the PikeVM, the slowest engine,
/// once per rule. `(?-u:\b)` is both the upstream meaning and a DFA-eligible
/// pattern.
///
/// `\W`, `\D` and `\S` would need the same treatment but appear nowhere in the
/// catalogue; a refresh that introduces one is caught by the compile test,
Expand All @@ -665,7 +671,11 @@ fn compile_regex(
/// Character classes are tracked because `\w` expands differently inside one:
/// `[\w-]` has to become `[0-9A-Za-z_-]`, never a nested class.
fn re2_compatible_regex(pattern: &str) -> Cow<'_, str> {
if !pattern.contains('{') && !pattern.contains(r"\w") {
if !pattern.contains('{')
&& !pattern.contains(r"\w")
&& !pattern.contains(r"\b")
&& !pattern.contains(r"\B")
{
return Cow::Borrowed(pattern);
}
let bytes = pattern.as_bytes();
Expand All @@ -683,6 +693,15 @@ fn re2_compatible_regex(pattern: &str) -> Cow<'_, str> {
index += 2;
continue;
}
if !in_class && matches!(bytes[index + 1], b'b' | b'B') {
rewritten.push_str(if bytes[index + 1] == b'b' {
"(?-u:\\b)"
} else {
"(?-u:\\B)"
});
index += 2;
continue;
}
// Any other escape pair is copied whole: its second byte is never
// structural, so it must not be re-examined.
let end = next_boundary(pattern, index + 1);
Expand Down Expand Up @@ -1374,10 +1393,15 @@ mod tests {
/// meaning both inside and outside a character class.
#[test]
fn re2_translation_preserves_meaning() {
for untouched in [r"[A-Z]{16}", r"sk-[A-Za-z0-9_-]{20,}", r"\bplain\b"] {
for untouched in [r"[A-Z]{16}", r"sk-[A-Za-z0-9_-]{20,}", r"plain"] {
assert_eq!(re2_compatible_regex(untouched), untouched);
}

// Word boundaries take RE2's ASCII meaning; inside a class `\b` is a
// backspace and is left alone.
assert_eq!(re2_compatible_regex(r"\bplain\B"), r"(?-u:\b)plain(?-u:\B)");
assert_eq!(re2_compatible_regex(r"[\b]"), r"[\b]");

assert_eq!(
re2_compatible_regex(r"^\$(?:\d+|{\d+})$"),
r"^\$(?:\d+|\{\d+})$"
Expand Down
Loading