From db1e3a6cbddcaa52c82979343d851757106eeefd Mon Sep 17 00:00:00 2001 From: Zach Vorhies Date: Sun, 13 Sep 2026 01:30:22 -0700 Subject: [PATCH] feat(text): bound name similarity behind the facade --- .github/workflows/ci.yml | 1 + Cargo.lock | 1 + Cargo.toml | 2 + ci/check_compilation_boundary_dependencies.py | 1 + docs/text-similarity.md | 24 ++++++++++ src/lib.rs | 2 + src/text.rs | 45 +++++++++++++++++++ tests/text_similarity.rs | 42 +++++++++++++++++ 8 files changed, 118 insertions(+) create mode 100644 docs/text-similarity.md create mode 100644 src/text.rs create mode 100644 tests/text_similarity.rs diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 8e669078..d66ed5af 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -162,6 +162,7 @@ jobs: - fs-watch - hash-sha256 - secure-random + - text-similarity - event-stream - http-server - http-client diff --git a/Cargo.lock b/Cargo.lock index aee31187..6c11b14c 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2577,6 +2577,7 @@ dependencies = [ "serde", "serde_json", "sha2", + "strsim", "sysinfo", "tar", "tauri", diff --git a/Cargo.toml b/Cargo.toml index 9739b257..7059a786 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -36,6 +36,7 @@ required-features = ["tauri-webview-test-support"] default = [] # Bounded OS entropy, independent of crash/profiling facilities. secure-random = ["dep:getrandom"] +text-similarity = ["dep:strsim"] # Advisory locking and modification-time setting are implemented natively # per host (see `src/platform_linux/fs.rs`, `src/platform_macos/fs.rs`, # `src/platform_win/fs.rs`) rather than through a wrapper crate, matching the @@ -217,6 +218,7 @@ flate2 = { version = "=1.1.9", optional = true } framehop = { version = "=0.13.3", optional = true } futures-core = { version = "=0.3.34", optional = true } getrandom = { version = "=0.4.3", optional = true } +strsim = { version = "=0.11.1", optional = true } globset = { version = "=0.4.18", optional = true } interprocess = { version = "=2.4.3", optional = true } jwalk = { version = "=0.8.1", optional = true } diff --git a/ci/check_compilation_boundary_dependencies.py b/ci/check_compilation_boundary_dependencies.py index 3d037ecc..53820f72 100644 --- a/ci/check_compilation_boundary_dependencies.py +++ b/ci/check_compilation_boundary_dependencies.py @@ -12,6 +12,7 @@ import sys CASES = ( + ("text-similarity", "strsim"), ("wasm-sketch-host", "wasmtime"), ("ipc", "interprocess"), ("tokio-console", "console-subscriber"), diff --git a/docs/text-similarity.md b/docs/text-similarity.md new file mode 100644 index 00000000..4844cc2a --- /dev/null +++ b/docs/text-similarity.md @@ -0,0 +1,24 @@ +# Bounded name similarity + +Enable `text-similarity` and call `kernal_api::text::name_similarity(left, right)`. +The facade privately pins strsim 0.11.1 for Jaro-Winkler scoring. It returns a +finite `f64` from 0 to 1, or a facade-owned resource error. No backend types are +public. This feature does not activate GUI, HTTP or sketch-host facilities. + +The fixed limits are 4096 UTF-8 bytes per input and a product of Unicode scalar +counts no greater than 1,048,576. Byte length is checked first, then scalar work, +before backend scoring or allocation. Even equal strings must satisfy these +limits. Limits are not configurable, so there is no invalid-configuration state. +This bounds comparison work and scratch storage, not a wall-clock deadline. +The function is synchronous and needs no runtime or cancellation mechanism. + +Scoring is case-sensitive and does not normalize Unicode or combine grapheme +clusters. Two empty strings score 1; exactly one empty string scores 0. The +algorithm gives matching prefixes a boost. Applications keep case folding, +substring preference, candidate enumeration, tie tolerance and user prompts. +Oversized input must be reported explicitly; do not truncate or substitute a +different score after a resource error. + +Issue #184 tracks FastLED adoption and exact published-release verification. +Removing a direct application dependency does not itself remove transitive +compilation or demonstrate a build-speed improvement. diff --git a/src/lib.rs b/src/lib.rs index 80071185..95df412c 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -27,6 +27,8 @@ pub mod hash; /// Bounded, fallible operating-system entropy without token-format policy. #[cfg(feature = "secure-random")] pub mod random; +#[cfg(feature = "text-similarity")] +pub mod text; /// Bounded extraction into caller-exclusive staging directories. #[cfg(feature = "archive")] diff --git a/src/text.rs b/src/text.rs new file mode 100644 index 00000000..cb848f19 --- /dev/null +++ b/src/text.rs @@ -0,0 +1,45 @@ +//! Bounded text comparison without application-specific normalization or ranking. + +/// Maximum UTF-8 bytes in either argument to [`name_similarity`]. +pub const MAX_SIMILARITY_INPUT_BYTES: usize = 4096; +/// Maximum product of the two Unicode scalar counts. +pub const MAX_SIMILARITY_WORK: usize = 1_048_576; + +/// Resource rejection before invoking the private scoring implementation. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum SimilarityError { + InputTooLarge, + WorkLimitExceeded, +} + +impl std::fmt::Display for SimilarityError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(match self { + Self::InputTooLarge => "similarity input exceeds the byte limit", + Self::WorkLimitExceeded => "similarity comparison exceeds the work limit", + }) + } +} + +impl std::error::Error for SimilarityError {} + +/// Score two names using Jaro-Winkler similarity, from 0 (unrelated) to 1 (equal). +/// +/// Operates on Unicode scalar values, not bytes or grapheme clusters. It performs +/// no case folding or Unicode normalization. Both empty inputs score 1; exactly +/// one empty input scores 0. Callers own candidate selection and tie policy. +/// +/// Both byte limits and the scalar-count product limit are checked before the +/// backend allocates matching flags or scores, including for equal inputs. +/// This bounds quadratic comparison work and linear scratch storage, not elapsed +/// time. This synchronous CPU operation has no cancellation or runtime dependency. +pub fn name_similarity(left: &str, right: &str) -> Result { + if left.len() > MAX_SIMILARITY_INPUT_BYTES || right.len() > MAX_SIMILARITY_INPUT_BYTES { + return Err(SimilarityError::InputTooLarge); + } + // Byte limits also bound scalar counts, so this product cannot overflow. + if left.chars().count() * right.chars().count() > MAX_SIMILARITY_WORK { + return Err(SimilarityError::WorkLimitExceeded); + } + Ok(strsim::jaro_winkler(left, right)) +} diff --git a/tests/text_similarity.rs b/tests/text_similarity.rs new file mode 100644 index 00000000..cb6afd17 --- /dev/null +++ b/tests/text_similarity.rs @@ -0,0 +1,42 @@ +#![cfg(feature = "text-similarity")] + +use kernal_api::text::{name_similarity, SimilarityError}; + +#[test] +fn known_scores_and_unicode_semantics() { + assert!((name_similarity("MARTHA", "MARHTA").unwrap() - 0.9611111111111111).abs() < 1e-12); + assert_eq!(name_similarity("", "").unwrap(), 1.0); + assert_eq!(name_similarity("", "x").unwrap(), 0.0); + assert_eq!(name_similarity("é🦀", "é🦀").unwrap(), 1.0); + assert_eq!(name_similarity("abc", "XYZ").unwrap(), 0.0); + assert!(name_similarity("é", "e\u{301}").unwrap() < 1.0); +} + +#[test] +fn rejects_storage_and_comparison_work_before_scoring() { + assert_eq!( + name_similarity(&"x".repeat(4097), "x"), + Err(SimilarityError::InputTooLarge) + ); + assert_eq!( + name_similarity(&"x".repeat(1025), &"y".repeat(1024)), + Err(SimilarityError::WorkLimitExceeded) + ); + assert!(name_similarity(&"🦀".repeat(1024), "🦀").unwrap() > 0.0); + assert_eq!( + name_similarity(&"x".repeat(1024), &"x".repeat(1024)).unwrap(), + 1.0 + ); + assert_eq!(name_similarity(&"x".repeat(4096), "").unwrap(), 0.0); +} + +#[test] +fn scores_are_finite_normalized_and_symmetric_for_representative_names() { + for left in ["", "Blink", "blnk", "Fire2012", "é🦀", "e\u{301}"] { + for right in ["", "Blink", "blnk", "Fire2012", "é🦀", "e\u{301}"] { + let score = name_similarity(left, right).unwrap(); + assert!(score.is_finite() && (0.0..=1.0).contains(&score)); + assert!((score - name_similarity(right, left).unwrap()).abs() < 1e-12); + } + } +}