From 1f67b3cb41ddc4f518c5387bc5340f4f2ffe6518 Mon Sep 17 00:00:00 2001 From: Hanwen Cheng Date: Sun, 12 Jul 2026 14:01:11 +0800 Subject: [PATCH] Modernize dependencies and Unicode support --- .github/dependabot.yml | 7 + .github/workflows/ci.yml | 21 +++ Cargo.toml | 7 +- README.md | 18 ++- benches/benchmark.rs | 35 +++-- src/lib.rs | 287 ++++++++++++++++++++------------------- tests/tests.rs | 59 ++++++-- 7 files changed, 260 insertions(+), 174 deletions(-) create mode 100644 .github/dependabot.yml create mode 100644 .github/workflows/ci.yml diff --git a/.github/dependabot.yml b/.github/dependabot.yml new file mode 100644 index 0000000..801662f --- /dev/null +++ b/.github/dependabot.yml @@ -0,0 +1,7 @@ +version: 2 +updates: + - package-ecosystem: cargo + directory: / + schedule: + interval: weekly + open-pull-requests-limit: 5 diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..4b0cb4c --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,21 @@ +name: CI + +on: + push: + pull_request: + +permissions: + contents: read + +jobs: + test: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v6 + - uses: dtolnay/rust-toolchain@stable + with: + components: clippy, rustfmt + - run: cargo fmt --all -- --check + - run: cargo clippy --all-targets --all-features -- -D warnings + - run: cargo test --all-features + - run: cargo bench --no-run diff --git a/Cargo.toml b/Cargo.toml index b5a8366..a6d9170 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,9 +1,10 @@ [package] name = "common_substrings" -description = "Finding all common strings" +description = "Find common substrings across a collection of strings" version = "1.0.0" authors = ["hanwencheng "] -edition = "2018" +edition = "2024" +rust-version = "1.86" license = "Apache-2.0" readme = "README.md" homepage = "https://github.com/hanwencheng/common_substrings_rust" @@ -12,7 +13,7 @@ repository = "https://github.com/hanwencheng/common_substrings_rust" [dependencies] [dev-dependencies] -criterion = "0.3" +criterion = "0.8.2" [[bench]] name = "benchmark" diff --git a/README.md b/README.md index ff0522c..9c83e44 100644 --- a/README.md +++ b/README.md @@ -1,14 +1,14 @@ # Find all common substrings -![versom](https://img.shields.io/crates/v/common_substrings) +![crates.io version](https://img.shields.io/crates/v/common_substrings) A method for finding all common strings. Check it on [Crates.io](https://crates.io/crates/common_substrings). -The algorithms uses a two dimension trie to get all the fragment. The vertical one is the standard suffix trie, but all the node of the last word in each suffix is linked, which I call them virtually horizontally linked. +The algorithm uses a two-dimensional trie to find the fragments. The vertical dimension is a standard suffix trie; nodes at the end of each suffix are also linked horizontally. ## Usage -Use the function `get_substrings` to get all the common strings in the strings list, +Use `get_substrings` to get the common substrings in a list of strings. ### Example ```rust @@ -17,7 +17,7 @@ let input_strings = vec!["java", "javascript", "typescript", "coffeescript", "co let result_substrings = get_substrings(input_strings, 2, 3); ``` -which gives the result list of +This produces results such as: ```shell Substring(sources: {2, 3}, name: escript, weight: 14) Substring(sources: {1, 0}, name: java, weight: 8) @@ -26,9 +26,13 @@ Substring(sources: {4, 3}, name: coffee, weight: 12) ### Arguments -* `input` - The target input string vector. -* `min_occurrences` The minimal occurrence of the captured common substrings. -* `min_length` The minimal length of the captured common substrings. +* `input` — The input strings. +* `min_occurrences` — The minimum number of input strings containing a result. +* `min_length` — The minimum result length, measured in Unicode scalar values. + +Both thresholds must be positive. When `min_occurrences` exceeds the input size, the result is empty. + +The minimum supported Rust version is 1.86. ## Algorithm diff --git a/benches/benchmark.rs b/benches/benchmark.rs index 7e71b51..f7d0225 100644 --- a/benches/benchmark.rs +++ b/benches/benchmark.rs @@ -1,18 +1,27 @@ -use criterion::{criterion_group, criterion_main, Criterion}; use common_substrings::get_substrings; +use criterion::{Criterion, criterion_group, criterion_main}; +use std::hint::black_box; -fn criterion_benchmark(c: &mut Criterion) { - c.bench_function("get substring", |b| b.iter(|| get_substrings(vec![ - "java", - "offe", - "coffescript", - "typescript", - "typed", - "javacoffie", - "fessss", - "fe", - ], 2, 3))); +fn criterion_benchmark(criterion: &mut Criterion) { + criterion.bench_function("get substring", |bencher| { + bencher.iter(|| { + get_substrings( + black_box(vec![ + "java", + "offe", + "coffescript", + "typescript", + "typed", + "javacoffie", + "fessss", + "fe", + ]), + 2, + 3, + ) + }) + }); } criterion_group!(benches, criterion_benchmark); -criterion_main!(benches); \ No newline at end of file +criterion_main!(benches); diff --git a/src/lib.rs b/src/lib.rs index fe62745..965a865 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -11,21 +11,22 @@ struct Node { horizontal: HashMap>>, label: String, } -/// The result common substring -#[derive(Clone)] + +/// A common substring and the input strings that contain it. +#[derive(Clone, Debug, Eq, PartialEq)] pub struct Substring { - /// sources indicates where it comes from + /// Zero-based indexes of the input strings containing this substring. pub sources: HashSet, - /// the name of the substring + /// The matching substring. pub name: String, - /// the weight = (sources number) * (name chars), which could used to sort the captured substrings. + /// The number of source strings multiplied by the character count. pub weight: usize, } impl Display for Substring { - fn fmt(&self, f: &mut Formatter<'_>) -> fmt::Result { + fn fmt(&self, formatter: &mut Formatter<'_>) -> fmt::Result { write!( - f, + formatter, "Substring(sources: {:?}, name: {}, weight: {})", self.sources, self.name, self.weight ) @@ -33,232 +34,244 @@ impl Display for Substring { } impl Node { - pub fn add_source(&mut self, source_index: usize) { - self.sources.insert(source_index); - } - - pub fn new(label: &str) -> Node { - Node { + fn new(label: &str) -> Self { + Self { sources: HashSet::new(), listed: false, nodes: HashMap::new(), horizontal: HashMap::new(), - label: String::from(label), + label: label.to_owned(), } } - - pub fn set_listing(&mut self, shoud_listing: bool) { - self.listed = shoud_listing; - } } impl Debug for Node { - fn fmt(&self, f: &mut Formatter<'_>) -> fmt::Result { + fn fmt(&self, formatter: &mut Formatter<'_>) -> fmt::Result { write!( - f, - "(sources: {:?}, listed: {:?}), label: {}, horizontal: {:?} with Map \n {:?}", + formatter, + "(sources: {:?}, listed: {:?}), label: {}, horizontal: {:?} with Map\n {:?}", self.sources, self.listed, self.label, self.horizontal, self.nodes ) } } -/// The function used to get all the common strings in the strings list, -/// # Arguments +/// Finds common substrings across the supplied strings. /// -/// * `input` - The target input string vector. -/// * `min_occurrences` The minimal occurrence of the captured common substrings. -/// * `min_leng` The minial length of the captured common substrings. +/// `min_occurrences` and `min_length` must both be positive. Length is measured +/// in Unicode scalar values rather than bytes. /// -/// # Example +/// # Examples /// /// ``` -/// use common_substrings::get_substrings; -/// let input_strings = vec!["java", "javascript", "typescript", "coffeescript", "coffee"]; -/// let result_substrings = get_substrings(input_strings, 2, 3); -/// ``` -/// which gives the result list of -/// ```shell -/// Substring(sources: {2, 3}, name: escript, weight: 14) -/// Substring(sources: {1, 0}, name: java, weight: 8) -/// Substring(sources: {4, 3}, name: coffee, weight: 12) +/// use common_substrings::get_substrings; +/// +/// let input_strings = vec!["java", "javascript", "typescript", "coffeescript", "coffee"]; +/// let result_substrings = get_substrings(input_strings, 2, 3); +/// assert_eq!(result_substrings.len(), 3); /// ``` +/// +/// # Panics +/// +/// Panics when either threshold is zero. pub fn get_substrings( input: Vec<&str>, min_occurrences: usize, min_length: usize, ) -> Vec { - let trie = Rc::new(RefCell::new(Node::new(""))); + assert!(min_occurrences > 0, "min_occurrences must be positive"); + assert!(min_length > 0, "min_length must be positive"); + + let trie_root = Rc::new(RefCell::new(Node::new(""))); let horizontal_trie_root = Rc::new(RefCell::new(Node::new(""))); - let mut result_vector = Vec::new(); + let mut result_substrings = Vec::new(); + for (word_index, word) in input.iter().enumerate() { - build_suffices(&word, &trie, word_index, &horizontal_trie_root, min_length); + build_suffixes( + word, + &trie_root, + word_index, + &horizontal_trie_root, + min_length, + ); } - accumulate_vertical(&trie, min_occurrences, min_length); + accumulate_vertical(&trie_root, min_occurrences, min_length); accumulate_horizontal(&horizontal_trie_root, min_occurrences); - list_sources(trie, &mut result_vector); - result_vector + list_sources(trie_root, &mut result_substrings); + result_substrings } -fn list_sources(trie_ref: Rc>, result_vec: &mut Vec) { - trie_ref - .borrow_mut() - .nodes - .iter_mut() - .for_each(|(_child_node_label, child_node_ref)| { - list_sources(child_node_ref.clone(), result_vec); +fn list_sources(trie_reference: Rc>, result_substrings: &mut Vec) { + let child_references: Vec<_> = trie_reference.borrow().nodes.values().cloned().collect(); - if child_node_ref.borrow().listed == true { - result_vec.push(Substring { - sources: child_node_ref.borrow().sources.clone(), - name: String::from(child_node_ref.borrow().label.clone()), - weight: child_node_ref.borrow().label.chars().count() - * child_node_ref.borrow().sources.len(), - }) - } - }) + for child_reference in child_references { + list_sources(child_reference.clone(), result_substrings); + + let child = child_reference.borrow(); + if child.listed { + result_substrings.push(Substring { + sources: child.sources.clone(), + name: child.label.clone(), + weight: child.label.chars().count() * child.sources.len(), + }); + } + } } fn accumulate_vertical( - trie_ref: &Rc>, + trie_reference: &Rc>, min_occurrences: usize, min_length: usize, ) -> HashSet { - let accumulated = trie_ref.borrow_mut().nodes.iter_mut().fold( - HashSet::new(), - |acc: HashSet, (_child_node_label, child_node_pointer)| { - let child_sources = - accumulate_vertical(child_node_pointer, min_occurrences, min_length); - if child_node_pointer.borrow().label.len() <= min_length { - acc - } else { - acc.union(&child_sources).cloned().collect() - } - }, - ); + let child_references: Vec<_> = trie_reference.borrow().nodes.values().cloned().collect(); + let accumulated_sources = + child_references + .iter() + .fold(HashSet::new(), |accumulated_sources, child_reference| { + let child_sources = + accumulate_vertical(child_reference, min_occurrences, min_length); + if child_reference.borrow().label.chars().count() <= min_length { + accumulated_sources + } else { + accumulated_sources.union(&child_sources).copied().collect() + } + }); - if trie_ref.borrow().label.len() < min_length { - return accumulated; + if trie_reference.borrow().label.chars().count() < min_length { + return accumulated_sources; } - let remained_occurrence: HashSet = trie_ref + let remaining_sources: HashSet = trie_reference .borrow() .sources - .difference(&accumulated) - .cloned() + .difference(&accumulated_sources) + .copied() .collect(); - if remained_occurrence.len() >= min_occurrences { - trie_ref.borrow_mut().set_listing(true); - trie_ref.borrow().sources.clone() + if remaining_sources.len() >= min_occurrences { + trie_reference.borrow_mut().listed = true; + trie_reference.borrow().sources.clone() } else { - accumulated + accumulated_sources } } fn accumulate_horizontal( - horizontal_parent_node_ref: &Rc>, + horizontal_parent_reference: &Rc>, min_occurrences: usize, ) -> HashSet { - let accumulated = horizontal_parent_node_ref.borrow().horizontal.iter().fold( - HashSet::new(), - |acc: HashSet, (_child_node_label, child_node_pointer)| { - let child_sources = accumulate_horizontal(child_node_pointer, min_occurrences); - acc.union(&child_sources).cloned().collect() - }, - ); + let child_references: Vec<_> = horizontal_parent_reference + .borrow() + .horizontal + .values() + .cloned() + .collect(); + let accumulated_sources = + child_references + .iter() + .fold(HashSet::new(), |accumulated_sources, child_reference| { + let child_sources = accumulate_horizontal(child_reference, min_occurrences); + accumulated_sources.union(&child_sources).copied().collect() + }); - let remained_occurrence: HashSet = horizontal_parent_node_ref + let remaining_sources: HashSet = horizontal_parent_reference .borrow() .sources - .difference(&accumulated) - .cloned() + .difference(&accumulated_sources) + .copied() .collect(); - if remained_occurrence.len() >= min_occurrences { - horizontal_parent_node_ref.borrow().sources.clone() + if remaining_sources.len() >= min_occurrences { + horizontal_parent_reference.borrow().sources.clone() } else { - horizontal_parent_node_ref.borrow_mut().set_listing(false); - accumulated + horizontal_parent_reference.borrow_mut().listed = false; + accumulated_sources } } -// for each word add all the suffices into the trie -fn build_suffices( +fn build_suffixes( word: &str, - trie: &Rc>, + trie_root: &Rc>, word_index: usize, horizontal_root: &Rc>, min_length: usize, ) { + let word_characters: Vec = word.chars().collect(); let mut last_suffix_leaves: LinkedList>> = LinkedList::new(); - if word.len() >= min_length { - for x in 0..(word.len() - min_length + 1) { + + if word_characters.len() >= min_length { + for first_character_index in 0..=(word_characters.len() - min_length) { build_suffix( - &word, - trie, + &word_characters, + trie_root, word_index, &mut last_suffix_leaves, - x, + first_character_index, horizontal_root, min_length, ); } } - assert!(last_suffix_leaves.len() == 0); + + assert!(last_suffix_leaves.is_empty()); } -// add one suffix of a word into trie, iterate each char of the suffix fn build_suffix( - word: &str, - trie: &Rc>, + word_characters: &[char], + trie_root: &Rc>, word_index: usize, last_suffix_leaves: &mut LinkedList>>, - first_char_index: usize, + first_character_index: usize, horizontal_root: &Rc>, min_length: usize, ) { - let suffix = &word[first_char_index..]; - suffix.chars().enumerate().fold( - trie.clone(), - |pointer: Rc>, (current_char_index, char)| { - let current_branch_length = current_char_index + 1; - let char_label = char.to_string(); - let current_branch_label = &suffix[0..current_branch_length]; - let insert_node = Node::new(current_branch_label); - let mut pointer_node = pointer.borrow_mut(); - let current_node_ref = pointer_node + let suffix = &word_characters[first_character_index..]; + + suffix.iter().enumerate().fold( + trie_root.clone(), + |current_reference, (current_character_index, character)| { + let current_branch_length = current_character_index + 1; + let character_label = character.to_string(); + let current_branch_label: String = suffix[..current_branch_length].iter().collect(); + let new_node = Node::new(¤t_branch_label); + let mut current_node = current_reference.borrow_mut(); + let current_node_reference = current_node .nodes - .entry(char_label) - .or_insert(Rc::new(RefCell::new(insert_node))); - //if branch not reach the list requirement + .entry(character_label) + .or_insert_with(|| Rc::new(RefCell::new(new_node))); + if current_branch_length < min_length { - return current_node_ref.clone(); + return current_node_reference.clone(); } - //consume last vertical - if first_char_index >= 1 { - // so that the last_suffix_leaves is not empty - let suffix_label_in_last_branch = - &word[(first_char_index - 1)..(first_char_index + current_branch_length)]; - let last_ptr = last_suffix_leaves.pop_front().unwrap(); - current_node_ref + if first_character_index >= 1 { + let last_branch_label: String = word_characters + [first_character_index - 1..first_character_index + current_branch_length] + .iter() + .collect(); + let last_reference = last_suffix_leaves + .pop_front() + .expect("suffix trie invariant violated"); + current_node_reference .borrow_mut() .horizontal - .insert(String::from(suffix_label_in_last_branch), last_ptr); + .insert(last_branch_label, last_reference); } - current_node_ref.borrow_mut().add_source(word_index); - let vertical_pointer = current_node_ref.clone(); + current_node_reference + .borrow_mut() + .sources + .insert(word_index); + let vertical_reference = current_node_reference.clone(); if current_branch_length > min_length { - last_suffix_leaves.push_back(vertical_pointer); - } else if current_branch_length == min_length { - // if it is the last min length suffix of the whole word, then add it to root + last_suffix_leaves.push_back(vertical_reference); + } else { horizontal_root .borrow_mut() .horizontal - .insert(String::from(current_branch_label), vertical_pointer); + .insert(current_branch_label, vertical_reference); } - current_node_ref.clone() + + current_node_reference.clone() }, ); } diff --git a/tests/tests.rs b/tests/tests.rs index 2de89bf..528c5fa 100644 --- a/tests/tests.rs +++ b/tests/tests.rs @@ -1,17 +1,48 @@ -#[cfg(test)] -mod tests { - use common_substrings::{get_substrings}; +use common_substrings::get_substrings; - #[test] - fn it_works() { - let test_samples = vec![ - "science", "typescript", "crisis", "kept", "javascript", "java" - ]; +#[test] +fn finds_expected_common_substrings() { + let test_samples = vec![ + "science", + "typescript", + "crisis", + "kept", + "javascript", + "java", + ]; - let result_substrings = get_substrings(test_samples, 2, 4); - result_substrings.iter().for_each(|it| { - println!("{}", it); - }); - assert_eq!(result_substrings.len(), 2); - } + let result_substrings = get_substrings(test_samples, 2, 4); + let mut names: Vec<_> = result_substrings + .iter() + .map(|substring| substring.name.as_str()) + .collect(); + names.sort_unstable(); + assert_eq!(names, vec!["java", "script"]); +} + +#[test] +fn supports_unicode_without_splitting_code_points() { + let result_substrings = get_substrings(vec!["go😀team", "hi😀team"], 2, 5); + + assert_eq!(result_substrings.len(), 1); + assert_eq!(result_substrings[0].name, "😀team"); + assert_eq!(result_substrings[0].weight, 10); +} + +#[test] +fn returns_no_result_when_min_occurrences_exceeds_input_size() { + let result_substrings = get_substrings(vec!["javascript", "typescript"], 3, 3); + assert!(result_substrings.is_empty()); +} + +#[test] +#[should_panic(expected = "min_length must be positive")] +fn rejects_zero_min_length() { + get_substrings(vec!["one", "two"], 2, 0); +} + +#[test] +#[should_panic(expected = "min_occurrences must be positive")] +fn rejects_zero_min_occurrences() { + get_substrings(vec!["one", "two"], 0, 2); }