diff --git a/src/score.rs b/src/score.rs index cef37f33..71dbbc34 100644 --- a/src/score.rs +++ b/src/score.rs @@ -2,90 +2,9 @@ /// the query and the choice. /// /// It is modeled after https://github.com/felipesere/icepick.git + use std::cmp::max; - -// return (start, matched_len) -pub fn compute_match_length(choice: &str, query: &str) -> Option<(usize, usize)> { - if query.len() <= 0 { - return Some((0, 0)); - } - - let impossible_match = choice.len() + 1; - let mut matched_start = impossible_match; - let mut matched_end = impossible_match; - - let mut choice_chars = choice.chars().enumerate().peekable(); - let mut query_chars = query.chars().enumerate().peekable(); - - loop { - if query_chars.peek() == None { - return Some((matched_start, matched_end - matched_start+1)); - } - - if choice_chars.peek() == None { - return None; - } - - let &(idx_choice, c) = choice_chars.peek().unwrap(); - let &(_, q) = query_chars.peek().unwrap(); - - - if c == q { - if matched_start == impossible_match { matched_start = idx_choice; } - let _ = query_chars.next(); - } - - matched_end = idx_choice; - let _ = choice_chars.next(); - } -} - - -// The fuzzy search algorith is inspired by https://github.com/forrestthewoods/lib_fts -// -// # Why is new Fuzzy Match algorithm needed? -// 1. fzf use the query to compose regex for non-greedy match. When use for file/path match, it did -// not work out for the best(e.g. Camel case characters should weight more) -// 2. fzf will only show the `range` be matched, while I think it is not enought sometimes. with -// this fuzzy_match algorithm we are able to get all matched indics. -// -// # The complexity is O(n^2) -// An intuitive way of fuzzy match is to find all occurences of pattern characters in the target -// string. Then find out all the choice composites and evaluate the score for each. which will take -// O(n!) for each input string. Then with dynamic programing, we can reduce the time to O(n^3) with -// addition O(m*n) space; -// -// I think that O(n^3) may not be acceptable in the use case of fzf-rs. so I made some assumptions -// and reduce the complexity to O(n^2). Note that lib_fts' complexity is O(n). -// -// # The algorithm -// The scoring core is the same to lib_fts. The difference part is that I search more composites -// than lib_fts for a better result. -// -// \ A B a b C c -// \------------------------ -// a| 20 x -6 -// b| x 20+5 x .Y. -// c| -// -// For each matched key(b for example), we need to find all the matched A before and find the best -// matched A that will get us the max score for current b. For example, the one marked as '.Y.' in -// the previous example will search the row before(i.e. 'a') and calculate the max score for '.Y.'. -// -// This process will take O(n^3), because each row will cause O(n^2); -// -// So as shown below: -// 1 2 3 4 -// ...a.........a.........a....b...a....b.... -// a -// b b1 b2 -// -// if we already know that b1 gets max score when we choose a2, we assume that b2 will get max -// score only if 'a' is chosen within [a2..a4]. That means a1 is not searched any more. -// -// The worst case of course is also O(n^3) but normally that won't happan. Of course I wonder -// whether O(n^3) algorithm is acceptable or not. - +use std::cell::RefCell; const BONUS_ADJACENCY: i32 = 5; const BONUS_SEPARATOR: i32 = 10; const BONUS_CAMEL: i32 = 10; @@ -94,10 +13,10 @@ const PENALTY_MAX_LEADING: i32 = -9; // maxing penalty for leading letters const PENALTY_UNMATCHED: i32 = -1; // judge how many scores the current index should get -fn fuzzy_score(string: &Vec, index: usize) -> i32 { +fn fuzzy_score(string: &Vec, index: usize, is_first: bool) -> i32 { let mut score = 0; if index == 0 { - return BONUS_SEPARATOR + BONUS_CAMEL; + return BONUS_SEPARATOR; } let prev = string[index-1]; @@ -113,6 +32,10 @@ fn fuzzy_score(string: &Vec, index: usize) -> i32 { score += BONUS_CAMEL; } + if is_first { + score += max((index as i32) * PENALTY_LEADING, PENALTY_MAX_LEADING); + } + score } @@ -122,58 +45,58 @@ pub fn fuzzy_match(choice: &str, pattern: &str) -> Option<(i32, Vec)>{ } let choice_chars: Vec = choice.chars().collect(); + let choice_lower = choice.to_lowercase(); let pattern_chars: Vec = pattern.to_lowercase().chars().collect(); - let mut scores = vec![vec![]]; + let mut scores = vec![]; let mut picked = vec![]; - // initialize the first row of scores - for choice_idx in 0..choice.len() { - if pattern_chars[0] == choice_chars[choice_idx].to_lowercase().next().unwrap() { - let score = fuzzy_score(&choice_chars, choice_idx) + max(choice_idx as i32 * PENALTY_LEADING, PENALTY_MAX_LEADING); - scores[0].push((choice_idx, score, 0)); //(char_idx, score, vec_idx back_ref) - } - } - - for pattern_idx in 1..pattern.len() { - scores.push(vec![]); - - if scores[pattern_idx-1].len() <= 0 { - return None; - } - - let mut best_idx = 0; // points to previous line - for choice_idx in 1..choice.len() { - let mut best_score = -10000; - - if pattern_chars[pattern_idx] != choice_chars[choice_idx].to_lowercase().next().unwrap() { - continue; - } - - for (vec_idx, &(idx, prev_score, _)) in scores[pattern_idx-1][best_idx..].iter().enumerate() { - if idx >= choice_idx { - break; - } - - let mut char_score = fuzzy_score(&choice_chars, choice_idx); - let adjacent_num = choice_idx - idx - 1; - char_score += if adjacent_num == 0 {BONUS_ADJACENCY} else {PENALTY_UNMATCHED * adjacent_num as i32}; - char_score += prev_score; - - if char_score > best_score { - best_score = char_score; - best_idx = vec_idx; + let mut prev_matched_idx = -1; // to ensure that the pushed char are able to match the pattern + for pattern_idx in 0..pattern_chars.len() { + let pattern_char = pattern_chars[pattern_idx]; + let vec_cell = RefCell::new(vec![]); + { + let mut vec = vec_cell.borrow_mut(); + for (idx, ch) in choice_lower.chars().enumerate() { + if ch == pattern_char && (idx as i32) > prev_matched_idx { + vec.push((idx, fuzzy_score(&choice_chars, idx, pattern_idx == 0), 0)); // (char_idx, score, vec_idx back_ref) } } - scores[pattern_idx].push((choice_idx, best_score, best_idx)); + if vec.len() <= 0 { + // not matched + return None; + } + prev_matched_idx = vec[0].0 as i32; + } + scores.push(vec_cell); + } + + for pattern_idx in 0..pattern.len()-1 { + let cur_row = scores[pattern_idx].borrow(); + let mut next_row = scores[pattern_idx+1].borrow_mut(); + + for idx in 0..next_row.len() { + let (next_char_idx, next_score, _) = next_row[idx]; +//(back_ref, &score) + let (back_ref, score) = cur_row.iter() + .take_while(|&&(idx, _, _)| idx < next_char_idx) + .map(|&(char_idx, score, _)| { + let adjacent_num = next_char_idx - char_idx - 1; + score + next_score + if adjacent_num == 0 {BONUS_ADJACENCY} else {PENALTY_UNMATCHED * adjacent_num as i32} + }) + .enumerate() + .max_by_key(|&(_, x)| x) + .unwrap(); + + next_row[idx] = (next_char_idx, score, back_ref); } } - let (mut next_col, &(_, score, _)) = scores[pattern.len()-1].iter().enumerate().max_by_key(|&(_, &x)| x.1).unwrap(); + let (mut next_col, &(_, score, _)) = scores[pattern.len()-1].borrow().iter().enumerate().max_by_key(|&(_, &x)| x.1).unwrap(); let mut pattern_idx = pattern.len() as i32 - 1; while pattern_idx >= 0 { - let (idx, _, next) = scores[pattern_idx as usize][next_col]; + let (idx, _, next) = scores[pattern_idx as usize].borrow()[next_col]; next_col = next; picked.push(idx); pattern_idx -= 1; @@ -207,7 +130,20 @@ mod test { #[test] fn teset_fuzzy_match() { - let choice_4 = "AaBbcc"; + // the score in this test doesn't actually matter, but the index matters. + let choice_1 = "1111121"; + let query_1 = "21"; + assert_eq!(fuzzy_match(&choice_1, &query_1), Some((-4, vec![5,6]))); + + let choice_2 = "Ca"; + let query_2 = "ac"; + assert_eq!(fuzzy_match(&choice_2, &query_2), None); + + let choice_3 = "."; + let query_3 = "s"; + assert_eq!(fuzzy_match(&choice_3, &query_3), None); + + let choice_4 = "AaBbCc"; let query_4 = "abc"; assert_eq!(fuzzy_match(&choice_4, &query_4), Some((28, vec![0,2,4]))); }