[score] refine the code for better understanding

This commit is contained in:
Mark Wallace 2016-06-25 15:44:26 +08:00
parent 1c7e90c6dd
commit c62afe4d94

View file

@ -2,90 +2,9 @@
/// the query and the choice.
///
/// It is modeled after https://github.com/felipesere/icepick.git
use std::cmp::max;
// return (start, matched_len)
pub fn compute_match_length(choice: &str, query: &str) -> Option<(usize, usize)> {
if query.len() <= 0 {
return Some((0, 0));
}
let impossible_match = choice.len() + 1;
let mut matched_start = impossible_match;
let mut matched_end = impossible_match;
let mut choice_chars = choice.chars().enumerate().peekable();
let mut query_chars = query.chars().enumerate().peekable();
loop {
if query_chars.peek() == None {
return Some((matched_start, matched_end - matched_start+1));
}
if choice_chars.peek() == None {
return None;
}
let &(idx_choice, c) = choice_chars.peek().unwrap();
let &(_, q) = query_chars.peek().unwrap();
if c == q {
if matched_start == impossible_match { matched_start = idx_choice; }
let _ = query_chars.next();
}
matched_end = idx_choice;
let _ = choice_chars.next();
}
}
// The fuzzy search algorith is inspired by https://github.com/forrestthewoods/lib_fts
//
// # Why is new Fuzzy Match algorithm needed?
// 1. fzf use the query to compose regex for non-greedy match. When use for file/path match, it did
// not work out for the best(e.g. Camel case characters should weight more)
// 2. fzf will only show the `range` be matched, while I think it is not enought sometimes. with
// this fuzzy_match algorithm we are able to get all matched indics.
//
// # The complexity is O(n^2)
// An intuitive way of fuzzy match is to find all occurences of pattern characters in the target
// string. Then find out all the choice composites and evaluate the score for each. which will take
// O(n!) for each input string. Then with dynamic programing, we can reduce the time to O(n^3) with
// addition O(m*n) space;
//
// I think that O(n^3) may not be acceptable in the use case of fzf-rs. so I made some assumptions
// and reduce the complexity to O(n^2). Note that lib_fts' complexity is O(n).
//
// # The algorithm
// The scoring core is the same to lib_fts. The difference part is that I search more composites
// than lib_fts for a better result.
//
// \ A B a b C c
// \------------------------
// a| 20 x -6
// b| x 20+5 x .Y.
// c|
//
// For each matched key(b for example), we need to find all the matched A before and find the best
// matched A that will get us the max score for current b. For example, the one marked as '.Y.' in
// the previous example will search the row before(i.e. 'a') and calculate the max score for '.Y.'.
//
// This process will take O(n^3), because each row will cause O(n^2);
//
// So as shown below:
// 1 2 3 4
// ...a.........a.........a....b...a....b....
// a
// b b1 b2
//
// if we already know that b1 gets max score when we choose a2, we assume that b2 will get max
// score only if 'a' is chosen within [a2..a4]. That means a1 is not searched any more.
//
// The worst case of course is also O(n^3) but normally that won't happan. Of course I wonder
// whether O(n^3) algorithm is acceptable or not.
use std::cell::RefCell;
const BONUS_ADJACENCY: i32 = 5;
const BONUS_SEPARATOR: i32 = 10;
const BONUS_CAMEL: i32 = 10;
@ -94,10 +13,10 @@ const PENALTY_MAX_LEADING: i32 = -9; // maxing penalty for leading letters
const PENALTY_UNMATCHED: i32 = -1;
// judge how many scores the current index should get
fn fuzzy_score(string: &Vec<char>, index: usize) -> i32 {
fn fuzzy_score(string: &Vec<char>, index: usize, is_first: bool) -> i32 {
let mut score = 0;
if index == 0 {
return BONUS_SEPARATOR + BONUS_CAMEL;
return BONUS_SEPARATOR;
}
let prev = string[index-1];
@ -113,6 +32,10 @@ fn fuzzy_score(string: &Vec<char>, index: usize) -> i32 {
score += BONUS_CAMEL;
}
if is_first {
score += max((index as i32) * PENALTY_LEADING, PENALTY_MAX_LEADING);
}
score
}
@ -122,58 +45,58 @@ pub fn fuzzy_match(choice: &str, pattern: &str) -> Option<(i32, Vec<usize>)>{
}
let choice_chars: Vec<char> = choice.chars().collect();
let choice_lower = choice.to_lowercase();
let pattern_chars: Vec<char> = pattern.to_lowercase().chars().collect();
let mut scores = vec![vec![]];
let mut scores = vec![];
let mut picked = vec![];
// initialize the first row of scores
for choice_idx in 0..choice.len() {
if pattern_chars[0] == choice_chars[choice_idx].to_lowercase().next().unwrap() {
let score = fuzzy_score(&choice_chars, choice_idx) + max(choice_idx as i32 * PENALTY_LEADING, PENALTY_MAX_LEADING);
scores[0].push((choice_idx, score, 0)); //(char_idx, score, vec_idx back_ref)
}
}
for pattern_idx in 1..pattern.len() {
scores.push(vec![]);
if scores[pattern_idx-1].len() <= 0 {
return None;
}
let mut best_idx = 0; // points to previous line
for choice_idx in 1..choice.len() {
let mut best_score = -10000;
if pattern_chars[pattern_idx] != choice_chars[choice_idx].to_lowercase().next().unwrap() {
continue;
}
for (vec_idx, &(idx, prev_score, _)) in scores[pattern_idx-1][best_idx..].iter().enumerate() {
if idx >= choice_idx {
break;
}
let mut char_score = fuzzy_score(&choice_chars, choice_idx);
let adjacent_num = choice_idx - idx - 1;
char_score += if adjacent_num == 0 {BONUS_ADJACENCY} else {PENALTY_UNMATCHED * adjacent_num as i32};
char_score += prev_score;
if char_score > best_score {
best_score = char_score;
best_idx = vec_idx;
let mut prev_matched_idx = -1; // to ensure that the pushed char are able to match the pattern
for pattern_idx in 0..pattern_chars.len() {
let pattern_char = pattern_chars[pattern_idx];
let vec_cell = RefCell::new(vec![]);
{
let mut vec = vec_cell.borrow_mut();
for (idx, ch) in choice_lower.chars().enumerate() {
if ch == pattern_char && (idx as i32) > prev_matched_idx {
vec.push((idx, fuzzy_score(&choice_chars, idx, pattern_idx == 0), 0)); // (char_idx, score, vec_idx back_ref)
}
}
scores[pattern_idx].push((choice_idx, best_score, best_idx));
if vec.len() <= 0 {
// not matched
return None;
}
prev_matched_idx = vec[0].0 as i32;
}
scores.push(vec_cell);
}
for pattern_idx in 0..pattern.len()-1 {
let cur_row = scores[pattern_idx].borrow();
let mut next_row = scores[pattern_idx+1].borrow_mut();
for idx in 0..next_row.len() {
let (next_char_idx, next_score, _) = next_row[idx];
//(back_ref, &score)
let (back_ref, score) = cur_row.iter()
.take_while(|&&(idx, _, _)| idx < next_char_idx)
.map(|&(char_idx, score, _)| {
let adjacent_num = next_char_idx - char_idx - 1;
score + next_score + if adjacent_num == 0 {BONUS_ADJACENCY} else {PENALTY_UNMATCHED * adjacent_num as i32}
})
.enumerate()
.max_by_key(|&(_, x)| x)
.unwrap();
next_row[idx] = (next_char_idx, score, back_ref);
}
}
let (mut next_col, &(_, score, _)) = scores[pattern.len()-1].iter().enumerate().max_by_key(|&(_, &x)| x.1).unwrap();
let (mut next_col, &(_, score, _)) = scores[pattern.len()-1].borrow().iter().enumerate().max_by_key(|&(_, &x)| x.1).unwrap();
let mut pattern_idx = pattern.len() as i32 - 1;
while pattern_idx >= 0 {
let (idx, _, next) = scores[pattern_idx as usize][next_col];
let (idx, _, next) = scores[pattern_idx as usize].borrow()[next_col];
next_col = next;
picked.push(idx);
pattern_idx -= 1;
@ -207,7 +130,20 @@ mod test {
#[test]
fn teset_fuzzy_match() {
let choice_4 = "AaBbcc";
// the score in this test doesn't actually matter, but the index matters.
let choice_1 = "1111121";
let query_1 = "21";
assert_eq!(fuzzy_match(&choice_1, &query_1), Some((-4, vec![5,6])));
let choice_2 = "Ca";
let query_2 = "ac";
assert_eq!(fuzzy_match(&choice_2, &query_2), None);
let choice_3 = ".";
let query_3 = "s";
assert_eq!(fuzzy_match(&choice_3, &query_3), None);
let choice_4 = "AaBbCc";
let query_4 = "abc";
assert_eq!(fuzzy_match(&choice_4, &query_4), Some((28, vec![0,2,4])));
}