From 7c94370e48e135aa393eeb6c6cad2bd797cb5c79 Mon Sep 17 00:00:00 2001 From: Yuri Karamian Date: Sun, 7 Jun 2026 17:44:05 +0200 Subject: [PATCH] Add CJK (Chinese/Japanese/Korean) tokenizer for full-text search --- src-tauri/src/search/mod.rs | 239 ++++++++++++++++++++++++++++++++---- 1 file changed, 218 insertions(+), 21 deletions(-) diff --git a/src-tauri/src/search/mod.rs b/src-tauri/src/search/mod.rs index 1bc2a70..1f9a645 100644 --- a/src-tauri/src/search/mod.rs +++ b/src-tauri/src/search/mod.rs @@ -9,11 +9,149 @@ use tantivy::collector::TopDocs; use tantivy::directory::MmapDirectory; #[cfg(mobile)] use tantivy::directory::RamDirectory; -use tantivy::query::{BooleanQuery, FuzzyTermQuery, Occur, PhrasePrefixQuery, Query}; +use tantivy::query::{BooleanQuery, FuzzyTermQuery, Occur, PhrasePrefixQuery, Query, TermQuery}; use tantivy::schema::*; +use tantivy::tokenizer::{LowerCaser, TextAnalyzer, Token, TokenStream, Tokenizer}; use tantivy::{Index, IndexWriter, TantivyDocument, Term}; use walkdir::WalkDir; +/// Bumped whenever the index schema or tokenizer changes, so the on-disk index is +/// wiped and rebuilt once on the next vault open (the index is derived from the +/// notes, so this never loses data). +const INDEX_SCHEMA_VERSION: &str = "2-cjk-bigram"; + +/// True for characters in the CJK / Japanese / Korean blocks, which are written +/// without spaces between words. These get uni/bi-gram tokenized so substring +/// search works; everything else keeps the default word-splitting behaviour. +fn is_cjk(c: char) -> bool { + matches!(c as u32, + 0x3400..=0x4DBF // CJK Unified Ideographs Extension A + | 0x4E00..=0x9FFF // CJK Unified Ideographs (Han) + | 0x3040..=0x309F // Hiragana + | 0x30A0..=0x30FF // Katakana + | 0xAC00..=0xD7AF // Hangul Syllables + | 0xF900..=0xFAFF // CJK Compatibility Ideographs + | 0xFF00..=0xFFEF // Halfwidth and Fullwidth Forms + ) +} + +/// Tokenize `text` so that CJK runs become overlapping unigrams + bigrams (recall +/// over precision: a query for any substring of a CJK run will match), while runs +/// of other alphanumeric characters become a single word token (same as tantivy's +/// SimpleTokenizer, so Latin/English indexing is unchanged). Lowercasing is applied +/// by a LowerCaser filter in the analyzer, not here. +fn cjk_tokens(text: &str) -> Vec { + let mut tokens: Vec = Vec::new(); + let mut position: usize = 0; + let mut cjk_run: Vec<(usize, char)> = Vec::new(); + let mut word = String::new(); + let mut word_start: usize = 0; + + macro_rules! flush_word { + () => {{ + if !word.is_empty() { + let len = word.len(); + tokens.push(Token { + offset_from: word_start, + offset_to: word_start + len, + position, + text: std::mem::take(&mut word), + position_length: 1, + }); + position += 1; + } + }}; + } + macro_rules! flush_cjk { + () => {{ + let n = cjk_run.len(); + for i in 0..n { + let (off, ch) = cjk_run[i]; + tokens.push(Token { + offset_from: off, + offset_to: off + ch.len_utf8(), + position, + text: ch.to_string(), + position_length: 1, + }); + position += 1; + } + for i in 0..n.saturating_sub(1) { + let (off1, c1) = cjk_run[i]; + let (off2, c2) = cjk_run[i + 1]; + let mut s = String::with_capacity(c1.len_utf8() + c2.len_utf8()); + s.push(c1); + s.push(c2); + tokens.push(Token { + offset_from: off1, + offset_to: off2 + c2.len_utf8(), + position, + text: s, + position_length: 1, + }); + position += 1; + } + cjk_run.clear(); + }}; + } + + for (offset, c) in text.char_indices() { + if is_cjk(c) { + flush_word!(); + cjk_run.push((offset, c)); + } else if c.is_alphanumeric() { + flush_cjk!(); + if word.is_empty() { + word_start = offset; + } + word.push(c); + } else { + flush_cjk!(); + flush_word!(); + } + } + flush_cjk!(); + flush_word!(); + let _ = position; // the final flush increments position but nothing reads it after + tokens +} + +/// A pre-computed token stream (all tokens collected up front by `cjk_tokens`). +struct PreTokenizedStream { + tokens: Vec, + idx: usize, +} + +impl TokenStream for PreTokenizedStream { + fn advance(&mut self) -> bool { + if self.idx < self.tokens.len() { + self.idx += 1; + true + } else { + false + } + } + fn token(&self) -> &Token { + &self.tokens[self.idx - 1] + } + fn token_mut(&mut self) -> &mut Token { + &mut self.tokens[self.idx - 1] + } +} + +#[derive(Clone)] +struct CjkTokenizer; + +impl Tokenizer for CjkTokenizer { + type TokenStream<'a> = PreTokenizedStream; + fn token_stream<'a>(&'a mut self, text: &'a str) -> PreTokenizedStream { + PreTokenizedStream { + tokens: cjk_tokens(text), + idx: 0, + } + } +} + pub struct SearchIndex { index: Index, writer: Mutex>, @@ -29,9 +167,16 @@ impl SearchIndex { pub fn new(vault_path: &str) -> Result { let mut schema_builder = Schema::builder(); let path_field = schema_builder.add_text_field("path", STRING | STORED); - let title_field = schema_builder.add_text_field("title", TEXT | STORED); - let body_field = schema_builder.add_text_field("body", TEXT); - let tags_field = schema_builder.add_text_field("tags", TEXT | STORED); + // CJK-aware tokenizer for the indexed text fields (keeps Latin/English + // behaviour identical; adds substring matching for Chinese/Japanese/Korean). + let cjk_indexing = TextFieldIndexing::default() + .set_tokenizer("cjk") + .set_index_option(IndexRecordOption::WithFreqsAndPositions); + let cjk_text = TextOptions::default().set_indexing_options(cjk_indexing.clone()); + let cjk_text_stored = cjk_text.clone().set_stored(); + let title_field = schema_builder.add_text_field("title", cjk_text_stored.clone()); + let body_field = schema_builder.add_text_field("body", cjk_text); + let tags_field = schema_builder.add_text_field("tags", cjk_text_stored); let schema = schema_builder.build(); // Mobile: use in-memory index (flock is unreliable on the sandboxed/FUSE filesystem) @@ -43,12 +188,42 @@ impl SearchIndex { }; #[cfg(desktop)] let index = { - let index_dir = helixnotes_dir(vault_path).join("search_index"); + let hn = helixnotes_dir(vault_path); + let index_dir = hn.join("search_index"); + let version_path = hn.join("search_index.version"); + // One-time wipe when the schema/tokenizer version changes, so a stale + // index built with the old tokenizer is rebuilt. open_vault always calls + // rebuild() afterwards, and the index is derived from the .md files, so + // wiping never loses data. + let version_ok = fs::read_to_string(&version_path) + .map(|v| v.trim() == INDEX_SCHEMA_VERSION) + .unwrap_or(false); + if !version_ok { + let _ = fs::remove_dir_all(&index_dir); + } fs::create_dir_all(&index_dir).map_err(|e| e.to_string())?; let dir = MmapDirectory::open(&index_dir).map_err(|e| e.to_string())?; - Index::open_or_create(dir, schema.clone()).map_err(|e| e.to_string())? + let idx = match Index::open_or_create(dir, schema.clone()) { + Ok(idx) => idx, + Err(_) => { + // Schema mismatch (older index) or corruption: wipe and recreate. + let _ = fs::remove_dir_all(&index_dir); + fs::create_dir_all(&index_dir).map_err(|e| e.to_string())?; + let dir = MmapDirectory::open(&index_dir).map_err(|e| e.to_string())?; + Index::open_or_create(dir, schema.clone()).map_err(|e| e.to_string())? + } + }; + let _ = fs::write(&version_path, INDEX_SCHEMA_VERSION); + idx }; + // Register the CJK-aware tokenizer (in-memory; must be done on every open, + // before the writer is created, so both indexing and querying use it). + index.tokenizers().register( + "cjk", + TextAnalyzer::builder(CjkTokenizer).filter(LowerCaser).build(), + ); + #[cfg(mobile)] let heap_size = 15_000_000; #[cfg(desktop)] @@ -153,28 +328,50 @@ impl SearchIndex { let searcher = reader.searcher(); let fields = vec![self.title_field, self.body_field, self.tags_field]; - let terms: Vec = query_str - .split_whitespace() - .map(|t| t.to_lowercase()) - .collect(); + // Tokenize the query with the SAME CJK-aware analyzer used for indexing, so a + // Chinese/Japanese/Korean query becomes the same uni/bigram tokens as the docs. + // (For pure-ASCII queries this yields the same lowercased word tokens as before.) + let mut analyzer = self + .index + .tokenizers() + .get("cjk") + .ok_or("cjk tokenizer not registered")?; + let terms: Vec = { + let mut out = Vec::new(); + let mut stream = analyzer.token_stream(query_str); + while stream.advance() { + out.push(stream.token().text.clone()); + } + out + }; - // For each search term, create prefix + fuzzy queries across all fields (OR), - // then AND all terms together + // For each term, OR queries across all fields, then AND all terms together. + // CJK tokens (uni/bigrams) already encode segmentation -> exact term match. + // Latin tokens keep prefix + fuzzy matching (unchanged English behaviour). let term_queries: Vec<(Occur, Box)> = terms .iter() .map(|term| { + let is_cjk_term = term.chars().any(is_cjk); let field_queries: Vec<(Occur, Box)> = fields .iter() .flat_map(|&field| { - let prefix: Box = Box::new(PhrasePrefixQuery::new( - vec![Term::from_field_text(field, term)], - )); - let fuzzy: Box = Box::new(FuzzyTermQuery::new( - Term::from_field_text(field, term), - 1, - true, - )); - vec![(Occur::Should, prefix), (Occur::Should, fuzzy)] + if is_cjk_term { + let exact: Box = Box::new(TermQuery::new( + Term::from_field_text(field, term), + IndexRecordOption::WithFreqs, + )); + vec![(Occur::Should, exact)] + } else { + let prefix: Box = Box::new(PhrasePrefixQuery::new( + vec![Term::from_field_text(field, term)], + )); + let fuzzy: Box = Box::new(FuzzyTermQuery::new( + Term::from_field_text(field, term), + 1, + true, + )); + vec![(Occur::Should, prefix), (Occur::Should, fuzzy)] + } }) .collect(); let combined: Box = Box::new(BooleanQuery::new(field_queries));