Add CJK (Chinese/Japanese/Korean) tokenizer for full-text search

This commit is contained in:
Yuri Karamian
2026-06-07 17:44:05 +02:00
parent ab39fc2ed6
commit 7c94370e48
+208 -11
View File
@@ -9,11 +9,149 @@ use tantivy::collector::TopDocs;
use tantivy::directory::MmapDirectory; use tantivy::directory::MmapDirectory;
#[cfg(mobile)] #[cfg(mobile)]
use tantivy::directory::RamDirectory; use tantivy::directory::RamDirectory;
use tantivy::query::{BooleanQuery, FuzzyTermQuery, Occur, PhrasePrefixQuery, Query}; use tantivy::query::{BooleanQuery, FuzzyTermQuery, Occur, PhrasePrefixQuery, Query, TermQuery};
use tantivy::schema::*; use tantivy::schema::*;
use tantivy::tokenizer::{LowerCaser, TextAnalyzer, Token, TokenStream, Tokenizer};
use tantivy::{Index, IndexWriter, TantivyDocument, Term}; use tantivy::{Index, IndexWriter, TantivyDocument, Term};
use walkdir::WalkDir; use walkdir::WalkDir;
/// Bumped whenever the index schema or tokenizer changes, so the on-disk index is
/// wiped and rebuilt once on the next vault open (the index is derived from the
/// notes, so this never loses data).
const INDEX_SCHEMA_VERSION: &str = "2-cjk-bigram";
/// True for characters in the CJK / Japanese / Korean blocks, which are written
/// without spaces between words. These get uni/bi-gram tokenized so substring
/// search works; everything else keeps the default word-splitting behaviour.
fn is_cjk(c: char) -> bool {
matches!(c as u32,
0x3400..=0x4DBF // CJK Unified Ideographs Extension A
| 0x4E00..=0x9FFF // CJK Unified Ideographs (Han)
| 0x3040..=0x309F // Hiragana
| 0x30A0..=0x30FF // Katakana
| 0xAC00..=0xD7AF // Hangul Syllables
| 0xF900..=0xFAFF // CJK Compatibility Ideographs
| 0xFF00..=0xFFEF // Halfwidth and Fullwidth Forms
)
}
/// Tokenize `text` so that CJK runs become overlapping unigrams + bigrams (recall
/// over precision: a query for any substring of a CJK run will match), while runs
/// of other alphanumeric characters become a single word token (same as tantivy's
/// SimpleTokenizer, so Latin/English indexing is unchanged). Lowercasing is applied
/// by a LowerCaser filter in the analyzer, not here.
fn cjk_tokens(text: &str) -> Vec<Token> {
let mut tokens: Vec<Token> = Vec::new();
let mut position: usize = 0;
let mut cjk_run: Vec<(usize, char)> = Vec::new();
let mut word = String::new();
let mut word_start: usize = 0;
macro_rules! flush_word {
() => {{
if !word.is_empty() {
let len = word.len();
tokens.push(Token {
offset_from: word_start,
offset_to: word_start + len,
position,
text: std::mem::take(&mut word),
position_length: 1,
});
position += 1;
}
}};
}
macro_rules! flush_cjk {
() => {{
let n = cjk_run.len();
for i in 0..n {
let (off, ch) = cjk_run[i];
tokens.push(Token {
offset_from: off,
offset_to: off + ch.len_utf8(),
position,
text: ch.to_string(),
position_length: 1,
});
position += 1;
}
for i in 0..n.saturating_sub(1) {
let (off1, c1) = cjk_run[i];
let (off2, c2) = cjk_run[i + 1];
let mut s = String::with_capacity(c1.len_utf8() + c2.len_utf8());
s.push(c1);
s.push(c2);
tokens.push(Token {
offset_from: off1,
offset_to: off2 + c2.len_utf8(),
position,
text: s,
position_length: 1,
});
position += 1;
}
cjk_run.clear();
}};
}
for (offset, c) in text.char_indices() {
if is_cjk(c) {
flush_word!();
cjk_run.push((offset, c));
} else if c.is_alphanumeric() {
flush_cjk!();
if word.is_empty() {
word_start = offset;
}
word.push(c);
} else {
flush_cjk!();
flush_word!();
}
}
flush_cjk!();
flush_word!();
let _ = position; // the final flush increments position but nothing reads it after
tokens
}
/// A pre-computed token stream (all tokens collected up front by `cjk_tokens`).
struct PreTokenizedStream {
tokens: Vec<Token>,
idx: usize,
}
impl TokenStream for PreTokenizedStream {
fn advance(&mut self) -> bool {
if self.idx < self.tokens.len() {
self.idx += 1;
true
} else {
false
}
}
fn token(&self) -> &Token {
&self.tokens[self.idx - 1]
}
fn token_mut(&mut self) -> &mut Token {
&mut self.tokens[self.idx - 1]
}
}
#[derive(Clone)]
struct CjkTokenizer;
impl Tokenizer for CjkTokenizer {
type TokenStream<'a> = PreTokenizedStream;
fn token_stream<'a>(&'a mut self, text: &'a str) -> PreTokenizedStream {
PreTokenizedStream {
tokens: cjk_tokens(text),
idx: 0,
}
}
}
pub struct SearchIndex { pub struct SearchIndex {
index: Index, index: Index,
writer: Mutex<Option<IndexWriter>>, writer: Mutex<Option<IndexWriter>>,
@@ -29,9 +167,16 @@ impl SearchIndex {
pub fn new(vault_path: &str) -> Result<Self, String> { pub fn new(vault_path: &str) -> Result<Self, String> {
let mut schema_builder = Schema::builder(); let mut schema_builder = Schema::builder();
let path_field = schema_builder.add_text_field("path", STRING | STORED); let path_field = schema_builder.add_text_field("path", STRING | STORED);
let title_field = schema_builder.add_text_field("title", TEXT | STORED); // CJK-aware tokenizer for the indexed text fields (keeps Latin/English
let body_field = schema_builder.add_text_field("body", TEXT); // behaviour identical; adds substring matching for Chinese/Japanese/Korean).
let tags_field = schema_builder.add_text_field("tags", TEXT | STORED); let cjk_indexing = TextFieldIndexing::default()
.set_tokenizer("cjk")
.set_index_option(IndexRecordOption::WithFreqsAndPositions);
let cjk_text = TextOptions::default().set_indexing_options(cjk_indexing.clone());
let cjk_text_stored = cjk_text.clone().set_stored();
let title_field = schema_builder.add_text_field("title", cjk_text_stored.clone());
let body_field = schema_builder.add_text_field("body", cjk_text);
let tags_field = schema_builder.add_text_field("tags", cjk_text_stored);
let schema = schema_builder.build(); let schema = schema_builder.build();
// Mobile: use in-memory index (flock is unreliable on the sandboxed/FUSE filesystem) // Mobile: use in-memory index (flock is unreliable on the sandboxed/FUSE filesystem)
@@ -43,11 +188,41 @@ impl SearchIndex {
}; };
#[cfg(desktop)] #[cfg(desktop)]
let index = { let index = {
let index_dir = helixnotes_dir(vault_path).join("search_index"); let hn = helixnotes_dir(vault_path);
let index_dir = hn.join("search_index");
let version_path = hn.join("search_index.version");
// One-time wipe when the schema/tokenizer version changes, so a stale
// index built with the old tokenizer is rebuilt. open_vault always calls
// rebuild() afterwards, and the index is derived from the .md files, so
// wiping never loses data.
let version_ok = fs::read_to_string(&version_path)
.map(|v| v.trim() == INDEX_SCHEMA_VERSION)
.unwrap_or(false);
if !version_ok {
let _ = fs::remove_dir_all(&index_dir);
}
fs::create_dir_all(&index_dir).map_err(|e| e.to_string())?;
let dir = MmapDirectory::open(&index_dir).map_err(|e| e.to_string())?;
let idx = match Index::open_or_create(dir, schema.clone()) {
Ok(idx) => idx,
Err(_) => {
// Schema mismatch (older index) or corruption: wipe and recreate.
let _ = fs::remove_dir_all(&index_dir);
fs::create_dir_all(&index_dir).map_err(|e| e.to_string())?; fs::create_dir_all(&index_dir).map_err(|e| e.to_string())?;
let dir = MmapDirectory::open(&index_dir).map_err(|e| e.to_string())?; let dir = MmapDirectory::open(&index_dir).map_err(|e| e.to_string())?;
Index::open_or_create(dir, schema.clone()).map_err(|e| e.to_string())? Index::open_or_create(dir, schema.clone()).map_err(|e| e.to_string())?
}
}; };
let _ = fs::write(&version_path, INDEX_SCHEMA_VERSION);
idx
};
// Register the CJK-aware tokenizer (in-memory; must be done on every open,
// before the writer is created, so both indexing and querying use it).
index.tokenizers().register(
"cjk",
TextAnalyzer::builder(CjkTokenizer).filter(LowerCaser).build(),
);
#[cfg(mobile)] #[cfg(mobile)]
let heap_size = 15_000_000; let heap_size = 15_000_000;
@@ -153,19 +328,40 @@ impl SearchIndex {
let searcher = reader.searcher(); let searcher = reader.searcher();
let fields = vec![self.title_field, self.body_field, self.tags_field]; let fields = vec![self.title_field, self.body_field, self.tags_field];
let terms: Vec<String> = query_str // Tokenize the query with the SAME CJK-aware analyzer used for indexing, so a
.split_whitespace() // Chinese/Japanese/Korean query becomes the same uni/bigram tokens as the docs.
.map(|t| t.to_lowercase()) // (For pure-ASCII queries this yields the same lowercased word tokens as before.)
.collect(); let mut analyzer = self
.index
.tokenizers()
.get("cjk")
.ok_or("cjk tokenizer not registered")?;
let terms: Vec<String> = {
let mut out = Vec::new();
let mut stream = analyzer.token_stream(query_str);
while stream.advance() {
out.push(stream.token().text.clone());
}
out
};
// For each search term, create prefix + fuzzy queries across all fields (OR), // For each term, OR queries across all fields, then AND all terms together.
// then AND all terms together // CJK tokens (uni/bigrams) already encode segmentation -> exact term match.
// Latin tokens keep prefix + fuzzy matching (unchanged English behaviour).
let term_queries: Vec<(Occur, Box<dyn Query>)> = terms let term_queries: Vec<(Occur, Box<dyn Query>)> = terms
.iter() .iter()
.map(|term| { .map(|term| {
let is_cjk_term = term.chars().any(is_cjk);
let field_queries: Vec<(Occur, Box<dyn Query>)> = fields let field_queries: Vec<(Occur, Box<dyn Query>)> = fields
.iter() .iter()
.flat_map(|&field| { .flat_map(|&field| {
if is_cjk_term {
let exact: Box<dyn Query> = Box::new(TermQuery::new(
Term::from_field_text(field, term),
IndexRecordOption::WithFreqs,
));
vec![(Occur::Should, exact)]
} else {
let prefix: Box<dyn Query> = Box::new(PhrasePrefixQuery::new( let prefix: Box<dyn Query> = Box::new(PhrasePrefixQuery::new(
vec![Term::from_field_text(field, term)], vec![Term::from_field_text(field, term)],
)); ));
@@ -175,6 +371,7 @@ impl SearchIndex {
true, true,
)); ));
vec![(Occur::Should, prefix), (Occur::Should, fuzzy)] vec![(Occur::Should, prefix), (Occur::Should, fuzzy)]
}
}) })
.collect(); .collect();
let combined: Box<dyn Query> = Box::new(BooleanQuery::new(field_queries)); let combined: Box<dyn Query> = Box::new(BooleanQuery::new(field_queries));