unified-storage: add feature flag to use ngram for indexing (#111265)

* unified-storage: add feature flag to use ngram instead of edge-ngram for indexing
This commit is contained in:
Will Assis
2025-09-24 10:36:50 +02:00
committed by GitHub
parent 2669e0a770
commit 33ff6dbb9e
14 changed files with 317 additions and 172 deletions
+21 -10
View File
@@ -4,6 +4,7 @@ import (
"github.com/blevesearch/bleve/v2/analysis/analyzer/custom"
"github.com/blevesearch/bleve/v2/analysis/token/edgengram"
"github.com/blevesearch/bleve/v2/analysis/token/lowercase"
"github.com/blevesearch/bleve/v2/analysis/token/ngram"
"github.com/blevesearch/bleve/v2/analysis/token/unique"
"github.com/blevesearch/bleve/v2/analysis/tokenizer/whitespace"
"github.com/blevesearch/bleve/v2/mapping"
@@ -11,23 +12,33 @@ import (
const TITLE_ANALYZER = "title_analyzer"
const EDGE_NGRAM_MIN_TOKEN = 3.0
const tokenFilterName = "ngram_filter"
func RegisterCustomAnalyzers(mapper *mapping.IndexMappingImpl) error {
return registerTitleAnalyzer(mapper)
func RegisterCustomAnalyzers(mapper *mapping.IndexMappingImpl, useFullNgram bool) error {
return registerTitleAnalyzer(mapper, useFullNgram)
}
// The registerTitleAnalyzer function defines a custom analyzer for the title field.
// The edgeNgramTokenFilter will create n-grams anchored to the front of each token.
// For example, the token "hello" will be tokenized into "hel", "hell", "hello".
func registerTitleAnalyzer(mapper *mapping.IndexMappingImpl) error {
// Define an N-Gram tokenizer (for substring search)
edgeNgramTokenFilter := map[string]interface{}{
// The registerTitleAnalyzer function defines a custom analyzer using edge n-gram or full n-gram
func registerTitleAnalyzer(mapper *mapping.IndexMappingImpl, useFullNgram bool) error {
// The edgengram tokenFilter will create grams anchored to the front of each token.
// For example, the token "hello" will be tokenized into "hel", "hell", "hello".
tokenFilter := map[string]interface{}{
"type": edgengram.Name,
"min": EDGE_NGRAM_MIN_TOKEN,
"max": 10.0,
"back": edgengram.FRONT,
}
err := mapper.AddCustomTokenFilter("edge_ngram_filter", edgeNgramTokenFilter)
if useFullNgram {
// The ngram tokenFilter will create additional grams in the middle of each token.
// For example, the token "hello" will be tokenized into "hel", "hell", "hello", "ell", "ello", "llo".
tokenFilter = map[string]interface{}{
"type": ngram.Name,
"min": EDGE_NGRAM_MIN_TOKEN,
"max": 10.0,
}
}
err := mapper.AddCustomTokenFilter(tokenFilterName, tokenFilter)
if err != nil {
return err
}
@@ -36,7 +47,7 @@ func registerTitleAnalyzer(mapper *mapping.IndexMappingImpl) error {
ngramAnalyzer := map[string]interface{}{
"type": custom.Name,
"tokenizer": whitespace.Name,
"token_filters": []string{"edge_ngram_filter", lowercase.Name, unique.Name},
"token_filters": []string{tokenFilterName, lowercase.Name, unique.Name},
//"char_filters": //TODO IF NEEDED
}