mirror of
https://github.com/go-gitea/gitea.git
synced 2026-09-25 22:23:42 +09:00
fix(indexer): index full file paths and real offsets in bleve (#39405)
The bleve path token filter added in https://github.com/go-gitea/gitea/pull/32210 never generated the full path of a file, so searching a file by its path (e.g. `potato/ham`) found nothing. It also gave the path tokens made-up offsets instead of their position in the path. The filter is replaced by a tokenizer that emits the raw path suffixes starting at each segment and word (`potato/ham.md`, `ham.md`). Paths below the root now match too, as do names containing `-` or spaces and dotfiles. An exact file name also ranks above files inside a directory with the same name. --------- Co-authored-by: silverwind <me@silverwind.io> Co-authored-by: wxiaoguang <wxiaoguang@gmail.com>
This commit is contained in:
co-authored by
silverwind
wxiaoguang
parent
8b46a956d8
commit
72243bead7
@@ -17,7 +17,6 @@ import (
|
||||
"gitea.dev/modules/git"
|
||||
"gitea.dev/modules/git/gitcmd"
|
||||
"gitea.dev/modules/indexer"
|
||||
path_filter "gitea.dev/modules/indexer/code/bleve/token/path"
|
||||
"gitea.dev/modules/indexer/code/internal"
|
||||
indexer_internal "gitea.dev/modules/indexer/internal"
|
||||
inner_bleve "gitea.dev/modules/indexer/internal/bleve"
|
||||
@@ -32,8 +31,8 @@ import (
|
||||
analyzer_keyword "github.com/blevesearch/bleve/v2/analysis/analyzer/keyword"
|
||||
"github.com/blevesearch/bleve/v2/analysis/token/lowercase"
|
||||
"github.com/blevesearch/bleve/v2/analysis/token/unicodenorm"
|
||||
"github.com/blevesearch/bleve/v2/analysis/tokenizer/unicode"
|
||||
"github.com/blevesearch/bleve/v2/mapping"
|
||||
"github.com/blevesearch/bleve/v2/registry"
|
||||
"github.com/blevesearch/bleve/v2/search/query"
|
||||
"github.com/go-enry/go-enry/v2"
|
||||
)
|
||||
@@ -69,7 +68,7 @@ const (
|
||||
repoIndexerAnalyzer = "repoIndexerAnalyzer"
|
||||
filenameIndexerAnalyzer = "filenameIndexerAnalyzer"
|
||||
repoIndexerDocType = "repoIndexerDocType"
|
||||
repoIndexerLatestVersion = 10
|
||||
repoIndexerLatestVersion = 11
|
||||
)
|
||||
|
||||
// generateBleveIndexMapping generates a bleve index mapping for the repo indexer
|
||||
@@ -114,8 +113,8 @@ func generateBleveIndexMapping() (mapping.IndexMapping, error) {
|
||||
if err := mapping.AddCustomAnalyzer(filenameIndexerAnalyzer, map[string]any{
|
||||
"type": analyzer_custom.Name,
|
||||
"char_filters": []string{},
|
||||
"tokenizer": unicode.Name,
|
||||
"token_filters": []string{unicodeNormalizeName, path_filter.Name, lowercase.Name},
|
||||
"tokenizer": pathTokenizerName,
|
||||
"token_filters": []string{unicodeNormalizeName, lowercase.Name},
|
||||
}); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -139,6 +138,13 @@ func (b *Indexer) SupportedSearchModes() []indexer.SearchMode {
|
||||
return indexer.SearchModesExactWords()
|
||||
}
|
||||
|
||||
func init() {
|
||||
// due to bleve's design problem, the "Register" must be done in the main goroutine, otherwise data-race
|
||||
util.MustNoError(registry.RegisterTokenizer(codeTokenizerName, codeTokenizerConstructor))
|
||||
util.MustNoError(registry.RegisterTokenFilter(codeTokenFilterName, codeTokenFilterConstructor))
|
||||
util.MustNoError(registry.RegisterTokenizer(pathTokenizerName, pathTokenizerConstructor))
|
||||
}
|
||||
|
||||
// NewIndexer creates a new bleve local indexer
|
||||
func NewIndexer(indexDir string) *Indexer {
|
||||
inner := inner_bleve.NewIndexer(indexDir, repoIndexerLatestVersion, generateBleveIndexMapping)
|
||||
|
||||
Reference in New Issue
Block a user