Files
72243bead7 fix(indexer): index full file paths and real offsets in bleve (#39405)
The bleve path token filter added in
https://github.com/go-gitea/gitea/pull/32210 never generated the full
path of a file, so searching a file by its path (e.g. `potato/ham`)
found nothing. It also gave the path tokens made-up offsets instead of
their position in the path.

The filter is replaced by a tokenizer that emits the raw path suffixes
starting at each segment and word (`potato/ham.md`, `ham.md`). Paths
below the root now match too, as do names containing `-` or spaces and
dotfiles. An exact file name also ranks above files inside a directory
with the same name.

---------

Co-authored-by: silverwind <me@silverwind.io>
Co-authored-by: wxiaoguang <wxiaoguang@gmail.com>
2026-09-24 07:57:42 +00:00

59 lines
1.6 KiB
Go

// Copyright 2026 The Gitea Authors. All rights reserved.
// SPDX-License-Identifier: MIT
package bleve
import (
"regexp"
"unicode"
"github.com/blevesearch/bleve/v2/analysis"
"github.com/blevesearch/bleve/v2/analysis/tokenizer/character"
"github.com/blevesearch/bleve/v2/registry"
)
const codeTokenizerName = "codeTokenizer"
func codeTokenizerConstructor(_ map[string]any, _ *registry.Cache) (analysis.Tokenizer, error) {
// Old code used "letter" tokenizer which doesn't support CJK.
// Here it still doesn't support CJK, since there is no usable CJK tokenizer at the moment.
return character.NewCharacterTokenizer(func(r rune) bool {
return unicode.IsLetter(r) || unicode.IsNumber(r)
}), nil
}
const codeTokenFilterName = "codeTokenFilter"
type codeTokenFilter struct {
re *regexp.Regexp
}
func (c codeTokenFilter) Filter(stream analysis.TokenStream) (ret analysis.TokenStream) {
// split one token to "letter" parts and "number" parts (to keep the old behavior).
// e.g.: input token="port123", then the output tokens are "port123", "port", "123"
for _, token := range stream {
ret = append(ret, token)
m := c.re.FindAllIndex(token.Term, -1)
if len(m) > 1 {
for _, it := range m {
p1, p2 := it[0], it[1]
t := &analysis.Token{
Start: token.Start + p1,
End: token.Start + p2,
Term: token.Term[p1:p2],
Position: token.Position,
Type: analysis.AlphaNumeric,
}
ret = append(ret, t)
}
}
}
return ret
}
func codeTokenFilterConstructor(_ map[string]any, _ *registry.Cache) (analysis.TokenFilter, error) {
return &codeTokenFilter{
re: regexp.MustCompile("[a-zA-Z]+|[0-9]+"),
}, nil
}