Update Chunk-Config mit Teilung der KBs und Tickets
release-tag / release-image (push) Successful in 1m37s

This commit is contained in:
2026-07-28 18:35:05 +02:00
parent 45b3d1b2df
commit de813aad3c
10 changed files with 183 additions and 30 deletions
+72 -21
View File
@@ -28,6 +28,7 @@ type ScoringConfig struct {
ChunkWords int
ChunkOverlap int
MaxChunksPerDoc int
MaxQueryChunks int
}
type Store struct {
@@ -56,7 +57,7 @@ type cacheFile struct {
}
func DefaultScoringConfig() ScoringConfig {
return ScoringConfig{SemanticWeight: .50, TitleWeight: .25, KeywordWeight: .15, CategoryWeight: .10, ChunkWords: 160, ChunkOverlap: 30, MaxChunksPerDoc: 24}
return ScoringConfig{SemanticWeight: .50, TitleWeight: .25, KeywordWeight: .15, CategoryWeight: .10, ChunkWords: 160, ChunkOverlap: 30, MaxChunksPerDoc: 24, MaxQueryChunks: 64}
}
func normalizeScoring(c ScoringConfig) ScoringConfig {
@@ -73,6 +74,9 @@ func normalizeScoring(c ScoringConfig) ScoringConfig {
if c.MaxChunksPerDoc <= 0 {
c.MaxChunksPerDoc = d.MaxChunksPerDoc
}
if c.MaxQueryChunks <= 0 {
c.MaxQueryChunks = d.MaxQueryChunks
}
return c
}
@@ -539,44 +543,78 @@ func (s *Store) Search(ctx context.Context, text string, topK int, categorySets
cats = categorySets[0]
}
var queryVector []float64
queryTitle, queryBody := splitQueryText(text)
queryChunks := chunkText(queryBody, scoreCfg.ChunkWords, scoreCfg.ChunkOverlap, scoreCfg.MaxQueryChunks)
if len(queryChunks) == 0 {
queryChunks = chunkText(text, scoreCfg.ChunkWords, scoreCfg.ChunkOverlap, scoreCfg.MaxQueryChunks)
}
var queryVectors [][]float64
var queryTitleVector []float64
if s.rag && s.embedder != nil {
q, err := s.embedder.Embed(ctx, []string{text})
if err != nil {
return nil, err
if len(queryChunks) > 0 {
q, err := s.embedTexts(ctx, queryChunks, 64)
if err != nil {
return nil, err
}
queryVectors = q
}
if len(q) > 0 {
queryVector = q[0]
if strings.TrimSpace(queryTitle) != "" {
tq, err := s.embedder.Embed(ctx, []string{queryTitle})
if err != nil {
return nil, err
}
if len(tq) > 0 {
queryTitleVector = tq[0]
}
}
}
hits := make([]model.KnowledgeHit, 0, len(docs))
for _, d := range docs {
semantic, bestChunk := 0.0, ""
semantic, bestChunk, bestQueryChunk := 0.0, "", ""
semanticAvailable := false
if len(queryVector) > 0 && len(chunkVecs[d.ID]) > 0 {
if len(queryVectors) > 0 && len(chunkVecs[d.ID]) > 0 {
semanticAvailable = true
for i, v := range chunkVecs[d.ID] {
score := clamp01(cosine(queryVector, v))
if score > semantic || bestChunk == "" {
semantic = score
if i < len(chunks[d.ID]) {
bestChunk = chunks[d.ID][i]
for qi, qv := range queryVectors {
for di, dv := range chunkVecs[d.ID] {
score := clamp01(cosine(qv, dv))
if score > semantic || bestChunk == "" {
semantic = score
if di < len(chunks[d.ID]) {
bestChunk = chunks[d.ID][di]
}
if qi < len(queryChunks) {
bestQueryChunk = queryChunks[qi]
}
}
}
}
} else if strings.TrimSpace(d.Text) != "" {
semanticAvailable = true
semantic = tokenF1(text, d.Text)
bestChunk = d.Text
docChunks := chunks[d.ID]
if len(docChunks) == 0 {
docChunks = []string{d.Text}
}
for _, qc := range queryChunks {
for _, dc := range docChunks {
score := tokenF1(qc, dc)
if score > semantic || bestChunk == "" {
semantic, bestChunk, bestQueryChunk = score, dc, qc
}
}
}
}
title := 0.0
titleAvailable := strings.TrimSpace(d.Title) != ""
if titleAvailable {
title = titleSimilarity(text, d.Title)
if len(queryVector) > 0 && len(titleVecs[d.ID]) > 0 {
title = math.Max(title, clamp01(cosine(queryVector, titleVecs[d.ID])))
titleQuery := queryTitle
if strings.TrimSpace(titleQuery) == "" {
titleQuery = text
}
title = titleSimilarity(titleQuery, d.Title)
if len(queryTitleVector) > 0 && len(titleVecs[d.ID]) > 0 {
title = math.Max(title, clamp01(cosine(queryTitleVector, titleVecs[d.ID])))
}
}
keyword, keywordAvailable := keywordSimilarity(text, d.Keywords)
@@ -587,7 +625,7 @@ func (s *Store) Search(ctx context.Context, text string, topK int, categorySets
scorePart{keyword, scoreCfg.KeywordWeight, keywordAvailable},
scorePart{category, scoreCfg.CategoryWeight, categoryAvailable},
)
hits = append(hits, model.KnowledgeHit{Doc: d, Score: total, SemanticScore: semantic, TitleScore: title, KeywordScore: keyword, CategoryScore: category, BestChunkExcerpt: excerpt(bestChunk, 280)})
hits = append(hits, model.KnowledgeHit{Doc: d, Score: total, SemanticScore: semantic, TitleScore: title, KeywordScore: keyword, CategoryScore: category, BestChunkExcerpt: excerpt(bestChunk, 280), BestQueryExcerpt: excerpt(bestQueryChunk, 280), QueryChunkCount: len(queryChunks), DocumentChunkCount: len(chunks[d.ID])})
}
sort.SliceStable(hits, func(i, j int) bool {
if hits[i].Score == hits[j].Score {
@@ -719,6 +757,19 @@ func loadCache(path string) cacheFile {
return cf
}
func splitQueryText(text string) (title, body string) {
text = strings.TrimSpace(text)
if text == "" {
return "", ""
}
if i := strings.IndexByte(text, '\n'); i >= 0 {
title = strings.TrimSpace(text[:i])
body = strings.TrimSpace(text[i+1:])
return title, body
}
return text, text
}
func chunkText(text string, words, overlap, maxChunks int) []string {
parts := strings.Fields(strings.TrimSpace(text))
if len(parts) == 0 {
+57
View File
@@ -201,3 +201,60 @@ func TestHybridScoringUsesChunksTitleKeywordsAndCategoryHints(t *testing.T) {
t.Fatalf("wrong best chunk: %q", h.BestChunkExcerpt)
}
}
type hashTestEmbedder struct{}
func (hashTestEmbedder) Embed(_ context.Context, texts []string) ([][]float64, error) {
out := make([][]float64, len(texts))
for i, text := range texts {
// Deterministic fixed-width vector: identical text -> identical vector;
// unrelated text is unlikely to point in the same direction.
v := make([]float64, 128)
for pos, r := range []byte(strings.ToLower(strings.Join(strings.Fields(text), " "))) {
idx := (int(r) + pos*31) % len(v)
if (int(r)+pos)%2 == 0 {
v[idx] += 1
} else {
v[idx] -= 1
}
}
out[i] = v
}
return out, nil
}
func TestLongQueryIsChunkedAndCanMatchIdenticalKnowledgeSection(t *testing.T) {
dir := t.TempDir()
data := t.TempDir()
parts := make([]string, 0, 400)
for i := 0; i < 400; i++ {
parts = append(parts, "Benutzerkonto Anmeldung Sperrung Active Directory Diagnose Schritt")
}
body := strings.Join(parts, " ")
doc := model.KnowledgeDoc{ID: "KB-LONG", Title: "Benutzerkonto gesperrt", Text: body, Source: "internal-kb", Language: "de-DE", CommunicationStyle: "formal"}
b, _ := json.Marshal(doc)
if err := os.WriteFile(filepath.Join(dir, "long.json"), b, 0o644); err != nil {
t.Fatal(err)
}
s, err := Load(context.Background(), dir, data, hashTestEmbedder{}, true, []string{"internal-kb"}, ScoringConfig{SemanticWeight: 1, ChunkWords: 80, ChunkOverlap: 20, MaxChunksPerDoc: 24})
if err != nil {
t.Fatal(err)
}
hits, err := s.Search(context.Background(), "Benutzerkonto gesperrt\n"+body, 1)
if err != nil {
t.Fatal(err)
}
if len(hits) != 1 {
t.Fatalf("hits=%d", len(hits))
}
h := hits[0]
if h.QueryChunkCount <= 1 || h.DocumentChunkCount <= 1 {
t.Fatalf("expected both sides to be chunked: %+v", h)
}
if h.SemanticScore < 0.999999 {
t.Fatalf("identical long body should contain an exact chunk match, got semantic=%f", h.SemanticScore)
}
if h.Score < 0.999999 {
t.Fatalf("semantic-only hybrid should be ~1, got %f", h.Score)
}
}