Update Chunk-Config mit Teilung der KBs und Tickets
release-tag / release-image (push) Successful in 1m37s
release-tag / release-image (push) Successful in 1m37s
This commit is contained in:
+72
-21
@@ -28,6 +28,7 @@ type ScoringConfig struct {
|
||||
ChunkWords int
|
||||
ChunkOverlap int
|
||||
MaxChunksPerDoc int
|
||||
MaxQueryChunks int
|
||||
}
|
||||
|
||||
type Store struct {
|
||||
@@ -56,7 +57,7 @@ type cacheFile struct {
|
||||
}
|
||||
|
||||
func DefaultScoringConfig() ScoringConfig {
|
||||
return ScoringConfig{SemanticWeight: .50, TitleWeight: .25, KeywordWeight: .15, CategoryWeight: .10, ChunkWords: 160, ChunkOverlap: 30, MaxChunksPerDoc: 24}
|
||||
return ScoringConfig{SemanticWeight: .50, TitleWeight: .25, KeywordWeight: .15, CategoryWeight: .10, ChunkWords: 160, ChunkOverlap: 30, MaxChunksPerDoc: 24, MaxQueryChunks: 64}
|
||||
}
|
||||
|
||||
func normalizeScoring(c ScoringConfig) ScoringConfig {
|
||||
@@ -73,6 +74,9 @@ func normalizeScoring(c ScoringConfig) ScoringConfig {
|
||||
if c.MaxChunksPerDoc <= 0 {
|
||||
c.MaxChunksPerDoc = d.MaxChunksPerDoc
|
||||
}
|
||||
if c.MaxQueryChunks <= 0 {
|
||||
c.MaxQueryChunks = d.MaxQueryChunks
|
||||
}
|
||||
return c
|
||||
}
|
||||
|
||||
@@ -539,44 +543,78 @@ func (s *Store) Search(ctx context.Context, text string, topK int, categorySets
|
||||
cats = categorySets[0]
|
||||
}
|
||||
|
||||
var queryVector []float64
|
||||
queryTitle, queryBody := splitQueryText(text)
|
||||
queryChunks := chunkText(queryBody, scoreCfg.ChunkWords, scoreCfg.ChunkOverlap, scoreCfg.MaxQueryChunks)
|
||||
if len(queryChunks) == 0 {
|
||||
queryChunks = chunkText(text, scoreCfg.ChunkWords, scoreCfg.ChunkOverlap, scoreCfg.MaxQueryChunks)
|
||||
}
|
||||
var queryVectors [][]float64
|
||||
var queryTitleVector []float64
|
||||
if s.rag && s.embedder != nil {
|
||||
q, err := s.embedder.Embed(ctx, []string{text})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
if len(queryChunks) > 0 {
|
||||
q, err := s.embedTexts(ctx, queryChunks, 64)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
queryVectors = q
|
||||
}
|
||||
if len(q) > 0 {
|
||||
queryVector = q[0]
|
||||
if strings.TrimSpace(queryTitle) != "" {
|
||||
tq, err := s.embedder.Embed(ctx, []string{queryTitle})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(tq) > 0 {
|
||||
queryTitleVector = tq[0]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
hits := make([]model.KnowledgeHit, 0, len(docs))
|
||||
for _, d := range docs {
|
||||
semantic, bestChunk := 0.0, ""
|
||||
semantic, bestChunk, bestQueryChunk := 0.0, "", ""
|
||||
semanticAvailable := false
|
||||
if len(queryVector) > 0 && len(chunkVecs[d.ID]) > 0 {
|
||||
if len(queryVectors) > 0 && len(chunkVecs[d.ID]) > 0 {
|
||||
semanticAvailable = true
|
||||
for i, v := range chunkVecs[d.ID] {
|
||||
score := clamp01(cosine(queryVector, v))
|
||||
if score > semantic || bestChunk == "" {
|
||||
semantic = score
|
||||
if i < len(chunks[d.ID]) {
|
||||
bestChunk = chunks[d.ID][i]
|
||||
for qi, qv := range queryVectors {
|
||||
for di, dv := range chunkVecs[d.ID] {
|
||||
score := clamp01(cosine(qv, dv))
|
||||
if score > semantic || bestChunk == "" {
|
||||
semantic = score
|
||||
if di < len(chunks[d.ID]) {
|
||||
bestChunk = chunks[d.ID][di]
|
||||
}
|
||||
if qi < len(queryChunks) {
|
||||
bestQueryChunk = queryChunks[qi]
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
} else if strings.TrimSpace(d.Text) != "" {
|
||||
semanticAvailable = true
|
||||
semantic = tokenF1(text, d.Text)
|
||||
bestChunk = d.Text
|
||||
docChunks := chunks[d.ID]
|
||||
if len(docChunks) == 0 {
|
||||
docChunks = []string{d.Text}
|
||||
}
|
||||
for _, qc := range queryChunks {
|
||||
for _, dc := range docChunks {
|
||||
score := tokenF1(qc, dc)
|
||||
if score > semantic || bestChunk == "" {
|
||||
semantic, bestChunk, bestQueryChunk = score, dc, qc
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
title := 0.0
|
||||
titleAvailable := strings.TrimSpace(d.Title) != ""
|
||||
if titleAvailable {
|
||||
title = titleSimilarity(text, d.Title)
|
||||
if len(queryVector) > 0 && len(titleVecs[d.ID]) > 0 {
|
||||
title = math.Max(title, clamp01(cosine(queryVector, titleVecs[d.ID])))
|
||||
titleQuery := queryTitle
|
||||
if strings.TrimSpace(titleQuery) == "" {
|
||||
titleQuery = text
|
||||
}
|
||||
title = titleSimilarity(titleQuery, d.Title)
|
||||
if len(queryTitleVector) > 0 && len(titleVecs[d.ID]) > 0 {
|
||||
title = math.Max(title, clamp01(cosine(queryTitleVector, titleVecs[d.ID])))
|
||||
}
|
||||
}
|
||||
keyword, keywordAvailable := keywordSimilarity(text, d.Keywords)
|
||||
@@ -587,7 +625,7 @@ func (s *Store) Search(ctx context.Context, text string, topK int, categorySets
|
||||
scorePart{keyword, scoreCfg.KeywordWeight, keywordAvailable},
|
||||
scorePart{category, scoreCfg.CategoryWeight, categoryAvailable},
|
||||
)
|
||||
hits = append(hits, model.KnowledgeHit{Doc: d, Score: total, SemanticScore: semantic, TitleScore: title, KeywordScore: keyword, CategoryScore: category, BestChunkExcerpt: excerpt(bestChunk, 280)})
|
||||
hits = append(hits, model.KnowledgeHit{Doc: d, Score: total, SemanticScore: semantic, TitleScore: title, KeywordScore: keyword, CategoryScore: category, BestChunkExcerpt: excerpt(bestChunk, 280), BestQueryExcerpt: excerpt(bestQueryChunk, 280), QueryChunkCount: len(queryChunks), DocumentChunkCount: len(chunks[d.ID])})
|
||||
}
|
||||
sort.SliceStable(hits, func(i, j int) bool {
|
||||
if hits[i].Score == hits[j].Score {
|
||||
@@ -719,6 +757,19 @@ func loadCache(path string) cacheFile {
|
||||
return cf
|
||||
}
|
||||
|
||||
func splitQueryText(text string) (title, body string) {
|
||||
text = strings.TrimSpace(text)
|
||||
if text == "" {
|
||||
return "", ""
|
||||
}
|
||||
if i := strings.IndexByte(text, '\n'); i >= 0 {
|
||||
title = strings.TrimSpace(text[:i])
|
||||
body = strings.TrimSpace(text[i+1:])
|
||||
return title, body
|
||||
}
|
||||
return text, text
|
||||
}
|
||||
|
||||
func chunkText(text string, words, overlap, maxChunks int) []string {
|
||||
parts := strings.Fields(strings.TrimSpace(text))
|
||||
if len(parts) == 0 {
|
||||
|
||||
@@ -201,3 +201,60 @@ func TestHybridScoringUsesChunksTitleKeywordsAndCategoryHints(t *testing.T) {
|
||||
t.Fatalf("wrong best chunk: %q", h.BestChunkExcerpt)
|
||||
}
|
||||
}
|
||||
|
||||
type hashTestEmbedder struct{}
|
||||
|
||||
func (hashTestEmbedder) Embed(_ context.Context, texts []string) ([][]float64, error) {
|
||||
out := make([][]float64, len(texts))
|
||||
for i, text := range texts {
|
||||
// Deterministic fixed-width vector: identical text -> identical vector;
|
||||
// unrelated text is unlikely to point in the same direction.
|
||||
v := make([]float64, 128)
|
||||
for pos, r := range []byte(strings.ToLower(strings.Join(strings.Fields(text), " "))) {
|
||||
idx := (int(r) + pos*31) % len(v)
|
||||
if (int(r)+pos)%2 == 0 {
|
||||
v[idx] += 1
|
||||
} else {
|
||||
v[idx] -= 1
|
||||
}
|
||||
}
|
||||
out[i] = v
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func TestLongQueryIsChunkedAndCanMatchIdenticalKnowledgeSection(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
data := t.TempDir()
|
||||
parts := make([]string, 0, 400)
|
||||
for i := 0; i < 400; i++ {
|
||||
parts = append(parts, "Benutzerkonto Anmeldung Sperrung Active Directory Diagnose Schritt")
|
||||
}
|
||||
body := strings.Join(parts, " ")
|
||||
doc := model.KnowledgeDoc{ID: "KB-LONG", Title: "Benutzerkonto gesperrt", Text: body, Source: "internal-kb", Language: "de-DE", CommunicationStyle: "formal"}
|
||||
b, _ := json.Marshal(doc)
|
||||
if err := os.WriteFile(filepath.Join(dir, "long.json"), b, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
s, err := Load(context.Background(), dir, data, hashTestEmbedder{}, true, []string{"internal-kb"}, ScoringConfig{SemanticWeight: 1, ChunkWords: 80, ChunkOverlap: 20, MaxChunksPerDoc: 24})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
hits, err := s.Search(context.Background(), "Benutzerkonto gesperrt\n"+body, 1)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(hits) != 1 {
|
||||
t.Fatalf("hits=%d", len(hits))
|
||||
}
|
||||
h := hits[0]
|
||||
if h.QueryChunkCount <= 1 || h.DocumentChunkCount <= 1 {
|
||||
t.Fatalf("expected both sides to be chunked: %+v", h)
|
||||
}
|
||||
if h.SemanticScore < 0.999999 {
|
||||
t.Fatalf("identical long body should contain an exact chunk match, got semantic=%f", h.SemanticScore)
|
||||
}
|
||||
if h.Score < 0.999999 {
|
||||
t.Fatalf("semantic-only hybrid should be ~1, got %f", h.Score)
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user