Cluster-Work

This commit is contained in:
2026-08-07 21:54:37 +02:00
parent f8bf8c944e
commit dc2af7ca75
761 changed files with 939 additions and 35364 deletions

View File

@@ -1120,7 +1120,7 @@ func mergeRunMetrics(metrics map[string]any, activity model.Activity) {
if metrics == nil {
return
}
for _, keys := range [][]string{{"comparisons", "candidate_comparisons"}, {"checked"}, {"relations_created"}, {"articles_created"}, {"articles_skipped"}, {"research_search_results", "result_count"}, {"research_fetched", "pages_fetched"}, {"research_accepted", "evidence_count"}, {"research_rejected"}, {"queries_executed"}, {"batch_count"}, {"duration_ms"}} {
for _, keys := range [][]string{{"comparisons", "candidate_comparisons"}, {"exact_comparisons"}, {"coarse_comparisons"}, {"candidate_pool"}, {"checked"}, {"relations_created"}, {"articles_created"}, {"articles_skipped"}, {"research_search_results", "result_count"}, {"research_fetched", "pages_fetched"}, {"research_accepted", "evidence_count"}, {"research_rejected"}, {"queries_executed"}, {"batch_count"}, {"duration_ms"}} {
name := keys[0]
value := metadataNumber(activity.Metadata, keys...)
if value == 0 {
@@ -1138,6 +1138,9 @@ func mergeRunMetrics(metrics map[string]any, activity model.Activity) {
if modelName := metadataString(activity.Metadata, "model"); modelName != "" {
metrics["model"] = modelName
}
if mode := metadataString(activity.Metadata, "processing_mode"); mode != "" {
metrics["processing_mode"] = mode
}
}
func metadataNumber(metadata map[string]any, keys ...string) float64 {

View File

@@ -0,0 +1,355 @@
package graph
import (
"container/heap"
"math/bits"
"sort"
"github.com/local/glpi-neural-brain/internal/model"
)
// ClusterSearchStats separates cheap semantic-hash work from expensive exact
// cosine work. CoarseComparisons are Hamming-distance operations; only
// ExactComparisons execute a full embedding dot product.
type ClusterSearchStats struct {
IndexedNodes int `json:"indexed_nodes"`
CoarseComparisons int `json:"coarse_comparisons"`
ExactComparisons int `json:"exact_comparisons"`
CandidatePool int `json:"candidate_pool"`
HashBits int `json:"hash_bits"`
HashTables int `json:"hash_tables"`
}
type semanticHashEntry struct {
node model.Node
vector []float32
signatures []uint64
}
type coarseCandidate struct {
index int
distance int
}
type coarseMaxHeap []coarseCandidate
func (h coarseMaxHeap) Len() int { return len(h) }
func (h coarseMaxHeap) Less(i, j int) bool {
if h[i].distance == h[j].distance {
return h[i].index > h[j].index
}
return h[i].distance > h[j].distance
}
func (h coarseMaxHeap) Swap(i, j int) { h[i], h[j] = h[j], h[i] }
func (h *coarseMaxHeap) Push(x any) { *h = append(*h, x.(coarseCandidate)) }
func (h *coarseMaxHeap) Pop() any {
old := *h
n := len(old)
x := old[n-1]
*h = old[:n-1]
return x
}
func normalizeClusterConfig(hashBits, hashTables, candidateLimit int) (int, int, int) {
if hashBits < 8 {
hashBits = 8
}
if hashBits > 63 {
hashBits = 63
}
if hashTables < 1 {
hashTables = 1
}
if hashTables > 4 {
hashTables = 4
}
if candidateLimit < 16 {
candidateLimit = 16
}
return hashBits, hashTables, candidateLimit
}
// sparseSemanticHash is a deterministic sparse random-projection hash. Each
// bit samples only six embedding dimensions, so building the coarse index is
// orders of magnitude cheaper than N full 768-dimensional cosine products.
func sparseSemanticHash(v []float32, table, hashBits int) uint64 {
if len(v) == 0 {
return 0
}
var signature uint64
seedBase := uint64(0x9e3779b97f4a7c15) ^ uint64(table+1)*0xbf58476d1ce4e5b9
for bit := 0; bit < hashBits; bit++ {
seed := mix64(seedBase ^ uint64(bit+1)*0x94d049bb133111eb)
var sum float32
for sample := 0; sample < 6; sample++ {
seed = mix64(seed + uint64(sample+1)*0x9e3779b97f4a7c15)
idx := int(seed % uint64(len(v)))
if seed&(1<<63) != 0 {
sum -= v[idx]
} else {
sum += v[idx]
}
}
if sum >= 0 {
signature |= 1 << bit
}
}
return signature
}
func mix64(x uint64) uint64 {
x ^= x >> 30
x *= 0xbf58476d1ce4e5b9
x ^= x >> 27
x *= 0x94d049bb133111eb
x ^= x >> 31
return x
}
func signaturesFor(v []float32, hashBits, hashTables int) []uint64 {
out := make([]uint64, hashTables)
for table := 0; table < hashTables; table++ {
out[table] = sparseSemanticHash(v, table, hashBits)
}
return out
}
func signatureDistance(a, b []uint64) int {
n := len(a)
if len(b) < n {
n = len(b)
}
distance := 0
for i := 0; i < n; i++ {
distance += bits.OnesCount64(a[i] ^ b[i])
}
return distance
}
func pushBestCoarse(h *coarseMaxHeap, candidate coarseCandidate, limit int) {
if h.Len() < limit {
heap.Push(h, candidate)
return
}
worst := (*h)[0]
if candidate.distance < worst.distance || (candidate.distance == worst.distance && candidate.index < worst.index) {
heap.Pop(h)
heap.Push(h, candidate)
}
}
// NextPairClusteredScopedDepth is the low-resource alternative to
// NextPairScopedDepth. It keeps the rotating anchor behaviour but ranks the
// corpus with cheap semantic hashes and computes exact cosine only on top-K.
func (s *Store) NextPairClusteredScopedDepth(min float64, anchorLimit int, filter NodeFilter, maxAIDepth, hashBits, hashTables, candidateLimit int) (model.Node, model.Node, float64, bool, ClusterSearchStats) {
hashBits, hashTables, candidateLimit = normalizeClusterConfig(hashBits, hashTables, candidateLimit)
stats := ClusterSearchStats{HashBits: hashBits, HashTables: hashTables}
s.mu.Lock()
defer s.mu.Unlock()
entries := make([]semanticHashEntry, 0, len(s.nodes))
for _, n := range s.nodes {
if n.Kind != "knowledge" && n.Kind != "ai-think" {
continue
}
if !filter.Matches(n) {
continue
}
if n.Kind == "ai-think" && maxAIDepth > 0 && graphNodeGenerationDepth(n) >= maxAIDepth {
continue
}
v, ok := s.vectors[n.ID]
if !ok || len(v) == 0 {
continue
}
entries = append(entries, semanticHashEntry{node: n, vector: v})
}
if len(entries) < 2 {
return model.Node{}, model.Node{}, 0, false, stats
}
sort.Slice(entries, func(i, j int) bool { return entries[i].node.ID < entries[j].node.ID })
for i := range entries {
entries[i].signatures = signaturesFor(entries[i].vector, hashBits, hashTables)
}
stats.IndexedNodes = len(entries)
if anchorLimit <= 0 || anchorLimit > len(entries) {
anchorLimit = len(entries)
}
start := s.pairCursor % len(entries)
anchorIDs := make(map[string]struct{}, anchorLimit)
for step := 0; step < anchorLimit; step++ {
anchorIDs[entries[(start+step)%len(entries)].node.ID] = struct{}{}
}
// Only materialise blocked neighbours for current anchors. This avoids a
// full pair-key allocation for every one of the ~180k graph edges.
blocked := make(map[string]map[string]struct{}, anchorLimit)
for _, edge := range s.edges {
if _, ok := anchorIDs[edge.Source]; ok {
if blocked[edge.Source] == nil {
blocked[edge.Source] = map[string]struct{}{}
}
blocked[edge.Source][edge.Target] = struct{}{}
}
if _, ok := anchorIDs[edge.Target]; ok {
if blocked[edge.Target] == nil {
blocked[edge.Target] = map[string]struct{}{}
}
blocked[edge.Target][edge.Source] = struct{}{}
}
}
best := -1.0
var bestA, bestB model.Node
evaluated := make(map[string]struct{}, anchorLimit*candidateLimit)
for step := 0; step < anchorLimit; step++ {
i := (start + step) % len(entries)
left := entries[i]
h := &coarseMaxHeap{}
heap.Init(h)
for j := range entries {
if i == j {
continue
}
right := entries[j]
if left.node.Kind == "ai-think" && right.node.Kind == "ai-think" {
continue
}
if neighbors := blocked[left.node.ID]; neighbors != nil {
if _, exists := neighbors[right.node.ID]; exists {
continue
}
}
stats.CoarseComparisons++
pushBestCoarse(h, coarseCandidate{index: j, distance: signatureDistance(left.signatures, right.signatures)}, candidateLimit)
}
candidates := make([]coarseCandidate, h.Len())
for k := len(candidates) - 1; k >= 0; k-- {
candidates[k] = heap.Pop(h).(coarseCandidate)
}
stats.CandidatePool += len(candidates)
for _, candidate := range candidates {
right := entries[candidate.index]
key := pairKey(left.node.ID, right.node.ID)
if _, seen := evaluated[key]; seen {
continue
}
evaluated[key] = struct{}{}
if len(left.vector) != len(right.vector) {
continue
}
stats.ExactComparisons++
score := cosine32(left.vector, right.vector)
if score >= min && score > best {
best, bestA, bestB = score, left.node, right.node
}
}
}
s.pairCursor = (start + anchorLimit) % len(entries)
return bestA, bestB, best, best >= 0, stats
}
// SimilarClusteredFiltered performs approximate nearest-neighbour retrieval by
// semantic hash followed by exact cosine on a bounded shortlist.
func (s *Store) SimilarClusteredFiltered(query []float64, limit, candidateLimit int, filter NodeFilter, maxAIDepth, hashBits, hashTables int) ([]model.Hit, ClusterSearchStats) {
hashBits, hashTables, candidateLimit = normalizeClusterConfig(hashBits, hashTables, candidateLimit)
if limit < 1 {
limit = 1
}
if candidateLimit < limit {
candidateLimit = limit
}
stats := ClusterSearchStats{HashBits: hashBits, HashTables: hashTables}
q := make([]float32, len(query))
for i, value := range query {
q[i] = float32(value)
}
qsig := signaturesFor(q, hashBits, hashTables)
s.mu.RLock()
defer s.mu.RUnlock()
entries := make([]semanticHashEntry, 0, len(s.nodes))
for _, n := range s.nodes {
if n.Kind != "knowledge" && n.Kind != "ai-think" {
continue
}
if !filter.Matches(n) {
continue
}
if n.Kind == "ai-think" && maxAIDepth > 0 && graphNodeGenerationDepth(n) >= maxAIDepth {
continue
}
v, ok := s.vectors[n.ID]
if !ok || len(v) != len(q) {
continue
}
entries = append(entries, semanticHashEntry{node: n, vector: v})
}
sort.Slice(entries, func(i, j int) bool { return entries[i].node.ID < entries[j].node.ID })
stats.IndexedNodes = len(entries)
h := &coarseMaxHeap{}
heap.Init(h)
for i := range entries {
entries[i].signatures = signaturesFor(entries[i].vector, hashBits, hashTables)
stats.CoarseComparisons++
pushBestCoarse(h, coarseCandidate{index: i, distance: signatureDistance(qsig, entries[i].signatures)}, candidateLimit)
}
candidates := make([]coarseCandidate, h.Len())
for i := len(candidates) - 1; i >= 0; i-- {
candidates[i] = heap.Pop(h).(coarseCandidate)
}
stats.CandidatePool = len(candidates)
hits := make([]model.Hit, 0, len(candidates))
for _, candidate := range candidates {
entry := entries[candidate.index]
stats.ExactComparisons++
hits = append(hits, model.Hit{NodeID: entry.node.ID, Label: entry.node.Label, Score: cosine32(q, entry.vector), Kind: entry.node.Kind, Status: entry.node.Status})
}
sort.Slice(hits, func(i, j int) bool {
if hits[i].Score == hits[j].Score {
return hits[i].NodeID < hits[j].NodeID
}
return hits[i].Score > hits[j].Score
})
if len(hits) > limit {
hits = hits[:limit]
}
return hits, stats
}
// NeighborScores returns only non-taxonomy graph links incident to the seeds,
// without allocating a full graph Snapshot.
func (s *Store) NeighborScores(seedIDs map[string]bool) map[string]float64 {
s.mu.RLock()
defer s.mu.RUnlock()
out := map[string]float64{}
for _, edge := range s.edges {
if edge.Status == "rejected" || clusterTaxonomyEdge(edge.Type) {
continue
}
weight := edge.Confidence
if edge.Weight > weight {
weight = edge.Weight
}
if weight < .2 {
weight = .2
}
if seedIDs[edge.Source] {
out[edge.Target] += weight
}
if seedIDs[edge.Target] {
out[edge.Source] += weight
}
}
return out
}
func clusterTaxonomyEdge(edgeType string) bool {
switch edgeType {
case "categorized_as", "mentions", "derived_from":
return true
default:
return false
}
}

View File

@@ -0,0 +1,54 @@
package graph
import (
"fmt"
"testing"
"github.com/local/glpi-neural-brain/internal/model"
)
func newSemanticTestStore(t *testing.T, count int) *Store {
t.Helper()
s := &Store{nodes: map[string]model.Node{}, edges: map[string]model.Edge{}, vectors: map[string][]float32{}, dirtyNodes: map[string]uint64{}, dirtyEdges: map[string]uint64{}, dirtyVectors: map[string]uint64{}, deletedNodes: map[string]uint64{}, deletedEdges: map[string]uint64{}, deletedVectors: map[string]uint64{}}
for i := 0; i < count; i++ {
id := fmt.Sprintf("n-%03d", i)
v := make([]float64, 16)
v[i%16] = 1
v[(i*7+3)%16] += float64((i%5)+1) * .03
s.UpsertNode(model.Node{ID: id, Kind: "knowledge", Status: "production", Label: id})
s.SetVector(id, v)
}
return s
}
func TestClusteredPairSearchUsesFarFewerExactCosines(t *testing.T) {
s := newSemanticTestStore(t, 240)
s.SetVector("n-000", []float64{1, .02, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0})
s.SetVector("n-001", []float64{1, .021, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0})
_, _, sim, ok, stats := s.NextPairClusteredScopedDepth(.95, 12, NodeFilter{}, 2, 24, 2, 24)
if !ok || sim < .95 {
t.Fatalf("expected clustered candidate, ok=%v sim=%.4f stats=%+v", ok, sim, stats)
}
if stats.ExactComparisons <= 0 || stats.ExactComparisons > 12*24 {
t.Fatalf("unexpected exact work: %+v", stats)
}
if stats.CoarseComparisons < 2000 {
t.Fatalf("expected cheap coarse scan over corpus, got %+v", stats)
}
if stats.ExactComparisons*5 >= stats.CoarseComparisons {
t.Fatalf("cluster mode did not reduce expensive comparisons enough: %+v", stats)
}
}
func TestClusteredSimilarFindsNearDuplicate(t *testing.T) {
s := newSemanticTestStore(t, 160)
target := []float64{.91, .31, .12, .04, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0}
s.SetVector("n-055", target)
hits, stats := s.SimilarClusteredFiltered(target, 8, 48, NodeFilter{}, 2, 24, 2)
if len(hits) == 0 || hits[0].NodeID != "n-055" || hits[0].Score < .999 {
t.Fatalf("near duplicate not ranked first: hits=%+v stats=%+v", hits, stats)
}
if stats.ExactComparisons > 48 {
t.Fatalf("too many exact comparisons: %+v", stats)
}
}