Agent-Mode mit Aufgabenteilung und automatischer Recherche
This commit is contained in:
@@ -139,7 +139,7 @@ func (e *Engine) synthesizeKnowledgeArticle(ctx context.Context, trigger string,
|
||||
researchStrategy := e.effectiveArticleResearchStrategy()
|
||||
freshness := detectArticleFreshnessNeed(plan, relation, selected)
|
||||
initialWebResearch := false
|
||||
if e.ResearchEnabledForRuntime() {
|
||||
if e.evidenceAcquisitionEnabled() {
|
||||
switch researchStrategy {
|
||||
case "always":
|
||||
collected, report, researchErr := e.collectResearchMaterialForArticle(ctx, trigger, plan.SourceNodeIDs, selected, plan, brief, researchResults)
|
||||
@@ -160,7 +160,7 @@ func (e *Engine) synthesizeKnowledgeArticle(ctx context.Context, trigger string,
|
||||
}
|
||||
}
|
||||
}
|
||||
e.Broker.Publish(model.Activity{Type: "article.research.strategy", Source: "brain", Phase: "knowledge-research-routing", NodeIDs: plan.SourceNodeIDs, Message: fmt.Sprintf("Artikelrecherche: %s · initiales Webmaterial: %t", researchStrategy, initialWebResearch), Strength: .44, Metadata: map[string]any{"trigger": trigger, "research_strategy": researchStrategy, "freshness_required": freshness.Required, "freshness_reason": freshness.Reason, "initial_web_research": initialWebResearch, "initial_research_queries": researchReport.Queries, "initial_research_fetched": researchReport.Fetched}})
|
||||
e.Broker.Publish(model.Activity{Type: "article.research.strategy", Source: "brain", Phase: "knowledge-research-routing", NodeIDs: plan.SourceNodeIDs, Message: fmt.Sprintf("Artikelrecherche: %s · initiales Webmaterial: %t", researchStrategy, initialWebResearch), Strength: .44, Metadata: map[string]any{"trigger": trigger, "research_strategy": researchStrategy, "freshness_required": freshness.Required, "freshness_reason": freshness.Reason, "initial_web_research": initialWebResearch, "initial_research_queries": researchReport.Queries, "source_inbox_results": researchReport.InboxResults, "initial_research_fetched": researchReport.Fetched}})
|
||||
|
||||
e.Broker.Publish(model.Activity{Type: "article.draft.started", Source: "brain", Phase: "knowledge-synthesis", NodeIDs: plan.SourceNodeIDs, Message: fmt.Sprintf("%s erstellt zuerst aus dem verfügbaren Evidenzsatz einen KB-Artikel; Webrecherche erfolgt nur bei Bedarf", e.Cfg.ArticleSynthesisModel), Strength: .95, Metadata: map[string]any{"trigger": trigger, "action": plan.Action, "article_type": plan.ArticleType, "target_article_id": plan.TargetArticleID, "source_count": len(selected), "research_material_count": len(researchResults), "research_rounds": researchReport.Rounds, "research_fetched": researchReport.Fetched, "research_strategy": researchStrategy, "freshness_required": freshness.Required, "synthesis_model": e.Cfg.ArticleSynthesisModel, "review_model": e.Cfg.ArticleReviewModel, "generation_depth": generationDepth}})
|
||||
|
||||
@@ -187,7 +187,7 @@ func (e *Engine) synthesizeKnowledgeArticle(ctx context.Context, trigger string,
|
||||
// In adaptive mode Gemma is allowed to say that the internal KB is not
|
||||
// sufficient. Only then do we pay for a focused Web round. This happens
|
||||
// before the reviewer, so an obvious evidence gap does not waste a Qwen call.
|
||||
if researchStrategy == "adaptive" && !authorResearchAttempted && !freshness.Required && e.ResearchEnabledForRuntime() && (content.ResearchNeeded || content.FreshnessSensitive) {
|
||||
if researchStrategy == "adaptive" && !authorResearchAttempted && !freshness.Required && e.evidenceAcquisitionEnabled() && (content.ResearchNeeded || content.FreshnessSensitive) {
|
||||
queries := append([]string{}, content.ResearchQueries...)
|
||||
if content.ResearchNeeded && len(queries) == 0 {
|
||||
queries = append(queries, plan.ResearchQuery)
|
||||
@@ -235,11 +235,11 @@ func (e *Engine) synthesizeKnowledgeArticle(ctx context.Context, trigger string,
|
||||
break
|
||||
}
|
||||
repairAttempts++
|
||||
if len(quality.MissingEvidenceQueries) > 0 && e.ResearchEnabledForRuntime() {
|
||||
if len(quality.MissingEvidenceQueries) > 0 && e.evidenceAcquisitionEnabled() {
|
||||
e.Broker.Publish(model.Activity{Type: "article.review.research.started", Source: "brain", Phase: "quality-repair-research", NodeIDs: draft.SourceNodeIDs, Message: "Der Reviewer hat konkrete unbelegte Aussagen gefunden · nur diese Punkte werden nachrecherchiert", Strength: .84, Metadata: map[string]any{"trigger": trigger, "repair_round": repairAttempts, "queries": quality.MissingEvidenceQueries, "review_model": e.Cfg.ArticleReviewModel}})
|
||||
additional, repairReport := e.collectReviewerRepairResearch(ctx, trigger, plan.SourceNodeIDs, quality.MissingEvidenceQueries, repairAttempts, attemptedRepairURLs)
|
||||
researchResults = uniqueResearchEvidence(append(researchResults, additional...))
|
||||
e.Broker.Publish(model.Activity{Type: "article.review.research.completed", Source: "brain", Phase: "quality-repair-research", NodeIDs: draft.SourceNodeIDs, Message: fmt.Sprintf("Gezielte Nachrecherche beendet · %d zusätzliche Volltextquellen", len(additional)), Strength: .82, Metadata: map[string]any{"trigger": trigger, "repair_round": repairAttempts, "new_material": len(additional), "queries": repairReport.Queries, "search_results": repairReport.SearchResults, "fetched": repairReport.Fetched}})
|
||||
e.Broker.Publish(model.Activity{Type: "article.review.research.completed", Source: "brain", Phase: "quality-repair-research", NodeIDs: draft.SourceNodeIDs, Message: fmt.Sprintf("Gezielte Nachrecherche beendet · %d zusätzliche Volltextquellen", len(additional)), Strength: .82, Metadata: map[string]any{"trigger": trigger, "repair_round": repairAttempts, "new_material": len(additional), "queries": repairReport.Queries, "source_inbox_results": repairReport.InboxResults, "search_results": repairReport.SearchResults, "fetched": repairReport.Fetched}})
|
||||
}
|
||||
reviewCopy := quality
|
||||
reviewFeedback = &reviewCopy
|
||||
|
||||
@@ -145,13 +145,27 @@ func (e *Engine) collectAdaptiveInitialResearch(ctx context.Context, trigger str
|
||||
out := make([]model.ResearchResult, 0)
|
||||
report := articleResearchReport{Rounds: 1}
|
||||
for i, query := range queries {
|
||||
report.Queries++
|
||||
freshnessSensitive := containsFreshnessLanguage(query)
|
||||
inbox := e.sourceInboxResearch(ctx, query, fetchCap, freshnessSensitive)
|
||||
if len(inbox) > 0 {
|
||||
report.InboxResults += len(inbox)
|
||||
report.Accepted += len(inbox)
|
||||
out = uniqueResearchEvidence(append(out, inbox...))
|
||||
}
|
||||
if len(inbox) >= e.Cfg.SourceInboxMinResults || !e.ResearchEnabledForRuntime() {
|
||||
continue
|
||||
}
|
||||
remaining := fetchCap - len(inbox)
|
||||
if remaining < 1 {
|
||||
remaining = 1
|
||||
}
|
||||
language := "en-US"
|
||||
if looksGermanResearchQuery(query) {
|
||||
language = "de-DE"
|
||||
}
|
||||
question := model.ResearchQuestion{GapID: fmt.Sprintf("ADAPTIVE-%d", i+1), Question: query, Critical: true, ExpectActionable: containsActionableLanguage(query)}
|
||||
report.Queries++
|
||||
items, stats := e.executeArticleResearchQueryForSynthesis(ctx, trigger, nodeIDs, question, query, language, 1, attemptedURLs, fetchCap)
|
||||
items, stats := e.executeArticleResearchQueryForSynthesis(ctx, trigger, nodeIDs, question, query, language, 1, attemptedURLs, remaining)
|
||||
report.SearchResults += stats.SearchResults
|
||||
report.Fetched += stats.Fetched
|
||||
report.Accepted += stats.Accepted
|
||||
@@ -307,7 +321,9 @@ func (e *Engine) materializeGroundedResearchEvidence(articleID string, sources [
|
||||
}
|
||||
categories := categoriesFromArticleSources(sources)
|
||||
ids := make([]string, 0, len(results))
|
||||
usedURLs := make([]string, 0, len(results))
|
||||
for _, result := range results {
|
||||
usedURLs = append(usedURLs, result.URL)
|
||||
id := graph.ID("external", result.URL)
|
||||
path, contentHash, err := e.queueResearchEvidence(result)
|
||||
if err != nil {
|
||||
@@ -322,5 +338,8 @@ func (e *Engine) materializeGroundedResearchEvidence(articleID string, sources [
|
||||
e.Graph.UpsertNode(node)
|
||||
ids = append(ids, id)
|
||||
}
|
||||
if e.SourceInbox != nil && len(usedURLs) > 0 {
|
||||
_ = e.SourceInbox.MarkUsed(context.Background(), usedURLs)
|
||||
}
|
||||
return unique(ids)
|
||||
}
|
||||
|
||||
@@ -25,7 +25,7 @@ func (e *Engine) collectResearchMaterialForArticle(ctx context.Context, trigger
|
||||
}
|
||||
}
|
||||
|
||||
if !e.ResearchEnabledForRuntime() {
|
||||
if !e.evidenceAcquisitionEnabled() {
|
||||
return material, report, nil
|
||||
}
|
||||
maxRounds := e.Cfg.ArticleResearchRounds
|
||||
@@ -56,6 +56,23 @@ func (e *Engine) collectResearchMaterialForArticle(ctx context.Context, trigger
|
||||
if report.Queries >= maxQueries {
|
||||
break
|
||||
}
|
||||
inboxMaterial := e.sourceInboxResearch(ctx, question.Question, e.Cfg.ArticleResearchFetchResults, containsFreshnessLanguage(question.Question))
|
||||
if len(inboxMaterial) > 0 {
|
||||
report.InboxResults += len(inboxMaterial)
|
||||
report.Accepted += len(inboxMaterial)
|
||||
for _, item := range inboxMaterial {
|
||||
key := canonicalResearchURL(item.URL)
|
||||
if key == "" || seenURLs[key] {
|
||||
continue
|
||||
}
|
||||
seenURLs[key] = true
|
||||
material = append(material, item)
|
||||
collectedThisRound++
|
||||
}
|
||||
}
|
||||
if len(inboxMaterial) >= e.Cfg.SourceInboxMinResults || !e.ResearchEnabledForRuntime() {
|
||||
continue
|
||||
}
|
||||
lease, reused, err := e.beginResearchIntent(ctx, "synthesis-material", question.Question)
|
||||
if err != nil {
|
||||
return material, report, err
|
||||
@@ -201,13 +218,26 @@ func (e *Engine) collectReviewerRepairResearch(ctx context.Context, trigger stri
|
||||
if query == "" {
|
||||
continue
|
||||
}
|
||||
report.Queries++
|
||||
inbox := e.sourceInboxResearch(ctx, query, fetchCap, containsFreshnessLanguage(query))
|
||||
if len(inbox) > 0 {
|
||||
report.InboxResults += len(inbox)
|
||||
report.Accepted += len(inbox)
|
||||
out = uniqueResearchEvidence(append(out, inbox...))
|
||||
}
|
||||
if len(inbox) >= e.Cfg.SourceInboxMinResults || !e.ResearchEnabledForRuntime() {
|
||||
continue
|
||||
}
|
||||
remaining := fetchCap - len(inbox)
|
||||
if remaining < 1 {
|
||||
remaining = 1
|
||||
}
|
||||
language := "en-US"
|
||||
if looksGermanResearchQuery(query) {
|
||||
language = "de-DE"
|
||||
}
|
||||
question := model.ResearchQuestion{GapID: fmt.Sprintf("REVIEW-%d", i+1), Question: query, Critical: true, ExpectActionable: containsActionableLanguage(query)}
|
||||
report.Queries++
|
||||
items, stats := e.executeArticleResearchQueryForSynthesis(ctx, trigger, nodeIDs, question, query, language, round, attemptedURLs, fetchCap)
|
||||
items, stats := e.executeArticleResearchQueryForSynthesis(ctx, trigger, nodeIDs, question, query, language, round, attemptedURLs, remaining)
|
||||
report.SearchResults += stats.SearchResults
|
||||
report.Fetched += stats.Fetched
|
||||
report.Accepted += stats.Accepted
|
||||
|
||||
@@ -26,6 +26,7 @@ type articleResearchReport struct {
|
||||
Rejected int
|
||||
FetchFailed int
|
||||
SearchFailed int
|
||||
InboxResults int
|
||||
}
|
||||
|
||||
type rankedResearchCandidate struct {
|
||||
@@ -492,6 +493,7 @@ type queryExecutionStats struct {
|
||||
Rejected int
|
||||
FetchFailed int
|
||||
SearchFailed int
|
||||
InboxResults int
|
||||
}
|
||||
|
||||
func (e *Engine) executeArticleResearchQuery(ctx context.Context, trigger string, nodeIDs []string, question model.ResearchQuestion, query, language string, round int, attemptedURLs map[string]bool, fetchCaps ...int) ([]model.ResearchResult, queryExecutionStats) {
|
||||
|
||||
@@ -24,6 +24,7 @@ import (
|
||||
"github.com/local/glpi-neural-brain/internal/ollama"
|
||||
"github.com/local/glpi-neural-brain/internal/persist"
|
||||
"github.com/local/glpi-neural-brain/internal/research"
|
||||
"github.com/local/glpi-neural-brain/internal/sourceagent"
|
||||
"github.com/local/glpi-neural-brain/internal/workqueue"
|
||||
)
|
||||
|
||||
@@ -63,6 +64,7 @@ type Engine struct {
|
||||
Scanner *ingest.KnowledgeScanner
|
||||
GLPIKB *ingest.GLPIKBSyncer
|
||||
Persistence *persist.Coordinator
|
||||
SourceInbox *sourceagent.Store
|
||||
|
||||
mu sync.Mutex
|
||||
stateMu sync.RWMutex
|
||||
@@ -329,6 +331,9 @@ func (e *Engine) Start(ctx context.Context) {
|
||||
}
|
||||
go e.idle(ctx)
|
||||
e.startAutonomousResearch(ctx)
|
||||
if e.SourceInbox != nil && e.Cfg.SourceInboxEnabled {
|
||||
go e.sourceInboxLoop(ctx)
|
||||
}
|
||||
}
|
||||
|
||||
func (e *Engine) enrichmentScheduler(ctx context.Context) {
|
||||
@@ -1041,6 +1046,11 @@ func (e *Engine) Status() map[string]any {
|
||||
}
|
||||
e.stateMu.RUnlock()
|
||||
status["autonomous_research"] = e.AutonomousResearchStatus(context.Background())
|
||||
if e.SourceInbox != nil {
|
||||
if inbox, err := e.SourceInbox.Stats(context.Background()); err == nil {
|
||||
status["source_inbox"] = inbox
|
||||
}
|
||||
}
|
||||
return status
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,120 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/local/glpi-neural-brain/internal/model"
|
||||
"github.com/local/glpi-neural-brain/internal/ollama"
|
||||
"github.com/local/glpi-neural-brain/internal/sourceagent"
|
||||
)
|
||||
|
||||
func (e *Engine) SetSourceInbox(store *sourceagent.Store) { e.SourceInbox = store }
|
||||
|
||||
func (e *Engine) evidenceAcquisitionEnabled() bool {
|
||||
return (e.SourceInbox != nil && e.Cfg.SourceInboxEnabled) || e.ResearchEnabledForRuntime()
|
||||
}
|
||||
|
||||
func (e *Engine) sourceInboxLoop(ctx context.Context) {
|
||||
ticker := time.NewTicker(e.Cfg.SourceInboxInterval)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-ticker.C:
|
||||
e.processSourceInbox(ctx)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (e *Engine) processSourceInbox(ctx context.Context) {
|
||||
if e.SourceInbox == nil || !e.Cfg.SourceInboxEnabled {
|
||||
return
|
||||
}
|
||||
items, err := e.SourceInbox.ClaimInbox(ctx, e.Cfg.SourceInboxBatchSize)
|
||||
if err != nil {
|
||||
slog.Warn("source inbox claim failed", "error", err)
|
||||
return
|
||||
}
|
||||
if len(items) == 0 {
|
||||
return
|
||||
}
|
||||
texts := make([]string, 0, len(items))
|
||||
for _, item := range items {
|
||||
text := item.Document.Title + "\n" + strings.Join(item.Document.Categories, " · ") + "\n" + item.Document.Text
|
||||
if len([]rune(text)) > 12000 {
|
||||
text = string([]rune(text)[:12000])
|
||||
}
|
||||
texts = append(texts, text)
|
||||
}
|
||||
cctx, cancel := context.WithTimeout(ollama.WithLowPriority(ctx), 4*time.Minute)
|
||||
vectors, embedErr := e.Ollama.Embed(cctx, texts)
|
||||
cancel()
|
||||
if embedErr != nil || len(vectors) != len(items) {
|
||||
for _, item := range items {
|
||||
_ = e.SourceInbox.ReleaseInbox(ctx, item.ID, fmt.Sprint(embedErr))
|
||||
}
|
||||
return
|
||||
}
|
||||
candidateCount := 0
|
||||
for i, item := range items {
|
||||
hits, stats := e.similarKnowledge(vectors[i], 1, e.effectiveLearningFilter(), 0)
|
||||
if len(hits) == 0 {
|
||||
_ = e.SourceInbox.ReleaseInbox(ctx, item.ID, "knowledge vectors are not ready")
|
||||
continue
|
||||
}
|
||||
relevance := 0.0
|
||||
matched := ""
|
||||
if len(hits) > 0 {
|
||||
relevance = hits[0].Score
|
||||
matched = hits[0].NodeID
|
||||
}
|
||||
status := "archived"
|
||||
if relevance >= e.Cfg.SourceInboxMinSimilarity {
|
||||
status = "candidate"
|
||||
candidateCount++
|
||||
}
|
||||
meta := make(map[string]any, len(item.Metadata)+4)
|
||||
for key, value := range item.Metadata {
|
||||
meta[key] = value
|
||||
}
|
||||
meta["processing_mode"] = e.RuntimeSettings().ProcessingMode
|
||||
meta["exact_comparisons"] = stats.ExactComparisons
|
||||
meta["coarse_comparisons"] = stats.CoarseComparisons
|
||||
meta["candidate_pool"] = stats.CandidatePool
|
||||
_ = e.SourceInbox.CompleteClassification(ctx, item.ID, status, relevance, matched, meta)
|
||||
}
|
||||
if e.Broker != nil {
|
||||
e.Broker.Publish(model.Activity{Type: "source.inbox.classified", Source: "brain", Phase: "source-inbox", Message: fmt.Sprintf("Source-Inbox: %d Dokumente geprüft · %d als Wissenskandidaten vorgemerkt", len(items), candidateCount), Strength: .36, Metadata: map[string]any{"documents": len(items), "candidates": candidateCount, "minimum_similarity": e.Cfg.SourceInboxMinSimilarity}})
|
||||
}
|
||||
}
|
||||
|
||||
func (e *Engine) sourceInboxResearch(ctx context.Context, query string, limit int, freshnessSensitive bool) []model.ResearchResult {
|
||||
if e.SourceInbox == nil || !e.Cfg.SourceInboxEnabled || limit < 1 {
|
||||
return nil
|
||||
}
|
||||
maxAge := time.Duration(0)
|
||||
if freshnessSensitive {
|
||||
maxAge = e.Cfg.SourceInboxFreshMaxAge
|
||||
}
|
||||
items, err := e.SourceInbox.SearchCandidates(ctx, query, limit, maxAge)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
out := make([]model.ResearchResult, 0, len(items))
|
||||
for _, item := range items {
|
||||
if item.QueryScore < .18 {
|
||||
continue
|
||||
}
|
||||
d := item.Document
|
||||
out = append(out, model.ResearchResult{Title: d.Title, URL: d.CanonicalURL, Snippet: clamp(d.Text, 1000), Content: d.Text, ContentType: d.ContentType, Query: query, Language: d.Language, Fetched: true, Relevant: true, Relevance: item.QueryScore, SourceQuality: "source_inbox", SourceQualityScore: .68, AssessmentReason: "Vorab durch Source-Agent gesammelt und vom Brain als thematisch passend zur Knowledgebase klassifiziert."})
|
||||
}
|
||||
if len(out) > 0 && e.Broker != nil {
|
||||
e.Broker.Publish(model.Activity{Type: "article.research.inbox", Source: "brain", Phase: "knowledge-research-routing", Message: fmt.Sprintf("Source-Inbox liefert %d bereits gecrawlte Kandidaten vor SearXNG", len(out)), Strength: .56, Metadata: map[string]any{"query": query, "results": len(out), "freshness_sensitive": freshnessSensitive}})
|
||||
}
|
||||
return out
|
||||
}
|
||||
Reference in New Issue
Block a user