mirror of
https://github.com/netbirdio/netbird.git
synced 2026-08-25 17:11:29 +02:00
[proxy,management] Conform the Agent Network endpoint to the LLM gateway protocol Reviewed the proxy against Claude Code's published gateway contract. The transport layer already held up; fourteen gaps sat one layer up, in the model catalog and in the non-inference endpoints clients call. Two of them cost money. The catalog carried no claude-opus-5 or claude-sonnet-5, so an operator could not authorise the models coding agents default to — those requests denied as not-routable, or priced at zero where a catch-all carried them. And gateway records pin ParserID "openai" while the same record serves /v1/messages, so Anthropic responses were read with the OpenAI parser, which never looks at message_start where input tokens live: input metered as roughly zero on every stream and cost was skipped entirely. The rest fix requests refused for structural rather than policy reasons: model discovery denied for every account with a model allowlist, token counting denied on Bedrock and mis-parsed on Vertex, startup probes refused and written into the access log at every session start, and denials rendered in a shape no LLM client parses. Two changes are additive by design — the deny body keeps every field it had and adds the vendor's error object alongside, and body-level identity injection is now gated on the request's dialect so it stops sending OpenAI-shape fields into Anthropic bodies that reject them. The end-to-end work turned up one more: the discovery filter treated any slash in a model id as a gateway prefix, which would have dropped every self-hosted "Qwen/..." model from the picker.
115 lines
5.8 KiB
Go
115 lines
5.8 KiB
Go
package middleware
|
|
|
|
// Metadata key namespace constants shared across the built-in
|
|
// middlewares. Each domain owns a prefix; middlewares declare their
|
|
// per-key allowlist drawn from these constants. Agents implementing
|
|
// the G2 middlewares import this file so the dashboard's expanded-row
|
|
// viewer and the access-log writer see a stable key surface.
|
|
//
|
|
// Key shape rules (enforced by the metadata accumulator):
|
|
// - Lowercase ASCII letters, digits, dot, underscore, hyphen.
|
|
// - At least one dot separating namespace from leaf.
|
|
// - Max length: MaxMetadataKeyBytes.
|
|
const (
|
|
// LLM request-side metadata (emitted by llm_request_parser).
|
|
KeyLLMProvider = "llm.provider"
|
|
KeyLLMModel = "llm.model"
|
|
KeyLLMStream = "llm.stream"
|
|
KeyLLMRequestPromptRaw = "llm.request_prompt_raw"
|
|
KeyLLMCaptureTruncated = "llm.capture_truncated"
|
|
// KeyLLMSessionID groups requests of the same conversation / coding
|
|
// session, read from the per-provider session marker in the request
|
|
// body. Empty for clients that don't send one.
|
|
KeyLLMSessionID = "llm.session_id"
|
|
|
|
// Sub-agent attribution (emitted by llm_request_parser from the
|
|
// client's request headers). A coding agent that spawns helpers
|
|
// stamps the spawned agent's id, and the spawning agent's id when
|
|
// the helper is itself nested, so cost within one session can be
|
|
// split across the agents that ran in parallel. These identify an
|
|
// agent, not a person or a device: never treat them as a user id.
|
|
KeyLLMAgentID = "llm.agent_id"
|
|
KeyLLMParentAgentID = "llm.parent_agent_id"
|
|
|
|
// LLM response-side metadata (emitted by llm_response_parser).
|
|
//nolint:gosec // metadata key name, not a credential
|
|
KeyLLMInputTokens = "llm.input_tokens"
|
|
//nolint:gosec // metadata key name, not a credential
|
|
KeyLLMOutputTokens = "llm.output_tokens"
|
|
//nolint:gosec // metadata key name, not a credential
|
|
KeyLLMTotalTokens = "llm.total_tokens"
|
|
// LLM cached-input bucket. For OpenAI it's the SUBSET of input
|
|
// tokens that hit the prompt cache (prompt_tokens_details.
|
|
// cached_tokens) — billed at the cached_input_per_1k rate when
|
|
// configured. For Anthropic it's cache_read_input_tokens, which
|
|
// is ADDITIVE to llm.input_tokens — billed at cache_read_per_1k.
|
|
// cost_meter switches formula on llm.provider.
|
|
//nolint:gosec // metadata key name, not a credential
|
|
KeyLLMCachedInputTokens = "llm.cached_input_tokens"
|
|
// LLM cache-creation bucket (Anthropic only). ADDITIVE to
|
|
// llm.input_tokens; billed at cache_creation_per_1k.
|
|
//nolint:gosec // metadata key name, not a credential
|
|
KeyLLMCacheCreationTokens = "llm.cache_creation_tokens"
|
|
KeyLLMResponseCompletion = "llm.response_completion"
|
|
|
|
// Guardrail outcomes (emitted by llm_guardrail). The guardrail
|
|
// also re-emits llm.request_prompt as a redacted variant of the
|
|
// raw prompt and drops llm.request_prompt_raw from the bag.
|
|
KeyLLMRequestPrompt = "llm.request_prompt"
|
|
KeyLLMPolicyDecision = "llm_policy.decision"
|
|
KeyLLMPolicyReason = "llm_policy.reason"
|
|
|
|
// LLM router routing decision (emitted by llm_router). The router
|
|
// stamps the resolved provider id so downstream middlewares and
|
|
// the access-log emitter can attribute the request without
|
|
// re-parsing the body.
|
|
KeyLLMResolvedProviderID = "llm.resolved_provider_id"
|
|
|
|
// LLM authorising groups for this request (emitted by llm_router
|
|
// on the allow path). Carries the comma-separated intersection of
|
|
// the caller's UserGroups with the resolved route's
|
|
// AllowedGroupIDs — i.e. the groups that actually authorise this
|
|
// specific request, NOT every group the peer happens to be in.
|
|
// Identity-stamping middlewares use this for per-request tag
|
|
// attribution so unrelated group memberships don't leak into
|
|
// downstream gateways' spend logs.
|
|
KeyLLMAuthorisingGroups = "llm.authorising_groups"
|
|
|
|
// LLM non-inference marker (emitted by llm_router on the allow path
|
|
// for endpoints that legitimately carry no model, such as model
|
|
// listing). The router still authorises these against the caller's
|
|
// groups; the marker only tells the limits gate that a per-model
|
|
// allowlist has nothing to evaluate, so an empty model must not be
|
|
// read as an undetermined one. Never derived from client input.
|
|
KeyLLMNonInference = "llm.non_inference"
|
|
|
|
// LLM policy attribution (emitted by llm_limit_check on the allow
|
|
// path). Names the policy that paid for this request and the
|
|
// dimension counters the post-flight llm_limit_record middleware
|
|
// must tick. Empty when no applicable policy has any caps
|
|
// configured (catch-all-allow attribution).
|
|
KeyLLMSelectedPolicyID = "llm.selected_policy_id"
|
|
KeyLLMAttributionGroupID = "llm.attribution_group_id"
|
|
KeyLLMAttributionWindowS = "llm.attribution_window_seconds"
|
|
|
|
// Cost metering (emitted by cost_meter). The four per-bucket keys are the
|
|
// base of the breakdown — one per token bucket the provider bills
|
|
// separately — and the two aggregates below are derived from them:
|
|
// usd_total is their sum, usd_cache is cached_input + cache_creation.
|
|
KeyCostUSDInput = "cost.usd_input"
|
|
// KeyCostUSDCachedInput is the cost of the cache-read bucket (Anthropic cache_read; OpenAI's discounted cached subset of input).
|
|
KeyCostUSDCachedInput = "cost.usd_cached_input"
|
|
// KeyCostUSDCacheCreation is the cost of the cache-write bucket. Zero for providers without one.
|
|
KeyCostUSDCacheCreation = "cost.usd_cache_creation"
|
|
KeyCostUSDOutput = "cost.usd_output"
|
|
KeyCostUSDTotal = "cost.usd_total"
|
|
// KeyCostUSDCache is the portion of cost.usd_total billed for prompt-cache buckets (cache read/creation, or OpenAI's cached input subset).
|
|
KeyCostUSDCache = "cost.usd_cache"
|
|
KeyCostSkipped = "cost.skipped"
|
|
|
|
// Framework-emitted error markers. Use the mw.<id>.* prefix to
|
|
// distinguish framework-injected entries from middleware-emitted
|
|
// metadata.
|
|
KeyFrameworkErrorKindFmt = "mw.%s.error_kind"
|
|
)
|