Files
netbird/proxy/internal/middleware/keys.go
mlsmaycon d928bcb630 [proxy] Exempt non-inference endpoints from the model allowlist gate
GET /v1/models carries no model, and management's per-model allowlist
fails closed on an undetermined one, so gateway model discovery denied
with model_blocked for every account that enables a model allowlist. The
client treats a failed discovery as silent and falls back to its built-in
list, so the operator sees an empty picker with no error to chase.

The router already classifies these paths and authorises the route against
the caller's groups before allowing them, so mark them non-inference there
and let the limits gate skip a pre-flight that has no model to evaluate
and no tokens to book. The marker comes from the router's own path
classification, never from client input, so an inference request cannot
set it to escape the allowlist.
2026-08-11 02:47:19 +00:00

106 lines
5.3 KiB
Go

package middleware
// Metadata key namespace constants shared across the built-in
// middlewares. Each domain owns a prefix; middlewares declare their
// per-key allowlist drawn from these constants. Agents implementing
// the G2 middlewares import this file so the dashboard's expanded-row
// viewer and the access-log writer see a stable key surface.
//
// Key shape rules (enforced by the metadata accumulator):
// - Lowercase ASCII letters, digits, dot, underscore, hyphen.
// - At least one dot separating namespace from leaf.
// - Max length: MaxMetadataKeyBytes.
const (
// LLM request-side metadata (emitted by llm_request_parser).
KeyLLMProvider = "llm.provider"
KeyLLMModel = "llm.model"
KeyLLMStream = "llm.stream"
KeyLLMRequestPromptRaw = "llm.request_prompt_raw"
KeyLLMCaptureTruncated = "llm.capture_truncated"
// KeyLLMSessionID groups requests of the same conversation / coding
// session, read from the per-provider session marker in the request
// body. Empty for clients that don't send one.
KeyLLMSessionID = "llm.session_id"
// LLM response-side metadata (emitted by llm_response_parser).
//nolint:gosec // metadata key name, not a credential
KeyLLMInputTokens = "llm.input_tokens"
//nolint:gosec // metadata key name, not a credential
KeyLLMOutputTokens = "llm.output_tokens"
//nolint:gosec // metadata key name, not a credential
KeyLLMTotalTokens = "llm.total_tokens"
// LLM cached-input bucket. For OpenAI it's the SUBSET of input
// tokens that hit the prompt cache (prompt_tokens_details.
// cached_tokens) — billed at the cached_input_per_1k rate when
// configured. For Anthropic it's cache_read_input_tokens, which
// is ADDITIVE to llm.input_tokens — billed at cache_read_per_1k.
// cost_meter switches formula on llm.provider.
//nolint:gosec // metadata key name, not a credential
KeyLLMCachedInputTokens = "llm.cached_input_tokens"
// LLM cache-creation bucket (Anthropic only). ADDITIVE to
// llm.input_tokens; billed at cache_creation_per_1k.
//nolint:gosec // metadata key name, not a credential
KeyLLMCacheCreationTokens = "llm.cache_creation_tokens"
KeyLLMResponseCompletion = "llm.response_completion"
// Guardrail outcomes (emitted by llm_guardrail). The guardrail
// also re-emits llm.request_prompt as a redacted variant of the
// raw prompt and drops llm.request_prompt_raw from the bag.
KeyLLMRequestPrompt = "llm.request_prompt"
KeyLLMPolicyDecision = "llm_policy.decision"
KeyLLMPolicyReason = "llm_policy.reason"
// LLM router routing decision (emitted by llm_router). The router
// stamps the resolved provider id so downstream middlewares and
// the access-log emitter can attribute the request without
// re-parsing the body.
KeyLLMResolvedProviderID = "llm.resolved_provider_id"
// LLM authorising groups for this request (emitted by llm_router
// on the allow path). Carries the comma-separated intersection of
// the caller's UserGroups with the resolved route's
// AllowedGroupIDs — i.e. the groups that actually authorise this
// specific request, NOT every group the peer happens to be in.
// Identity-stamping middlewares use this for per-request tag
// attribution so unrelated group memberships don't leak into
// downstream gateways' spend logs.
KeyLLMAuthorisingGroups = "llm.authorising_groups"
// LLM non-inference marker (emitted by llm_router on the allow path
// for endpoints that legitimately carry no model, such as model
// listing). The router still authorises these against the caller's
// groups; the marker only tells the limits gate that a per-model
// allowlist has nothing to evaluate, so an empty model must not be
// read as an undetermined one. Never derived from client input.
KeyLLMNonInference = "llm.non_inference"
// LLM policy attribution (emitted by llm_limit_check on the allow
// path). Names the policy that paid for this request and the
// dimension counters the post-flight llm_limit_record middleware
// must tick. Empty when no applicable policy has any caps
// configured (catch-all-allow attribution).
KeyLLMSelectedPolicyID = "llm.selected_policy_id"
KeyLLMAttributionGroupID = "llm.attribution_group_id"
KeyLLMAttributionWindowS = "llm.attribution_window_seconds"
// Cost metering (emitted by cost_meter). The four per-bucket keys are the
// base of the breakdown — one per token bucket the provider bills
// separately — and the two aggregates below are derived from them:
// usd_total is their sum, usd_cache is cached_input + cache_creation.
KeyCostUSDInput = "cost.usd_input"
// KeyCostUSDCachedInput is the cost of the cache-read bucket (Anthropic cache_read; OpenAI's discounted cached subset of input).
KeyCostUSDCachedInput = "cost.usd_cached_input"
// KeyCostUSDCacheCreation is the cost of the cache-write bucket. Zero for providers without one.
KeyCostUSDCacheCreation = "cost.usd_cache_creation"
KeyCostUSDOutput = "cost.usd_output"
KeyCostUSDTotal = "cost.usd_total"
// KeyCostUSDCache is the portion of cost.usd_total billed for prompt-cache buckets (cache read/creation, or OpenAI's cached input subset).
KeyCostUSDCache = "cost.usd_cache"
KeyCostSkipped = "cost.skipped"
// Framework-emitted error markers. Use the mw.<id>.* prefix to
// distinguish framework-injected entries from middleware-emitted
// metadata.
KeyFrameworkErrorKindFmt = "mw.%s.error_kind"
)