From 47cb96c01028a6c784be5fcd5623bc46927cb77f Mon Sep 17 00:00:00 2001 From: mlsmaycon Date: Wed, 19 Aug 2026 06:31:30 +0000 Subject: [PATCH] [e2e] Spike the vendors' own model-discovery endpoints MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Management could populate the provider-config model picker from live vendor data instead of the hand-curated catalog, which today carries comments tracking which models a vendor retired on which date and which cannot know what a particular account may actually invoke. Whether that is a small feature or a large one turns on one unknown: ListInferenceProfiles is a control-plane operation on bedrock..amazonaws.com, while a provider record's upstream must be bedrock-runtime. for InvokeModel. Nothing documents whether a Bedrock API key authorises the control plane, and if it does not, live discovery for Bedrock needs SigV4 signing in management. Probe each vendor directly — no proxy, no tunnel, no containers — under the credential an operator would actually supply, and print a verdict table. Bedrock gets three probes so the control-plane answer is read against the runtime host we already know 404s, rather than as a lone data point; Vertex gets three because which path serves the publisher listing is itself part of the question. Vertex reuses the same JWT mint and cloud-platform scope llm_router uses in production, so a result here transfers. No status code is asserted: the status is the finding. A probe fails only when its credential is present and the request could not be made at all, which is a broken probe rather than an answer. Add a test_pattern dispatch input so this can run on its own in about a minute instead of behind the sixteen-minute container suite. It reaches the shell through an env var, since a dispatch input interpolated into a run script is a script-injection seam. --- .github/workflows/agent-network-e2e.yml | 13 +- e2e/providerdiscovery/discovery_spike_test.go | 371 ++++++++++++++++++ 2 files changed, 383 insertions(+), 1 deletion(-) create mode 100644 e2e/providerdiscovery/discovery_spike_test.go diff --git a/.github/workflows/agent-network-e2e.yml b/.github/workflows/agent-network-e2e.yml index 88b98293d..42df456b0 100644 --- a/.github/workflows/agent-network-e2e.yml +++ b/.github/workflows/agent-network-e2e.yml @@ -12,6 +12,13 @@ on: AWS issues it. Leave empty for the Sonnet 4.6 default. required: false default: "" + test_pattern: + description: >- + Package pattern to run. Defaults to the whole suite; narrow it to one + package (e.g. ./e2e/providerdiscovery/...) when a run only needs that + package's answer and not the sixteen minutes the container suite costs. + required: false + default: "./e2e/..." concurrency: group: ${{ github.workflow }}-${{ github.ref }} @@ -77,4 +84,8 @@ jobs: GOOGLE_VERTEX_PROJECT: ${{ secrets.E2E_GOOGLE_VERTEX_PROJECT }} GOOGLE_VERTEX_REGION: ${{ secrets.E2E_GOOGLE_VERTEX_REGION }} GOOGLE_VERTEX_MODEL: ${{ secrets.E2E_GOOGLE_VERTEX_MODEL }} - run: go test -tags e2e -timeout 40m -v ./e2e/... + # Read through an env var rather than interpolated into the run + # script: a dispatch input reaching a shell command directly is a + # script-injection seam, however trusted the dispatcher. + TEST_PATTERN: ${{ inputs.test_pattern || './e2e/...' }} + run: go test -tags e2e -timeout 40m -v "$TEST_PATTERN" diff --git a/e2e/providerdiscovery/discovery_spike_test.go b/e2e/providerdiscovery/discovery_spike_test.go new file mode 100644 index 000000000..f8fe4134b --- /dev/null +++ b/e2e/providerdiscovery/discovery_spike_test.go @@ -0,0 +1,371 @@ +//go:build e2e + +// Package providerdiscovery is a credential-driven spike, not a regression +// suite. It calls each vendor's model-discovery endpoint DIRECTLY — no proxy, +// no tunnel, no containers — to answer a question the mock upstream cannot: +// what does each vendor return to a caller holding only the credential an +// operator gave us, and in what shape? +// +// That is the call management would have to make to populate the provider-config +// model picker from live data instead of the hand-maintained catalog in +// management/internals/modules/agentnetwork/catalog. Today that catalog is +// curated by hand — it carries comments tracking which models a vendor retired +// on which date — and it cannot know what a particular account may actually +// invoke. +// +// The load-bearing unknown is Bedrock. ListInferenceProfiles is a CONTROL PLANE +// operation on bedrock..amazonaws.com, while a provider record's +// upstream is bedrock-runtime. because that is what InvokeModel needs. +// Whether a Bedrock API key (AWS_BEARER_TOKEN_BEDROCK) authorises the control +// plane at all is undocumented, and the answer decides whether live discovery +// for Bedrock is a small feature or needs SigV4 signing in management. The +// other three surfaces are here to corroborate the shape. +// +// NOTHING HERE ASSERTS A STATUS CODE, deliberately: the status is the finding. +// A probe fails the test only when its credential is present and the request +// could not be made at all, which is a harness problem rather than an answer. +package providerdiscovery + +import ( + "context" + "encoding/base64" + "encoding/json" + "fmt" + "io" + "net/http" + "os" + "strings" + "testing" + "time" + + "golang.org/x/oauth2/google" + + "github.com/stretchr/testify/require" +) + +// probeTimeout bounds a single vendor call. Generous for a discovery GET, and +// short enough that a hanging endpoint reports rather than stalls the job. +const probeTimeout = 20 * time.Second + +// maxLoggedBody bounds what a probe echoes into the job log. A full model +// listing runs to tens of kilobytes and the useful part is the front; an error +// body is short and is reproduced whole. +const maxLoggedBody = 1500 + +// gcpScope matches the scope llm_router mints Vertex tokens under, so this +// probe exercises the same credential the proxy already uses in production +// rather than a differently-scoped one that might succeed where it fails. +const gcpScope = "https://www.googleapis.com/auth/cloud-platform" + +// probe is one discovery endpoint to try. +type probe struct { + // surface is the provider-config entry this informs (openai_api, …). + surface string + // variant distinguishes several candidate endpoints for one surface, + // because part of the spike is finding out WHICH path answers. + variant string + url string + headers map[string]string +} + +// result is what came back, including the failure cases — those are answers too. +type result struct { + probe + status int + // shape names the JSON envelope the ids were found under, so the eventual + // implementation knows which parser each surface needs. + shape string + ids []string + body string + err error +} + +// TestProviderModelDiscoverySpike probes every configured vendor and prints a +// verdict table. Read the log, not the pass/fail. +func TestProviderModelDiscoverySpike(t *testing.T) { + probes := configuredProbes(t) + if len(probes) == 0 { + t.Skip("no provider credentials set; source ~/.llm-keys to run the discovery spike") + } + + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Minute) + defer cancel() + + results := make([]result, 0, len(probes)) + for _, p := range probes { + t.Run(p.surface+"/"+p.variant, func(t *testing.T) { + res := runProbe(ctx, p) + results = append(results, res) + + // A transport error means we never got an answer — that is a broken + // probe, not a finding about the vendor. + require.NoError(t, res.err, "%s %s: request could not be made", p.surface, p.variant) + + t.Logf("[spike] %s %s -> %d", p.surface, p.variant, res.status) + t.Logf("[spike] %s %s url: %s", p.surface, p.variant, p.url) + if res.shape != "" { + t.Logf("[spike] %s %s shape=%s ids=%d: %s", + p.surface, p.variant, res.shape, len(res.ids), strings.Join(sample(res.ids, 12), ", ")) + } + t.Logf("[spike] %s %s body: %s", p.surface, p.variant, truncate(res.body, maxLoggedBody)) + }) + } + + t.Log("[spike] ==================== VERDICT ====================") + for _, r := range results { + t.Logf("[spike] %-28s %-42s %3d %s", + r.surface+"/"+r.variant, verdict(r), r.status, shapeNote(r)) + } + t.Log("[spike] =================================================") +} + +// configuredProbes assembles the probe list from whichever credentials are +// present, mirroring the env-var gating the rest of the e2e suite uses. +func configuredProbes(t *testing.T) []probe { + t.Helper() + var ps []probe + + if k := os.Getenv("OPENAI_TOKEN"); k != "" { + ps = append(ps, probe{ + surface: "openai_api", variant: "v1-models", + url: "https://api.openai.com/v1/models", + headers: map[string]string{"Authorization": "Bearer " + k}, + }) + } + + if k := os.Getenv("ANTHROPIC_TOKEN"); k != "" { + // limit=1000 because the default page is small and a picker wants the + // whole catalogue in one call if it can get it. + ps = append(ps, probe{ + surface: "anthropic_api", variant: "v1-models", + url: "https://api.anthropic.com/v1/models?limit=1000", + headers: map[string]string{ + "x-api-key": k, + "anthropic-version": "2023-06-01", + }, + }) + } + + ps = append(ps, bedrockProbes()...) + ps = append(ps, vertexProbes(t)...) + return ps +} + +// bedrockProbes covers the three candidate hosts/paths. The runtime probe is +// the control: we already know it 404s, and having it in the same table makes +// the control-plane result unambiguous rather than a lone data point. +func bedrockProbes() []probe { + k := os.Getenv("AWS_BEARER_TOKEN_BEDROCK") + if k == "" { + return nil + } + region := os.Getenv("AWS_REGION") + if region == "" { + region = "eu-central-1" + } + auth := map[string]string{"Authorization": "Bearer " + k} + return []probe{ + { + surface: "bedrock_api", variant: "control-inference-profiles", + url: "https://bedrock." + region + ".amazonaws.com/inference-profiles", + headers: auth, + }, + { + surface: "bedrock_api", variant: "control-foundation-models", + url: "https://bedrock." + region + ".amazonaws.com/foundation-models", + headers: auth, + }, + { + // The control: the record's real upstream, which we expect to + // answer . + surface: "bedrock_api", variant: "runtime-inference-profiles", + url: "https://bedrock-runtime." + region + ".amazonaws.com/inference-profiles", + headers: auth, + }, + } +} + +// vertexProbes covers the publisher-model listing under both API versions and +// both the global and project-scoped forms, because which one answers is itself +// part of what the spike is for. +func vertexProbes(t *testing.T) []probe { + t.Helper() + sa := os.Getenv("GOOGLE_VERTEX_SA_BASE64") + project := os.Getenv("GOOGLE_VERTEX_PROJECT") + if sa == "" || project == "" { + return nil + } + region := os.Getenv("GOOGLE_VERTEX_REGION") + if region == "" { + region = "global" + } + host := "aiplatform.googleapis.com" + if region != "global" { + host = region + "-aiplatform.googleapis.com" + } + + token, err := mintGCPToken(sa) + if err != nil { + // Report rather than fail: a credential we cannot mint from is a + // finding about the credential, and the other surfaces still have + // something to say. + t.Logf("[spike] vertex_ai_api: could not mint an OAuth token from the service-account key: %v", err) + return nil + } + auth := map[string]string{"Authorization": "Bearer " + token} + + return []probe{ + { + surface: "vertex_ai_api", variant: "v1-publishers", + url: "https://" + host + "/v1/publishers/anthropic/models", + headers: auth, + }, + { + surface: "vertex_ai_api", variant: "v1beta1-publishers", + url: "https://" + host + "/v1beta1/publishers/anthropic/models", + headers: auth, + }, + { + surface: "vertex_ai_api", variant: "v1-project-scoped", + url: "https://" + host + "/v1/projects/" + project + + "/locations/" + region + "/publishers/anthropic/models", + headers: auth, + }, + } +} + +// mintGCPToken builds an access token from the base64 service-account key, +// the same way llm_router does at request time. +func mintGCPToken(saKeyB64 string) (string, error) { + jsonKey, err := base64.StdEncoding.DecodeString(strings.TrimSpace(saKeyB64)) + if err != nil { + return "", fmt.Errorf("decode service-account key: %w", err) + } + conf, err := google.JWTConfigFromJSON(jsonKey, gcpScope) + if err != nil { + return "", fmt.Errorf("parse service-account key: %w", err) + } + ctx, cancel := context.WithTimeout(context.Background(), probeTimeout) + defer cancel() + tok, err := conf.TokenSource(ctx).Token() + if err != nil { + return "", fmt.Errorf("mint token: %w", err) + } + return tok.AccessToken, nil +} + +// runProbe issues one request and extracts whatever model ids it can find. +func runProbe(ctx context.Context, p probe) result { + res := result{probe: p} + + reqCtx, cancel := context.WithTimeout(ctx, probeTimeout) + defer cancel() + + req, err := http.NewRequestWithContext(reqCtx, http.MethodGet, p.url, nil) + if err != nil { + res.err = err + return res + } + for k, v := range p.headers { + req.Header.Set(k, v) + } + req.Header.Set("Accept", "application/json") + + resp, err := http.DefaultClient.Do(req) + if err != nil { + res.err = err + return res + } + defer func() { _ = resp.Body.Close() }() + + body, err := io.ReadAll(io.LimitReader(resp.Body, 4<<20)) + if err != nil { + res.err = err + return res + } + res.status = resp.StatusCode + res.body = string(body) + res.shape, res.ids = extractIDs(body) + return res +} + +// listingShapes maps a response envelope to the field naming the model id +// inside it. Each vendor invented its own; a picker has to read all of them. +var listingShapes = []struct { + envelope string + idField string +}{ + {"data", "id"}, // OpenAI, Anthropic + {"inferenceProfileSummaries", "inferenceProfileId"}, // Bedrock control plane + {"modelSummaries", "modelId"}, // Bedrock foundation models + {"publisherModels", "name"}, // Vertex publisher models + {"models", "name"}, // Vertex, older shape +} + +// extractIDs returns the envelope that matched and the ids under it. An empty +// shape means the body is not a listing this spike recognises — which for an +// error response is the expected outcome. +func extractIDs(body []byte) (string, []string) { + var doc map[string]json.RawMessage + if err := json.Unmarshal(body, &doc); err != nil { + return "", nil + } + for _, shape := range listingShapes { + raw, ok := doc[shape.envelope] + if !ok { + continue + } + var entries []map[string]json.RawMessage + if err := json.Unmarshal(raw, &entries); err != nil { + continue + } + ids := make([]string, 0, len(entries)) + for _, entry := range entries { + var id string + if err := json.Unmarshal(entry[shape.idField], &id); err == nil && id != "" { + ids = append(ids, id) + } + } + return shape.envelope, ids + } + return "", nil +} + +// verdict renders the one-line answer for the summary table. +func verdict(r result) string { + switch { + case r.err != nil: + return "REQUEST FAILED" + case r.status == http.StatusOK && len(r.ids) > 0: + return "USABLE — listing returned" + case r.status == http.StatusOK: + return "200 but no ids parsed" + case r.status == http.StatusUnauthorized, r.status == http.StatusForbidden: + return "CREDENTIAL REJECTED" + case r.status == http.StatusNotFound: + return "NOT SERVED HERE" + default: + return "UNEXPECTED" + } +} + +func shapeNote(r result) string { + if r.shape == "" { + return "" + } + return fmt.Sprintf("shape=%s ids=%d", r.shape, len(r.ids)) +} + +func sample(ids []string, n int) []string { + if len(ids) <= n { + return ids + } + return append(append([]string{}, ids[:n]...), fmt.Sprintf("… +%d more", len(ids)-n)) +} + +func truncate(s string, limit int) string { + if len(s) <= limit { + return s + } + return s[:limit] + fmt.Sprintf("… (%d more bytes)", len(s)-limit) +}