{
  "schemaVersion": 1,
  "published": "2026-10-07",
  "updated": "2026-10-07",
  "reviewed": "2026-10-07",
  "author": "OneQuill Research",
  "affiliation": "Published by OneQuill, developer of OneVir. This is a documentary review of selected public sources, not an independent vendor security audit.",
  "methodology": "Product coverage is a researched snapshot, not an exhaustive inventory. Category and deployment mode describe the cited offering; editions, contracts and configuration can change the scope. Selected cases illustrate evaluation questions. Advisory counts are not security rankings.",
  "providers": [
    {
      "id": "agentgateway",
      "name": "agentgateway",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://agentgateway.dev/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://agentgateway.dev/docs/standalone/latest/documentation/configuration/resiliency/rate-limits/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Rust gateway for AI and MCP. Local rate limits are held in memory; a remote rate-limit service supplies a different sharing boundary.",
      "evaluationQuestion": "Who can author policies and reference credentials across namespaces?",
      "relatedCases": [
        "agentgateway-namespace-isolation"
      ]
    },
    {
      "id": "aimlapi",
      "name": "AI/ML API",
      "category": "Hosted routers",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "AIMLAPI",
        "AI ML API"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://aimlapi.com/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.aimlapi.com/api-references/service-endpoints/api-key-management"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Hosted model API. Service key management does not replace review of provider processing and retention terms.",
      "evaluationQuestion": "Can keys be scoped and revoked without interrupting all applications?",
      "relatedCases": []
    },
    {
      "id": "aihubmix",
      "name": "AIHubMix",
      "category": "Hosted routers",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "AI Hub Mix"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://aihubmix.com/developers"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.aihubmix.com/en/terms-and-privacy/Privacy"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Hosted model aggregation. Review the service policy alongside the actual selected upstream.",
      "evaluationQuestion": "What provider, geography and content-retention terms govern this model?",
      "relatedCases": []
    },
    {
      "id": "apisix",
      "name": "Apache APISIX",
      "category": "API platforms",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [
        "APISIX"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://apisix.apache.org/ai-gateway/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://apisix.apache.org/docs/apisix/plugins/ai-proxy/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "AI plugins extend the API gateway. Documented streaming timeouts can close a stream without a DONE event.",
      "evaluationQuestion": "How does the client detect an incomplete stream and avoid unsafe replay?",
      "relatedCases": []
    },
    {
      "id": "apigee",
      "name": "Apigee / Model Armor",
      "category": "Cloud gateways",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "Google Cloud",
        "GCP"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://docs.cloud.google.com/model-armor/model-armor-apigee-integration"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.cloud.google.com/apigee/docs/api-platform/tutorials/using-model-armor-policies"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Model Armor is a configured integration. Inspection limits, supported regions and service latency affect coverage.",
      "evaluationQuestion": "What is the explicit fail-open or fail-closed policy for inspection errors?",
      "relatedCases": []
    },
    {
      "id": "aws-agentcore",
      "name": "AWS Bedrock AgentCore Gateway",
      "category": "Cloud gateways",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "AWS",
        "Amazon",
        "Bedrock"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://docs.aws.amazon.com/bedrock-agentcore/latest/devguide/gateway.html"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.aws.amazon.com/bedrock-agentcore/latest/devguide/gateway-targets-inference.html"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "AgentCore Gateway now documents inference targets as well as tool integration. Policy-language and target support still have limits.",
      "evaluationQuestion": "Which inference and tool targets are in scope for each policy?",
      "relatedCases": []
    },
    {
      "id": "azure-apim",
      "name": "Azure API Management",
      "category": "Cloud gateways",
      "deploymentModes": [
        "Managed",
        "Self-hosted"
      ],
      "aliases": [
        "Microsoft",
        "Azure",
        "APIM"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://learn.microsoft.com/en-us/azure/api-management/genai-gateway-capabilities"
        },
        {
          "label": "Documentation / policy",
          "url": "https://learn.microsoft.com/en-us/azure/api-management/llm-token-limit-policy"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Token-limit counters are independent across specified gateway, region and workspace boundaries. A newer AI gateway preview has separate limitations.",
      "evaluationQuestion": "Is your budget global, or is it enforced independently per region?",
      "relatedCases": []
    },
    {
      "id": "bifrost",
      "name": "Bifrost",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Self-hosted",
        "Managed"
      ],
      "aliases": [
        "Maxim"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://www.getbifrost.ai/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.getbifrost.ai/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Gateway with MCP and custom-plugin administration. The September advisories distinguish dynamically linked builds from published static Docker images.",
      "evaluationQuestion": "Is the management API authenticated and isolated from application callers?",
      "relatedCases": [
        "bifrost-management-api-boundaries"
      ]
    },
    {
      "id": "braintrust",
      "name": "Braintrust AI Proxy",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Managed",
        "Customer-hosted"
      ],
      "aliases": [
        "Braintrust proxy"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://www.braintrust.dev/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://www.braintrust.dev/docs/platform/architecture"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Proxy and observability sit within Braintrust's architecture. Data-plane hosting and logging choices need to be assessed together.",
      "evaluationQuestion": "Where are request logs and evaluation datasets stored?",
      "relatedCases": []
    },
    {
      "id": "cloudflare",
      "name": "Cloudflare AI Gateway",
      "category": "Cloud gateways",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "Cloudflare"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://developers.cloudflare.com/ai-gateway/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://developers.cloudflare.com/ai-gateway/configuration/authentication/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Account-level AI Gateway permissions can cover all gateways, including stored BYOK credentials. Logging behaviour changed for new customers from 24 September 2026.",
      "evaluationQuestion": "How are tenants separated and which logging generation applies to your account?",
      "relatedCases": []
    },
    {
      "id": "cometapi",
      "name": "CometAPI",
      "category": "Hosted routers",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "Comet API"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://www.cometapi.com/about/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://www.cometapi.com/privacy-policy/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Hosted model aggregator. Published privacy claims need a deployment-specific contractual scope.",
      "evaluationQuestion": "Which upstreams receive content and which contractual exceptions apply?",
      "relatedCases": []
    },
    {
      "id": "databricks",
      "name": "Databricks AI Gateway",
      "category": "Cloud gateways",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "Unity AI Gateway",
        "Mosaic"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://docs.databricks.com/aws/en/ai-gateway/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.databricks.com/gcp/en/ai-gateway/model-services"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Unity-oriented gateway services and legacy Mosaic endpoints have different controls. Inference-table payload capture is a separate configuration.",
      "evaluationQuestion": "Which service generation and payload-capture settings are enabled?",
      "relatedCases": []
    },
    {
      "id": "eden",
      "name": "Eden AI",
      "category": "Hosted routers",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "Eden AI"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://www.edenai.co/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://www.edenai.co/dpa"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Hosted aggregation across AI services. The data-processing agreement and provider exceptions matter alongside security descriptions.",
      "evaluationQuestion": "Does the DPA cover each selected service and subprocessor?",
      "relatedCases": []
    },
    {
      "id": "envoy-ai-gateway",
      "name": "Envoy AI Gateway / Agent Router",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [
        "Envoy",
        "Agent Router"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://aigateway.envoyproxy.io/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://github.com/theagentrouter/agent-router"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Project naming and repository have moved towards Agent Router. Advisories retain the Envoy AI Gateway name and component scope.",
      "evaluationQuestion": "Which release and MCP request-processing path are actually deployed?",
      "relatedCases": [
        "envoy-ai-gateway-mcp-message-smuggling",
        "envoy-ai-gateway-mcp-request-limits"
      ]
    },
    {
      "id": "f5-nginx",
      "name": "F5 NGINX Gateway Fabric",
      "category": "API platforms",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [
        "NGINX",
        "F5"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://docs.nginx.com/nginx-gateway-fabric/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.nginx.com/nginx-gateway-fabric/how-to/f5-ai-guardrails/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "AI Guardrails integration is a configured stack feature, not a property of every NGINX proxy. Verify policy acceptance and request-size handling.",
      "evaluationQuestion": "Which guardrail integration and enforcement status are active?",
      "relatedCases": []
    },
    {
      "id": "fastrouter",
      "name": "FastRouter",
      "category": "Hosted routers",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "Fast Router"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://fastrouter.ai/features"
        },
        {
          "label": "Documentation / policy",
          "url": "https://fastrouter.ai/privacy"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Hosted router with routing and privacy descriptions. Confirm the effective configuration rather than infer it from feature names.",
      "evaluationQuestion": "What happens to data and cost when a route falls back?",
      "relatedCases": []
    },
    {
      "id": "gitlab",
      "name": "GitLab AI Gateway",
      "category": "Related products",
      "deploymentModes": [
        "Managed",
        "Self-hosted"
      ],
      "aliases": [
        "GitLab Duo"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://docs.gitlab.com/administration/gitlab_duo/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.gitlab.com/releases/patches/other-patches/patch-release-gitlab-ai-gateway-19-4-1-released/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Application-specific gateway for GitLab Duo, not a general interchangeable model router. Hosted and self-hosted patch responsibilities differ.",
      "evaluationQuestion": "Which Duo access and flow-template capabilities are exposed?",
      "relatedCases": [
        "gitlab-ai-gateway-template-sandbox"
      ]
    },
    {
      "id": "gravitee",
      "name": "Gravitee",
      "category": "API platforms",
      "deploymentModes": [
        "Self-hosted",
        "Managed"
      ],
      "aliases": [],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://www.gravitee.io/platform/ai-gateway"
        },
        {
          "label": "Documentation / policy",
          "url": "https://documentation.gravitee.io/apim/create-and-configure-apis/apply-policies/policy-reference/data-logging-masking"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "AI policies and masking are edition and flow-order dependent. Payload logging introduces its own memory and retention needs.",
      "evaluationQuestion": "Is masking applied before every logger and export in this edition?",
      "relatedCases": []
    },
    {
      "id": "helicone",
      "name": "Helicone",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Managed",
        "Self-hosted"
      ],
      "aliases": [],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://www.helicone.ai/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.helicone.ai/getting-started/quick-start"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Observability can capture requests automatically. Retention and self-hosting depend on the selected service and configuration.",
      "evaluationQuestion": "Which content fields are logged, for how long, and under whose account?",
      "relatedCases": []
    },
    {
      "id": "higress",
      "name": "Higress",
      "category": "API platforms",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [
        "Alibaba Higress"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://higress.io/en/ai-gateway"
        },
        {
          "label": "Documentation / policy",
          "url": "https://github.com/higress-group/higress/security"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Gateway with AI provider and security-guard plugins. Plugin selection and policy order define the effective behaviour.",
      "evaluationQuestion": "Which request and response paths are covered by the selected plugins?",
      "relatedCases": []
    },
    {
      "id": "huggingface",
      "name": "Hugging Face Inference Providers",
      "category": "Hosted routers",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "HF",
        "HuggingFace"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://huggingface.co/docs/inference-providers/en/index"
        },
        {
          "label": "Documentation / policy",
          "url": "https://huggingface.co/docs/inference-providers/en/security"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Hugging Face's own request handling and metadata retention differ from upstream provider policies. Dedicated Inference Endpoints are a separate service.",
      "evaluationQuestion": "Which provider receives the request, and are its retention terms approved?",
      "relatedCases": []
    },
    {
      "id": "ibm-api-connect",
      "name": "IBM API Connect",
      "category": "API platforms",
      "deploymentModes": [
        "Managed",
        "Self-hosted"
      ],
      "aliases": [
        "IBM",
        "API Connect"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://www.ibm.com/docs/en/api-connect/cloud/saas?topic=applications-using-ai-gateway-support-watsonxai-apis"
        },
        {
          "label": "Documentation / policy",
          "url": "https://www.ibm.com/docs/en/api-connect/software/12.1.1?topic=overview-known-limitations"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "AI support and limits vary between SaaS and software versions, including watsonx integrations.",
      "evaluationQuestion": "Which AI features and limitations apply to your deployment and release?",
      "relatedCases": []
    },
    {
      "id": "kgateway",
      "name": "kgateway",
      "category": "API platforms",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [
        "K Gateway"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://kgateway.dev/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://kgateway.dev/docs/envoy/2.1.x/ai/about/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Kubernetes gateway with AI extensions. Envoy-based documentation is versioned; prompt guards and rate limits need configured policies.",
      "evaluationQuestion": "What happens if an external guard or rate-limit service is unavailable?",
      "relatedCases": []
    },
    {
      "id": "kong",
      "name": "Kong AI Gateway",
      "category": "API platforms",
      "deploymentModes": [
        "Managed",
        "Self-hosted"
      ],
      "aliases": [
        "Kong",
        "AI Proxy"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://developer.konghq.com/index/ai-gateway/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://developer.konghq.com/plugins/ai-proxy/changelog/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Traditional AI Proxy plugins and newer AI Gateway offerings have distinct version and deployment scopes.",
      "evaluationQuestion": "How is streaming usage reconciled against provider invoices?",
      "relatedCases": [
        "kong-gemini-streaming-token-accounting"
      ]
    },
    {
      "id": "langdb",
      "name": "LangDB",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Managed",
        "Customer-hosted"
      ],
      "aliases": [],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://langdb.ai/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.langdb.ai/enterprise/resources/configuring-data-retention/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "AI gateway with observability. Documented ClickHouse TTL deletion runs through asynchronous merges.",
      "evaluationQuestion": "What deletion latency applies to logs, backups and exports?",
      "relatedCases": []
    },
    {
      "id": "leanroute",
      "name": "Leanroute",
      "category": "Hosted routers",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "Lean Route"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://leanroute.dev/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://leanroute.dev/privacy"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Policy dated 1 October 2026 distinguishes database storage from response and semantic caches, which default to 15 minutes; no-persistence is a separate choice.",
      "evaluationQuestion": "Are response caching and semantic caching disabled for sensitive routes?",
      "relatedCases": []
    },
    {
      "id": "litellm",
      "name": "LiteLLM",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Self-hosted",
        "Managed"
      ],
      "aliases": [],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://www.litellm.ai/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.litellm.ai/docs/proxy/prod"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Python proxy and SDK distribution are distinct from the vendor's pinned proxy Docker distribution.",
      "evaluationQuestion": "What package or image digest is running, and how are credentials rotated after an incident?",
      "relatedCases": [
        "litellm-march-2026-package-incident"
      ]
    },
    {
      "id": "llmgateway",
      "name": "LLM Gateway",
      "category": "Hosted routers",
      "deploymentModes": [
        "Managed",
        "Self-hosted"
      ],
      "aliases": [
        "LLMGateway",
        "theopenco"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://llmgateway.io/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://llmgateway.io/legal/privacy"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Metadata-only logging is described as the default; payload logging is an opt-in setting. Upstream retention remains separate.",
      "evaluationQuestion": "Which content logging options and upstream terms are enabled?",
      "relatedCases": []
    },
    {
      "id": "lm-studio",
      "name": "LM Studio",
      "category": "Inference engines",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [
        "LMStudio",
        "LM Studio llmster",
        "llmster",
        "lms"
      ],
      "officialLinks": [
        {
          "label": "Product",
          "url": "https://lmstudio.ai/"
        },
        {
          "label": "API-token authentication",
          "url": "https://lmstudio.ai/docs/developer/core/authentication"
        },
        {
          "label": "Network server settings",
          "url": "https://lmstudio.ai/docs/developer/core/server/serve-on-network"
        },
        {
          "label": "LM Link remote execution",
          "url": "https://lmstudio.ai/docs/developer/core/lmlink"
        },
        {
          "label": "Offline operation",
          "url": "https://lmstudio.ai/docs/app/offline"
        }
      ],
      "scope": "Desktop and headless model runtime with an API server. Authentication is optional; network serving and LM Link alter the access and execution boundary.",
      "evaluationQuestion": "Are API-token permissions enforced, which machine executes each model, and what model-loading and eviction behaviour does the application rely on?",
      "relatedCases": [
        "lm-studio-network-authentication-and-lifecycle"
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "status": "published"
    },
    {
      "id": "lunar",
      "name": "Lunar.dev",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Self-hosted",
        "Managed"
      ],
      "aliases": [
        "Lunar"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://www.lunar.dev/product/ai-gateway"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.lunar.dev/api-gateway/lunar-dev-in-production/lunar-logs"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Gateway and traffic-management offering. Production logs and data-plane placement need their own review.",
      "evaluationQuestion": "Which data leaves the gateway through logs or management integrations?",
      "relatedCases": []
    },
    {
      "id": "martian",
      "name": "Martian",
      "category": "Hosted routers",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "With Martian"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://withmartian.com/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://gateway-docs.withmartian.com/api-reference/authentication"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Hosted routing gateway. Authentication documentation establishes an API boundary, not every enterprise data-control claim.",
      "evaluationQuestion": "What route decision evidence and data-processing terms can be supplied?",
      "relatedCases": []
    },
    {
      "id": "mlflow",
      "name": "MLflow AI Gateway",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [
        "MLflow"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://mlflow.org/docs/latest/genai/governance/ai-gateway"
        },
        {
          "label": "Documentation / policy",
          "url": "https://www.mlflow.org/docs/latest/genai/governance/ai-gateway/api-keys/key-rotation/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Current gateway documentation distinguishes development credential-encryption defaults from production key management and SQL-backed rotation.",
      "evaluationQuestion": "Is a production encryption secret configured and is rotation rehearsed?",
      "relatedCases": []
    },
    {
      "id": "new-api",
      "name": "New API",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [
        "QuantumNous",
        "new-api"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://docs.newapi.pro/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.newapi.pro/en/docs/guide/wiki/changelog"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Project with quota and billing functions. Release-candidate version scope matters in the quota-overflow advisory.",
      "evaluationQuestion": "Can credits, reservations and final charges be reconciled under concurrent requests?",
      "relatedCases": [
        "new-api-quota-billing-overflow"
      ]
    },
    {
      "id": "nexos",
      "name": "nexos.ai",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "Nexos"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://nexos.ai/ai-gateway/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://nexos.ai/legal/privacy-policy/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Hosted gateway and AI platform. Procurement needs the service contract, provider list and configured routing policy.",
      "evaluationQuestion": "What geography and retention commitments cover every fallback?",
      "relatedCases": []
    },
    {
      "id": "not-diamond",
      "name": "Not Diamond",
      "category": "Related products",
      "deploymentModes": [
        "Managed",
        "Customer-hosted"
      ],
      "aliases": [
        "NotDiamond"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://docs.notdiamond.ai/docs/what-is-not-diamond"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.notdiamond.ai/docs/privacy-security-and-local-deployments"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Routing and model-selection API. It is listed as a related product rather than assumed to implement a full gateway boundary.",
      "evaluationQuestion": "Which controls reside in your calling application and which in the routing service?",
      "relatedCases": []
    },
    {
      "id": "ollama",
      "name": "Ollama",
      "category": "Inference engines",
      "deploymentModes": [
        "Self-hosted",
        "Managed"
      ],
      "aliases": [
        "Ollama local server",
        "Ollama Cloud"
      ],
      "officialLinks": [
        {
          "label": "Project",
          "url": "https://ollama.com/"
        },
        {
          "label": "Local and cloud API authentication",
          "url": "https://docs.ollama.com/api/authentication"
        },
        {
          "label": "Network, cloud and concurrency settings",
          "url": "https://docs.ollama.com/faq"
        },
        {
          "label": "Context and memory",
          "url": "https://docs.ollama.com/context-length"
        }
      ],
      "scope": "Local model runtime and API, with optional cloud-model features. The local API does not require authentication; cloud API credentials and cloud processing are a separate mode.",
      "evaluationQuestion": "Is execution local or cloud, who can reach inference and model-management endpoints, and how do context, parallel requests and model residency fit memory?",
      "relatedCases": [
        "ollama-model-import-memory-boundary"
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "status": "published"
    },
    {
      "id": "one-api",
      "name": "One API",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [
        "songquanpeng"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://github.com/songquanpeng/one-api"
        },
        {
          "label": "Documentation / policy",
          "url": "https://github.com/songquanpeng/one-api"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Separate project from New API. Model mappings and protocol reconstruction can affect unsupported fields.",
      "evaluationQuestion": "Which request fields survive translation, including tools and usage options?",
      "relatedCases": []
    },
    {
      "id": "onevir",
      "name": "OneVir",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [
        "OneQuill"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://onevir.onequill.dev/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://onevir.onequill.dev/#capabilities"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "OneQuill's own product combines gateway controls and local execution paths. This profile is affiliated and is not an independent security assessment.",
      "evaluationQuestion": "Which controls are enabled in your installed version, and which remain the inference backend's responsibility?",
      "relatedCases": []
    },
    {
      "id": "openrouter",
      "name": "OpenRouter",
      "category": "Hosted routers",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://openrouter.ai/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://openrouter.ai/docs/guides/features/guardrails/overview"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Provider routing, fallback and endpoint-level privacy controls need to be configured together. ZDR and geography are route properties.",
      "evaluationQuestion": "Can every selected endpoint and fallback satisfy the same privacy constraints?",
      "relatedCases": []
    },
    {
      "id": "openziti",
      "name": "OpenZiti LLM Gateway",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [
        "Ziti"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://blog.openziti.io/ai-secops-why-your-ai-infrastructure-has-a-network-shaped-blind-spot"
        },
        {
          "label": "Documentation / policy",
          "url": "https://openziti.io/docs/learn/introduction/features/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Identity-based overlay networking is relevant to gateway access. Platform network features do not establish every application-level safeguard.",
      "evaluationQuestion": "How are service identities issued, revoked and separated between tenants?",
      "relatedCases": []
    },
    {
      "id": "opper",
      "name": "Opper",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://opper.ai/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://opper.ai/security-overview"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Security overview updated 5 October 2026: upstream inference is not EEA-restricted by default. Tracing, backups and provider retention have separate rules.",
      "evaluationQuestion": "Does the selected route meet geography and retention needs, including backups?",
      "relatedCases": []
    },
    {
      "id": "orq",
      "name": "orq.ai",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Managed",
        "Customer-hosted"
      ],
      "aliases": [
        "Orq"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://orq.ai/platform/ai-gateway"
        },
        {
          "label": "Documentation / policy",
          "url": "https://orq.ai/legal/security"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Gateway within an orchestration platform. Vendor security claims and logging retention must be matched to the purchased deployment.",
      "evaluationQuestion": "Which attestations and retention settings cover this service and edition?",
      "relatedCases": []
    },
    {
      "id": "plano",
      "name": "Plano",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [
        "Arch",
        "Arch Gateway"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://planoai.dev/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.planoai.dev/get_started/overview.html"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Agent proxy project formerly named Arch. Establish the current component boundaries and supported deployment before comparing controls.",
      "evaluationQuestion": "Which tool, model and policy paths pass through Plano?",
      "relatedCases": []
    },
    {
      "id": "portkey",
      "name": "Portkey / Prisma AIRS AI Gateway",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Self-hosted",
        "Managed"
      ],
      "aliases": [
        "Portkey AI",
        "Prisma AIRS"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://portkey.ai/features/ai-gateway"
        },
        {
          "label": "Documentation / policy",
          "url": "https://www.paloaltonetworks.com/blog/2026/07/announcing-general-availability-of-prisma-airs-ai-gateway/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Portkey commercial offerings have moved into Prisma AIRS. The selected advisory applies to the open-source Portkey gateway; it does not establish managed-service exposure.",
      "evaluationQuestion": "Which product, build and custom-host policy are you evaluating?",
      "relatedCases": [
        "portkey-custom-host-ssrf"
      ]
    },
    {
      "id": "requesty",
      "name": "Requesty",
      "category": "Hosted routers",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://www.requesty.ai/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://www.requesty.ai/privacy"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Gateway logging, caching and upstream retention are separate. EU router processing does not establish the upstream inference location.",
      "evaluationQuestion": "Does this plan and route prohibit training and retention at every layer?",
      "relatedCases": []
    },
    {
      "id": "sglang",
      "name": "SGLang",
      "category": "Inference engines",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [
        "SGLang Runtime",
        "SGLang serving framework",
        "sgl-project"
      ],
      "officialLinks": [
        {
          "label": "Project",
          "url": "https://github.com/sgl-project/sglang"
        },
        {
          "label": "Server arguments and access controls",
          "url": "https://docs.sglang.io/docs/advanced_features/server_arguments"
        },
        {
          "label": "Session-aware radix cache",
          "url": "https://github.com/sgl-project/sglang/blob/main/docs/docs/advanced_features/session_radix_cache.mdx"
        }
      ],
      "scope": "LLM and multimodal serving framework with distributed workers and prefix-cache reuse. The inference API, administrative controls and worker channels have distinct trust boundaries.",
      "evaluationQuestion": "Which API and admin keys, worker networks, model-loading features and cache isolation are configured in the installed release?",
      "relatedCases": [
        "sglang-worker-and-management-trust"
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "status": "published"
    },
    {
      "id": "tensorzero",
      "name": "TensorZero",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://www.tensorzero.com/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://www.tensorzero.com/docs/operations/set-up-auth-for-tensorzero"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Gateway authentication is an explicit operational configuration with PostgreSQL support. Dashboard access is a separate boundary.",
      "evaluationQuestion": "Are both the gateway and dashboard authenticated for this deployment?",
      "relatedCases": []
    },
    {
      "id": "tetrate",
      "name": "Tetrate Agent Router",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Managed",
        "Customer-hosted"
      ],
      "aliases": [
        "Agent Router Service",
        "TARS",
        "Tetrate Enterprise"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://www.tetrate.io/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.tetrate.ai/product-architecture/data-flows/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Hosted Agent Router Service and enterprise customer-hosted data plane have different data flows. The management plane is separately hosted.",
      "evaluationQuestion": "What crosses between your data plane and the hosted management plane?",
      "relatedCases": []
    },
    {
      "id": "traefik",
      "name": "Traefik Hub",
      "category": "API platforms",
      "deploymentModes": [
        "Managed",
        "Self-hosted"
      ],
      "aliases": [
        "Traefik"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://doc.traefik.io/traefik-hub/ai-gateway/overview"
        },
        {
          "label": "Documentation / policy",
          "url": "https://doc.traefik.io/traefik-hub/ai-gateway/middlewares/token-rate-limit"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Hub AI middlewares include Redis-backed shared token limits. Interrupted streams and response-completed events affect usage measurement.",
      "evaluationQuestion": "Does incomplete-stream usage reach the budget ledger?",
      "relatedCases": []
    },
    {
      "id": "truefoundry",
      "name": "TrueFoundry",
      "category": "Specialist gateways",
      "deploymentModes": [
        "Managed",
        "Customer-hosted"
      ],
      "aliases": [],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://www.truefoundry.com/ai-gateway"
        },
        {
          "label": "Documentation / policy",
          "url": "https://www.truefoundry.com/docs/ai-gateway/feedback-for-traces"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Gateway, traces and deployment platform. Residency-aware fallback is a vendor-described configuration, not proof of your route policy.",
      "evaluationQuestion": "Can the vendor show that all retries and fallback targets satisfy your residency policy?",
      "relatedCases": []
    },
    {
      "id": "tyk",
      "name": "Tyk AI Studio / MCP Gateway",
      "category": "API platforms",
      "deploymentModes": [
        "Managed",
        "Self-hosted"
      ],
      "aliases": [
        "Tyk AI Studio",
        "Tyk MCP"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://tyk.io/tyk-mcp-gateway/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://tyk.io/blog/introducing-tyk-ai-studio-welcome-to-the-future-of-ai-governance-powered-by-ai-chat-gateway-and-portal/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "AI Studio, chat and MCP offerings are separate capabilities in the Tyk ecosystem.",
      "evaluationQuestion": "Which model-routing and tool-access controls are available in the purchased product?",
      "relatedCases": []
    },
    {
      "id": "vercel",
      "name": "Vercel AI Gateway",
      "category": "Hosted routers",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "Vercel"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://vercel.com/ai-gateway"
        },
        {
          "label": "Documentation / policy",
          "url": "https://vercel.com/i/secure-ai-gateway"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Gateway content deletion and provider ZDR arrangements have separate scope and eligibility.",
      "evaluationQuestion": "Which providers and plan-specific agreements implement the required ZDR setting?",
      "relatedCases": []
    },
    {
      "id": "vllm",
      "name": "vLLM",
      "category": "Inference engines",
      "deploymentModes": [
        "Self-hosted"
      ],
      "aliases": [
        "vLLM Production Stack"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://docs.vllm.ai/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://docs.vllm.ai/en/latest/usage/security/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Inference engine and serving stack. API authentication, internal networks, model loading and GPU scheduling are separate boundaries from a gateway.",
      "evaluationQuestion": "Which engine, features, model format and cluster topology are deployed?",
      "relatedCases": [
        "vllm-kv-transfer-network-isolation",
        "vllm-model-loading-python-optimisation",
        "vllm-multimodal-input-validation",
        "vllm-gguf-gpu-memory-isolation"
      ]
    },
    {
      "id": "wso2",
      "name": "WSO2 API Manager",
      "category": "API platforms",
      "deploymentModes": [
        "Self-hosted",
        "Managed"
      ],
      "aliases": [
        "WSO2",
        "APIM"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://apim.docs.wso2.com/en/latest/ai-gateway/rate-limiting/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://apim.docs.wso2.com/en/latest/api-design-manage/design/rate-limiting/protect-backend-services/"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "AI token policies and backend throttling have different enforcement paths. Local per-node limits differ from distributed Traffic Manager policies.",
      "evaluationQuestion": "Which counters are shared across replicas and regions?",
      "relatedCases": []
    },
    {
      "id": "zenmux",
      "name": "ZenMux",
      "category": "Hosted routers",
      "deploymentModes": [
        "Managed"
      ],
      "aliases": [
        "Zen Mux"
      ],
      "officialLinks": [
        {
          "label": "Product / project",
          "url": "https://zenmux.ai/"
        },
        {
          "label": "Documentation / policy",
          "url": "https://zenmux.ai/docs/guide/advanced/data-services.html"
        }
      ],
      "reviewed": "2026-10-07",
      "evidenceType": "documented-behaviour",
      "scope": "Turning off Data Services changes ZenMux logging and related features. It does not independently establish upstream zero retention.",
      "evaluationQuestion": "Which data services are disabled and what provider retention remains?",
      "relatedCases": []
    }
  ],
  "cases": [
    {
      "id": "portkey-custom-host-ssrf",
      "provider": "portkey",
      "title": "Portkey: custom-host routing and private-network access",
      "topic": "security",
      "evidenceType": "security-advisory",
      "sourcePublished": "2025-12-01",
      "reviewed": "2026-10-07",
      "sources": [
        {
          "label": "GHSA-hhh5-2cvx-vmfp · vendor advisory",
          "url": "https://github.com/Portkey-AI/gateway/security/advisories/GHSA-hhh5-2cvx-vmfp"
        }
      ],
      "scope": "Open-source Portkey gateway custom-host routing",
      "affected": "< 1.14.0",
      "patched": "1.14.0",
      "conditions": "An application caller can influence a custom upstream host on an affected open-source gateway. A useful attack also depends on the gateway being able to reach a private destination or metadata service.",
      "impact": "The documented SSRF lets the gateway make requests to destinations that should not be selectable by an application caller. The business concern is the gateway's network authority: credentials and access intended for a trusted operator can become reachable through an untrusted request.",
      "vendorResponse": "The vendor identifies 1.14.0 as patched. Upgrade the affected distribution, restrict custom-host selection and enforce outbound network rules that also cover resolved addresses and redirects.",
      "lesson": "A provider name in a route is only the start of the decision. Check the actual host and network destination at dispatch time. A tenant should not gain the gateway's access to internal services by changing an endpoint header.",
      "questions": [
        "Can an application key set a custom host or base URL?",
        "Are private, loopback and metadata destinations blocked after DNS resolution?",
        "Do redirects and fallback endpoints pass the same destination checks?",
        "Which installed build and regression evidence establish remediation?"
      ],
      "limits": "The advisory does not establish that every managed Portkey or Prisma AIRS deployment was affected. A product-family name is insufficient evidence of exposure.",
      "identifierNotes": [
        "Primary identifier: GHSA-hhh5-2cvx-vmfp. Vendor mapping: CVE-2025-66405."
      ],
      "status": "published",
      "findingIds": [
        "GHSA-hhh5-2cvx-vmfp"
      ]
    },
    {
      "id": "bifrost-management-api-boundaries",
      "provider": "bifrost",
      "title": "Bifrost: management APIs are an execution boundary",
      "topic": "security",
      "evidenceType": "security-advisory",
      "sourcePublished": "2026-09-23",
      "reviewed": "2026-10-07",
      "sources": [
        {
          "label": "Stdio MCP advisory · GHSA-86gf-xh3g-rvxq",
          "url": "https://github.com/maximhq/bifrost/security/advisories/GHSA-86gf-xh3g-rvxq"
        },
        {
          "label": "Remote plugin advisory · GHSA-2qp8-4xgm-fw6g",
          "url": "https://github.com/maximhq/bifrost/security/advisories/GHSA-2qp8-4xgm-fw6g"
        },
        {
          "label": "Bifrost 2.1.0 changelog",
          "url": "https://docs.getbifrost.ai/changelogs/v2.1.0"
        }
      ],
      "scope": "Reachable management APIs with authentication disabled; stdio MCP registration and remote custom-plugin loading",
      "affected": "Stdio advisory: < 1.5.27. Remote-plugin advisory: < 1.6.3.",
      "patched": "Both advisories identify 2.1.0.",
      "conditions": "The management API must be reachable with authentication disabled. Stdio MCP registration can start processes as the service user. The remote-plugin code-execution path additionally requires a dynamically linked build produced with DYNAMIC=1.",
      "impact": "These capabilities have greater authority than model inference. Registering a process or loading native code can cross directly into the host execution boundary. The vendor distinguishes the published static Docker builds: plugin.Open fails there, while the remote fetch still creates a blind-SSRF concern.",
      "vendorResponse": "The vendor's 2.1.0 notes corroborate authentication protections for stdio registration and private-address handling for remote plugins. Use the supported patched release, authenticate administration and keep management access separate from model callers.",
      "lesson": "An API key that permits inference should not automatically permit plugin installation or process registration. Test administration with the credentials used by an ordinary client, including after a proxy or ingress is added.",
      "questions": [
        "Can an application caller reach MCP registration or plugin installation?",
        "Is the deployed artifact static or dynamically linked?",
        "Which identity and filesystem permissions apply to spawned processes?",
        "Does outbound fetching reject private destinations and redirects?"
      ],
      "limits": "Two mechanisms are discussed in one case because they share an administrative trust boundary. The remote-plugin RCE finding must not be applied to every static image.",
      "identifierNotes": [
        "Version ranges belong to separate advisories; do not combine them into one affected interval."
      ],
      "status": "published",
      "findingIds": [
        "GHSA-86gf-xh3g-rvxq",
        "GHSA-2qp8-4xgm-fw6g"
      ]
    },
    {
      "id": "envoy-ai-gateway-mcp-message-smuggling",
      "provider": "envoy-ai-gateway",
      "title": "Envoy AI Gateway: one message, two interpretations",
      "topic": "security",
      "evidenceType": "security-advisory",
      "sourcePublished": "2026-05-13",
      "reviewed": "2026-10-07",
      "sources": [
        {
          "label": "GHSA-4gph-2hhr-5mwg · vendor advisory",
          "url": "https://github.com/theagentrouter/agent-router/security/advisories/GHSA-4gph-2hhr-5mwg"
        }
      ],
      "scope": "MCP JSON-RPC parsing in the Envoy AI Gateway / Agent Router project",
      "affected": "< 0.6.0",
      "patched": "0.6.0",
      "conditions": "The affected MCP path interprets a message differently across parsing and forwarding stages. The finding concerns protocol interpretation, not a generic vulnerability in every Envoy deployment.",
      "impact": "If the policy stage checks one interpretation and the destination acts on another, an authorised-looking envelope can conceal an action the policy did not approve. Clients need assurance that tool name, arguments and the forwarded message are bound to the same interpretation.",
      "vendorResponse": "The vendor identifies 0.6.0 as patched. Update the affected project and check the deployed MCP path, including policy plugins and any intermediate normalisation.",
      "lesson": "Protocol compatibility includes security semantics. Evaluate ambiguous input handling and rejection behaviour as well as successful tool calls. A parser repair should be verified on the exact path that enforces tool policy.",
      "questions": [
        "Does policy inspect exactly the message sent to the tool?",
        "Are ambiguous or duplicate protocol fields rejected consistently?",
        "Which gateway release and parser are on the MCP route?",
        "Can the operator show a regression result without sharing exploit payloads?"
      ],
      "limits": "The project repository now uses Agent Router naming. This case does not transfer the finding to the whole Envoy proxy ecosystem.",
      "identifierNotes": [],
      "status": "published",
      "findingIds": [
        "GHSA-4gph-2hhr-5mwg"
      ]
    },
    {
      "id": "envoy-ai-gateway-mcp-request-limits",
      "provider": "envoy-ai-gateway",
      "title": "Envoy AI Gateway: enforce limits before buffering",
      "topic": "reliability",
      "evidenceType": "security-advisory",
      "sourcePublished": "2026-09-26",
      "reviewed": "2026-10-07",
      "sources": [
        {
          "label": "GHSA-43xg-mvg9-qwpq · vendor advisory",
          "url": "https://github.com/theagentrouter/agent-router/security/advisories/GHSA-43xg-mvg9-qwpq"
        }
      ],
      "scope": "MCP POST-body buffering in the external-processing component",
      "affected": "0.4.0 through 0.7.0",
      "patched": "1.0.0",
      "conditions": "The affected path reads the full MCP POST body before applying the relevant limits. An authenticated caller may still be able to send a body large enough to exhaust component memory.",
      "impact": "Request limits applied after allocation protect downstream work but may leave the gateway itself exposed to memory exhaustion. Availability depends on where a limit runs, how many bodies can be buffered concurrently and whether cancellation releases the allocated resources.",
      "vendorResponse": "The vendor identifies 1.0.0 as patched. Upgrade and configure request-size, concurrency and timeout controls at the earliest ingress layer and the processing component.",
      "lesson": "A small per-request limit is not a complete capacity model. Multiply buffering by admitted concurrency and account for protocol expansion, such as decoding, before choosing a memory budget.",
      "questions": [
        "Where is the body-size limit enforced relative to allocation?",
        "What is the maximum simultaneously buffered input?",
        "Do authenticated tenants share the same processing memory?",
        "How are rejection, cancellation and recovery observed?"
      ],
      "limits": "This is a documented MCP component issue. It is not evidence that all large LLM requests or every Envoy data plane will fail.",
      "identifierNotes": [],
      "status": "published",
      "findingIds": [
        "GHSA-43xg-mvg9-qwpq"
      ]
    },
    {
      "id": "agentgateway-namespace-isolation",
      "provider": "agentgateway",
      "title": "agentgateway: a patched release can still need a policy setting",
      "topic": "security",
      "evidenceType": "security-advisory",
      "sourcePublished": "2026-06-29",
      "reviewed": "2026-10-07",
      "sources": [
        {
          "label": "GHSA-jwm2-83f3-52xc · vendor advisory",
          "url": "https://github.com/agentgateway/agentgateway/security/advisories/GHSA-jwm2-83f3-52xc"
        }
      ],
      "scope": "Cross-namespace backend references authored by Kubernetes namespace administrators",
      "affected": "< 1.3.0; configuration also matters after upgrade.",
      "patched": "1.3.0 plus AGW_BACKEND_REF_GRANT_MODE=route-and-policy",
      "conditions": "A tenant namespace administrator can author Gateway, HTTPRoute or Policy resources that refer to credential-bearing backends in another namespace. The vendor describes a multi-tenant control-plane issue and no remotely exploitable surface.",
      "impact": "A route that is syntactically valid can still cross an ownership boundary. Credential-backed resources require authorisation for every route and policy reference, not just for the namespace that supplies the frontend.",
      "vendorResponse": "The vendor requires the patched release together with route-and-policy backend-reference grant enforcement. The default route mode alone is insufficient for the described policy-reference condition.",
      "lesson": "Patch status is a release-and-configuration claim. Record the environment setting, accepted resources and cross-namespace authorisation evidence beside the version number.",
      "questions": [
        "Who may create routes and policies in each namespace?",
        "Are route and policy backend references both subject to grants?",
        "Is AGW_BACKEND_REF_GRANT_MODE set to route-and-policy?",
        "Can a tenant use a backend credential owned by another namespace?"
      ],
      "limits": "An internet caller without control-plane permissions is outside the described prerequisite. Do not present this case as an unauthenticated network attack.",
      "identifierNotes": [],
      "status": "published",
      "findingIds": [
        "GHSA-jwm2-83f3-52xc"
      ]
    },
    {
      "id": "new-api-quota-billing-overflow",
      "provider": "new-api",
      "title": "New API: quota arithmetic is part of the trust boundary",
      "topic": "budgets",
      "evidenceType": "security-advisory",
      "sourcePublished": "2026-07-09",
      "reviewed": "2026-10-07",
      "sources": [
        {
          "label": "GHSA-8r8v-xf7q-rcpr · vendor advisory",
          "url": "https://github.com/QuantumNous/new-api/security/advisories/GHSA-8r8v-xf7q-rcpr"
        }
      ],
      "scope": "Quota settlement in affected New API release candidates",
      "affected": "≤ 1.0.0-rc.17",
      "patched": "≥ 1.0.0-rc.18",
      "conditions": "The caller must satisfy normal pre-consumption requirements through a positive balance or valid subscription. Integer overflow during settlement can then turn a charge into a credit. Self-registration with free credits increases exposure when configured.",
      "impact": "An admission check and final bill can disagree even when both work in ordinary examples. Numeric bounds, rounding and the sign of the final ledger entry deserve the same scrutiny as authentication.",
      "vendorResponse": "The vendor reports exploitation observed on 6 July and an emergency fix on 7 July, with the advisory published on 9 July. rc.18 is the stated patch; rc.19 adds logging context. Preserve the release-candidate qualifier.",
      "lesson": "Evaluate the ledger as a state transition: reserve, dispatch, settle or release. Concurrency, very large usage values, cancellation and repeated settlement should not create spendable credit.",
      "questions": [
        "What upper bounds apply to tokens, prices and multiplication?",
        "Is the reservation atomic under competing requests?",
        "Can a failed or repeated settlement increase a user's balance?",
        "How are anomalous credits reconciled with provider invoices?"
      ],
      "limits": "The finding belongs to New API. It must not be assigned to the separate One API project merely because the projects share naming conventions.",
      "identifierNotes": [],
      "status": "published",
      "findingIds": [
        "GHSA-8r8v-xf7q-rcpr"
      ]
    },
    {
      "id": "kong-gemini-streaming-token-accounting",
      "provider": "kong",
      "title": "Kong: a streaming usage fix deserves billing regression tests",
      "topic": "budgets",
      "evidenceType": "release-fix",
      "sourcePublished": "2026-09-15",
      "reviewed": "2026-10-07",
      "sources": [
        {
          "label": "Kong AI Proxy plugin changelog",
          "url": "https://developer.konghq.com/plugins/ai-proxy/changelog/"
        }
      ],
      "scope": "Kong AI Proxy plugin handling of Gemini streaming usage",
      "affected": "The cited release notes describe the fix; no security-advisory affected interval is supplied here.",
      "patched": "3.16.0.0 includes the stated fix.",
      "conditions": "Streaming Gemini responses can repeat running usage metadata or omit metadata in some chunks. The release notes address overcounting repeated cumulative values and resetting usage when metadata is missing.",
      "impact": "Reliable accounting requires understanding whether a value is cumulative, incremental or absent. Summing every reported total can overstate a charge; treating absence as zero can erase evidence of prior work.",
      "vendorResponse": "The vendor's 3.16.0.0 changelog records the correction. Confirm which AI Proxy plugin is deployed and reproduce the usage pattern with provider-side accounting evidence.",
      "lesson": "Use release notes as evaluation evidence without promoting every correction to a CVE. A billing test should include complete streams, disconnected streams, missing final events and retries.",
      "questions": [
        "Is each usage field incremental or cumulative?",
        "What remains recorded if the client disconnects?",
        "Which plugin version processes the Gemini response?",
        "How are gateway measurements compared with the provider bill?"
      ],
      "limits": "This is labelled a release fix. The cited entry does not establish an independent vulnerability or apply automatically to every newer Kong AI offering.",
      "identifierNotes": [],
      "status": "published",
      "findingIds": [
        "Kong AI Proxy changelog: 3.16.0.0"
      ]
    },
    {
      "id": "litellm-march-2026-package-incident",
      "provider": "litellm",
      "title": "LiteLLM: package provenance and incident response",
      "topic": "supply-chain",
      "evidenceType": "incident-report",
      "sourcePublished": "2026-03-24",
      "reviewed": "2026-10-07",
      "sources": [
        {
          "label": "LiteLLM March 2026 security update",
          "url": "https://docs.litellm.ai/blog/security-update-march-2026"
        }
      ],
      "scope": "Malicious PyPI distributions, distinct from the vendor's official pinned Proxy Docker images",
      "affected": "PyPI LiteLLM 1.82.7 and 1.82.8 during the reported publication window.",
      "patched": "Vendor describes a clean 1.83 release and pipeline v2 on 30 March; incident response also requires containment and credential review.",
      "conditions": "The organisation installed or executed one of the affected PyPI versions. The vendor reports a roughly 40-minute window beginning at 10:39 UTC on 24 March. The vendor states its official Proxy Docker distribution was unaffected because it pinned requirements.",
      "impact": "A dependency update is a change to executable code with the authority of the installation environment. Exposure assessment needs the actual artifact and install history, not just the top-level application name.",
      "vendorResponse": "The vendor removed the affected packages, investigated the incident and rebuilt the release pipeline. Its account of a connection to a Trivy-related compromise is expressed as a belief, not independently established attribution.",
      "lesson": "An upgrade removes a malicious artifact from the future path; it does not establish that previously accessible secrets are safe. Preserve artifact evidence, isolate affected environments and rotate credentials according to the incident investigation.",
      "questions": [
        "What package versions and image digests were installed during the window?",
        "Which secrets and network permissions were available to that environment?",
        "Can builds be reproduced from a trusted, pinned dependency set?",
        "Who owns containment, rotation and verification before service resumes?"
      ],
      "limits": "This article summarises the vendor's incident account. It does not reproduce malicious code, assert every LiteLLM installation was compromised or equate a package incident with the security of all deployments.",
      "identifierNotes": [],
      "status": "published",
      "findingIds": [
        "LiteLLM PyPI incident: 2026-03-24"
      ]
    },
    {
      "id": "gitlab-ai-gateway-template-sandbox",
      "provider": "gitlab",
      "title": "GitLab AI Gateway: flow templates and sandbox assumptions",
      "topic": "security",
      "evidenceType": "security-advisory",
      "sourcePublished": null,
      "reviewed": "2026-10-07",
      "sources": [
        {
          "label": "GitLab AI Gateway patch release notice",
          "url": "https://docs.gitlab.com/releases/patches/other-patches/patch-release-gitlab-ai-gateway-19-4-1-released/"
        }
      ],
      "scope": "GitLab Duo Agent Platform flow-template processing",
      "affected": "18.1.6 to < 19.2.4; 19.3 to < 19.3.2; 19.4 to < 19.4.1",
      "patched": "19.2.4, 19.3.2 and 19.4.1",
      "conditions": "An authenticated user with access to the Duo Agent Platform can submit a crafted flow template to the affected template-processing path. This is an application-specific prerequisite.",
      "impact": "The vendor describes a template sandbox escape. Authoring a workflow is therefore a security-sensitive capability even when the user does not hold host-administration rights.",
      "vendorResponse": "The vendor released branch-specific patches and states its hosted gateways were already fixed. Self-hosted operators retain their upgrade responsibility and should use the patch appropriate to their release branch.",
      "lesson": "Ask what an author can cause a workflow runtime to execute, read and contact. A template language or sandbox needs a defined permission boundary and a plan for patching that runtime.",
      "questions": [
        "Who can create and submit flow templates?",
        "Which AI Gateway branch and patch are installed?",
        "What host permissions and secrets are available to the runtime?",
        "Does a hosted-service patch cover your self-hosted components?"
      ],
      "limits": "This case describes GitLab's application-specific gateway. It is not a generic finding against interchangeable AI model gateways.",
      "identifierNotes": [
        "Vendor identifier: CVE-2026-90970. Keep branch-specific intervals separate."
      ],
      "status": "published",
      "findingIds": [
        "CVE-2026-90970"
      ]
    },
    {
      "id": "vllm-kv-transfer-network-isolation",
      "provider": "vllm",
      "title": "vLLM: isolate KV-transfer and distributed communication",
      "topic": "inference",
      "evidenceType": "security-advisory",
      "sourcePublished": "2025-05-20",
      "reviewed": "2026-10-07",
      "sources": [
        {
          "label": "GHSA-hjq4-87xh-g4fv · vendor advisory",
          "url": "https://github.com/vllm-project/vllm/security/advisories/GHSA-hjq4-87xh-g4fv"
        },
        {
          "label": "vLLM security guidance",
          "url": "https://docs.vllm.ai/en/latest/usage/security/"
        }
      ],
      "scope": "PyNcclPipe KV-cache transfer in the V0 engine",
      "affected": "0.6.5 through 0.8.4",
      "patched": "0.8.5",
      "conditions": "The affected V0 PyNcclPipe transfer path receives untrusted serialised network messages. Network reachability to the relevant communication service and use of that backend define exposure; this is not a malicious-model-loading case.",
      "impact": "An internal cluster channel can carry execution-sensitive messages without being the public inference API. Restricting /v1 access alone does not control a service bound on a different address or inter-node port.",
      "vendorResponse": "The vendor identifies 0.8.5 as patched and discusses binding and network isolation. Current security guidance separately warns about insecure defaults in several distributed communication paths.",
      "lesson": "Map public API, model-management and cluster networks independently. Keep worker communication on trusted isolated networks and verify bind addresses, firewall rules and the backend actually in use.",
      "questions": [
        "Is the deployed engine V0 and is PyNcclPipe KV transfer enabled?",
        "Which addresses and ports are reachable from untrusted networks?",
        "What authentication and encryption apply to each distributed channel?",
        "Which release and topology changes establish remediation?"
      ],
      "limits": "The current documentation must be read for the chosen distributed backend. It does not justify saying every vLLM deployment sends plaintext prompts, weights and outputs through ZeroMQ.",
      "identifierNotes": [
        "Primary identifier: GHSA-hjq4-87xh-g4fv; vendor mapping CVE-2025-47277."
      ],
      "status": "published",
      "findingIds": [
        "GHSA-hjq4-87xh-g4fv"
      ]
    },
    {
      "id": "vllm-model-loading-python-optimisation",
      "provider": "vllm",
      "title": "vLLM: model configuration and Python optimisation",
      "topic": "inference",
      "evidenceType": "security-advisory",
      "sourcePublished": "2026-06-14",
      "reviewed": "2026-10-07",
      "sources": [
        {
          "label": "GHSA-q8gq-377p-jq3r · vendor advisory",
          "url": "https://github.com/vllm-project/vllm/security/advisories/GHSA-q8gq-377p-jq3r"
        }
      ],
      "scope": "Model-configuration loading when Python assertion checks are disabled",
      "affected": "< 0.22.0",
      "patched": "≥ 0.22.0",
      "conditions": "The operator loads malicious model configuration while Python runs with python -O or PYTHONOPTIMIZE=1. The advisory describes an activation-function import path whose assertion checks disappear in that mode.",
      "impact": "A model bundle is part of the software supply chain. Configuration can influence executable behaviour, so authorisation to add a model is more powerful than permission to send an ordinary inference prompt.",
      "vendorResponse": "The vendor identifies 0.22.0 as patched. Upgrade the affected loader, establish model provenance and review the Python execution environment before importing model assets.",
      "lesson": "Keep runtime optimisation and trust validation separate. vLLM's own performance or compilation levels are not the same setting as Python's -O flag.",
      "questions": [
        "Who can approve and load model configurations?",
        "Is python -O or PYTHONOPTIMIZE used in the service environment?",
        "Are model revision and configuration hashes recorded?",
        "Which validation remains active under every supported runtime mode?"
      ],
      "limits": "The prerequisite is malicious model configuration plus Python optimisation. The advisory is not evidence that every ordinary inference request can execute code.",
      "identifierNotes": [
        "Primary identifier: GHSA-q8gq-377p-jq3r; vendor mapping CVE-2026-41523."
      ],
      "status": "published",
      "findingIds": [
        "GHSA-q8gq-377p-jq3r"
      ]
    },
    {
      "id": "vllm-multimodal-input-validation",
      "provider": "vllm",
      "title": "vLLM: input features need separate validation and capacity limits",
      "topic": "inference",
      "evidenceType": "security-advisory",
      "sourcePublished": "2026-07-27",
      "reviewed": "2026-10-07",
      "sources": [
        {
          "label": "Prompt embedding advisory · GHSA-mcmc-2m55-j8jj",
          "url": "https://github.com/vllm-project/vllm/security/advisories/GHSA-mcmc-2m55-j8jj"
        },
        {
          "label": "Concurrency follow-up · GHSA-pr7f-p5mw-fc87",
          "url": "https://github.com/vllm-project/vllm/security/advisories/GHSA-pr7f-p5mw-fc87"
        },
        {
          "label": "Video frames · GHSA-pq5c-rjhq-qp7p",
          "url": "https://github.com/vllm-project/vllm/security/advisories/GHSA-pq5c-rjhq-qp7p"
        },
        {
          "label": "Red Hat video-frame record · CVE-2026-5497",
          "url": "https://access.redhat.com/security/cve/cve-2026-5497"
        }
      ],
      "scope": "Optional prompt embeddings and base64 video/jpeg frame processing; separate findings collected in one feature-validation case",
      "affected": "Original embedding header: ≥ 0.10.2 and < 0.11.1. Follow-up header: ≥ 0.21.0. Video header: ≥ 0.7.0. These are source-specific statements.",
      "patched": "Original embeddings: 0.13.0. Concurrency follow-up: ≥ 0.26.0. Video frames: 0.19.0.",
      "conditions": "Prompt-embedding advisories depend on enabling the feature; the follow-up explicitly requires --enable-prompt-embeds, which is off by default. The video issue concerns base64 video/jpeg frames rather than every binary video loader.",
      "impact": "Schema validation and resource budgeting are different controls. A structurally accepted tensor can still need concurrency-safe checks, while a frame list can consume excessive decode and processing work before inference starts.",
      "vendorResponse": "The upstream advisories list separate fixes. The follow-up demonstrates a concurrency-related invariant-check bypass but explicitly does not establish a live-server crash, unsafe conversion, GPU memory corruption or code execution.",
      "lesson": "Maintain an input-feature inventory. Limit decoded dimensions, frame counts, aggregate input size and concurrent work; disable unsupported input types rather than rely on the text-only request limit.",
      "questions": [
        "Are prompt embeddings and video input enabled on the public route?",
        "Which limits apply after base64 decoding and tensor construction?",
        "Are invariant checks isolated between concurrent requests?",
        "What evidence confirms each separate patched path?"
      ],
      "limits": "These findings share an evaluation theme but are not one vulnerability. The follow-up's unproven outcomes must not be reported as demonstrated RCE.",
      "identifierNotes": [
        "The original embedding advisory header lists ≥ 0.10.2 and < 0.11.1 while naming 0.13.0 as patched; do not silently widen the interval.",
        "The upstream video advisory uses CVE-2026-34755. Red Hat describes the frame issue under CVE-2026-5497. The relationship is not counted as two independently confirmed flaws. Primary finding identifier remains GHSA-pq5c-rjhq-qp7p."
      ],
      "status": "published",
      "findingIds": [
        "GHSA-mcmc-2m55-j8jj",
        "GHSA-pr7f-p5mw-fc87",
        "GHSA-pq5c-rjhq-qp7p"
      ]
    },
    {
      "id": "vllm-gguf-gpu-memory-isolation",
      "provider": "vllm",
      "title": "vLLM: a GGUF kernel defect is distinct from cache policy",
      "topic": "inference",
      "evidenceType": "security-advisory",
      "sourcePublished": "2026-06-11",
      "reviewed": "2026-10-07",
      "sources": [
        {
          "label": "GHSA-5jv2-g5wq-cmr4 · vendor advisory",
          "url": "https://github.com/vllm-project/vllm/security/advisories/GHSA-5jv2-g5wq-cmr4"
        },
        {
          "label": "vLLM prefix-cache isolation",
          "url": "https://docs.vllm.ai/en/latest/design/prefix_caching/"
        }
      ],
      "scope": "Integer truncation in specific GGUF dequantisation kernels",
      "affected": "Vendor header: ≥ 0.5.5; affected kernel path and tensor shape are prerequisites.",
      "patched": "Current vendor patched-version field: ≥ 0.24.0",
      "conditions": "The model and execution path use the affected GGUF dequantisation kernels with dimensions that encounter integer truncation. The defect can leave part of an allocated output tensor uninitialised.",
      "impact": "In a shared GPU environment, uninitialised output can expose data from earlier GPU tensors. This is a kernel-memory correctness concern, distinct from whether a prefix cache is intentionally shared across tenants.",
      "vendorResponse": "The current vendor advisory lists 0.24.0 as patched. Upgrade the affected runtime and evaluate model-format and tenant-isolation choices. Do not replace the current primary patch statement with an older registry value.",
      "lesson": "Per-request cache salting addresses cache reuse and timing inference between trust groups. It does not repair this kernel defect or establish process and GPU memory isolation by itself.",
      "questions": [
        "Are GGUF models and the affected kernels used?",
        "Which tenants share a process, GPU and memory allocator?",
        "Is the deployed version consistent with the current vendor patch statement?",
        "What isolation evidence covers kernels separately from prefix-cache policy?"
      ],
      "limits": "The advisory is not a general finding that every user's KV cache is left unwiped. It concerns particular dequantisation kernels and partially uninitialised output tensors.",
      "identifierNotes": [
        "Primary identifier: GHSA-5jv2-g5wq-cmr4; vendor mapping CVE-2026-53923.",
        "An older registry patch value differs; the current primary advisory's ≥ 0.24.0 statement is retained."
      ],
      "status": "published",
      "findingIds": [
        "GHSA-5jv2-g5wq-cmr4"
      ]
    },
    {
      "id": "sglang-worker-and-management-trust",
      "provider": "sglang",
      "title": "SGLang: isolate worker channels and model-management paths",
      "topic": "inference",
      "evidenceType": "security-advisory",
      "sourcePublished": "2026-03-12",
      "sources": [
        {
          "label": "CERT/CC VU#665416 · March findings and April update",
          "url": "https://kb.cert.org/vuls/id/665416"
        },
        {
          "label": "CERT/CC VU#281278 · separate July findings",
          "url": "https://kb.cert.org/vuls/id/281278"
        },
        {
          "label": "CVE-2026-3059 · source-specific version record",
          "url": "https://www.cve.org/CVERecord?id=CVE-2026-3059"
        },
        {
          "label": "Upstream replay-dump fix, PR #20904",
          "url": "https://github.com/sgl-project/sglang/pull/20904"
        },
        {
          "label": "Current API and administration settings",
          "url": "https://docs.sglang.io/docs/advanced_features/server_arguments"
        }
      ],
      "scope": "Feature-dependent worker serialization, replay tooling and administrative/model-loading interfaces",
      "affected": "The March worker issues require multimodal generation or encoder parallel disaggregation and reachable affected channels. The July notice describes separate feature/configuration conditions without one unified version range.",
      "patched": "March CERT/CC update: 0.5.10, with conflicting CVE-2026-3059 metadata. A fixed release for the July group is not established by the cited July notice.",
      "conditions": "CVE-2026-3059 and CVE-2026-3060 concern unsafe deserialization in enabled worker paths reachable by an attacker. CVE-2026-3989 instead requires replaying a malicious dump. The later July findings concern different endpoints and optional settings.",
      "impact": "A protected public chat route can coexist with execution-sensitive worker or management interfaces. The useful boundary is which actor can reach each interface and supply executable or serialization-sensitive material.",
      "vendorResponse": "CERT/CC’s 7 April update identifies 0.5.10 for the March findings; upstream PR #20904 records the replay-dump change. The July notice reported no patches at its publication and no vendor statement in that note. Current documentation describes API and admin keys, but documentation alone does not establish remediation of every historical path.",
      "lesson": "Keep worker channels isolated, separate inference from administration, approve model and adapter sources, and verify the exact release and enabled endpoint inventory. Treat patch statements for different findings independently.",
      "questions": [
        "Can an untrusted caller reach the worker or disaggregation ports?",
        "Which model-update, adapter, dumper and replay paths are enabled?",
        "Do API and admin credentials protect the intended endpoint matrix?",
        "Which artifact or commit proves each selected finding is remediated?"
      ],
      "limits": "The March and July notices describe different findings. Their conditions and patch status cannot be merged into a claim about every SGLang deployment or the current release.",
      "identifierNotes": [
        "CERT/CC VU#665416 covers CVE-2026-3059, CVE-2026-3060 and CVE-2026-3989. Its April update calls 0.5.10 a remediation, while the CVE-2026-3059 record has listed 0.5.10 as affected; retain both statements and request artifact-specific confirmation.",
        "The separate 30 July VU#281278 covers CVE-2026-15969, CVE-2026-15971, CVE-2026-15974, CVE-2026-15976, CVE-2026-15977 and CVE-2026-15978. No current fixed version is inferred from that historical notice."
      ],
      "findingIds": [
        "VU#665416",
        "CVE-2026-3059",
        "CVE-2026-3060",
        "CVE-2026-3989",
        "VU#281278",
        "CVE-2026-15969",
        "CVE-2026-15971",
        "CVE-2026-15974",
        "CVE-2026-15976",
        "CVE-2026-15977",
        "CVE-2026-15978"
      ],
      "reviewed": "2026-10-07",
      "status": "published"
    },
    {
      "id": "ollama-model-import-memory-boundary",
      "provider": "ollama",
      "title": "Ollama: model import is a separate memory and access boundary",
      "topic": "inference",
      "evidenceType": "security-advisory",
      "sourcePublished": "2026-05-04",
      "sources": [
        {
          "label": "GHSA-x8qc-fggm-mpqg / CVE-2026-7482 · reviewed registry record",
          "url": "https://github.com/advisories/GHSA-x8qc-fggm-mpqg"
        },
        {
          "label": "Upstream tensor-size validation, PR #14406",
          "url": "https://github.com/ollama/ollama/pull/14406"
        },
        {
          "label": "Upstream v0.17.1 release",
          "url": "https://github.com/ollama/ollama/releases/tag/v0.17.1"
        },
        {
          "label": "Local versus cloud API authentication",
          "url": "https://docs.ollama.com/api/authentication"
        },
        {
          "label": "Network binding and deployment settings",
          "url": "https://docs.ollama.com/faq"
        }
      ],
      "scope": "GGUF tensor-size validation during model creation and quantization",
      "affected": "The reviewed registry identifies versions before 0.17.1. Exploitation depends on supplying a malformed GGUF to the affected model-creation path; the reported export scenario also uses model push.",
      "patched": "Registry patched-version field: 0.17.1. Confirm inclusion of the upstream tensor-size fix in the deployed artifact.",
      "conditions": "The record concerns attacker-controlled GGUF tensor offsets and sizes processed during model creation/quantization. It is not a finding that an ordinary prompt in every local Ollama installation exposes memory.",
      "impact": "Model-management permissions can be more sensitive than inference permissions. A shared service should establish who may import, create, quantize and export models, alongside who may submit prompts.",
      "vendorResponse": "Upstream PR #14406 validates expected tensor sizes during model creation. The registry identifies 0.17.1 as patched. The linked release is dated 24 February while the PR merge is dated 25 February; those records alone do not prove that a particular 0.17.1 artifact contains the change.",
      "lesson": "Patch the affected path with artifact-level evidence, restrict model-management operations and keep the service on approved networks. A gateway access policy supports admission but cannot repair an inference runtime’s model loader.",
      "questions": [
        "Can application users invoke create, import, quantization or push operations?",
        "What installed artifact and commit establish the tensor-size fix?",
        "Does the local service remain on loopback or have an authenticated network boundary?",
        "Are model sources and export destinations approved separately?"
      ],
      "limits": "The registry’s version field is attributed to the registry. This documentary review did not reproduce the reported memory disclosure or verify every release artifact.",
      "identifierNotes": [
        "Finding identifier: GHSA-x8qc-fggm-mpqg; CVE mapping: CVE-2026-7482.",
        "Preserve the registry’s 0.17.1 patch statement alongside the upstream release/merge chronology. Request build or backport evidence rather than widening an affected range or declaring the artifact verified."
      ],
      "findingIds": [
        "GHSA-x8qc-fggm-mpqg",
        "CVE-2026-7482"
      ],
      "reviewed": "2026-10-07",
      "status": "published"
    },
    {
      "id": "lm-studio-network-authentication-and-lifecycle",
      "provider": "lm-studio",
      "title": "LM Studio: choose authentication, execution location and model lifecycle",
      "topic": "inference",
      "evidenceType": "documented-behaviour",
      "sourcePublished": null,
      "sources": [
        {
          "label": "Native authentication and token permissions",
          "url": "https://lmstudio.ai/docs/developer/core/authentication"
        },
        {
          "label": "Serving on a network",
          "url": "https://lmstudio.ai/docs/developer/core/server/serve-on-network"
        },
        {
          "label": "LM Link execution location",
          "url": "https://lmstudio.ai/docs/developer/core/lmlink"
        },
        {
          "label": "Idle TTL and Auto-Evict",
          "url": "https://lmstudio.ai/docs/developer/core/ttl-and-auto-evict"
        },
        {
          "label": "Offline operations and runtime updates",
          "url": "https://lmstudio.ai/docs/app/offline"
        }
      ],
      "scope": "Documented API authentication, network binding, remote model resolution and JIT model residency",
      "affected": "Configuration-dependent behaviour. Native API-token authentication is documented for LM Studio 0.4.0 or newer; this case does not identify a vulnerable version range.",
      "patched": "Enable required authentication and scoped token permissions for shared access; approve bind addresses and remote devices; configure model residency intentionally.",
      "conditions": "Authentication is optional by default. Enabling network serving makes the API reachable beyond localhost. With LM Link, a localhost API request can be served by a model on a linked remote machine.",
      "impact": "The application’s API address does not fully describe who can call the server or where model execution occurs. Model switching can also introduce cold-load latency and memory pressure.",
      "vendorResponse": "The vendor provides native API tokens and permissions, recommends authentication for network binds, documents LM Link routing, and exposes idle TTL and Auto-Evict controls.",
      "lesson": "Verify effective server settings, token permissions and execution location. Measure first-load and repeated-request behaviour under the model lifecycle settings your application actually uses.",
      "questions": [
        "Do requests without a valid token fail on the deployed API?",
        "Which inference, model-management and tool permissions does each token have?",
        "Does LM Link resolve the model to an approved device?",
        "How do JIT loading, explicit loads, idle TTL and Auto-Evict affect latency and memory?"
      ],
      "limits": "This is documented configuration behaviour, not a security advisory, CVE or an observed compromise. The offline documentation describes local operation; enabled remote links and integrations require their own data-flow review.",
      "identifierNotes": [
        "No CVE is assigned by this case. Evidence type: documented behaviour.",
        "The vendor distinguishes JIT-loaded models from explicitly loaded models. Auto-Evict does not apply indiscriminately to every model in memory."
      ],
      "findingIds": [],
      "reviewed": "2026-10-07",
      "status": "published"
    }
  ]
}
