From 4ea8f849d4d52ce11ace3ad7c66e4faefef18797 Mon Sep 17 00:00:00 2001 From: sjungwon03 Date: Tue, 6 Oct 2026 14:40:34 +0900 Subject: [PATCH] gateway: support model output modality lists Refs: #472 Assisted-by: Codex Signed-off-by: sjungwon03 --- contracts/model-discovery-filters.md | 6 +- contracts/model-output-filters.md | 15 ++ docs/PRD.md | 8 +- docs/acceptance.md | 8 +- docs/architecture.md | 8 +- docs/openrouter-compatibility.md | 6 +- docs/plans/472-model-output-filters.md | 44 ++++ src/gateway/chat-handler.ts | 8 +- src/gateway/model-list-query.ts | 18 +- test/model-discovery-filters.test.ts | 2 +- test/model-output-filters.test.ts | 243 +++++++++++++++++++++++ test/sdk-model-discovery-filters.test.ts | 78 +++++++- 12 files changed, 428 insertions(+), 16 deletions(-) create mode 100644 contracts/model-output-filters.md create mode 100644 docs/plans/472-model-output-filters.md create mode 100644 test/model-output-filters.test.ts diff --git a/contracts/model-discovery-filters.md b/contracts/model-discovery-filters.md index 0b7ada2..92e1a17 100644 --- a/contracts/model-discovery-filters.md +++ b/contracts/model-discovery-filters.md @@ -2,16 +2,16 @@ Issue #456. Plan: [456-model-discovery-filters](../docs/plans/456-model-discovery-filters.md). -GET /api/v1/models accepts optional singleton output_modalities, supported_parameters and context alongside offset/limit and the [input/search/order extension](model-discovery-exploration.md). Output modality is one of text/image/embeddings/audio/video/rerank/decisions/speech/transcription, or all. Supported parameter is one exact lower_snake_case identifier of 1..128 characters. Context is a canonical positive decimal safe integer. Duplicate keys, comma lists, blanks, unknown fields, casing/whitespace variants and invalid bounds reject safely before catalog reads. These are local syntax restrictions, not bounds invented in the source schema. /v1 continues rejecting every nonempty query. +GET /api/v1/models accepts optional output_modalities lists, singleton supported_parameters and context alongside offset/limit and the [input/search/order extension](model-discovery-exploration.md). Output modalities are one to nine distinct values from text/image/embeddings/audio/video/rerank/decisions/speech/transcription, matching any, or standalone all; see [output list contract](model-output-filters.md). Supported parameter is one exact lower_snake_case identifier of 1..128 characters. Context is a canonical positive decimal safe integer. Duplicate keys/items, unsupported input/parameter comma lists, blanks, unknown fields, casing/whitespace variants and invalid bounds reject safely before catalog reads. These are local syntax restrictions, not bounds invented in the source schema. /v1 continues rejecting every nonempty query. No filter is implicit. Filter-only requests return all matching aliases without a 500-item page default. Supplying offset or limit uses existing offset=0/limit=500 defaults and limit1..1000. The complete current catalog is validated first, including disabled/denied/out-of-page records. Only enabled aliases passing model AND final-provider IAM are eligible; then all supplied metadata predicates must hold, then optional stable discovery ordering precedes paging; omitted sort retains catalog order. -Use immutable administrator-published metadata snapshots: exact output/parameter array membership, and known context_length >= context. Missing metadata fails asserted conditions; null context_length fails a context condition. output_modalities=all imposes no modality condition and retains basic aliases if no other condition excludes them. Conjunction is a documented local subset; upstream multi-filter combination behavior is unspecified. Do not infer capabilities from aliases, route kind, top-provider limits, defaults or string modality labels. Model-level parameter metadata may be a union across providers, so a match cannot establish support on an authorized or selected endpoint. +Use immutable administrator-published metadata snapshots: any-member output and exact singleton parameter array membership, and known context_length >= context. Missing metadata fails asserted conditions; null context_length fails a context condition. output_modalities=all imposes no modality condition and retains basic aliases if no other condition excludes them. Conjunction is a documented local subset; upstream multi-filter combination behavior is unspecified. Do not infer capabilities from aliases, route kind, top-provider limits, defaults or string modality labels. Model-level parameter metadata may be a union across providers, so a match cannot establish support on an authorized or selected endpoint. total_count counts only currently authorized matching aliases. Next links have fixed relative /api/v1/models destination and validated offset/limit plus every accepted filter/search/order field. No incoming host, hidden model metadata or unknown filter can influence the link. Full lists, exhausted/empty pages and beyond-end offsets have next=null. Each continuation authenticates, rereads current catalog and reevaluates IAM; changes can shift offsets without snapshot/cursor guarantees. Required models-listed audit records only returned count and established attribution before output. Invalid queries and catalog/audit failure have sanitized existing errors/events; neither filter values nor private metadata enters operational records. Listing invokes no inference routes, provider secrets, limit or usage ports. Informational metadata does not grant invocation rights or certify live capability/prices. Installed OpenRouter 1.4.18 socket tests verify scalar query serialization, filtered pagination and fresh denial; OpenAI 7.23.0 sockets can consume the standard list envelope through the same filtered URL. -The version-30 chat pin stays unchanged. Separate model-query pin version2 structurally tracks the eight implemented query objects. Bounded input/search/order behavior is defined in its extension contract; multi-values/category/other sorts/provider/region filters, metadata refresh/provisioning, broader external-client workflows, #116 and unresolved #7 remain open. +The version-31 chat pin stays unchanged by the output-list extension. Separate model-query pin version2 structurally tracks the eight implemented query objects. Bounded input/search/order behavior is defined in its extension contract; input/parameter multi-values/category/other sorts/provider/region filters, metadata refresh/provisioning, broader external-client workflows, #116 and unresolved #7 remain open. Sources: [official models reference](https://openrouter.ai/docs/api/api-reference/models/get-models), [official schema](https://openrouter.ai/openapi.json). diff --git a/contracts/model-output-filters.md b/contracts/model-output-filters.md new file mode 100644 index 0000000..5ad1151 --- /dev/null +++ b/contracts/model-output-filters.md @@ -0,0 +1,15 @@ +# Authorized output modality lists + +Issue #472. Plan: [472-model-output-filters](../docs/plans/472-model-output-filters.md). + +GET /api/v1/models accepts output_modalities as one comma-separated string containing one to nine distinct values from text/image/embeddings/audio/video/rerank/decisions/speech/transcription. A model matches if its captured administrator-published output_modalities contains any requested value. all remains a standalone sentinel imposing no output condition; mixing it with a modality rejects. Omission retains the complete current authorized list without an implicit text default. Missing/empty metadata cannot satisfy an explicit list. + +Empty items, duplicate items, unknown values, casing/whitespace variants and repeated query keys reject400 before catalog reads. This strict syntax is a local subset: the source schema is a plain string, and the upstream accepts duplicates. Input modalities and supported parameters remain singleton. Different supplied fields still combine conjunctively, followed by established ordering and optional paging. + +Validate the entire catalog, then enabled model AND final-provider IAM, before any-member filtering. Denied/disabled aliases never affect matching totals or offsets. Fixed relative continuation links retain every accepted field and the output list in its original validated order as one URL-encoded scalar. Each page authenticates and reevaluates current catalog/IAM without snapshot guarantees. Filter-only lists retain full-list behavior above500; paging bounds/defaults remain unchanged. /v1 still rejects query strings. + +Required sanitized audit precedes output. Listing calls no route, inference, provider-secret, limit or usage ports. Errors/events exclude filters and private metadata; whole-catalog, audit failure and immutable metadata guarantees remain shared. A metadata match grants no invocation permission or selected-provider capability guarantee. + +Installed OpenRouter1.4.18 and OpenAI7.23.0 socket cases exercise union selection, retained pagination and fresh Deny. All three source pins remain unchanged. Input/parameter lists, other discovery filters, metadata refresh, full #116 and unresolved #7 remain open. + +Sources checked2026-10-06: [official models guide](https://openrouter.ai/docs/guides/overview/models), [official models reference](https://openrouter.ai/docs/api/api-reference/models/get-models). Union semantics are supported by the guide example and credential-free official response observations containing both text-only and image-only models; exact implementation and local syntax bounds are not encoded in the structural schema. diff --git a/docs/PRD.md b/docs/PRD.md index ad784d9..fae33f7 100644 --- a/docs/PRD.md +++ b/docs/PRD.md @@ -916,7 +916,7 @@ Version 30 adds only the raw nullable service_tier enum and unknown-value extens ## Authorized model discovery filters (#456) -GET /api/v1/models supports bounded singleton output_modalities (one documented modality or all), supported_parameters (one exact lower_snake_case token, at most 128 characters), and context (canonical positive safe integer minimum). Omitted filters preserve the current list without an implicit text default. Multi-values and undocumented normalization/combination behavior remain unimplemented; conjunction is an explicit local subset. Filter-only requests retain full-list behavior; offset/limit keep existing defaults and bounds. /v1 still rejects queries. +GET /api/v1/models supports bounded output_modalities (one to nine distinct documented modalities, matching any, or standalone all; extended by #472), supported_parameters (one exact lower_snake_case token, at most 128 characters), and context (canonical positive safe integer minimum). Omitted filters preserve the current list without an implicit text default. Input/parameter multi-values and undocumented normalization remain unimplemented; conjunction across different filters is an explicit local subset. Output list union is defined in the [output contract](../contracts/model-output-filters.md). Filter-only requests retain full-list behavior; offset/limit keep existing defaults and bounds. /v1 still rejects queries. Validate the whole catalog and fresh enabled model/final-provider IAM before applying every asserted condition to frozen administrator metadata and then paging in catalog order. Missing metadata/null context cannot establish a filtered capability; output all imposes no modality condition. total_count and fixed relative continuation links describe only authorized matching aliases and preserve all accepted filters. Required audit precedes delivery; filters/metadata stay out of operational events/errors. Listing calls no secrets, limits, inference routes or usage ports, and cannot grant routing authority. Metadata may describe a model-level union across providers; it does not certify feature support on the selected authorized endpoint. @@ -973,3 +973,9 @@ All three source pins remain byte-identical. Jev text disclosure still rejects i Measure OpenCode 1.18.5 local PNG attachments through delegated OpenRouter and managed OpenAI/Anthropic/Gemini on both bases. Use an explicitly image-capable fixture alias, omitted detail and exact inline/native bytes; require rendered text and image-preserving actual read-function/result continuation. Verify initial model/provider Deny, fresh result-follow-up provider Deny, disconnect accounting and private metadata. Keep isolated temporary configuration, fixed mocked hosts and bounded children. This expands named-client conformance and fixes advisory session-header scope; runtime image policy and unsupported native body session_id remain unchanged. It does not certify live models, remote images, general image capability or complete #116. See [contract](../contracts/opencode-inline-images.md). OpenCode automatically sends X-Session-Id. Validate and capture bounded selected headers before awaits, but turn header-only identifiers into session_id only for delegated OpenRouter routes after resolving the approved route. Managed calls omit header-derived identifiers, while explicit native body session_id still rejects before secrets. This header never controls authentication, IAM, route choice, limits or accounting identity. + +## Authorized output modality lists (#472) + +GET /api/v1/models accepts one comma-separated output_modalities value with one to nine distinct exact modalities. Select any matching captured published output after enabled model AND final-provider IAM; missing/empty metadata cannot establish a match. Keep all standalone and omitted-filter behavior without an implicit text default. Reject empty/duplicate/unknown/case/whitespace items, mixed all and repeated query keys before catalog reads. Duplicate rejection and finite vocabulary bounds are explicit local restrictions. + +Retain conjunction with other fields, search/order/paging, full-list behavior above500, and fixed relative continuation links preserving validated list order. Each page reevaluates current catalog/IAM. Whole-catalog validation, required private audit, metadata capture, safe dependency errors and no inference/secret/limit/usage calls remain enforced. Installed OpenRouter1.4.18/OpenAI7.23.0 socket tests cover union, pagination and fresh Deny. All three source pins remain byte-identical; metadata freshness, input/parameter lists, full #116 and unresolved #7 remain open. See [plan](plans/472-model-output-filters.md) and [contract](../contracts/model-output-filters.md). diff --git a/docs/acceptance.md b/docs/acceptance.md index 1c797b6..affecdc 100644 --- a/docs/acceptance.md +++ b/docs/acceptance.md @@ -1312,7 +1312,7 @@ Version 30 adds only the raw nullable service_tier enum and unknown-value extens ## Authorized model discovery filters (#456) -GET /api/v1/models supports bounded singleton output_modalities (one documented modality or all), supported_parameters (one exact lower_snake_case token, at most 128 characters), and context (canonical positive safe integer minimum). Omitted filters preserve the current list without an implicit text default. Multi-values and undocumented normalization/combination behavior remain unimplemented; conjunction is an explicit local subset. Filter-only requests retain full-list behavior; offset/limit keep existing defaults and bounds. /v1 still rejects queries. +GET /api/v1/models supports bounded output_modalities (one to nine distinct documented modalities, matching any, or standalone all; extended by #472), supported_parameters (one exact lower_snake_case token, at most 128 characters), and context (canonical positive safe integer minimum). Omitted filters preserve the current list without an implicit text default. Input/parameter multi-values and undocumented normalization remain unimplemented; conjunction across different filters is an explicit local subset. Output list union is defined in the [output contract](../contracts/model-output-filters.md). Filter-only requests retain full-list behavior; offset/limit keep existing defaults and bounds. /v1 still rejects queries. Validate the whole catalog and fresh enabled model/final-provider IAM before applying every asserted condition to frozen administrator metadata and then paging in catalog order. Missing metadata/null context cannot establish a filtered capability; output all imposes no modality condition. total_count and fixed relative continuation links describe only authorized matching aliases and preserve all accepted filters. Required audit precedes delivery; filters/metadata stay out of operational events/errors. Listing calls no secrets, limits, inference routes or usage ports, and cannot grant routing authority. Metadata may describe a model-level union across providers; it does not certify feature support on the selected authorized endpoint. @@ -1378,3 +1378,9 @@ All three source pins remain byte-identical. Jev text disclosure still rejects i - Default CI verifies socket/config/assertion cases; the explicit installed-client gate adds 48 image probes to the existing fifty. No live inference, automatic discovery, signed-image combinations or complete #116 certification is inferred; #7 remains unresolved. - A bounded header-only X-Session-Id request succeeds natively without a session_id body field; explicit native body identifiers still reject before secrets, including with an invalid unselected header. Selected overlong headers reject before route lookup. Delegated body/header precedence and captured values survive asynchronous mutation. Header-only native success, auth/IAM/limit denial, missing usage, upstream failure and required persistence failures preserve their existing controls. + +## Authorized output modality lists (#472) + +GET /api/v1/models accepts one comma-separated output_modalities value with one to nine distinct exact modalities. Select any matching captured published output after enabled model AND final-provider IAM; missing/empty metadata cannot establish a match. Keep all standalone and omitted-filter behavior without an implicit text default. Reject empty/duplicate/unknown/case/whitespace items, mixed all and repeated query keys before catalog reads. Duplicate rejection and finite vocabulary bounds are explicit local restrictions. + +Retain conjunction with other fields, search/order/paging, full-list behavior above500, and fixed relative continuation links preserving validated list order. Each page reevaluates current catalog/IAM. Whole-catalog validation, required private audit, metadata capture, safe dependency errors and no inference/secret/limit/usage calls remain enforced. Installed OpenRouter1.4.18/OpenAI7.23.0 socket tests cover union, pagination and fresh Deny. All three source pins remain byte-identical; metadata freshness, input/parameter lists, full #116 and unresolved #7 remain open. See [plan](plans/472-model-output-filters.md) and [contract](../contracts/model-output-filters.md). diff --git a/docs/architecture.md b/docs/architecture.md index eb19fcc..f54c61a 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -1121,7 +1121,7 @@ Version 30 adds only the raw nullable service_tier enum and unknown-value extens ## Authorized model discovery filters (#456) -GET /api/v1/models supports bounded singleton output_modalities (one documented modality or all), supported_parameters (one exact lower_snake_case token, at most 128 characters), and context (canonical positive safe integer minimum). Omitted filters preserve the current list without an implicit text default. Multi-values and undocumented normalization/combination behavior remain unimplemented; conjunction is an explicit local subset. Filter-only requests retain full-list behavior; offset/limit keep existing defaults and bounds. /v1 still rejects queries. +GET /api/v1/models supports bounded output_modalities (one to nine distinct documented modalities, matching any, or standalone all; extended by #472), supported_parameters (one exact lower_snake_case token, at most 128 characters), and context (canonical positive safe integer minimum). Omitted filters preserve the current list without an implicit text default. Input/parameter multi-values and undocumented normalization remain unimplemented; conjunction across different filters is an explicit local subset. Output list union is defined in the [output contract](../contracts/model-output-filters.md). Filter-only requests retain full-list behavior; offset/limit keep existing defaults and bounds. /v1 still rejects queries. Validate the whole catalog and fresh enabled model/final-provider IAM before applying every asserted condition to frozen administrator metadata and then paging in catalog order. Missing metadata/null context cannot establish a filtered capability; output all imposes no modality condition. total_count and fixed relative continuation links describe only authorized matching aliases and preserve all accepted filters. Required audit precedes delivery; filters/metadata stay out of operational events/errors. Listing calls no secrets, limits, inference routes or usage ports, and cannot grant routing authority. Metadata may describe a model-level union across providers; it does not certify feature support on the selected authorized endpoint. @@ -1180,3 +1180,9 @@ All three source pins remain byte-identical. Jev text disclosure still rejects i The isolated runner attaches a valid synthetic one-pixel PNG using the pinned CLI file option and explicit fixture text/image input modalities. Reuse the real Node gateway and existing fixed-host native/delegated transport fixtures. Inspect image-bearing captured requests for exact omitted-detail URL/native MIME/base64, and require the attachment on every tool-result continuation. Image probes supply a fixed title to suppress auxiliary title generation and require exactly one image on every captured request. Existing non-image probes retain auxiliary calls. Cover six modes on four route/provider combinations and both bases while retaining child/config isolation, fixture-only tool permissions, authentication/IAM/limits and private usage/audit. The HTTP header fallback is scoped to delegated routes after approved route resolution; explicit native body session_id remains unsupported. No native session equivalent, discovery, schema-pin or authorization policy change is introduced. See [plan](plans/470-opencode-inline-images.md). Preserve the body-presence marker from parsed JSON before asynchronous route resolution. Existing normalization validates/captures the selected identifier once. After resolving a managed route, remove only a header-derived session_id from the immutable-content request copy before shared invocation. Explicit body identifiers retain native rejection; delegated body/header precedence and exact forwarding remain unchanged. Do not forward arbitrary headers or substitute native user/cache/metadata. + +## Authorized output modality lists (#472) + +GET /api/v1/models accepts one comma-separated output_modalities value with one to nine distinct exact modalities. Select any matching captured published output after enabled model AND final-provider IAM; missing/empty metadata cannot establish a match. Keep all standalone and omitted-filter behavior without an implicit text default. Reject empty/duplicate/unknown/case/whitespace items, mixed all and repeated query keys before catalog reads. Duplicate rejection and finite vocabulary bounds are explicit local restrictions. + +Retain conjunction with other fields, search/order/paging, full-list behavior above500, and fixed relative continuation links preserving validated list order. Each page reevaluates current catalog/IAM. Whole-catalog validation, required private audit, metadata capture, safe dependency errors and no inference/secret/limit/usage calls remain enforced. Installed OpenRouter1.4.18/OpenAI7.23.0 socket tests cover union, pagination and fresh Deny. All three source pins remain byte-identical; metadata freshness, input/parameter lists, full #116 and unresolved #7 remain open. See [plan](plans/472-model-output-filters.md) and [contract](../contracts/model-output-filters.md). diff --git a/docs/openrouter-compatibility.md b/docs/openrouter-compatibility.md index 7839e78..41bd3b8 100644 --- a/docs/openrouter-compatibility.md +++ b/docs/openrouter-compatibility.md @@ -25,7 +25,7 @@ Tools with a hardcoded openrouter.ai host need a configurable endpoint or an int | Area | Current state | Remaining acceptance gate | | --- | --- | --- | | Base paths and Bearer token | /api/v1 chat/models aliases; shared proxy authorization; pinned OpenAI SDK smoke tests | Broader direct streaming and named-client workflows; OpenCode 1.18.5 explicit custom-provider registration/text is verified on delegated and managed OpenAI/Anthropic/Gemini fixture routes | -| Model discovery | IAM-filtered aliases; optional administrator-published SDK-required discovery metadata and bounded offset/limit paging and singleton input/output/parameter/minimum-context filters, literal name/slug/alias search and bounded creation/context ordering on /api/v1; separate supported-query structural pin | Metadata provisioning/refresh, multi-value and remaining filters/sorts, broader optional fields and complete discovery workflows | +| Model discovery | IAM-filtered aliases; optional administrator-published SDK-required discovery metadata and bounded offset/limit paging and singleton input/parameter/minimum-context filters and bounded any-member output lists, literal name/slug/alias search and bounded creation/context ordering on /api/v1; separate supported-query structural pin | Metadata provisioning/refresh, input/parameter multi-value and remaining filters/sorts, broader optional fields and complete discovery workflows | | Non-streaming text chat | One normalized choice with portable output/sampling controls, provider-bounded verbosity/effort, delegated min_p/top_a/repetition_penalty, bounded OpenAI/OpenRouter logprobs/top_logprobs with content/refusal token alternatives and bytes, and Gemini chosen/alternative probabilities with unavailable bytes | Anthropic probability mapping, remaining request/response schema, sampling and capability metadata | | Streaming | Managed OpenAI/delegated OpenRouter nullable probability controls and bounded choice-level content/refusal token alternatives/bytes, including private final usage-choice probabilities after required handoffs; managed Gemini HTTP text/function streams with exact initial version identity, complete objects with bounded same-part signatures through raw HTTP/OpenAI SDK and explicitly configured OpenCode, dense call indices, clean framed EOF, final reported totals and actual SDK/persisted direct/dual checks on both bases; managed Anthropic HTTP text/function streams with bounded JSON arguments, generated persisted direct/dual wiring, cumulative aggregate usage and actual OpenAI/OpenRouter SDK checks on both bases; managed OpenAI HTTP text/refusal/indexed function streams with per-attempt usage/audit, generated persisted direct/dual wiring and actual SDK checks on both bases; delegated HTTP text/refusal/scalar/detail reasoning and indexed function streams with bounded validation, awaited delivery, cancellation, final usage metadata and interruption audit; both installed SDKs exercise function workflows on both bases; OpenCode 1.18.5 reads a fixture through a streamed function and completes correlated results on delegated and managed OpenAI/Anthropic/Gemini fixture routes | Gemini partial-argument streams and Anthropic/Gemini thinking/server-tool streams, native custom/multimodal and other tool variants, unselected schema targets, additional stream option fields and full named external-client conformance | | Tool calling | Validated function-tool requests and complete text-only result histories; delegated OpenRouter nonstream/stream assistant calls and direct OpenAI nonstream/stream calls and bounded managed Anthropic nonstream/stream declarations/choices/calls/correlated results; bounded managed Gemini nonstream/stream declarations/choices/calls/results with thinking disabled, plus official nonstream/stream same-part tool-call signature replay through raw HTTP/OpenAI SDK and explicitly configured OpenCode; SDK socket tests cover two-function continuations and fresh IAM | Server/custom tools, rich content, other native mappings and broader named external-tool workflows and application configurations | @@ -985,3 +985,7 @@ All three source pins remain byte-identical. Jev text disclosure still rejects i The pinned named application attaches a valid synthetic PNG through its actual CLI, with explicit image input modalities and omitted detail. Forty-eight new probes cover delegated OpenRouter and managed OpenAI/Anthropic/Gemini on both bases across text, actual read-result continuation, initial model/provider Deny, fresh follow-up Deny and cancellation. Exact inline/native bytes and private metadata are checked; no live inference or signed-image/client-version/general image certification follows. See [plan](plans/470-opencode-inline-images.md), [contract](../contracts/opencode-inline-images.md) and [reproduction guide](opencode-conformance.md). The #452 advisory-header regression is corrected: header-only X-Session-Id is validated/captured before awaits and converted to session_id only for delegated OpenRouter after route resolution. Managed bodies omit that header-derived field; explicit body values remain unsupported natively. Authentication/IAM/routing/limits/accounting identity remain independent. + +## Output modality list extension (#472) + +Compatible model discovery now accepts bounded any-member output lists as one comma string, retaining standalone all, omitted-filter behavior, cross-filter conjunction, search/order/paging, current IAM and required audit. Public and installed OpenRouter/OpenAI SDK cases cover union, continuation retention and fresh Deny. Local bounds/duplicate rejection remain stricter than the upstream plain-string schema; input/parameter lists and full discovery remain open. See [plan](plans/472-model-output-filters.md) and [contract](../contracts/model-output-filters.md). Existing three source pins are unchanged. diff --git a/docs/plans/472-model-output-filters.md b/docs/plans/472-model-output-filters.md new file mode 100644 index 0000000..09da8a7 --- /dev/null +++ b/docs/plans/472-model-output-filters.md @@ -0,0 +1,44 @@ +# Model output modality list plan + +## Issue and problem + +- Issue #472; release gate #116 and unresolved #7 remain open. +- The compatible model list rejects documented output_modalities=text,image. Existing singleton filters cannot discover multiple output types in one request. + +## Scope and expected behavior + +- GET /api/v1/models accepts one to nine distinct exact output modalities in one comma-separated string; match any listed modality. Keep all standalone and preserve omitted-filter behavior with no implicit text default. +- Keep input_modalities and supported_parameters singleton. Preserve conjunction across different filter fields, search, ordering, paging defaults and legacy /v1 query rejection. +- Reject blank/duplicate/unknown/case/whitespace items, mixed all and repeated keys before catalog reads. Duplicate rejection and finite vocabulary bounds are explicit local restrictions; upstream accepts duplicate values. +- Evaluate enabled model and final-provider IAM before metadata filtering. Missing metadata fails any explicit output list. Validate the whole catalog first. Fresh catalog/IAM applies independently to each page. +- Required sanitized audit precedes output; filters/metadata stay out of operational events/errors. Listing invokes no inference, route, credential, limit or usage ports. + +## Design + +- Capture a frozen validated output list in the pure query parser; use any exact membership against captured administrator metadata. Continuations join the validated list into one encoded scalar and retain its order. +- Official [models guide](https://openrouter.ai/docs/guides/overview/models) documents text,image. Credential-free official list probes on 2026-10-06 returned both text-only and image-only entries for text,image and image,text, supporting union semantics by empirical inference. all,image rejected; text,text was accepted upstream. Official schema uses a plain string, without local bounds or combination rules. +- Installed OpenRouter1.4.18 serializes a comma string as one query value and retains it during iteration. No new dependency, schema selection, metadata provisioner, routing policy, glossary term or ADR is needed. +- Leave other multi-value semantics and model capability/metadata freshness unresolved. Update [PRD](../PRD.md), [architecture](../architecture.md), [acceptance](../acceptance.md), [compatibility](../openrouter-compatibility.md), and [contract](../../contracts/model-output-filters.md). + +## TDD plan + +- First public regression expects text-only and image-only authorized aliases for text,image; current parser returns400. Installed SDK requests reproduce rejection before production edits. +- Cover reversed order, all nine values, exact URL encoding, conjunction/search/order, full lists over500, paged retention, fresh Deny, hidden/disabled metadata, missing/empty metadata, invalid syntax, legacy queries, authentication, whole-catalog and required audit failures, audit-time metadata mutation and operational privacy. +- Smallest change: output list validation/capture, any-member predicate and scalar continuation encoding. Keep other fields unchanged. +- Format changed files; run focused public/SDK discovery tests and npm run check. Existing three source pins remain byte-identical; runtime tests do not certify complete schemas or live capabilities. + +## Delivery + +- Issue, new branch and plan precede coding; record meaningful red/green. Publish a focused PR, inspect exact diff, wait for both exact-head check jobs, review as sjungwon03-ai and squash merge as sjungwon03. +- Rollback is code-only. Published metadata can be stale and model-level capability unions cannot guarantee selected-provider support. Offset pages can shift when policy/catalog changes; no snapshot guarantee. Basic aliases remain outside rich OpenRouter SDK metadata validation. + +## Verification evidence + +GET /api/v1/models previously rejected documented output_modalities=text,image. It now accepts one to nine distinct exact output modalities as one comma-separated value and selects any matching captured published output after enabled model AND final-provider IAM. Standalone all and omitted-filter behavior are preserved. Different fields retain conjunction; search/order/paging and fixed relative continuations keep validated list order. Input/parameter multi-values remain unsupported. + +Red evidence: 38 new public regressions initially produced14 expected failures (valid lists returned400) and24 invariant passes; four new actual SDK socket cases all failed on gateway400 before production edits. Green evidence: all42 new cases and186 focused discovery tests pass. Coverage includes both route kinds, all nine outputs, union/reversed order/no-match, missing/empty metadata, combined search/order/filters, filter-only501 rows, exact continuation encoding, authentication, model/provider/implicit/explicit Deny and fresh SDK page denial, malformed whole-catalog metadata, required audit/catalog failures, source mutation and sanitized operational records. Installed OpenRouter1.4.18 injects default paging; OpenAI7.23.0 retains filter-only requests. Ten added successful SDK requests exercise union, iteration and policy changes. + +Official guide and credential-free official list observations support output union; that semantic evidence is distinct from the existing plain-string structural pin. Duplicate rejection and finite vocabulary bounds are explicit local restrictions (upstream accepts duplicates). Omitted filters still have no implicit text default. All three source pins remain byte-identical. Published metadata can be stale and cannot guarantee selected-provider capability. Offset pages can shift with current policy/catalog; there is no snapshot guarantee. Rich OpenRouter SDK metadata requirements, input/parameter lists, other discovery filters, metadata refresh, complete #116 and unresolved #7 remain open. Listing still invokes no inference, route, secret, limit or usage ports. + + +Full npm run check passes strict types, lint, 5595 tests with one existing PostgreSQL skip, planning/contracts/fixture checks and offline schema integrity. diff --git a/src/gateway/chat-handler.ts b/src/gateway/chat-handler.ts index d1beb7d..d1ba98d 100644 --- a/src/gateway/chat-handler.ts +++ b/src/gateway/chat-handler.ts @@ -1018,9 +1018,11 @@ export function createChatHandler( [model.alias, published?.name, published?.canonical_slug].some( (text) => text?.toLowerCase().includes(search) === true, )) && - (query.outputModality === undefined || - query.outputModality === 'all' || - published?.architecture.output_modalities.includes(query.outputModality) === true) && + (query.outputModalities === undefined || + query.outputModalities.includes('all') || + query.outputModalities.some( + (modality) => published?.architecture.output_modalities.includes(modality) === true, + )) && (query.supportedParameter === undefined || published?.supported_parameters.includes(query.supportedParameter) === true) && (query.minimumContextLength === undefined || diff --git a/src/gateway/model-list-query.ts b/src/gateway/model-list-query.ts index 1c3aa16..bfe6569 100644 --- a/src/gateway/model-list-query.ts +++ b/src/gateway/model-list-query.ts @@ -1,7 +1,7 @@ export interface ModelListQuery { readonly offset: number; readonly limit?: number; - readonly outputModality?: string; + readonly outputModalities?: readonly string[]; readonly supportedParameter?: string; readonly minimumContextLength?: number; readonly inputModality?: string; @@ -31,7 +31,14 @@ export function parseModelListQuery(url: URL, compatible: boolean): ModelListQue for (const [key, value] of params) { if (params.getAll(key).length !== 1) return undefined; if (key === 'output_modalities') { - if (!outputModalities.includes(value)) return undefined; + const selected = value.split(','); + if ( + selected.length > 9 || + new Set(selected).size !== selected.length || + selected.some((item) => !outputModalities.includes(item)) || + (selected.includes('all') && selected.length !== 1) + ) + return undefined; } else if (key === 'input_modalities') { if (!['text', 'image', 'audio', 'file'].includes(value)) return undefined; } else if (key === 'q') { @@ -67,7 +74,9 @@ export function parseModelListQuery(url: URL, compatible: boolean): ModelListQue return { offset, ...(params.has('offset') || params.has('limit') ? { limit } : {}), - ...(outputModality === null ? {} : { outputModality }), + ...(outputModality === null + ? {} + : { outputModalities: Object.freeze(outputModality.split(',')) }), ...(supportedParameter === null ? {} : { supportedParameter }), ...(context === null ? {} : { minimumContextLength: Number(context) }), ...(inputModality === null ? {} : { inputModality }), @@ -80,7 +89,8 @@ export function parseModelListQuery(url: URL, compatible: boolean): ModelListQue export function modelListContinuation(query: ModelListQuery, offset: number): string | null { if (query.limit === undefined) return null; const params = new URLSearchParams({ offset: String(offset), limit: String(query.limit) }); - if (query.outputModality !== undefined) params.set('output_modalities', query.outputModality); + if (query.outputModalities !== undefined) + params.set('output_modalities', query.outputModalities.join(',')); if (query.supportedParameter !== undefined) params.set('supported_parameters', query.supportedParameter); if (query.minimumContextLength !== undefined) diff --git a/test/model-discovery-filters.test.ts b/test/model-discovery-filters.test.ts index e78c15d..90b7ea5 100644 --- a/test/model-discovery-filters.test.ts +++ b/test/model-discovery-filters.test.ts @@ -164,7 +164,7 @@ for (const query of [ '?output_modalities=null', '?output_modalities=TEXT', '?output_modalities=unknown', - '?output_modalities=text,image', + '?output_modalities=text,unknown', '?output_modalities=all,text', '?output_modalities=%20text', '?output_modalities=text&output_modalities=text', diff --git a/test/model-output-filters.test.ts b/test/model-output-filters.test.ts new file mode 100644 index 0000000..446ff80 --- /dev/null +++ b/test/model-output-filters.test.ts @@ -0,0 +1,243 @@ +import assert from 'node:assert/strict'; +import test from 'node:test'; +import { + discoveryAllow, + discoveryFixture, + discoveryModel, + discoveryPrivacy, +} from './model-discovery-filters-fixture.ts'; + +type List = { + data: Array<{ id: string; architecture?: { output_modalities: string[] } }>; + total_count: number; + links: { next: string | null }; +}; +const outputs = [ + 'text', + 'image', + 'embeddings', + 'audio', + 'video', + 'rerank', + 'decisions', + 'speech', + 'transcription', +]; +for (const [value, expected] of [ + ['text,image', ['small', 'multi', 'unknown', 'large', 'image']], + ['image,text', ['small', 'multi', 'unknown', 'large', 'image']], + ['image,audio', ['multi', 'large', 'image']], + ['audio,video', []], +] as const) + test(`output list matches any authorized published modality: ${value}`, async () => { + const f = discoveryFixture(); + const response = await f.handler(f.request(`?output_modalities=${value}`)); + assert.equal(response.status, 200); + const body = (await response.json()) as List; + assert.deepEqual( + body.data.map((m) => m.id), + expected, + ); + assert.equal(body.total_count, expected.length); + assert.equal(body.links.next, null); + assert.equal(f.reads(), 1); + discoveryPrivacy(f); + }); +for (const kind of ['managed', 'delegated'] as const) + test(`all nine output values match exact metadata on ${kind} without inferring missing capability`, async () => { + const f = discoveryFixture([ + ...outputs.map((output) => discoveryModel(output, { outputs: [output], kind })), + discoveryModel('basic', { basic: true, kind }), + discoveryModel('empty', { outputs: [], kind }), + discoveryModel('other', { outputs: ['future-output'], kind }), + ]); + const response = await f.handler(f.request(`?output_modalities=${outputs.join(',')}`)); + assert.equal(response.status, 200); + assert.deepEqual( + ((await response.json()) as List).data.map((m) => m.id), + outputs, + ); + discoveryPrivacy(f); + }); +test('output union combines with input, parameter, context, literal search and stable ordering before paging', async () => { + const f = discoveryFixture([ + discoveryModel('Atlas image', { outputs: ['image'], created: 1 }), + discoveryModel('Atlas text', { outputs: ['text'], created: 2 }), + discoveryModel('Atlas wrong-input', { inputs: ['audio'] }), + discoveryModel('Atlas wrong-parameter', { parameters: [] }), + discoveryModel('Atlas small', { context: 1 }), + discoveryModel('Atlas wrong-output', { outputs: ['audio'] }), + discoveryModel('not-search', { name: 'another', slug: 'another' }), + discoveryModel('Atlas disabled', { enabled: false }), + discoveryModel('private-denied'), + discoveryModel('Atlas provider-denied', { provider: 'private-provider' }), + ]); + const query = + '?output_modalities=image%2Ctext&input_modalities=text&supported_parameters=tools&context=8192&q=Atlas&sort=newest&limit=1'; + const response = await f.handler(f.request(query)); + assert.equal(response.status, 200); + const first = (await response.json()) as List; + assert.deepEqual( + first.data.map((m) => m.id), + ['Atlas text'], + ); + assert.equal(first.total_count, 2); + assert.ok(first.links.next); + const url = new URL(first.links.next, 'http://localhost'); + assert.equal(url.pathname, '/api/v1/models'); + assert.equal(url.searchParams.get('output_modalities'), 'image,text'); + assert.equal(url.searchParams.getAll('output_modalities').length, 1); + for (const key of ['input_modalities', 'supported_parameters', 'context', 'q', 'sort', 'limit']) + assert.equal(url.searchParams.get(key), new URLSearchParams(query).get(key)); + assert.equal(url.searchParams.size, 8); + assert.doesNotMatch(first.links.next, /untrusted|private/u); + const second = await f.handler(f.request(url.search)); + assert.equal(second.status, 200); + const last = (await second.json()) as List; + assert.deepEqual( + last.data.map((m) => m.id), + ['Atlas image'], + ); + assert.equal(last.total_count, 2); + assert.equal(last.links.next, null); + discoveryPrivacy(f); +}); +test('output union without paging returns every matching alias above500', async () => { + const f = discoveryFixture( + Array.from({ length: 501 }, (_, i) => + discoveryModel(`alias-${i}`, { outputs: [i % 2 ? 'image' : 'text'] }), + ), + ); + const response = await f.handler(f.request('?output_modalities=text,image')); + assert.equal(response.status, 200); + const body = (await response.json()) as List; + assert.equal(body.data.length, 501); + assert.equal(body.total_count, 501); + assert.equal(body.links.next, null); +}); +for (const value of [ + 'text,', + ',text', + 'text,,image', + 'text,text', + 'image,text,image', + 'all,text', + 'text,all', + 'all,all', + 'text,IMAGE', + 'text,unknown', + 'text,%20image', + 'text,image%20', + 'text,%09image', + 'text,%00image', + outputs.concat('text').join(','), + Array(1000).fill('text').join(','), +]) + test(`invalid output list fails before catalog reads: ${value.slice(0, 60)}`, async () => { + const f = discoveryFixture(); + const response = await f.handler(f.request(`?output_modalities=${value}`)); + assert.equal(response.status, 400); + assert.equal(f.reads(), 0); + assert.equal((f.events[0] as { kind: string }).kind, 'request-denied'); + assert.doesNotMatch(await response.text(), /output_modalities|unknown|IMAGE/u); + discoveryPrivacy(f); + }); +for (const query of [ + '?output_modalities=text,image&output_modalities=text,image', + '?output_modalities=text,image&input_modalities=text,image', + '?output_modalities=text,image&supported_parameters=tools,temperature', + '?output_modalities=text,image&provider=private-filter', +]) + test(`output lists do not broaden other query acceptance: ${query}`, async () => { + const f = discoveryFixture(); + assert.equal((await f.handler(f.request(query))).status, 400); + assert.equal(f.reads(), 0); + discoveryPrivacy(f); + }); +for (const query of ['?output_modalities=text,image', '?output_modalities=text,unknown']) + test(`authentication precedes output list parsing: ${query}`, async () => { + const f = discoveryFixture(undefined, { authenticated: false }); + assert.equal((await f.handler(f.request(query))).status, 401); + assert.equal(f.reads(), 0); + discoveryPrivacy(f); + }); +test('legacy output lists reject before catalog reads', async () => { + const f = discoveryFixture(); + assert.equal((await f.handler(f.request('?output_modalities=text,image', '/v1'))).status, 400); + assert.equal(f.reads(), 0); +}); +for (const statements of [[], [{ effect: 'Deny' as const, actions: ['*'], resources: ['*'] }]]) + test(`implicit or explicit denial hides output union totals: ${statements.length}`, async () => { + const f = discoveryFixture(undefined, { statements }); + const response = await f.handler(f.request('?output_modalities=text,image&limit=1')); + assert.equal(response.status, 200); + assert.deepEqual(await response.json(), { + object: 'list', + data: [], + total_count: 0, + links: { next: null }, + }); + discoveryPrivacy(f); + }); +test('output list continuation reevaluates both model and provider Deny', async () => { + const f = discoveryFixture( + [ + discoveryModel('first'), + discoveryModel('second'), + discoveryModel('third', { provider: 'other' }), + ], + { statements: [discoveryAllow] }, + ); + const response = await f.handler(f.request('?output_modalities=text,image&limit=1')); + assert.equal(response.status, 200); + const first = (await response.json()) as List; + assert.ok(first.links.next); + f.statements.push({ + effect: 'Deny', + actions: ['*'], + resources: ['model:second', 'provider:other'], + }); + const next = await f.handler(f.request(new URL(first.links.next, 'http://localhost').search)); + assert.equal(next.status, 200); + const body = (await next.json()) as List; + assert.deepEqual(body.data, []); + assert.equal(body.total_count, 1); + assert.equal(body.links.next, null); + discoveryPrivacy(f); +}); +for (const options of [{ catalogFailure: true }, { auditFailure: true }]) + test(`output lists retain safe required dependency failure: ${JSON.stringify(options)}`, async () => { + const f = discoveryFixture(undefined, options); + const response = await f.handler(f.request('?output_modalities=text,image')); + assert.equal(response.status, 503); + assert.doesNotMatch(await response.text(), /private|text,image|output_modalities/u); + discoveryPrivacy(f); + }); +test('output lists validate malformed denied metadata before filtering and paging', async () => { + const invalid = { + ...discoveryModel('private-denied'), + openRouterMetadata: { name: 'private malformed' }, + } as unknown as ReturnType; + const f = discoveryFixture([discoveryModel('valid'), invalid]); + assert.equal((await f.handler(f.request('?output_modalities=text,image&limit=1'))).status, 503); + discoveryPrivacy(f); +}); +test('output list matching and response capture survive required audit source mutation', async () => { + const model = discoveryModel('stable', { outputs: ['image'] }); + const metadata = model.openRouterMetadata; + assert.ok(metadata); + const f = discoveryFixture([model], { + audit: () => { + (metadata.architecture.output_modalities as string[]).splice(0); + }, + }); + const response = await f.handler(f.request('?output_modalities=text,image')); + assert.equal(response.status, 200); + const body = (await response.json()) as List; + assert.deepEqual( + body.data.map((m) => m.id), + ['stable'], + ); + assert.deepEqual(body.data[0]?.architecture?.output_modalities, ['image']); + discoveryPrivacy(f); +}); diff --git a/test/sdk-model-discovery-filters.test.ts b/test/sdk-model-discovery-filters.test.ts index d770023..dd76fa2 100644 --- a/test/sdk-model-discovery-filters.test.ts +++ b/test/sdk-model-discovery-filters.test.ts @@ -131,7 +131,7 @@ test('installed OpenRouter SDK next page reevaluates current IAM under the same }); for (const [fields, query] of [ [{ context: 0 }, { context: 0 }], - [{ outputModalities: 'text,image' }, { output_modalities: 'text,image' }], + [{ outputModalities: 'text,unknown' }, { output_modalities: 'text,unknown' }], [{ supportedParameters: 'tools,temperature' }, { supported_parameters: 'tools,temperature' }], ] as const) test(`installed SDK invalid bounded filters reject at the gateway: ${JSON.stringify(fields)}`, async () => { @@ -151,3 +151,79 @@ for (const [fields, query] of [ discoveryPrivacy(f); }); }); +for (const value of ['text,image', 'image,text']) + test(`installed SDKs consume any-member output lists as one scalar: ${value}`, async () => { + const f = discoveryFixture(); + await socket(f, async (sdk, openai, queries) => { + const router = await sdk.models.list({ outputModalities: value }); + const ai = await openai.models.list({ query: { output_modalities: value } }); + const expected = ['small', 'multi', 'unknown', 'large', 'image']; + assert.deepEqual( + router.result.data.map((m) => m.id), + expected, + ); + assert.equal(router.result.totalCount, 5); + assert.equal(router.result.links.next, null); + assert.deepEqual( + ai.data.map((m) => m.id), + expected, + ); + assert.equal(queries.length, 2); + for (const [index, query] of queries.entries()) { + assert.deepEqual(query.getAll('output_modalities'), [value]); + // OpenRouter injects its paging defaults; OpenAI retains a filter-only URL. + assert.equal(query.size, index === 0 ? 3 : 1); + if (index === 0) { + assert.equal(query.get('offset'), '0'); + assert.equal(query.get('limit'), '500'); + } + } + discoveryPrivacy(f); + }); + }); +test('installed OpenRouter output-list iteration retains conjunction and list order through terminal pages', async () => { + const f = discoveryFixture(); + await socket(f, async (sdk, _openai, queries) => { + const ids: string[] = []; + for await (const page of await sdk.models.list({ + outputModalities: 'image,text', + supportedParameters: 'tools', + context: 8192, + limit: 1, + })) { + ids.push(...page.result.data.map((m) => m.id)); + assert.equal(page.result.totalCount, 3); + } + assert.deepEqual(ids, ['multi', 'large', 'image']); + assert.equal(queries.length, 4); + for (const [index, query] of queries.entries()) { + assert.deepEqual(query.getAll('output_modalities'), ['image,text']); + assert.equal(query.get('supported_parameters'), 'tools'); + assert.equal(query.get('context'), '8192'); + assert.equal(query.get('offset'), String(index)); + } + discoveryPrivacy(f); + }); +}); +test('installed OpenRouter output-list continuation reevaluates current explicit Deny', async () => { + const f = discoveryFixture(); + await socket(f, async (sdk, _openai, queries) => { + const first = await sdk.models.list({ outputModalities: 'text,image', limit: 1 }); + assert.deepEqual( + first.result.data.map((m) => m.id), + ['small'], + ); + f.statements.splice(0, f.statements.length, { + effect: 'Deny', + actions: ['*'], + resources: ['*'], + }); + const next = await first.next(); + assert.ok(next); + assert.deepEqual(next.result.data, []); + assert.equal(next.result.totalCount, 0); + assert.equal(next.result.links.next, null); + assert.deepEqual(queries[1]?.getAll('output_modalities'), ['text,image']); + discoveryPrivacy(f); + }); +});