Files
Pat Sukprasert 33a06cc0a4 [models] Add intent resolver contracts (#3443)
* feat(models): add intent resolver contracts

Define stable model intents and provider-neutral metadata for capabilities, context windows, cost tiers, and wire APIs. Capability support is tri-state so incomplete provider listings cannot be mistaken for positive support.

Add deterministic resolution precedence for explicit choices, configured defaults, live catalogs, and documented static fallbacks. Catalog order remains the tie-breaker, while provider-specific preference policies can override ranking without changing callers.

Expose normalized metadata through model catalog entries and payloads without changing any executor or routing defaults in this slice.

Tests: 108 focused resolver, catalog, and smart-routing tests; staged pre-commit hooks.

Part of #3426

Signed-off-by: Pat Sukprasert <pat.sukprasert@databricks.com>

* refactor(models): keep resolver intents caller-backed

Limit the public model intent vocabulary to default, fast, balanced, and powerful because those are the only purposes represented by current callers.

Express tool use, image generation, structured output, and similar requirements through explicit capabilities instead of speculative intent-to-capability mappings. Remove the unused large-context ranking path and update resolver tests and migration guidance accordingly.

Tests: uv run --no-sync pytest -q tests/test_model_resolver.py tests/test_model_catalog.py tests/server/test_smart_routing.py; pre-commit run
Signed-off-by: Pat Sukprasert <pat.sukprasert@databricks.com>

* refactor(models): complete wire API contract

Cover every model-endpoint request shape implemented by the provider adapters by adding Bedrock Converse and naming Gemini generateContent explicitly. Keep native CLI and ACP transports outside the model wire protocol vocabulary.

Clarify that explicit model overrides bypass compatibility constraints, intent tiers are best-effort ranking preferences, and uncatalogued explicit resolutions have unknown family and metadata. Add regression coverage for those semantics and for the complete wire API vocabulary.

Tests: uv run --no-sync pytest -q tests/test_model_resolver.py tests/test_model_catalog.py tests/server/test_smart_routing.py tests/llms/test_openai_adapter.py tests/llms/test_anthropic_adapter.py tests/llms/test_gemini_adapter.py tests/llms/test_vertex_adapter.py tests/llms/test_bedrock_adapter.py tests/llms/test_databricks_adapter.py; pre-commit run
Signed-off-by: Pat Sukprasert <pat.sukprasert@databricks.com>

---------

Signed-off-by: Pat Sukprasert <pat.sukprasert@databricks.com>
2026-07-29 05:18:34 +00:00

236 lines
7.1 KiB
Python

"""Unit tests for provider-neutral model-intent resolution."""
from __future__ import annotations
from collections.abc import Sequence
import pytest
from omnigent.model_catalog import ModelEntry
from omnigent.model_metadata import (
ModelCapability,
ModelCostTier,
ModelIntent,
ModelMetadata,
ModelWireAPI,
)
from omnigent.model_resolver import (
ModelCandidate,
ModelPreferencePolicy,
ModelResolutionError,
ModelResolutionRequest,
ModelResolutionSource,
resolve_model,
)
def _candidate(
model_id: str,
*,
family: str = "provider-family",
supported: frozenset[ModelCapability] = frozenset(),
unsupported: frozenset[ModelCapability] = frozenset(),
context_window: int | None = None,
cost_tier: ModelCostTier | None = None,
wire_apis: frozenset[ModelWireAPI] = frozenset(),
) -> ModelEntry:
return ModelEntry(
id=model_id,
family=family,
metadata=ModelMetadata(
supported_capabilities=supported,
unsupported_capabilities=unsupported,
context_window=context_window,
cost_tier=cost_tier,
wire_apis=wire_apis,
),
)
def test_explicit_model_bypasses_constraints_without_catalog_metadata() -> None:
resolution = resolve_model(
ModelResolutionRequest(
explicit_model="user-choice",
required_capabilities=frozenset({ModelCapability.TOOL_USE}),
minimum_context_window=200_000,
required_wire_api=ModelWireAPI.BEDROCK_CONVERSE,
allowed_families=frozenset({"required-family"}),
),
[_candidate("catalog-choice")],
)
assert resolution.model_id == "user-choice"
assert resolution.source == ModelResolutionSource.EXPLICIT
assert resolution.metadata == ModelMetadata()
assert resolution.family is None
def test_configured_default_wins_before_live_catalog() -> None:
models = [_candidate("provider-default"), _candidate("catalog-choice")]
resolution = resolve_model(
ModelResolutionRequest(configured_default="provider-default"),
models,
)
assert resolution.model_id == "provider-default"
assert resolution.source == ModelResolutionSource.CONFIGURED_DEFAULT
def test_capability_requirement_skips_unverified_configured_default() -> None:
coding_model = _candidate(
"tool-capable",
supported=frozenset({ModelCapability.TOOL_USE}),
)
resolution = resolve_model(
ModelResolutionRequest(
configured_default="unlisted-default",
required_capabilities=frozenset({ModelCapability.TOOL_USE}),
),
[coding_model],
)
assert resolution.model_id == "tool-capable"
assert resolution.source == ModelResolutionSource.LIVE_CATALOG
def test_unknown_capability_does_not_satisfy_requirement() -> None:
unknown = _candidate("unknown-tools")
supported = _candidate(
"known-tools",
supported=frozenset({ModelCapability.TOOL_USE}),
)
resolution = resolve_model(
ModelResolutionRequest(required_capabilities=frozenset({ModelCapability.TOOL_USE})),
[unknown, supported],
)
assert resolution.model_id == "known-tools"
@pytest.mark.parametrize(
("intent", "expected"),
[
(ModelIntent.FAST, "economy"),
(ModelIntent.BALANCED, "standard"),
(ModelIntent.POWERFUL, "premium"),
],
)
def test_cost_intents_rank_by_normalized_tier(intent: ModelIntent, expected: str) -> None:
models = [
_candidate("premium", cost_tier=ModelCostTier.PREMIUM),
_candidate("economy", cost_tier=ModelCostTier.ECONOMY),
_candidate("standard", cost_tier=ModelCostTier.STANDARD),
]
resolution = resolve_model(ModelResolutionRequest(intent=intent), models)
assert resolution.model_id == expected
def test_cost_intent_returns_best_available_tier() -> None:
resolution = resolve_model(
ModelResolutionRequest(intent=ModelIntent.FAST),
[_candidate("premium", cost_tier=ModelCostTier.PREMIUM)],
)
assert resolution.model_id == "premium"
assert resolution.metadata.cost_tier == ModelCostTier.PREMIUM
def test_wire_api_vocabulary_covers_model_endpoint_adapters() -> None:
assert {wire_api.value for wire_api in ModelWireAPI} == {
"anthropic-messages",
"bedrock-converse",
"gemini-generate-content",
"openai-chat",
"openai-responses",
}
def test_constraints_filter_family_context_capability_and_wire_api() -> None:
required_capabilities = frozenset(
{ModelCapability.REASONING, ModelCapability.STRUCTURED_OUTPUT}
)
compatible = _candidate(
"compatible",
family="allowed",
supported=required_capabilities,
context_window=250_000,
wire_apis=frozenset({ModelWireAPI.OPENAI_RESPONSES}),
)
wrong_wire = _candidate(
"wrong-wire",
family="allowed",
supported=required_capabilities,
context_window=250_000,
wire_apis=frozenset({ModelWireAPI.OPENAI_CHAT}),
)
resolution = resolve_model(
ModelResolutionRequest(
required_capabilities=required_capabilities,
minimum_context_window=200_000,
required_wire_api=ModelWireAPI.OPENAI_RESPONSES,
allowed_families=frozenset({"allowed"}),
),
[wrong_wire, compatible],
)
assert resolution.model_id == "compatible"
def test_static_fallback_is_used_only_after_live_candidates_fail() -> None:
fallback = _candidate(
"fallback",
supported=frozenset({ModelCapability.IMAGE_GENERATION}),
)
resolution = resolve_model(
ModelResolutionRequest(
required_capabilities=frozenset({ModelCapability.IMAGE_GENERATION})
),
[_candidate("live-with-unknown-capabilities")],
static_fallbacks=[fallback],
)
assert resolution.model_id == "fallback"
assert resolution.source == ModelResolutionSource.STATIC_FALLBACK
def test_provider_preference_policy_can_override_default_order() -> None:
class _ReversePolicy(ModelPreferencePolicy):
def rank(
self,
intent: ModelIntent,
candidates: Sequence[ModelCandidate],
) -> Sequence[ModelCandidate]:
del intent
return tuple(reversed(candidates))
resolution = resolve_model(
ModelResolutionRequest(),
[_candidate("first"), _candidate("second")],
preference_policy=_ReversePolicy(),
)
assert resolution.model_id == "second"
def test_no_compatible_model_raises_clear_error() -> None:
with pytest.raises(ModelResolutionError, match=r"default.*tool-use"):
resolve_model(
ModelResolutionRequest(required_capabilities=frozenset({ModelCapability.TOOL_USE})),
[_candidate("unknown-tools")],
)
def test_metadata_rejects_conflicting_capability_facts() -> None:
with pytest.raises(ValueError, match="both supported and unsupported"):
ModelMetadata(
supported_capabilities=frozenset({ModelCapability.VISION}),
unsupported_capabilities=frozenset({ModelCapability.VISION}),
)