From 7ab9522c871a33d9dc489f7319836b21634b7cc3 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sat, 26 Sep 2026 05:00:42 +0900 Subject: [PATCH 1/3] fix(discovery): Bytez unfiltered fallback, Experiential key spellings, OpenCode Go endpoint table --- ...der-discovery-bytez-explabs-opencode-go.md | 1 + contextual_orchestrator/opencode_headers.py | 97 +++++++ contextual_orchestrator/provider_bootstrap.py | 6 +- contextual_orchestrator/review_gateway.py | 6 +- scripts/ci/serve_seeded_gateway.py | 13 +- tests/test_provider_model_discovery_fixes.py | 261 ++++++++++++++++++ 6 files changed, 376 insertions(+), 8 deletions(-) create mode 100644 CHANGELOG.d/provider-discovery-bytez-explabs-opencode-go.md create mode 100644 contextual_orchestrator/opencode_headers.py create mode 100644 tests/test_provider_model_discovery_fixes.py diff --git a/CHANGELOG.d/provider-discovery-bytez-explabs-opencode-go.md b/CHANGELOG.d/provider-discovery-bytez-explabs-opencode-go.md new file mode 100644 index 000000000..cbd04151a --- /dev/null +++ b/CHANGELOG.d/provider-discovery-bytez-explabs-opencode-go.md @@ -0,0 +1 @@ +Fixed three provider-discovery gaps. Bytez discovery now makes one last unfiltered catalog request when both task-filtered list calls fail or come back empty, keeping only rows whose own `task` is `chat` or `text-generation`, so a Bytez HTTP 500 on the filtered endpoint no longer empties the pool. The Experiential Labs key is read from `EXPERIENTIAL_LABS_API_KEY` first, then the historical `EXPERIENTAL_LABS_API_KEY`, then the provider-documented `EXPLABS_API_KEY`, in provider bootstrap, the review gateway, and the seeded CI gateway (the KV label is unchanged). OpenCode Go now carries the documented per-model endpoint table (`OPENCODE_GO_MODEL_ENDPOINTS`, from https://opencode.ai/docs/go/#endpoints): `chat/completions` models are served, while `responses` and `messages` models stay evidence-only until those adapters exist. Added `contextual_orchestrator.opencode_headers.opencode_request_headers`, which builds the identifying `user-agent` and stable per-conversation `x-opencode-session` header the OpenCode Go docs ask for; wiring it into the `ModelClient` send paths lands separately with the orchestrator fallback change. diff --git a/contextual_orchestrator/opencode_headers.py b/contextual_orchestrator/opencode_headers.py new file mode 100644 index 000000000..970d85f6c --- /dev/null +++ b/contextual_orchestrator/opencode_headers.py @@ -0,0 +1,97 @@ +"""Client identification headers OpenCode Zen and Go ask integrators to send. + +The OpenCode Go documentation (https://opencode.ai/docs/go/#where-can-i-use-it, +checked 2026-09-26) asks every client to identify itself with its own user +agent instead of a generic HTTP-library name, and to send a stable session ID +in ``x-opencode-session`` for each conversation so OpenCode can route and +prompt-cache it. Requests to any other provider get no extra headers. +""" + +from __future__ import annotations + +import hashlib +import json +from collections.abc import Mapping +from typing import Any +from urllib.parse import urlsplit + +OPENCODE_PROVIDER_NAMES = frozenset({"opencode_zen", "opencode_go"}) +OPENCODE_HOST = "opencode.ai" +OPENCODE_SESSION_HEADER = "x-opencode-session" +OPENCODE_USER_AGENT = ( + "contextual-orchestrator/0.2.0 " + "(+https://github.com/ContextualWisdomLab/contextual-orchestrator)" +) +_MAX_CALLER_SESSION_CHARS = 128 + + +def _is_opencode_agent(agent: Any) -> bool: + if getattr(agent, "provider_name", None) in OPENCODE_PROVIDER_NAMES: + return True + base_url = getattr(agent, "base_url", "") + if not isinstance(base_url, str) or not base_url: + return False + host = (urlsplit(base_url).hostname or "").casefold() + return host == OPENCODE_HOST or host.endswith("." + OPENCODE_HOST) + + +def _caller_session_id(payload: Mapping[str, Any]) -> str | None: + """Return a caller-chosen conversation key when the request carries one.""" + for field in ("prompt_cache_key", "session_id"): + value = payload.get(field) + if isinstance(value, str) and value.strip(): + return value.strip()[:_MAX_CALLER_SESSION_CHARS] + metadata = payload.get("metadata") + if isinstance(metadata, Mapping): + value = metadata.get("session_id") + if isinstance(value, str) and value.strip(): + return value.strip()[:_MAX_CALLER_SESSION_CHARS] + return None + + +def _conversation_prefix(payload: Mapping[str, Any]) -> Any: + """Return the part of a request that stays fixed across one conversation. + + Every turn of a chat conversation resends the same opening system and + first user message, so hashing that prefix yields the same ID on each + turn without storing any state. Responses-style ``instructions`` and the + first ``input`` item play the same role there. + """ + messages = payload.get("messages") + if isinstance(messages, list) and messages: + prefix: list[Any] = [] + for message in messages: + prefix.append(message) + if isinstance(message, Mapping) and message.get("role") == "user": + break + return {"messages": prefix} + input_value = payload.get("input") + if isinstance(input_value, list) and input_value: + input_value = input_value[0] + return {"instructions": payload.get("instructions"), "input": input_value} + + +def opencode_session_id(payload: Mapping[str, Any]) -> str: + """Return a stable, non-reversible session ID for one conversation.""" + caller = _caller_session_id(payload) + if caller is not None: + return caller + canonical = json.dumps( + _conversation_prefix(payload), + sort_keys=True, + separators=(",", ":"), + ensure_ascii=False, + default=str, + ) + digest = hashlib.sha256(canonical.encode("utf-8")).hexdigest()[:32] + return f"co-{digest}" + + +def opencode_request_headers(agent: Any, payload: Mapping[str, Any]) -> dict[str, str]: + """Return the OpenCode identification headers for one provider request.""" + if not _is_opencode_agent(agent): + return {} + return { + "user-agent": OPENCODE_USER_AGENT, + OPENCODE_SESSION_HEADER: opencode_session_id(payload), + } diff --git a/contextual_orchestrator/provider_bootstrap.py b/contextual_orchestrator/provider_bootstrap.py index 1bc6074f6..a0c081f9d 100644 --- a/contextual_orchestrator/provider_bootstrap.py +++ b/contextual_orchestrator/provider_bootstrap.py @@ -32,6 +32,7 @@ _currency_is_comparable, agent_from_discovered, agent_id_for, + bootstrap_credential_value, discover_all_models, discovery_tool_call_tags, privacy_tags_for_discovered, @@ -116,9 +117,8 @@ def collect_provider_credentials( values: dict[str, str] = {} missing: list[str] = [] for name in PROVIDER_ACCEPTED_CREDENTIAL_NAMES: - raw = environ.get(name, "") - value = _strip_mounted_line_endings(raw) if isinstance(raw, str) else "" - if value and value.strip(): + value = bootstrap_credential_value(environ, name) + if value: values[name] = value elif name in PROVIDER_CREDENTIAL_NAMES: missing.append(name) diff --git a/contextual_orchestrator/review_gateway.py b/contextual_orchestrator/review_gateway.py index 219341bb7..cc3e4def8 100644 --- a/contextual_orchestrator/review_gateway.py +++ b/contextual_orchestrator/review_gateway.py @@ -24,6 +24,7 @@ from .model_discovery import ( DiscoveredModel, agent_from_discovered, + bootstrap_credential_value, discover_all_models, general_free_serving_candidates, ) @@ -158,9 +159,8 @@ def register_review_credentials( requested_names = _validated_credential_names(credential_names) registered: list[str] = [] for name in requested_names: - raw_value = environment.get(name, "") - value = raw_value.rstrip("\r\n") if isinstance(raw_value, str) else "" - if value and value.strip(): + value = bootstrap_credential_value(environment, name) + if value: register_credential(name, value) registered.append(name) raw_auth_value = environment.get(REVIEW_AUTH_CREDENTIAL_NAME, "") diff --git a/scripts/ci/serve_seeded_gateway.py b/scripts/ci/serve_seeded_gateway.py index 4c156615a..f79af5e79 100644 --- a/scripts/ci/serve_seeded_gateway.py +++ b/scripts/ci/serve_seeded_gateway.py @@ -23,7 +23,10 @@ register_credential, set_backend, ) -from contextual_orchestrator.model_discovery import PROVIDER_MODEL_SOURCES +from contextual_orchestrator.model_discovery import ( + PROVIDER_MODEL_SOURCES, + credential_env_names, +) PROVIDER_KEY_ENV_NAMES = tuple( dict.fromkeys(source.credential_name for source in PROVIDER_MODEL_SOURCES) @@ -36,7 +39,13 @@ def seed_credentials_from_bootstrap_env() -> list[str]: set_backend(InMemoryCredentialBackend()) seeded: list[str] = [] for credential_name in (*PROVIDER_KEY_ENV_NAMES, SERVER_AUTH_ENV_NAME): - value = os.environ.pop(credential_name, None) + # Pop every accepted spelling so no alias lingers in the environment; + # the first non-empty one in documented order wins. + value = None + for env_name in credential_env_names(credential_name): + candidate = os.environ.pop(env_name, None) + if candidate and not value: + value = candidate # Bootstrap transport only: the trusted CI job injects each value into # this process environment; nothing reads os.environ again after this. if value: diff --git a/tests/test_provider_model_discovery_fixes.py b/tests/test_provider_model_discovery_fixes.py new file mode 100644 index 000000000..89e2c2be7 --- /dev/null +++ b/tests/test_provider_model_discovery_fixes.py @@ -0,0 +1,261 @@ +"""Bytez unfiltered fallback, Experiential key spellings, and OpenCode Go protocol/session contracts. + +Every provider response here is a fixture; no test reaches a real provider. +""" + +from __future__ import annotations + +import os +import urllib.error +from types import SimpleNamespace +from unittest.mock import patch + +import pytest + +from contextual_orchestrator.credentials import register_credential +from contextual_orchestrator.model_discovery import ( + OPENCODE_GO_MODEL_ENDPOINTS, + PROVIDER_MODEL_SOURCES, + ProviderDiscoveryError, + _parse_openai_compatible, + bootstrap_credential_value, + credential_env_names, + discover_provider_models, + opencode_go_model_endpoint, +) +from contextual_orchestrator.opencode_headers import ( + OPENCODE_SESSION_HEADER, + opencode_request_headers, + opencode_session_id, +) +from contextual_orchestrator.provider_bootstrap import collect_provider_credentials +from contextual_orchestrator.review_gateway import register_review_credentials + +SOURCES = {source.provider_name: source for source in PROVIDER_MODEL_SOURCES} +EXPLABS_KV = "EXPERIENTAL_LABS_API_KEY" + + +class _Response: + def __init__(self, payload): + import json + + self._body = json.dumps(payload).encode("utf-8") + + def read(self, *_args): + body, self._body = self._body, b"" + return body + + def __enter__(self): + return self + + def __exit__(self, *_exc): + return False + + def close(self): + return None + + +# --- Bytez --------------------------------------------------------------- + + +def _http_500(url: str) -> urllib.error.HTTPError: + return urllib.error.HTTPError(url, 500, "Internal Server Error", hdrs=None, fp=None) + + +def test_bytez_unfiltered_fallback_keeps_only_chat_task_rows() -> None: + """Both task filters answering 500 must not end discovery when the plain list works.""" + source = SOURCES["bytez"] + register_credential("BYTEZ_API_KEY", "bytez-secret") + seen: list[str] = [] + + def urlopen(request, timeout=None, **_kwargs): + seen.append(request.full_url) + if "task=" in request.full_url: + raise _http_500(request.full_url) + return _Response( + { + "error": None, + "output": [ + {"modelId": "Qwen/Qwen3-4B", "task": "text-generation", "meterPrice": "0 / sec"}, + {"modelId": "openai/whisper-large-v3", "task": "automatic-speech-recognition"}, + {"modelId": "no-task/model"}, + ], + } + ) + + with ( + patch("contextual_orchestrator.model_discovery._open_trusted_discovery_request", side_effect=urlopen), + patch("contextual_orchestrator.model_discovery.time.sleep"), + ): + discovered = discover_provider_models(source) + + assert [model.model_id for model in discovered] == ["Qwen/Qwen3-4B"] + assert seen[-1] == "https://api.bytez.com/models/v2/list/models" + assert any(url.endswith("task=chat") for url in seen) + assert any(url.endswith("task=text-generation") for url in seen) + + +def test_bytez_unfiltered_fallback_without_chat_rows_is_empty_catalog() -> None: + source = SOURCES["bytez"] + register_credential("BYTEZ_API_KEY", "bytez-secret") + + def urlopen(request, timeout=None, **_kwargs): + if "task=" in request.full_url: + return _Response({"error": None, "output": []}) + return _Response({"error": None, "output": [{"modelId": "a/asr", "task": "automatic-speech-recognition"}]}) + + with ( + patch("contextual_orchestrator.model_discovery._open_trusted_discovery_request", side_effect=urlopen), + pytest.raises(ProviderDiscoveryError) as excinfo, + ): + discover_provider_models(source) + assert excinfo.value.error_code == "empty_provider_catalog" + + +def test_bytez_all_three_failures_report_the_last_http_status() -> None: + source = SOURCES["bytez"] + register_credential("BYTEZ_API_KEY", "bytez-secret") + seen: list[str] = [] + + def urlopen(request, timeout=None, **_kwargs): + seen.append(request.full_url) + raise _http_500(request.full_url) + + with ( + patch("contextual_orchestrator.model_discovery._open_trusted_discovery_request", side_effect=urlopen), + patch("contextual_orchestrator.model_discovery.time.sleep"), + pytest.raises(ProviderDiscoveryError) as excinfo, + ): + discover_provider_models(source) + assert excinfo.value.error_code == "http_status_500" + assert "https://api.bytez.com/models/v2/list/models" in seen + + +# --- Experiential Labs key spellings ------------------------------------ + + +def test_explabs_env_names_prefer_correct_spelling() -> None: + assert credential_env_names(EXPLABS_KV) == ( + "EXPERIENTIAL_LABS_API_KEY", + "EXPERIENTAL_LABS_API_KEY", + "EXPLABS_API_KEY", + ) + assert credential_env_names("BYTEZ_API_KEY") == ("BYTEZ_API_KEY",) + assert SOURCES["experiential_labs"].credential_name == EXPLABS_KV + + +@pytest.mark.parametrize( + ("environ", "expected"), + [ + ({"EXPERIENTIAL_LABS_API_KEY": "right", "EXPERIENTAL_LABS_API_KEY": "typo", "EXPLABS_API_KEY": "doc"}, "right"), + ({"EXPERIENTIAL_LABS_API_KEY": " ", "EXPERIENTAL_LABS_API_KEY": "typo\r\n", "EXPLABS_API_KEY": "doc"}, "typo"), + ({"EXPLABS_API_KEY": "doc"}, "doc"), + ({}, ""), + ], +) +def test_explabs_bootstrap_value_order(environ, expected) -> None: + assert bootstrap_credential_value(environ, EXPLABS_KV) == expected + + +def test_provider_bootstrap_registers_explabs_from_documented_name() -> None: + values = collect_provider_credentials({"EXPLABS_API_KEY": "doc-key"}, require_all=False) + assert values == {EXPLABS_KV: "doc-key"} + + +def test_review_gateway_registers_explabs_from_correct_spelling() -> None: + registered = register_review_credentials( + {"EXPERIENTIAL_LABS_API_KEY": "right-key", "EXPERIENTAL_LABS_API_KEY": "typo-key"}, + credential_names=(EXPLABS_KV,), + ) + assert registered == (EXPLABS_KV,) + from contextual_orchestrator.credentials import get_credential + + assert get_credential(EXPLABS_KV) == "right-key" + + +def test_seeded_gateway_pops_every_explabs_spelling(monkeypatch) -> None: + import importlib.util + from pathlib import Path + + path = Path(__file__).resolve().parents[1] / "scripts" / "ci" / "serve_seeded_gateway.py" + spec = importlib.util.spec_from_file_location("serve_seeded_gateway_under_test", path) + module = importlib.util.module_from_spec(spec) + assert spec.loader is not None + spec.loader.exec_module(module) + for name in ("EXPERIENTIAL_LABS_API_KEY", "EXPERIENTAL_LABS_API_KEY", "EXPLABS_API_KEY"): + monkeypatch.delenv(name, raising=False) + monkeypatch.setenv("EXPERIENTAL_LABS_API_KEY", "typo-key") + monkeypatch.setenv("EXPLABS_API_KEY", "doc-key") + + seeded = module.seed_credentials_from_bootstrap_env() + + from contextual_orchestrator.credentials import get_credential + + assert EXPLABS_KV in seeded + assert get_credential(EXPLABS_KV) == "typo-key" + assert "EXPLABS_API_KEY" not in os.environ + assert "EXPERIENTAL_LABS_API_KEY" not in os.environ + + +# --- OpenCode Go protocol table and session header ---------------------- + + +def test_go_table_matches_documented_protocols() -> None: + assert opencode_go_model_endpoint("glm-5.3") == "chat/completions" + assert opencode_go_model_endpoint("deepseek-v4.1-flash") == "chat/completions" + assert opencode_go_model_endpoint("mimo-v2.6-pro") == "chat/completions" + assert opencode_go_model_endpoint("grok-4.7") == "responses" + assert opencode_go_model_endpoint("gpt-6-luna") == "responses" + assert opencode_go_model_endpoint("minimax-m3") == "messages" + assert opencode_go_model_endpoint("qwen3.8-max") == "messages" + assert opencode_go_model_endpoint("unlisted-model") is None + assert set(OPENCODE_GO_MODEL_ENDPOINTS.values()) == {"chat/completions", "responses", "messages"} + + +def test_go_serves_only_chat_completions_rows() -> None: + rows = _parse_openai_compatible( + {"data": [{"id": "deepseek-v4.1-flash"}, {"id": "grok-4.7"}, {"id": "qwen3.8-max"}, {"id": "glm-5"}]}, + SOURCES["opencode_go"], + ) + assert {row.model_id: row.evidence_only for row in rows} == { + "deepseek-v4.1-flash": False, + "grok-4.7": True, + "qwen3.8-max": True, + "glm-5": True, + } + + +def _agent(provider_name: str, base_url: str = "https://example.invalid/v1"): + return SimpleNamespace(provider_name=provider_name, base_url=base_url) + + +def test_opencode_headers_only_for_opencode() -> None: + payload = {"messages": [{"role": "user", "content": "hi"}]} + assert opencode_request_headers(_agent("openrouter"), payload) == {} + for agent in (_agent("opencode_go"), _agent("opencode_zen"), _agent("custom", "https://opencode.ai/zen/go/v1")): + headers = opencode_request_headers(agent, payload) + assert headers["user-agent"].startswith("contextual-orchestrator/") + assert headers[OPENCODE_SESSION_HEADER].startswith("co-") + assert opencode_request_headers(_agent("custom", "https://notopencode.ai/v1"), payload) == {} + + +def test_session_id_is_stable_across_turns_and_distinct_across_conversations() -> None: + first_turn = {"messages": [{"role": "system", "content": "s"}, {"role": "user", "content": "task A"}]} + later_turn = { + "messages": [ + {"role": "system", "content": "s"}, + {"role": "user", "content": "task A"}, + {"role": "assistant", "content": "ok"}, + {"role": "user", "content": "more"}, + ] + } + other = {"messages": [{"role": "system", "content": "s"}, {"role": "user", "content": "task B"}]} + assert opencode_session_id(first_turn) == opencode_session_id(later_turn) + assert opencode_session_id(first_turn) != opencode_session_id(other) + assert "task A" not in opencode_session_id(first_turn) + + +def test_session_id_prefers_caller_key() -> None: + assert opencode_session_id({"prompt_cache_key": "conv-1", "messages": []}) == "conv-1" + assert opencode_session_id({"metadata": {"session_id": "conv-2"}}) == "conv-2" + assert opencode_session_id({"prompt_cache_key": "x" * 500}) == "x" * 128 From acda0fa6f06b40c15e6fbce5b2b1b549c6baa823 Mon Sep 17 00:00:00 2001 From: Resource-Investigator Date: Sat, 26 Sep 2026 09:21:06 +0900 Subject: [PATCH 2/3] fix(discovery): add model_discovery.py changes --- contextual_orchestrator/model_discovery.py | 134 +++++++++++++++++++-- 1 file changed, 126 insertions(+), 8 deletions(-) diff --git a/contextual_orchestrator/model_discovery.py b/contextual_orchestrator/model_discovery.py index 10ff032eb..b8b439e0b 100644 --- a/contextual_orchestrator/model_discovery.py +++ b/contextual_orchestrator/model_discovery.py @@ -14,6 +14,7 @@ from __future__ import annotations from decimal import Decimal +from types import MappingProxyType import hashlib import json import logging @@ -89,15 +90,95 @@ # safe to send on every request, authenticated or not. _HTTP_USER_AGENT = "contextual-orchestrator/0.2.0 (+https://github.com/ContextualWisdomLab/contextual-orchestrator)" _CAPABILITY_NAMES = {"embeddings": "embedding"} -# The live Go catalog exposes only id/object/created/owned_by. Keep the one -# serving allowlist to the official Chat Completions table; every other row is -# retained as evidence-only until the API reports a protocol or an adapter exists. -_OPENCODE_GO_CHAT_MODELS = frozenset({ - "glm-5.3-flash", "glm-5.3", "glm-5.2", "glm-5.1", "kimi-k3", - "kimi-k2.7-code", "kimi-k2.6", "longcat-2.0", "deepseek-v4-pro", - "deepseek-v4-flash", "deepseek-v4-flash-vision-exp", "mimo-v2.5", - "mimo-v2.5-pro", "hy4-preview", "hy3", +# The live Go catalog exposes only id/object/created/owned_by, so the wire +# protocol per model comes from the official endpoint table at +# https://opencode.ai/docs/go/#endpoints (checked 2026-09-26). Only +# ``chat/completions`` rows are served today; ``responses`` and ``messages`` +# rows stay evidence-only until an adapter for that protocol exists. A model +# the live catalog lists but the table does not name has no known protocol and +# also stays evidence-only. +OPENCODE_GO_MODEL_ENDPOINTS: Mapping[str, str] = MappingProxyType({ + "grok-4.7": "responses", + "grok-4.6": "responses", + "gpt-6-luna": "responses", + "gpt-5.6-luna": "responses", + "muse-spark-1.3-contributor": "responses", + "muse-spark-1.2-contributor": "responses", + "glm-5.3-flash": "chat/completions", + "glm-5.3": "chat/completions", + "glm-5.2": "chat/completions", + "glm-5.1": "chat/completions", + "kimi-k3": "chat/completions", + "kimi-k2.7-code": "chat/completions", + "kimi-k2.6": "chat/completions", + "longcat-2.0": "chat/completions", + "deepseek-v4.1-flash": "chat/completions", + "deepseek-v4-pro": "chat/completions", + "deepseek-v4-flash": "chat/completions", + "deepseek-v4-flash-vision-exp": "chat/completions", + "mimo-v2.6-flash": "chat/completions", + "mimo-v2.6-pro": "chat/completions", + "mimo-v2.5": "chat/completions", + "mimo-v2.5-pro": "chat/completions", + "hy4-preview": "chat/completions", + "hy3": "chat/completions", + "space-bunny-free": "chat/completions", + "minimax-m3": "messages", + "minimax-m2.7": "messages", + "minimax-m2.5": "messages", + "qwen3.8-max": "messages", + "qwen3.8-flash": "messages", + "qwen3.7-max": "messages", + "qwen3.7-plus": "messages", + "qwen3.6-plus": "messages", }) +_OPENCODE_GO_CHAT_MODELS = frozenset( + model_id + for model_id, endpoint in OPENCODE_GO_MODEL_ENDPOINTS.items() + if endpoint == "chat/completions" +) + + +def opencode_go_model_endpoint(model_id: str) -> str | None: + """Return the documented OpenCode Go endpoint path for one model, if known.""" + return OPENCODE_GO_MODEL_ENDPOINTS.get(model_id) + + +# Deployments have used three spellings for the Experiential Labs key. The KV +# label stays ``EXPERIENTAL_LABS_API_KEY`` (existing stored secrets and pool +# policy use it), while bootstrap reads the environment in this order: the +# correct spelling first, then the historical typo, then the provider's own +# documented ``EXPLABS_API_KEY``. +CREDENTIAL_ENV_ALIASES: Mapping[str, tuple[str, ...]] = MappingProxyType({ + "EXPERIENTAL_LABS_API_KEY": ( + "EXPERIENTIAL_LABS_API_KEY", + "EXPERIENTAL_LABS_API_KEY", + "EXPLABS_API_KEY", + ), +}) + + +def credential_env_names(credential_name: str) -> tuple[str, ...]: + """Return the environment names that may carry one KV credential, in order.""" + return CREDENTIAL_ENV_ALIASES.get(credential_name, (credential_name,)) + + +def bootstrap_credential_value( + environ: Mapping[str, str], credential_name: str +) -> str: + """Return the first non-blank environment value for one KV credential. + + Only trailing CR/LF bytes from mounted secret files are removed; every + other byte is preserved. Returns ``""`` when no spelling is set. + """ + for env_name in credential_env_names(credential_name): + raw = environ.get(env_name, "") + value = raw.rstrip("\r\n") if isinstance(raw, str) else "" + if value and value.strip(): + return value + return "" + + DISCOVERY_TOOL_CALL_SINGLE_TAG = "discovery:tool_call:single" DISCOVERY_TOOL_CALL_MULTI_TAG = "discovery:tool_call:multi" @@ -1730,6 +1811,43 @@ def _discover_bytez_task_catalog( ) if discovered: return discovered + # Last resort: Bytez has answered HTTP 500 to both task-filtered list calls + # while the key was valid (an invalid key answers 401). Ask once for the + # unfiltered catalog and keep only rows whose own ``task`` field names one + # of the chat-compatible tasks above, so no non-chat model slips in. + payload, failure = _fetch_provider_json_with_retry( + fetch, + source.list_url, + timeout=timeout, + fetch_kwargs=fetch_kwargs, + ) + if failure is not None: + last_exc = failure + _LOGGER.info( + "discovery_task_result account=%s task=unfiltered outcome=failed error_code=%s", + source.provider_name, + _provider_discovery_error_code(failure), + ) + else: + allowed_tasks = {task for task in task_filters if task} + rows = payload.get("output") if isinstance(payload, dict) else None + filtered_payload = { + "output": [ + row + for row in (rows if isinstance(rows, list) else []) + if isinstance(row, dict) and row.get("task") in allowed_tasks + ] + } + discovered = _parse_bytez(filtered_payload, source) + _LOGGER.info( + "discovery_task_result account=%s task=unfiltered outcome=%s model_count=%d", + source.provider_name, + "succeeded" if discovered else "empty", + len(discovered), + ) + if discovered: + return discovered + last_exc = None if last_exc is not None: raise ProviderDiscoveryError( source.provider_name, From f310f2444c07f093573bba4fc4f587db900c9c6e Mon Sep 17 00:00:00 2001 From: seonghobae <8172694+seonghobae@users.noreply.github.com> Date: Sat, 26 Sep 2026 14:47:44 +0900 Subject: [PATCH 3/3] fix(discovery): keep Bytez fallback failures honest, read only the registered Experiential key --- ...der-discovery-bytez-explabs-opencode-go.md | 2 +- contextual_orchestrator/model_discovery.py | 147 ++++--- .../current-main-provider-bootstrap.md | 8 +- scripts/ci/serve_seeded_gateway.py | 14 +- tests/test_provider_model_discovery_fixes.py | 416 ++++++++++++++++-- 5 files changed, 465 insertions(+), 122 deletions(-) diff --git a/CHANGELOG.d/provider-discovery-bytez-explabs-opencode-go.md b/CHANGELOG.d/provider-discovery-bytez-explabs-opencode-go.md index cbd04151a..3dda93c1e 100644 --- a/CHANGELOG.d/provider-discovery-bytez-explabs-opencode-go.md +++ b/CHANGELOG.d/provider-discovery-bytez-explabs-opencode-go.md @@ -1 +1 @@ -Fixed three provider-discovery gaps. Bytez discovery now makes one last unfiltered catalog request when both task-filtered list calls fail or come back empty, keeping only rows whose own `task` is `chat` or `text-generation`, so a Bytez HTTP 500 on the filtered endpoint no longer empties the pool. The Experiential Labs key is read from `EXPERIENTIAL_LABS_API_KEY` first, then the historical `EXPERIENTAL_LABS_API_KEY`, then the provider-documented `EXPLABS_API_KEY`, in provider bootstrap, the review gateway, and the seeded CI gateway (the KV label is unchanged). OpenCode Go now carries the documented per-model endpoint table (`OPENCODE_GO_MODEL_ENDPOINTS`, from https://opencode.ai/docs/go/#endpoints): `chat/completions` models are served, while `responses` and `messages` models stay evidence-only until those adapters exist. Added `contextual_orchestrator.opencode_headers.opencode_request_headers`, which builds the identifying `user-agent` and stable per-conversation `x-opencode-session` header the OpenCode Go docs ask for; wiring it into the `ModelClient` send paths lands separately with the orchestrator fallback change. +Fixed three provider-discovery gaps. Bytez discovery now makes one last unfiltered catalog request when both task-filtered list calls fail or come back empty, keeping only rows whose own `task` is the string `chat` or `text-generation` (rows with a list or dict `task` are skipped instead of raising and aborting discovery for every provider), so a Bytez HTTP 500 on the filtered endpoint no longer empties the pool. The fallback is skipped when a filtered call answers 401 or 403, and if it fails or finds no chat rows the first filtered-call HTTP status (for example `http_status_500`) is still reported. The unfiltered catalog may exceed the 8 MiB discovery response limit; that has not been measured with a live key. The Experiential Labs key is read only from the registered organization Secret name `EXPERIENTAL_LABS_API_KEY`; other spellings such as `EXPERIENTIAL_LABS_API_KEY` or `EXPLABS_API_KEY` are ignored. Provider bootstrap, the review gateway and the seeded CI gateway now share one normalization, so the seeded CI gateway also treats blank or whitespace-only values as unset and strips trailing CR/LF. OpenCode Go now carries the documented per-model endpoint table (`OPENCODE_GO_MODEL_ENDPOINTS`, from https://opencode.ai/docs/go/#endpoints): `chat/completions` models are served, while `responses` and `messages` models stay evidence-only until those adapters exist. Added `contextual_orchestrator.opencode_headers.opencode_request_headers`, which builds the identifying `user-agent` and stable per-conversation `x-opencode-session` header the OpenCode Go docs ask for; wiring it into the `ModelClient` send paths lands separately with the orchestrator fallback change. diff --git a/contextual_orchestrator/model_discovery.py b/contextual_orchestrator/model_discovery.py index b8b439e0b..fac696fdf 100644 --- a/contextual_orchestrator/model_discovery.py +++ b/contextual_orchestrator/model_discovery.py @@ -144,38 +144,24 @@ def opencode_go_model_endpoint(model_id: str) -> str | None: return OPENCODE_GO_MODEL_ENDPOINTS.get(model_id) -# Deployments have used three spellings for the Experiential Labs key. The KV -# label stays ``EXPERIENTAL_LABS_API_KEY`` (existing stored secrets and pool -# policy use it), while bootstrap reads the environment in this order: the -# correct spelling first, then the historical typo, then the provider's own -# documented ``EXPLABS_API_KEY``. -CREDENTIAL_ENV_ALIASES: Mapping[str, tuple[str, ...]] = MappingProxyType({ - "EXPERIENTAL_LABS_API_KEY": ( - "EXPERIENTIAL_LABS_API_KEY", - "EXPERIENTAL_LABS_API_KEY", - "EXPLABS_API_KEY", - ), -}) - - -def credential_env_names(credential_name: str) -> tuple[str, ...]: - """Return the environment names that may carry one KV credential, in order.""" - return CREDENTIAL_ENV_ALIASES.get(credential_name, (credential_name,)) - - def bootstrap_credential_value( environ: Mapping[str, str], credential_name: str ) -> str: - """Return the first non-blank environment value for one KV credential. - - Only trailing CR/LF bytes from mounted secret files are removed; every - other byte is preserved. Returns ``""`` when no spelling is set. + """Return the non-blank bootstrap environment value for one KV credential. + + Only the exact registered name is read -- no alternate spelling. For + Experiential Labs that is ``EXPERIENTAL_LABS_API_KEY``, the organization + Secret name as registered (see + ``docs/doctoring/current-main-provider-bootstrap.md``); reading another + spelling first could pick up a different key from a developer shell and + spend that account's paid credits. Only trailing CR/LF bytes from mounted + secret files are removed; every other byte is preserved. Returns ``""`` + when the name is unset or blank. """ - for env_name in credential_env_names(credential_name): - raw = environ.get(env_name, "") - value = raw.rstrip("\r\n") if isinstance(raw, str) else "" - if value and value.strip(): - return value + raw = environ.get(credential_name, "") + value = raw.rstrip("\r\n") if isinstance(raw, str) else "" + if value and value.strip(): + return value return "" @@ -1737,6 +1723,11 @@ def _url_with_task_filter(url: str, task_filter: str) -> str: ) +# Bytez answers an invalid or unauthorized key with one of these; retrying the +# same key against the unfiltered catalog cannot succeed. +_BYTEZ_AUTH_FAILURE_STATUSES = frozenset({401, 403}) + + def _fetch_provider_json_with_retry( fetch: Callable[..., Any], url: str, @@ -1783,6 +1774,12 @@ def _discover_bytez_task_catalog( dict.fromkeys((source.task_filter, *source.fallback_task_filters)) ) last_exc: Exception | None = None + # The first HTTP failure from a task-filtered call is the error this + # provider reports if nothing is discovered: the unfiltered fallback below + # must never replace a real ``http_status_500`` with its own timeout, + # oversized/invalid body, different status, or empty result. + first_http_failure: urllib.error.HTTPError | None = None + auth_rejected = False for task_filter in task_filters: if not task_filter: continue @@ -1794,6 +1791,11 @@ def _discover_bytez_task_catalog( ) if failure is not None: last_exc = failure + if isinstance(failure, urllib.error.HTTPError): + if first_http_failure is None: + first_http_failure = failure + if failure.code in _BYTEZ_AUTH_FAILURE_STATUSES: + auth_rejected = True _LOGGER.info( "discovery_task_result account=%s task=%s outcome=failed error_code=%s", source.provider_name, @@ -1811,47 +1813,60 @@ def _discover_bytez_task_catalog( ) if discovered: return discovered - # Last resort: Bytez has answered HTTP 500 to both task-filtered list calls - # while the key was valid (an invalid key answers 401). Ask once for the - # unfiltered catalog and keep only rows whose own ``task`` field names one - # of the chat-compatible tasks above, so no non-chat model slips in. - payload, failure = _fetch_provider_json_with_retry( - fetch, - source.list_url, - timeout=timeout, - fetch_kwargs=fetch_kwargs, - ) - if failure is not None: - last_exc = failure - _LOGGER.info( - "discovery_task_result account=%s task=unfiltered outcome=failed error_code=%s", - source.provider_name, - _provider_discovery_error_code(failure), - ) - else: - allowed_tasks = {task for task in task_filters if task} - rows = payload.get("output") if isinstance(payload, dict) else None - filtered_payload = { - "output": [ - row - for row in (rows if isinstance(rows, list) else []) - if isinstance(row, dict) and row.get("task") in allowed_tasks - ] - } - discovered = _parse_bytez(filtered_payload, source) - _LOGGER.info( - "discovery_task_result account=%s task=unfiltered outcome=%s model_count=%d", - source.provider_name, - "succeeded" if discovered else "empty", - len(discovered), + if not auth_rejected: + # Last resort: Bytez has answered HTTP 500 to both task-filtered list + # calls while the key was valid. Ask once for the unfiltered catalog + # and keep only rows whose own ``task`` field is one of the + # chat-compatible tasks above, so no non-chat model slips in. A + # 401/403 means the key itself was refused, so the fallback is + # skipped rather than spending another call on the same key. The + # unfiltered catalog is large and may exceed + # MAX_DISCOVERY_RESPONSE_BYTES; that surfaces as a fallback failure + # and the original filtered-call error is still reported. + payload, failure = _fetch_provider_json_with_retry( + fetch, + source.list_url, + timeout=timeout, + fetch_kwargs=fetch_kwargs, ) - if discovered: - return discovered - last_exc = None - if last_exc is not None: + if failure is not None: + if last_exc is None: + last_exc = failure + _LOGGER.info( + "discovery_task_result account=%s task=unfiltered outcome=failed error_code=%s", + source.provider_name, + _provider_discovery_error_code(failure), + ) + else: + allowed_tasks = {task for task in task_filters if task} + rows = payload.get("output") if isinstance(payload, dict) else None + filtered_payload = { + "output": [ + row + for row in (rows if isinstance(rows, list) else []) + # A list/dict ``task`` is unhashable; the ``str`` check + # keeps the set lookup from raising TypeError, which is + # not a ProviderDiscoveryError and would abort discovery + # for every provider. + if isinstance(row, dict) + and isinstance(row.get("task"), str) + and row["task"] in allowed_tasks + ] + } + discovered = _parse_bytez(filtered_payload, source) + _LOGGER.info( + "discovery_task_result account=%s task=unfiltered outcome=%s model_count=%d", + source.provider_name, + "succeeded" if discovered else "empty", + len(discovered), + ) + if discovered: + return discovered + reported_exc = first_http_failure if first_http_failure is not None else last_exc + if reported_exc is not None: raise ProviderDiscoveryError( source.provider_name, - _provider_discovery_error_code(last_exc), + _provider_discovery_error_code(reported_exc), source.credential_name, ) from None raise ProviderDiscoveryError( diff --git a/docs/doctoring/current-main-provider-bootstrap.md b/docs/doctoring/current-main-provider-bootstrap.md index dc2507a30..7524cdc6c 100644 --- a/docs/doctoring/current-main-provider-bootstrap.md +++ b/docs/doctoring/current-main-provider-bootstrap.md @@ -133,8 +133,12 @@ Nielsen, S., Cetin, E., Schwendeman, P., Sun, Q., Xu, J., & Tang, Y. (2025). `experiential_labs` is an optional provider. Bootstrap accepts the organization Secret name `EXPERIENTAL_LABS_API_KEY` exactly as registered; the spelling is -intentional. The catalog-sync workflow transports it into the existing encrypted -KV. Runtime discovery reads that KV name and sends Bearer authentication to +intentional. Provider bootstrap, the review gateway and the seeded CI gateway +read only that name: `EXPERIENTIAL_LABS_API_KEY`, `EXPLABS_API_KEY` or any +other spelling in the environment is ignored, so a different key in a developer +shell cannot be picked up and spend that account's credits. The catalog-sync +workflow transports it into the existing encrypted KV. Runtime discovery reads +that KV name and sends Bearer authentication to `https://api.experientiallabs.ai/v1/models`; chat uses the same `/v1` base. Existing deployments do not need this additional credential. diff --git a/scripts/ci/serve_seeded_gateway.py b/scripts/ci/serve_seeded_gateway.py index f79af5e79..dde70322f 100644 --- a/scripts/ci/serve_seeded_gateway.py +++ b/scripts/ci/serve_seeded_gateway.py @@ -25,7 +25,7 @@ ) from contextual_orchestrator.model_discovery import ( PROVIDER_MODEL_SOURCES, - credential_env_names, + bootstrap_credential_value, ) PROVIDER_KEY_ENV_NAMES = tuple( @@ -39,13 +39,11 @@ def seed_credentials_from_bootstrap_env() -> list[str]: set_backend(InMemoryCredentialBackend()) seeded: list[str] = [] for credential_name in (*PROVIDER_KEY_ENV_NAMES, SERVER_AUTH_ENV_NAME): - # Pop every accepted spelling so no alias lingers in the environment; - # the first non-empty one in documented order wins. - value = None - for env_name in credential_env_names(credential_name): - candidate = os.environ.pop(env_name, None) - if candidate and not value: - value = candidate + # Same normalization as provider bootstrap and the review gateway: + # only trailing CR/LF is removed, and a blank or whitespace-only value + # counts as unset. Pop the name either way so it never lingers. + value = bootstrap_credential_value(os.environ, credential_name) + os.environ.pop(credential_name, None) # Bootstrap transport only: the trusted CI job injects each value into # this process environment; nothing reads os.environ again after this. if value: diff --git a/tests/test_provider_model_discovery_fixes.py b/tests/test_provider_model_discovery_fixes.py index 89e2c2be7..c8f6aaf5b 100644 --- a/tests/test_provider_model_discovery_fixes.py +++ b/tests/test_provider_model_discovery_fixes.py @@ -1,4 +1,4 @@ -"""Bytez unfiltered fallback, Experiential key spellings, and OpenCode Go protocol/session contracts. +"""Bytez unfiltered fallback, Experiential registered key name, and OpenCode Go protocol/session contracts. Every provider response here is a fixture; no test reaches a real provider. """ @@ -14,12 +14,13 @@ from contextual_orchestrator.credentials import register_credential from contextual_orchestrator.model_discovery import ( + MAX_DISCOVERY_RESPONSE_BYTES, OPENCODE_GO_MODEL_ENDPOINTS, PROVIDER_MODEL_SOURCES, ProviderDiscoveryError, _parse_openai_compatible, bootstrap_credential_value, - credential_env_names, + discover_all_models, discover_provider_models, opencode_go_model_endpoint, ) @@ -36,10 +37,10 @@ class _Response: - def __init__(self, payload): + def __init__(self, payload=None, *, raw: bytes | None = None): import json - self._body = json.dumps(payload).encode("utf-8") + self._body = raw if raw is not None else json.dumps(payload).encode("utf-8") def read(self, *_args): body, self._body = self._body, b"" @@ -76,15 +77,25 @@ def urlopen(request, timeout=None, **_kwargs): { "error": None, "output": [ - {"modelId": "Qwen/Qwen3-4B", "task": "text-generation", "meterPrice": "0 / sec"}, - {"modelId": "openai/whisper-large-v3", "task": "automatic-speech-recognition"}, + { + "modelId": "Qwen/Qwen3-4B", + "task": "text-generation", + "meterPrice": "0 / sec", + }, + { + "modelId": "openai/whisper-large-v3", + "task": "automatic-speech-recognition", + }, {"modelId": "no-task/model"}, ], } ) with ( - patch("contextual_orchestrator.model_discovery._open_trusted_discovery_request", side_effect=urlopen), + patch( + "contextual_orchestrator.model_discovery._open_trusted_discovery_request", + side_effect=urlopen, + ), patch("contextual_orchestrator.model_discovery.time.sleep"), ): discovered = discover_provider_models(source) @@ -102,10 +113,20 @@ def test_bytez_unfiltered_fallback_without_chat_rows_is_empty_catalog() -> None: def urlopen(request, timeout=None, **_kwargs): if "task=" in request.full_url: return _Response({"error": None, "output": []}) - return _Response({"error": None, "output": [{"modelId": "a/asr", "task": "automatic-speech-recognition"}]}) + return _Response( + { + "error": None, + "output": [ + {"modelId": "a/asr", "task": "automatic-speech-recognition"} + ], + } + ) with ( - patch("contextual_orchestrator.model_discovery._open_trusted_discovery_request", side_effect=urlopen), + patch( + "contextual_orchestrator.model_discovery._open_trusted_discovery_request", + side_effect=urlopen, + ), pytest.raises(ProviderDiscoveryError) as excinfo, ): discover_provider_models(source) @@ -122,7 +143,10 @@ def urlopen(request, timeout=None, **_kwargs): raise _http_500(request.full_url) with ( - patch("contextual_orchestrator.model_discovery._open_trusted_discovery_request", side_effect=urlopen), + patch( + "contextual_orchestrator.model_discovery._open_trusted_discovery_request", + side_effect=urlopen, + ), patch("contextual_orchestrator.model_discovery.time.sleep"), pytest.raises(ProviderDiscoveryError) as excinfo, ): @@ -131,70 +155,342 @@ def urlopen(request, timeout=None, **_kwargs): assert "https://api.bytez.com/models/v2/list/models" in seen -# --- Experiential Labs key spellings ------------------------------------ +_BYTEZ_UNFILTERED_URL = "https://api.bytez.com/models/v2/list/models" +_OPEN = "contextual_orchestrator.model_discovery._open_trusted_discovery_request" +_SLEEP = "contextual_orchestrator.model_discovery.time.sleep" + + +def _http_error(url: str, code: int) -> urllib.error.HTTPError: + return urllib.error.HTTPError(url, code, "error", hdrs=None, fp=None) + + +@pytest.mark.parametrize( + "bad_task", + [["chat"], {"name": "text-generation"}], + ids=["list-task", "dict-task"], +) +def test_bytez_unfiltered_fallback_skips_unhashable_task_rows(bad_task) -> None: + """A list/dict ``task`` must be skipped, not raise ``TypeError: unhashable``.""" + source = SOURCES["bytez"] + register_credential("BYTEZ_API_KEY", "bytez-secret") + + def urlopen(request, timeout=None, **_kwargs): + if "task=" in request.full_url: + raise _http_500(request.full_url) + return _Response( + { + "error": None, + "output": [ + {"modelId": "bad/unhashable-task", "task": bad_task}, + {"modelId": "Qwen/Qwen3-4B", "task": "chat"}, + ], + } + ) + + with patch(_OPEN, side_effect=urlopen), patch(_SLEEP): + discovered = discover_provider_models(source) + + assert [model.model_id for model in discovered] == ["Qwen/Qwen3-4B"] + + +@pytest.mark.parametrize( + "bad_task", + [["chat"], {"name": "chat"}], + ids=["list-task", "dict-task"], +) +def test_bytez_unhashable_task_rows_do_not_abort_other_providers(bad_task) -> None: + """Malformed Bytez rows must not escape discover_all_models and drop every provider.""" + register_credential("BYTEZ_API_KEY", "bytez-secret") + register_credential(EXPLABS_KV, "experiential-secret") + + def urlopen(request, timeout=None, **_kwargs): + url = request.full_url + if url.startswith("https://api.experientiallabs.ai/"): + return _Response( + { + "object": "list", + "data": [{"id": "explabs-chat-1", "object": "model"}], + } + ) + if "task=" in url: + raise _http_500(url) + # Only unhashable-task rows: Bytez ends up with no chat models. + return _Response( + {"error": None, "output": [{"modelId": "bad/row", "task": bad_task}]} + ) + + with patch(_OPEN, side_effect=urlopen), patch(_SLEEP): + models, errors = discover_all_models( + (SOURCES["bytez"], SOURCES["experiential_labs"]), + discovery_deadline=None, + ) + + assert [(model.provider_name, model.model_id) for model in models] == [ + ("experiential_labs", "explabs-chat-1") + ] + assert [(error.provider_name, error.error_code) for error in errors] == [ + ("bytez", "http_status_500") + ] + + +def _fallback_failure_timeout(url: str): + raise TimeoutError("fixture timeout") + + +def _fallback_failure_503(url: str): + raise _http_error(url, 503) + + +def _fallback_invalid_json(url: str): + return _Response(raw=b"not json") + + +def _fallback_oversized(url: str): + return _Response(raw=b" " * (MAX_DISCOVERY_RESPONSE_BYTES + 1)) -def test_explabs_env_names_prefer_correct_spelling() -> None: - assert credential_env_names(EXPLABS_KV) == ( - "EXPERIENTIAL_LABS_API_KEY", - "EXPERIENTAL_LABS_API_KEY", - "EXPLABS_API_KEY", +def _fallback_no_chat_rows(url: str): + return _Response( + { + "error": None, + "output": [{"modelId": "a/asr", "task": "automatic-speech-recognition"}], + } ) - assert credential_env_names("BYTEZ_API_KEY") == ("BYTEZ_API_KEY",) + + +def _fallback_empty(url: str): + return _Response({"error": None, "output": []}) + + +@pytest.mark.parametrize( + "fallback", + [ + _fallback_failure_timeout, + _fallback_failure_503, + _fallback_invalid_json, + _fallback_oversized, + _fallback_no_chat_rows, + _fallback_empty, + ], + ids=[ + "timeout", + "http-503", + "invalid-json", + "oversized-body", + "no-chat-rows", + "empty", + ], +) +def test_bytez_failed_fallback_keeps_original_http_status(fallback) -> None: + """Filtered 500 + failing/empty unfiltered fallback must still report http_status_500.""" + source = SOURCES["bytez"] + register_credential("BYTEZ_API_KEY", "bytez-secret") + seen: list[str] = [] + + def urlopen(request, timeout=None, **_kwargs): + seen.append(request.full_url) + if "task=" in request.full_url: + raise _http_500(request.full_url) + return fallback(request.full_url) + + with ( + patch(_OPEN, side_effect=urlopen), + patch(_SLEEP), + pytest.raises(ProviderDiscoveryError) as excinfo, + ): + discover_provider_models(source) + + assert excinfo.value.error_code == "http_status_500" + assert _BYTEZ_UNFILTERED_URL in seen + + +def test_bytez_first_http_failure_wins_over_later_filtered_failure() -> None: + source = SOURCES["bytez"] + register_credential("BYTEZ_API_KEY", "bytez-secret") + + def urlopen(request, timeout=None, **_kwargs): + url = request.full_url + if url.endswith("task=chat"): + raise _http_500(url) + if "task=" in url: + raise TimeoutError("fixture timeout") + raise _http_error(url, 502) + + with ( + patch(_OPEN, side_effect=urlopen), + patch(_SLEEP), + pytest.raises(ProviderDiscoveryError) as excinfo, + ): + discover_provider_models(source) + assert excinfo.value.error_code == "http_status_500" + + +@pytest.mark.parametrize("status", [401, 403]) +def test_bytez_auth_failure_skips_unfiltered_fallback(status) -> None: + """A refused key must not spend another call on the unfiltered catalog.""" + source = SOURCES["bytez"] + register_credential("BYTEZ_API_KEY", "bytez-secret") + seen: list[str] = [] + + def urlopen(request, timeout=None, **_kwargs): + seen.append(request.full_url) + if "task=" in request.full_url: + raise _http_error(request.full_url, status) + return _Response( + {"error": None, "output": [{"modelId": "Qwen/Qwen3-4B", "task": "chat"}]} + ) + + with ( + patch(_OPEN, side_effect=urlopen), + patch(_SLEEP), + pytest.raises(ProviderDiscoveryError) as excinfo, + ): + discover_provider_models(source) + + assert excinfo.value.error_code == f"http_status_{status}" + assert _BYTEZ_UNFILTERED_URL not in seen + assert all("task=" in url for url in seen) + + +def test_bytez_non_http_filtered_failure_is_kept_when_fallback_is_empty() -> None: + source = SOURCES["bytez"] + register_credential("BYTEZ_API_KEY", "bytez-secret") + + def urlopen(request, timeout=None, **_kwargs): + if "task=" in request.full_url: + raise TimeoutError("fixture timeout") + return _Response({"error": None, "output": []}) + + with ( + patch(_OPEN, side_effect=urlopen), + patch(_SLEEP), + pytest.raises(ProviderDiscoveryError) as excinfo, + ): + discover_provider_models(source) + assert excinfo.value.error_code == "timeout" + + +# --- Experiential Labs registered key name ------------------------------ + +_UNREGISTERED_EXPLABS_SPELLINGS = ("EXPERIENTIAL_LABS_API_KEY", "EXPLABS_API_KEY") + + +def test_explabs_kv_name_is_the_registered_spelling() -> None: assert SOURCES["experiential_labs"].credential_name == EXPLABS_KV @pytest.mark.parametrize( ("environ", "expected"), [ - ({"EXPERIENTIAL_LABS_API_KEY": "right", "EXPERIENTAL_LABS_API_KEY": "typo", "EXPLABS_API_KEY": "doc"}, "right"), - ({"EXPERIENTIAL_LABS_API_KEY": " ", "EXPERIENTAL_LABS_API_KEY": "typo\r\n", "EXPLABS_API_KEY": "doc"}, "typo"), - ({"EXPLABS_API_KEY": "doc"}, "doc"), + ({"EXPERIENTAL_LABS_API_KEY": "registered"}, "registered"), + ({"EXPERIENTAL_LABS_API_KEY": "registered\r\n"}, "registered"), + ( + { + "EXPERIENTAL_LABS_API_KEY": "registered", + "EXPERIENTIAL_LABS_API_KEY": "other", + "EXPLABS_API_KEY": "doc", + }, + "registered", + ), + ({"EXPERIENTIAL_LABS_API_KEY": "other", "EXPLABS_API_KEY": "doc"}, ""), + ({"EXPERIENTAL_LABS_API_KEY": " ", "EXPLABS_API_KEY": "doc"}, ""), ({}, ""), ], ) -def test_explabs_bootstrap_value_order(environ, expected) -> None: +def test_explabs_bootstrap_reads_only_registered_name(environ, expected) -> None: assert bootstrap_credential_value(environ, EXPLABS_KV) == expected -def test_provider_bootstrap_registers_explabs_from_documented_name() -> None: - values = collect_provider_credentials({"EXPLABS_API_KEY": "doc-key"}, require_all=False) - assert values == {EXPLABS_KV: "doc-key"} +def test_provider_bootstrap_ignores_unregistered_explabs_spellings() -> None: + environ = {name: "shell-key" for name in _UNREGISTERED_EXPLABS_SPELLINGS} + environ["BYTEZ_API_KEY"] = "bytez-key" + assert collect_provider_credentials(environ, require_all=False) == { + "BYTEZ_API_KEY": "bytez-key" + } + del environ["BYTEZ_API_KEY"] + environ[EXPLABS_KV] = "registered-key" + assert collect_provider_credentials(environ, require_all=False) == { + EXPLABS_KV: "registered-key" + } -def test_review_gateway_registers_explabs_from_correct_spelling() -> None: +def test_review_gateway_registers_only_registered_explabs_name() -> None: + from contextual_orchestrator.credentials import get_credential + registered = register_review_credentials( - {"EXPERIENTIAL_LABS_API_KEY": "right-key", "EXPERIENTAL_LABS_API_KEY": "typo-key"}, + {name: "shell-key" for name in _UNREGISTERED_EXPLABS_SPELLINGS}, credential_names=(EXPLABS_KV,), ) - assert registered == (EXPLABS_KV,) - from contextual_orchestrator.credentials import get_credential + assert registered == () - assert get_credential(EXPLABS_KV) == "right-key" + registered = register_review_credentials( + {EXPLABS_KV: "registered-key", "EXPERIENTIAL_LABS_API_KEY": "shell-key"}, + credential_names=(EXPLABS_KV,), + ) + assert registered == (EXPLABS_KV,) + assert get_credential(EXPLABS_KV) == "registered-key" -def test_seeded_gateway_pops_every_explabs_spelling(monkeypatch) -> None: +def _load_seeded_gateway(): import importlib.util from pathlib import Path - path = Path(__file__).resolve().parents[1] / "scripts" / "ci" / "serve_seeded_gateway.py" - spec = importlib.util.spec_from_file_location("serve_seeded_gateway_under_test", path) + path = ( + Path(__file__).resolve().parents[1] + / "scripts" + / "ci" + / "serve_seeded_gateway.py" + ) + spec = importlib.util.spec_from_file_location( + "serve_seeded_gateway_under_test", path + ) module = importlib.util.module_from_spec(spec) assert spec.loader is not None spec.loader.exec_module(module) - for name in ("EXPERIENTIAL_LABS_API_KEY", "EXPERIENTAL_LABS_API_KEY", "EXPLABS_API_KEY"): + return module + + +def _clear_seeded_env(monkeypatch, module) -> None: + for name in ( + *module.PROVIDER_KEY_ENV_NAMES, + module.SERVER_AUTH_ENV_NAME, + *_UNREGISTERED_EXPLABS_SPELLINGS, + ): monkeypatch.delenv(name, raising=False) - monkeypatch.setenv("EXPERIENTAL_LABS_API_KEY", "typo-key") + + +def test_seeded_gateway_reads_only_registered_explabs_name(monkeypatch) -> None: + from contextual_orchestrator.credentials import get_credential + + module = _load_seeded_gateway() + _clear_seeded_env(monkeypatch, module) + monkeypatch.setenv("EXPERIENTIAL_LABS_API_KEY", "shell-key") monkeypatch.setenv("EXPLABS_API_KEY", "doc-key") + assert EXPLABS_KV not in module.seed_credentials_from_bootstrap_env() + + monkeypatch.setenv(EXPLABS_KV, "registered-key\r\n") seeded = module.seed_credentials_from_bootstrap_env() + assert EXPLABS_KV in seeded + assert get_credential(EXPLABS_KV) == "registered-key" + assert EXPLABS_KV not in os.environ + +@pytest.mark.parametrize("blank", ["", " ", "\t", "\r\n", " \r\n"]) +def test_seeded_gateway_skips_blank_values_and_pops_them(monkeypatch, blank) -> None: from contextual_orchestrator.credentials import get_credential - assert EXPLABS_KV in seeded - assert get_credential(EXPLABS_KV) == "typo-key" - assert "EXPLABS_API_KEY" not in os.environ - assert "EXPERIENTAL_LABS_API_KEY" not in os.environ + module = _load_seeded_gateway() + _clear_seeded_env(monkeypatch, module) + monkeypatch.setenv("BYTEZ_API_KEY", blank) + monkeypatch.setenv(EXPLABS_KV, "registered-key") + + seeded = module.seed_credentials_from_bootstrap_env() + + assert "BYTEZ_API_KEY" not in seeded + assert get_credential("BYTEZ_API_KEY") is None + assert seeded == [EXPLABS_KV] + assert "BYTEZ_API_KEY" not in os.environ # --- OpenCode Go protocol table and session header ---------------------- @@ -209,12 +505,23 @@ def test_go_table_matches_documented_protocols() -> None: assert opencode_go_model_endpoint("minimax-m3") == "messages" assert opencode_go_model_endpoint("qwen3.8-max") == "messages" assert opencode_go_model_endpoint("unlisted-model") is None - assert set(OPENCODE_GO_MODEL_ENDPOINTS.values()) == {"chat/completions", "responses", "messages"} + assert set(OPENCODE_GO_MODEL_ENDPOINTS.values()) == { + "chat/completions", + "responses", + "messages", + } def test_go_serves_only_chat_completions_rows() -> None: rows = _parse_openai_compatible( - {"data": [{"id": "deepseek-v4.1-flash"}, {"id": "grok-4.7"}, {"id": "qwen3.8-max"}, {"id": "glm-5"}]}, + { + "data": [ + {"id": "deepseek-v4.1-flash"}, + {"id": "grok-4.7"}, + {"id": "qwen3.8-max"}, + {"id": "glm-5"}, + ] + }, SOURCES["opencode_go"], ) assert {row.model_id: row.evidence_only for row in rows} == { @@ -232,15 +539,27 @@ def _agent(provider_name: str, base_url: str = "https://example.invalid/v1"): def test_opencode_headers_only_for_opencode() -> None: payload = {"messages": [{"role": "user", "content": "hi"}]} assert opencode_request_headers(_agent("openrouter"), payload) == {} - for agent in (_agent("opencode_go"), _agent("opencode_zen"), _agent("custom", "https://opencode.ai/zen/go/v1")): + for agent in ( + _agent("opencode_go"), + _agent("opencode_zen"), + _agent("custom", "https://opencode.ai/zen/go/v1"), + ): headers = opencode_request_headers(agent, payload) assert headers["user-agent"].startswith("contextual-orchestrator/") assert headers[OPENCODE_SESSION_HEADER].startswith("co-") - assert opencode_request_headers(_agent("custom", "https://notopencode.ai/v1"), payload) == {} + assert ( + opencode_request_headers(_agent("custom", "https://notopencode.ai/v1"), payload) + == {} + ) def test_session_id_is_stable_across_turns_and_distinct_across_conversations() -> None: - first_turn = {"messages": [{"role": "system", "content": "s"}, {"role": "user", "content": "task A"}]} + first_turn = { + "messages": [ + {"role": "system", "content": "s"}, + {"role": "user", "content": "task A"}, + ] + } later_turn = { "messages": [ {"role": "system", "content": "s"}, @@ -249,13 +568,20 @@ def test_session_id_is_stable_across_turns_and_distinct_across_conversations() - {"role": "user", "content": "more"}, ] } - other = {"messages": [{"role": "system", "content": "s"}, {"role": "user", "content": "task B"}]} + other = { + "messages": [ + {"role": "system", "content": "s"}, + {"role": "user", "content": "task B"}, + ] + } assert opencode_session_id(first_turn) == opencode_session_id(later_turn) assert opencode_session_id(first_turn) != opencode_session_id(other) assert "task A" not in opencode_session_id(first_turn) def test_session_id_prefers_caller_key() -> None: - assert opencode_session_id({"prompt_cache_key": "conv-1", "messages": []}) == "conv-1" + assert ( + opencode_session_id({"prompt_cache_key": "conv-1", "messages": []}) == "conv-1" + ) assert opencode_session_id({"metadata": {"session_id": "conv-2"}}) == "conv-2" assert opencode_session_id({"prompt_cache_key": "x" * 500}) == "x" * 128