Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Fixed three provider-discovery gaps. Bytez discovery now makes one last unfiltered catalog request when both task-filtered list calls fail or come back empty, keeping only rows whose own `task` is the string `chat` or `text-generation` (rows with a list or dict `task` are skipped instead of raising and aborting discovery for every provider), so a Bytez HTTP 500 on the filtered endpoint no longer empties the pool. The fallback is skipped when a filtered call answers 401 or 403, and if it fails or finds no chat rows the first filtered-call HTTP status (for example `http_status_500`) is still reported. The unfiltered catalog may exceed the 8 MiB discovery response limit; that has not been measured with a live key. The Experiential Labs key is read only from the registered organization Secret name `EXPERIENTAL_LABS_API_KEY`; other spellings such as `EXPERIENTIAL_LABS_API_KEY` or `EXPLABS_API_KEY` are ignored. Provider bootstrap, the review gateway and the seeded CI gateway now share one normalization, so the seeded CI gateway also treats blank or whitespace-only values as unset and strips trailing CR/LF. OpenCode Go now carries the documented per-model endpoint table (`OPENCODE_GO_MODEL_ENDPOINTS`, from https://opencode.ai/docs/go/#endpoints): `chat/completions` models are served, while `responses` and `messages` models stay evidence-only until those adapters exist. Added `contextual_orchestrator.opencode_headers.opencode_request_headers`, which builds the identifying `user-agent` and stable per-conversation `x-opencode-session` header the OpenCode Go docs ask for; wiring it into the `ModelClient` send paths lands separately with the orchestrator fallback change.
153 changes: 143 additions & 10 deletions contextual_orchestrator/model_discovery.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@
from __future__ import annotations

from decimal import Decimal
from types import MappingProxyType
import hashlib
import json
import logging
Expand Down Expand Up @@ -89,15 +90,81 @@
# safe to send on every request, authenticated or not.
_HTTP_USER_AGENT = "contextual-orchestrator/0.2.0 (+https://github.com/ContextualWisdomLab/contextual-orchestrator)"
_CAPABILITY_NAMES = {"embeddings": "embedding"}
# The live Go catalog exposes only id/object/created/owned_by. Keep the one
# serving allowlist to the official Chat Completions table; every other row is
# retained as evidence-only until the API reports a protocol or an adapter exists.
_OPENCODE_GO_CHAT_MODELS = frozenset({
"glm-5.3-flash", "glm-5.3", "glm-5.2", "glm-5.1", "kimi-k3",
"kimi-k2.7-code", "kimi-k2.6", "longcat-2.0", "deepseek-v4-pro",
"deepseek-v4-flash", "deepseek-v4-flash-vision-exp", "mimo-v2.5",
"mimo-v2.5-pro", "hy4-preview", "hy3",
# The live Go catalog exposes only id/object/created/owned_by, so the wire
# protocol per model comes from the official endpoint table at
# https://opencode.ai/docs/go/#endpoints (checked 2026-09-26). Only
# ``chat/completions`` rows are served today; ``responses`` and ``messages``
# rows stay evidence-only until an adapter for that protocol exists. A model
# the live catalog lists but the table does not name has no known protocol and
# also stays evidence-only.
OPENCODE_GO_MODEL_ENDPOINTS: Mapping[str, str] = MappingProxyType({
"grok-4.7": "responses",
"grok-4.6": "responses",
"gpt-6-luna": "responses",
"gpt-5.6-luna": "responses",
"muse-spark-1.3-contributor": "responses",
"muse-spark-1.2-contributor": "responses",
"glm-5.3-flash": "chat/completions",
"glm-5.3": "chat/completions",
"glm-5.2": "chat/completions",
"glm-5.1": "chat/completions",
"kimi-k3": "chat/completions",
"kimi-k2.7-code": "chat/completions",
"kimi-k2.6": "chat/completions",
"longcat-2.0": "chat/completions",
"deepseek-v4.1-flash": "chat/completions",
"deepseek-v4-pro": "chat/completions",
"deepseek-v4-flash": "chat/completions",
"deepseek-v4-flash-vision-exp": "chat/completions",
"mimo-v2.6-flash": "chat/completions",
"mimo-v2.6-pro": "chat/completions",
"mimo-v2.5": "chat/completions",
"mimo-v2.5-pro": "chat/completions",
"hy4-preview": "chat/completions",
"hy3": "chat/completions",
"space-bunny-free": "chat/completions",
"minimax-m3": "messages",
"minimax-m2.7": "messages",
"minimax-m2.5": "messages",
"qwen3.8-max": "messages",
"qwen3.8-flash": "messages",
"qwen3.7-max": "messages",
"qwen3.7-plus": "messages",
"qwen3.6-plus": "messages",
})
_OPENCODE_GO_CHAT_MODELS = frozenset(
model_id
for model_id, endpoint in OPENCODE_GO_MODEL_ENDPOINTS.items()
if endpoint == "chat/completions"
)


def opencode_go_model_endpoint(model_id: str) -> str | None:
"""Return the documented OpenCode Go endpoint path for one model, if known."""
return OPENCODE_GO_MODEL_ENDPOINTS.get(model_id)


def bootstrap_credential_value(
environ: Mapping[str, str], credential_name: str
) -> str:
"""Return the non-blank bootstrap environment value for one KV credential.

Only the exact registered name is read -- no alternate spelling. For
Experiential Labs that is ``EXPERIENTAL_LABS_API_KEY``, the organization
Secret name as registered (see
``docs/doctoring/current-main-provider-bootstrap.md``); reading another
spelling first could pick up a different key from a developer shell and
spend that account's paid credits. Only trailing CR/LF bytes from mounted
secret files are removed; every other byte is preserved. Returns ``""``
when the name is unset or blank.
"""
raw = environ.get(credential_name, "")
value = raw.rstrip("\r\n") if isinstance(raw, str) else ""
if value and value.strip():
return value
return ""


DISCOVERY_TOOL_CALL_SINGLE_TAG = "discovery:tool_call:single"
DISCOVERY_TOOL_CALL_MULTI_TAG = "discovery:tool_call:multi"

Expand Down Expand Up @@ -1656,6 +1723,11 @@ def _url_with_task_filter(url: str, task_filter: str) -> str:
)


# Bytez answers an invalid or unauthorized key with one of these; retrying the
# same key against the unfiltered catalog cannot succeed.
_BYTEZ_AUTH_FAILURE_STATUSES = frozenset({401, 403})


def _fetch_provider_json_with_retry(
fetch: Callable[..., Any],
url: str,
Expand Down Expand Up @@ -1702,6 +1774,12 @@ def _discover_bytez_task_catalog(
dict.fromkeys((source.task_filter, *source.fallback_task_filters))
)
last_exc: Exception | None = None
# The first HTTP failure from a task-filtered call is the error this
# provider reports if nothing is discovered: the unfiltered fallback below
# must never replace a real ``http_status_500`` with its own timeout,
# oversized/invalid body, different status, or empty result.
first_http_failure: urllib.error.HTTPError | None = None
auth_rejected = False
for task_filter in task_filters:
if not task_filter:
continue
Expand All @@ -1713,6 +1791,11 @@ def _discover_bytez_task_catalog(
)
if failure is not None:
last_exc = failure
if isinstance(failure, urllib.error.HTTPError):
if first_http_failure is None:
first_http_failure = failure
if failure.code in _BYTEZ_AUTH_FAILURE_STATUSES:
auth_rejected = True
_LOGGER.info(
"discovery_task_result account=%s task=%s outcome=failed error_code=%s",
source.provider_name,
Expand All @@ -1730,10 +1813,60 @@ def _discover_bytez_task_catalog(
)
if discovered:
return discovered
if last_exc is not None:
if not auth_rejected:
# Last resort: Bytez has answered HTTP 500 to both task-filtered list
# calls while the key was valid. Ask once for the unfiltered catalog
# and keep only rows whose own ``task`` field is one of the
# chat-compatible tasks above, so no non-chat model slips in. A
# 401/403 means the key itself was refused, so the fallback is
# skipped rather than spending another call on the same key. The
# unfiltered catalog is large and may exceed
# MAX_DISCOVERY_RESPONSE_BYTES; that surfaces as a fallback failure
# and the original filtered-call error is still reported.
payload, failure = _fetch_provider_json_with_retry(
fetch,
source.list_url,
timeout=timeout,
fetch_kwargs=fetch_kwargs,
)
if failure is not None:
if last_exc is None:
last_exc = failure
_LOGGER.info(
"discovery_task_result account=%s task=unfiltered outcome=failed error_code=%s",
source.provider_name,
_provider_discovery_error_code(failure),
)
else:
allowed_tasks = {task for task in task_filters if task}
rows = payload.get("output") if isinstance(payload, dict) else None
filtered_payload = {
"output": [
row
for row in (rows if isinstance(rows, list) else [])
# A list/dict ``task`` is unhashable; the ``str`` check
# keeps the set lookup from raising TypeError, which is
# not a ProviderDiscoveryError and would abort discovery
# for every provider.
if isinstance(row, dict)
and isinstance(row.get("task"), str)
and row["task"] in allowed_tasks
]
}
discovered = _parse_bytez(filtered_payload, source)
_LOGGER.info(
"discovery_task_result account=%s task=unfiltered outcome=%s model_count=%d",
source.provider_name,
"succeeded" if discovered else "empty",
len(discovered),
)
if discovered:
return discovered
reported_exc = first_http_failure if first_http_failure is not None else last_exc
if reported_exc is not None:
raise ProviderDiscoveryError(
source.provider_name,
_provider_discovery_error_code(last_exc),
_provider_discovery_error_code(reported_exc),
source.credential_name,
) from None
raise ProviderDiscoveryError(
Expand Down
97 changes: 97 additions & 0 deletions contextual_orchestrator/opencode_headers.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,97 @@
"""Client identification headers OpenCode Zen and Go ask integrators to send.

The OpenCode Go documentation (https://opencode.ai/docs/go/#where-can-i-use-it,
checked 2026-09-26) asks every client to identify itself with its own user
agent instead of a generic HTTP-library name, and to send a stable session ID
in ``x-opencode-session`` for each conversation so OpenCode can route and
prompt-cache it. Requests to any other provider get no extra headers.
"""

from __future__ import annotations

import hashlib
import json
from collections.abc import Mapping
from typing import Any
from urllib.parse import urlsplit

OPENCODE_PROVIDER_NAMES = frozenset({"opencode_zen", "opencode_go"})
OPENCODE_HOST = "opencode.ai"
OPENCODE_SESSION_HEADER = "x-opencode-session"
OPENCODE_USER_AGENT = (
"contextual-orchestrator/0.2.0 "
"(+https://github.com/ContextualWisdomLab/contextual-orchestrator)"
)
_MAX_CALLER_SESSION_CHARS = 128


def _is_opencode_agent(agent: Any) -> bool:
if getattr(agent, "provider_name", None) in OPENCODE_PROVIDER_NAMES:
return True
base_url = getattr(agent, "base_url", "")
if not isinstance(base_url, str) or not base_url:
return False
host = (urlsplit(base_url).hostname or "").casefold()
return host == OPENCODE_HOST or host.endswith("." + OPENCODE_HOST)


def _caller_session_id(payload: Mapping[str, Any]) -> str | None:
"""Return a caller-chosen conversation key when the request carries one."""
for field in ("prompt_cache_key", "session_id"):
value = payload.get(field)
if isinstance(value, str) and value.strip():
return value.strip()[:_MAX_CALLER_SESSION_CHARS]
metadata = payload.get("metadata")
if isinstance(metadata, Mapping):
value = metadata.get("session_id")
if isinstance(value, str) and value.strip():
return value.strip()[:_MAX_CALLER_SESSION_CHARS]
return None


def _conversation_prefix(payload: Mapping[str, Any]) -> Any:
"""Return the part of a request that stays fixed across one conversation.

Every turn of a chat conversation resends the same opening system and
first user message, so hashing that prefix yields the same ID on each
turn without storing any state. Responses-style ``instructions`` and the
first ``input`` item play the same role there.
"""
messages = payload.get("messages")
if isinstance(messages, list) and messages:
prefix: list[Any] = []
for message in messages:
prefix.append(message)
if isinstance(message, Mapping) and message.get("role") == "user":
break
return {"messages": prefix}
input_value = payload.get("input")
if isinstance(input_value, list) and input_value:
input_value = input_value[0]
return {"instructions": payload.get("instructions"), "input": input_value}


def opencode_session_id(payload: Mapping[str, Any]) -> str:
"""Return a stable, non-reversible session ID for one conversation."""
caller = _caller_session_id(payload)
if caller is not None:
return caller
canonical = json.dumps(
_conversation_prefix(payload),
sort_keys=True,
separators=(",", ":"),
ensure_ascii=False,
default=str,
)
digest = hashlib.sha256(canonical.encode("utf-8")).hexdigest()[:32]
return f"co-{digest}"


def opencode_request_headers(agent: Any, payload: Mapping[str, Any]) -> dict[str, str]:
"""Return the OpenCode identification headers for one provider request."""
if not _is_opencode_agent(agent):
return {}
return {
"user-agent": OPENCODE_USER_AGENT,
OPENCODE_SESSION_HEADER: opencode_session_id(payload),
}
6 changes: 3 additions & 3 deletions contextual_orchestrator/provider_bootstrap.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,7 @@
_currency_is_comparable,
agent_from_discovered,
agent_id_for,
bootstrap_credential_value,
discover_all_models,
discovery_tool_call_tags,
privacy_tags_for_discovered,
Expand Down Expand Up @@ -116,9 +117,8 @@ def collect_provider_credentials(
values: dict[str, str] = {}
missing: list[str] = []
for name in PROVIDER_ACCEPTED_CREDENTIAL_NAMES:
raw = environ.get(name, "")
value = _strip_mounted_line_endings(raw) if isinstance(raw, str) else ""
if value and value.strip():
value = bootstrap_credential_value(environ, name)
if value:
values[name] = value
elif name in PROVIDER_CREDENTIAL_NAMES:
missing.append(name)
Expand Down
6 changes: 3 additions & 3 deletions contextual_orchestrator/review_gateway.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@
from .model_discovery import (
DiscoveredModel,
agent_from_discovered,
bootstrap_credential_value,
discover_all_models,
general_free_serving_candidates,
)
Expand Down Expand Up @@ -158,9 +159,8 @@ def register_review_credentials(
requested_names = _validated_credential_names(credential_names)
registered: list[str] = []
for name in requested_names:
raw_value = environment.get(name, "")
value = raw_value.rstrip("\r\n") if isinstance(raw_value, str) else ""
if value and value.strip():
value = bootstrap_credential_value(environment, name)
if value:
register_credential(name, value)
registered.append(name)
raw_auth_value = environment.get(REVIEW_AUTH_CREDENTIAL_NAME, "")
Expand Down
8 changes: 6 additions & 2 deletions docs/doctoring/current-main-provider-bootstrap.md
Original file line number Diff line number Diff line change
Expand Up @@ -133,8 +133,12 @@ Nielsen, S., Cetin, E., Schwendeman, P., Sun, Q., Xu, J., & Tang, Y. (2025).

`experiential_labs` is an optional provider. Bootstrap accepts the organization
Secret name `EXPERIENTAL_LABS_API_KEY` exactly as registered; the spelling is
intentional. The catalog-sync workflow transports it into the existing encrypted
KV. Runtime discovery reads that KV name and sends Bearer authentication to
intentional. Provider bootstrap, the review gateway and the seeded CI gateway
read only that name: `EXPERIENTIAL_LABS_API_KEY`, `EXPLABS_API_KEY` or any
other spelling in the environment is ignored, so a different key in a developer
shell cannot be picked up and spend that account's credits. The catalog-sync
workflow transports it into the existing encrypted KV. Runtime discovery reads
that KV name and sends Bearer authentication to
`https://api.experientiallabs.ai/v1/models`; chat uses the same `/v1` base.
Existing deployments do not need this additional credential.

Expand Down
11 changes: 9 additions & 2 deletions scripts/ci/serve_seeded_gateway.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,7 +23,10 @@
register_credential,
set_backend,
)
from contextual_orchestrator.model_discovery import PROVIDER_MODEL_SOURCES
from contextual_orchestrator.model_discovery import (
PROVIDER_MODEL_SOURCES,
bootstrap_credential_value,
)

PROVIDER_KEY_ENV_NAMES = tuple(
dict.fromkeys(source.credential_name for source in PROVIDER_MODEL_SOURCES)
Expand All @@ -36,7 +39,11 @@ def seed_credentials_from_bootstrap_env() -> list[str]:
set_backend(InMemoryCredentialBackend())
seeded: list[str] = []
for credential_name in (*PROVIDER_KEY_ENV_NAMES, SERVER_AUTH_ENV_NAME):
value = os.environ.pop(credential_name, None)
# Same normalization as provider bootstrap and the review gateway:
# only trailing CR/LF is removed, and a blank or whitespace-only value
# counts as unset. Pop the name either way so it never lingers.
value = bootstrap_credential_value(os.environ, credential_name)
os.environ.pop(credential_name, None)
# Bootstrap transport only: the trusted CI job injects each value into
# this process environment; nothing reads os.environ again after this.
if value:
Expand Down
Loading
Loading