Skip to content
Draft
Show file tree
Hide file tree
Changes from 2 commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Fixed three provider-discovery gaps. Bytez discovery now makes one last unfiltered catalog request when both task-filtered list calls fail or come back empty, keeping only rows whose own `task` is `chat` or `text-generation`, so a Bytez HTTP 500 on the filtered endpoint no longer empties the pool. The Experiential Labs key is read from `EXPERIENTIAL_LABS_API_KEY` first, then the historical `EXPERIENTAL_LABS_API_KEY`, then the provider-documented `EXPLABS_API_KEY`, in provider bootstrap, the review gateway, and the seeded CI gateway (the KV label is unchanged). OpenCode Go now carries the documented per-model endpoint table (`OPENCODE_GO_MODEL_ENDPOINTS`, from https://opencode.ai/docs/go/#endpoints): `chat/completions` models are served, while `responses` and `messages` models stay evidence-only until those adapters exist. Added `contextual_orchestrator.opencode_headers.opencode_request_headers`, which builds the identifying `user-agent` and stable per-conversation `x-opencode-session` header the OpenCode Go docs ask for; wiring it into the `ModelClient` send paths lands separately with the orchestrator fallback change.
134 changes: 126 additions & 8 deletions contextual_orchestrator/model_discovery.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@
from __future__ import annotations

from decimal import Decimal
from types import MappingProxyType
import hashlib
import json
import logging
Expand Down Expand Up @@ -89,15 +90,95 @@
# safe to send on every request, authenticated or not.
_HTTP_USER_AGENT = "contextual-orchestrator/0.2.0 (+https://github.com/ContextualWisdomLab/contextual-orchestrator)"
_CAPABILITY_NAMES = {"embeddings": "embedding"}
# The live Go catalog exposes only id/object/created/owned_by. Keep the one
# serving allowlist to the official Chat Completions table; every other row is
# retained as evidence-only until the API reports a protocol or an adapter exists.
_OPENCODE_GO_CHAT_MODELS = frozenset({
"glm-5.3-flash", "glm-5.3", "glm-5.2", "glm-5.1", "kimi-k3",
"kimi-k2.7-code", "kimi-k2.6", "longcat-2.0", "deepseek-v4-pro",
"deepseek-v4-flash", "deepseek-v4-flash-vision-exp", "mimo-v2.5",
"mimo-v2.5-pro", "hy4-preview", "hy3",
# The live Go catalog exposes only id/object/created/owned_by, so the wire
# protocol per model comes from the official endpoint table at
# https://opencode.ai/docs/go/#endpoints (checked 2026-09-26). Only
# ``chat/completions`` rows are served today; ``responses`` and ``messages``
# rows stay evidence-only until an adapter for that protocol exists. A model
# the live catalog lists but the table does not name has no known protocol and
# also stays evidence-only.
OPENCODE_GO_MODEL_ENDPOINTS: Mapping[str, str] = MappingProxyType({
"grok-4.7": "responses",
"grok-4.6": "responses",
"gpt-6-luna": "responses",
"gpt-5.6-luna": "responses",
"muse-spark-1.3-contributor": "responses",
"muse-spark-1.2-contributor": "responses",
"glm-5.3-flash": "chat/completions",
"glm-5.3": "chat/completions",
"glm-5.2": "chat/completions",
"glm-5.1": "chat/completions",
"kimi-k3": "chat/completions",
"kimi-k2.7-code": "chat/completions",
"kimi-k2.6": "chat/completions",
"longcat-2.0": "chat/completions",
"deepseek-v4.1-flash": "chat/completions",
"deepseek-v4-pro": "chat/completions",
"deepseek-v4-flash": "chat/completions",
"deepseek-v4-flash-vision-exp": "chat/completions",
"mimo-v2.6-flash": "chat/completions",
"mimo-v2.6-pro": "chat/completions",
"mimo-v2.5": "chat/completions",
"mimo-v2.5-pro": "chat/completions",
"hy4-preview": "chat/completions",
"hy3": "chat/completions",
"space-bunny-free": "chat/completions",
"minimax-m3": "messages",
"minimax-m2.7": "messages",
"minimax-m2.5": "messages",
"qwen3.8-max": "messages",
"qwen3.8-flash": "messages",
"qwen3.7-max": "messages",
"qwen3.7-plus": "messages",
"qwen3.6-plus": "messages",
})
_OPENCODE_GO_CHAT_MODELS = frozenset(
model_id
for model_id, endpoint in OPENCODE_GO_MODEL_ENDPOINTS.items()
if endpoint == "chat/completions"
)


def opencode_go_model_endpoint(model_id: str) -> str | None:
"""Return the documented OpenCode Go endpoint path for one model, if known."""
return OPENCODE_GO_MODEL_ENDPOINTS.get(model_id)


# Deployments have used three spellings for the Experiential Labs key. The KV
# label stays ``EXPERIENTAL_LABS_API_KEY`` (existing stored secrets and pool
# policy use it), while bootstrap reads the environment in this order: the
# correct spelling first, then the historical typo, then the provider's own
# documented ``EXPLABS_API_KEY``.
CREDENTIAL_ENV_ALIASES: Mapping[str, tuple[str, ...]] = MappingProxyType({
"EXPERIENTAL_LABS_API_KEY": (
"EXPERIENTIAL_LABS_API_KEY",
"EXPERIENTAL_LABS_API_KEY",
"EXPLABS_API_KEY",
),
})


def credential_env_names(credential_name: str) -> tuple[str, ...]:
"""Return the environment names that may carry one KV credential, in order."""
return CREDENTIAL_ENV_ALIASES.get(credential_name, (credential_name,))


def bootstrap_credential_value(
environ: Mapping[str, str], credential_name: str
) -> str:
"""Return the first non-blank environment value for one KV credential.

Only trailing CR/LF bytes from mounted secret files are removed; every
other byte is preserved. Returns ``""`` when no spelling is set.
"""
for env_name in credential_env_names(credential_name):
raw = environ.get(env_name, "")
value = raw.rstrip("\r\n") if isinstance(raw, str) else ""
if value and value.strip():
return value
return ""


DISCOVERY_TOOL_CALL_SINGLE_TAG = "discovery:tool_call:single"
DISCOVERY_TOOL_CALL_MULTI_TAG = "discovery:tool_call:multi"

Expand Down Expand Up @@ -1730,6 +1811,43 @@ def _discover_bytez_task_catalog(
)
if discovered:
return discovered
# Last resort: Bytez has answered HTTP 500 to both task-filtered list calls
# while the key was valid (an invalid key answers 401). Ask once for the
# unfiltered catalog and keep only rows whose own ``task`` field names one
# of the chat-compatible tasks above, so no non-chat model slips in.
payload, failure = _fetch_provider_json_with_retry(
fetch,
source.list_url,
timeout=timeout,
fetch_kwargs=fetch_kwargs,
)
if failure is not None:
last_exc = failure
_LOGGER.info(
"discovery_task_result account=%s task=unfiltered outcome=failed error_code=%s",
source.provider_name,
_provider_discovery_error_code(failure),
)
else:
allowed_tasks = {task for task in task_filters if task}
rows = payload.get("output") if isinstance(payload, dict) else None
filtered_payload = {
"output": [
row
for row in (rows if isinstance(rows, list) else [])
if isinstance(row, dict) and row.get("task") in allowed_tasks
]
}
discovered = _parse_bytez(filtered_payload, source)
_LOGGER.info(
"discovery_task_result account=%s task=unfiltered outcome=%s model_count=%d",
source.provider_name,
"succeeded" if discovered else "empty",
len(discovered),
)
if discovered:
return discovered
last_exc = None
if last_exc is not None:
raise ProviderDiscoveryError(
source.provider_name,
Expand Down
97 changes: 97 additions & 0 deletions contextual_orchestrator/opencode_headers.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,97 @@
"""Client identification headers OpenCode Zen and Go ask integrators to send.

The OpenCode Go documentation (https://opencode.ai/docs/go/#where-can-i-use-it,
checked 2026-09-26) asks every client to identify itself with its own user
agent instead of a generic HTTP-library name, and to send a stable session ID
in ``x-opencode-session`` for each conversation so OpenCode can route and
prompt-cache it. Requests to any other provider get no extra headers.
"""

from __future__ import annotations

import hashlib
import json
from collections.abc import Mapping
from typing import Any
from urllib.parse import urlsplit

OPENCODE_PROVIDER_NAMES = frozenset({"opencode_zen", "opencode_go"})
OPENCODE_HOST = "opencode.ai"
OPENCODE_SESSION_HEADER = "x-opencode-session"
OPENCODE_USER_AGENT = (
"contextual-orchestrator/0.2.0 "
"(+https://github.com/ContextualWisdomLab/contextual-orchestrator)"
)
_MAX_CALLER_SESSION_CHARS = 128


def _is_opencode_agent(agent: Any) -> bool:
if getattr(agent, "provider_name", None) in OPENCODE_PROVIDER_NAMES:
return True
base_url = getattr(agent, "base_url", "")
if not isinstance(base_url, str) or not base_url:
return False
host = (urlsplit(base_url).hostname or "").casefold()
return host == OPENCODE_HOST or host.endswith("." + OPENCODE_HOST)


def _caller_session_id(payload: Mapping[str, Any]) -> str | None:
"""Return a caller-chosen conversation key when the request carries one."""
for field in ("prompt_cache_key", "session_id"):
value = payload.get(field)
if isinstance(value, str) and value.strip():
return value.strip()[:_MAX_CALLER_SESSION_CHARS]
metadata = payload.get("metadata")
if isinstance(metadata, Mapping):
value = metadata.get("session_id")
if isinstance(value, str) and value.strip():
return value.strip()[:_MAX_CALLER_SESSION_CHARS]
return None


def _conversation_prefix(payload: Mapping[str, Any]) -> Any:
"""Return the part of a request that stays fixed across one conversation.

Every turn of a chat conversation resends the same opening system and
first user message, so hashing that prefix yields the same ID on each
turn without storing any state. Responses-style ``instructions`` and the
first ``input`` item play the same role there.
"""
messages = payload.get("messages")
if isinstance(messages, list) and messages:
prefix: list[Any] = []
for message in messages:
prefix.append(message)
if isinstance(message, Mapping) and message.get("role") == "user":
break
return {"messages": prefix}
input_value = payload.get("input")
if isinstance(input_value, list) and input_value:
input_value = input_value[0]
return {"instructions": payload.get("instructions"), "input": input_value}


def opencode_session_id(payload: Mapping[str, Any]) -> str:
"""Return a stable, non-reversible session ID for one conversation."""
caller = _caller_session_id(payload)
if caller is not None:
return caller
canonical = json.dumps(
_conversation_prefix(payload),
sort_keys=True,
separators=(",", ":"),
ensure_ascii=False,
default=str,
)
digest = hashlib.sha256(canonical.encode("utf-8")).hexdigest()[:32]
return f"co-{digest}"


def opencode_request_headers(agent: Any, payload: Mapping[str, Any]) -> dict[str, str]:
"""Return the OpenCode identification headers for one provider request."""
if not _is_opencode_agent(agent):
return {}
return {
"user-agent": OPENCODE_USER_AGENT,
OPENCODE_SESSION_HEADER: opencode_session_id(payload),
}
6 changes: 3 additions & 3 deletions contextual_orchestrator/provider_bootstrap.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,7 @@
_currency_is_comparable,
agent_from_discovered,
agent_id_for,
bootstrap_credential_value,
discover_all_models,
discovery_tool_call_tags,
privacy_tags_for_discovered,
Expand Down Expand Up @@ -116,9 +117,8 @@ def collect_provider_credentials(
values: dict[str, str] = {}
missing: list[str] = []
for name in PROVIDER_ACCEPTED_CREDENTIAL_NAMES:
raw = environ.get(name, "")
value = _strip_mounted_line_endings(raw) if isinstance(raw, str) else ""
if value and value.strip():
value = bootstrap_credential_value(environ, name)
if value:
values[name] = value
elif name in PROVIDER_CREDENTIAL_NAMES:
missing.append(name)
Expand Down
6 changes: 3 additions & 3 deletions contextual_orchestrator/review_gateway.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@
from .model_discovery import (
DiscoveredModel,
agent_from_discovered,
bootstrap_credential_value,
discover_all_models,
general_free_serving_candidates,
)
Expand Down Expand Up @@ -158,9 +159,8 @@ def register_review_credentials(
requested_names = _validated_credential_names(credential_names)
registered: list[str] = []
for name in requested_names:
raw_value = environment.get(name, "")
value = raw_value.rstrip("\r\n") if isinstance(raw_value, str) else ""
if value and value.strip():
value = bootstrap_credential_value(environment, name)
if value:
register_credential(name, value)
registered.append(name)
raw_auth_value = environment.get(REVIEW_AUTH_CREDENTIAL_NAME, "")
Expand Down
13 changes: 11 additions & 2 deletions scripts/ci/serve_seeded_gateway.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,7 +23,10 @@
register_credential,
set_backend,
)
from contextual_orchestrator.model_discovery import PROVIDER_MODEL_SOURCES
from contextual_orchestrator.model_discovery import (
PROVIDER_MODEL_SOURCES,
credential_env_names,
)

PROVIDER_KEY_ENV_NAMES = tuple(
dict.fromkeys(source.credential_name for source in PROVIDER_MODEL_SOURCES)
Expand All @@ -36,7 +39,13 @@ def seed_credentials_from_bootstrap_env() -> list[str]:
set_backend(InMemoryCredentialBackend())
seeded: list[str] = []
for credential_name in (*PROVIDER_KEY_ENV_NAMES, SERVER_AUTH_ENV_NAME):
value = os.environ.pop(credential_name, None)
# Pop every accepted spelling so no alias lingers in the environment;
# the first non-empty one in documented order wins.
value = None
for env_name in credential_env_names(credential_name):
candidate = os.environ.pop(env_name, None)
if candidate and not value:
value = candidate
Comment thread
coderabbitai[bot] marked this conversation as resolved.
Outdated
# Bootstrap transport only: the trusted CI job injects each value into
# this process environment; nothing reads os.environ again after this.
if value:
Expand Down
Loading
Loading