From dd76b36a4924b7a76b395222adf3838483444e9c Mon Sep 17 00:00:00 2001 From: bdchatham Date: Thu, 1 Oct 2026 07:31:05 -0700 Subject: [PATCH 1/5] feat(xreview-scout-codex): run on gpt-6.1-sol at high reasoning effort The scout pinned gpt-5.6-sol. gpt-6.1-sol is now the Codex CLI's default model, released 2026-09-29. The scout also set no reasoning effort, so it ran at the harness default. It now sets `executor.reasoning_effort: high`. The deployed server (omnigent 0.16.0.dev0) reads `executor.reasoning_effort`. The bundle check in CI pins omnigent 0.9.0, which parses the bundle but drops the field, so CI does not assert it. Co-Authored-By: Claude Opus 5.5 --- agents/xreview-scout-codex/config.yaml | 14 +++++--------- 1 file changed, 5 insertions(+), 9 deletions(-) diff --git a/agents/xreview-scout-codex/config.yaml b/agents/xreview-scout-codex/config.yaml index 7151685f..8a766c60 100644 --- a/agents/xreview-scout-codex/config.yaml +++ b/agents/xreview-scout-codex/config.yaml @@ -28,15 +28,11 @@ executor: # vendor-neutral gateway, read live off a process that was routed at # api.openai.com. Misreading it produced three wrong diagnoses. # - # gpt-5.6-sol is what ai-review's codex pass resolves to: that workflow pins no - # model and takes openai/codex-action's default, measured on a passing run. - # Matching it is the point — a scout on a different model than the reference - # reviewer is not the second opinion we mean to reproduce. - # - # Dotted, not dashed: codex spells it gpt-5.6-sol where the catalog writes - # gpt-5-6-sol. This value goes verbatim to `codex -c model=`, and the dotted - # form is the one the API accepts. - model: gpt-5.6-sol + # gpt-6.1-sol is the Codex CLI's default model. Dotted, not dashed: this value + # goes verbatim to `codex -c model=`, and the dotted form is the one the API + # accepts. + model: gpt-6.1-sol + reasoning_effort: high config: # codex, and that is the point. The bundle is what fixes the harness, # so this is the one field that makes this a genuinely different reader From 926726788420eed430c852bbe382617058f4e787 Mon Sep 17 00:00:00 2001 From: bdchatham Date: Thu, 1 Oct 2026 07:47:13 -0700 Subject: [PATCH 2/5] ci(agent-bundles): parse with omnigent 0.16.0 and assert the scout's effort The bundle check pinned omnigent 0.9.0, which parses a bundle but drops executor.reasoning_effort. It now pins 0.16.0, the release the deployed server tracks, and fails if xreview-scout-codex does not read effort high. Co-Authored-By: Claude Opus 5.5 --- .github/workflows/verify-agent-bundles.yml | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/.github/workflows/verify-agent-bundles.yml b/.github/workflows/verify-agent-bundles.yml index 1419860b..0f217a6b 100644 --- a/.github/workflows/verify-agent-bundles.yml +++ b/.github/workflows/verify-agent-bundles.yml @@ -37,7 +37,7 @@ jobs: - name: Install the parser # No PyPI release matches the fork build the server runs, so this gate # approximates it. ecr-server.yml parses with what production has. - run: pip install --disable-pip-version-check 'omnigent==0.9.0' + run: pip install --disable-pip-version-check 'omnigent==0.16.0' - name: Parse every bundle and assert its contract run: | @@ -55,7 +55,7 @@ jobs: # A scout carries no method and takes none from the host: the driver's # per-run prompt is its whole instruction, so a discovered skill would # be capability it was never asked to use. - 'xreview-scout-codex': {'filter': 'none', 'skills': 0}, + 'xreview-scout-codex': {'filter': 'none', 'skills': 0, 'effort': 'high'}, # Same contract on a different harness. Listed here because a bundle # with no entry fails below, and the deployment cannot reach its # harness yet: this gate is the only thing asserting it stays parsable @@ -102,6 +102,12 @@ jobs: print(f'FAIL {d.name}: non-{pref} skills vendored in: {stray}') failed = True + # A parser too old to read the field reports None, which fails here too. + effort = getattr(spec.executor, 'reasoning_effort', None) + if 'effort' in exp and effort != exp['effort']: + print(f'FAIL {d.name}: reasoning effort is {effort!r}, expected {exp["effort"]!r}') + failed = True + # With `none`, host discovery must return nothing even from a directory # that has host skills to offer. if spec.skills_filter == 'none': From 0f62953ed7121e2e2d406a4d9342b32013deadde Mon Sep 17 00:00:00 2001 From: bdchatham Date: Thu, 1 Oct 2026 08:06:36 -0700 Subject: [PATCH 3/5] docs(xreview-scout-codex): give the model's own reason for gpt-6.1-sol The comment named the Codex CLI's default as the reason, but this bundle runs the codex harness without the Codex CLI. Co-Authored-By: Claude Opus 5.5 --- agents/xreview-scout-codex/config.yaml | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/agents/xreview-scout-codex/config.yaml b/agents/xreview-scout-codex/config.yaml index 8a766c60..dbdbef91 100644 --- a/agents/xreview-scout-codex/config.yaml +++ b/agents/xreview-scout-codex/config.yaml @@ -28,9 +28,10 @@ executor: # vendor-neutral gateway, read live off a process that was routed at # api.openai.com. Misreading it produced three wrong diagnoses. # - # gpt-6.1-sol is the Codex CLI's default model. Dotted, not dashed: this value - # goes verbatim to `codex -c model=`, and the dotted form is the one the API - # accepts. + # gpt-6.1-sol is OpenAI's balanced GPT-6 model: near-Astra on complex work at a + # lower cost, which suits one reading per pull request. Dotted, not dashed: this + # value goes verbatim to `codex -c model=`, and the dotted form is the one the + # API accepts. model: gpt-6.1-sol reasoning_effort: high config: From dcf0cbbac53b3d71525b24fed00fa9dfe1f54a6b Mon Sep 17 00:00:00 2001 From: bdchatham Date: Thu, 1 Oct 2026 08:25:05 -0700 Subject: [PATCH 4/5] ci(ecr-server): assert the scout's reasoning effort in the publish gate The PR gate's EXPECT map now requires effort high for xreview-scout-codex, but the publish gate's map did not, so a later drop of reasoning_effort could pass the in-image parse that production uses. Both gates now assert it. Co-Authored-By: Claude Opus 5.5 --- .github/workflows/ecr-server.yml | 14 +++++++++----- 1 file changed, 9 insertions(+), 5 deletions(-) diff --git a/.github/workflows/ecr-server.yml b/.github/workflows/ecr-server.yml index ec504e75..3645122e 100644 --- a/.github/workflows/ecr-server.yml +++ b/.github/workflows/ecr-server.yml @@ -97,11 +97,12 @@ jobs: # headless, so host-skill discovery would put ~/.claude/skills in its # reach. Do not "fix" this back to "all" when the gate trips; read the # comment on the bundle before changing this value. + # The last column is the reasoning effort, None where a bundle sets none. EXPECT = { - "sei-spec": ("none", 8, "speckit-"), - "root-cause": ("none", 1, None), - "xreview-scout-codex": ("none", 0, None), - "xreview-scout-cursor": ("none", 0, None), + "sei-spec": ("none", 8, "speckit-", None), + "root-cause": ("none", 1, None, None), + "xreview-scout-codex": ("none", 0, None, "high"), + "xreview-scout-cursor": ("none", 0, None, None), } root = Path("/opt/sei-omnigent/agents") dirs = sorted(p for p in root.iterdir() if p.is_dir()) @@ -110,7 +111,7 @@ jobs: bad = False for d in dirs: spec = parse(d) - want_filter, want_skills, want_prefix = EXPECT[d.name] + want_filter, want_skills, want_prefix, want_effort = EXPECT[d.name] names = sorted(s.name for s in spec.skills) n = len(names) if spec.name != d.name: @@ -123,6 +124,9 @@ jobs: stray = [x for x in names if not x.startswith(want_prefix)] if stray: print(f"FAIL {d.name}: non-{want_prefix} skills vendored in: {stray}"); bad = True + effort = getattr(spec.executor, "reasoning_effort", None) + if want_effort is not None and effort != want_effort: + print(f"FAIL {d.name}: reasoning effort {effort!r} != {want_effort!r}"); bad = True if spec.skills_filter == "none" and discover_host_skills(d, spec.skills_filter): print(f"FAIL {d.name}: host scope leaked despite skills: none"); bad = True print(f" ok {d.name} filter={spec.skills_filter!r} skills={n}") From fd4cefba12e22c24c760164a047e6dc9609c3358 Mon Sep 17 00:00:00 2001 From: bdchatham Date: Thu, 1 Oct 2026 08:29:20 -0700 Subject: [PATCH 5/5] docs(ecr-server): say that a None effort skips the check Co-Authored-By: Claude Opus 5.5 --- .github/workflows/ecr-server.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/ecr-server.yml b/.github/workflows/ecr-server.yml index 3645122e..085bd8ad 100644 --- a/.github/workflows/ecr-server.yml +++ b/.github/workflows/ecr-server.yml @@ -97,7 +97,7 @@ jobs: # headless, so host-skill discovery would put ~/.claude/skills in its # reach. Do not "fix" this back to "all" when the gate trips; read the # comment on the bundle before changing this value. - # The last column is the reasoning effort, None where a bundle sets none. + # The last column is the expected reasoning effort; None skips the check. EXPECT = { "sei-spec": ("none", 8, "speckit-", None), "root-cause": ("none", 1, None, None),