diff --git a/.github/workflows/ecr-server.yml b/.github/workflows/ecr-server.yml index ec504e75..085bd8ad 100644 --- a/.github/workflows/ecr-server.yml +++ b/.github/workflows/ecr-server.yml @@ -97,11 +97,12 @@ jobs: # headless, so host-skill discovery would put ~/.claude/skills in its # reach. Do not "fix" this back to "all" when the gate trips; read the # comment on the bundle before changing this value. + # The last column is the expected reasoning effort; None skips the check. EXPECT = { - "sei-spec": ("none", 8, "speckit-"), - "root-cause": ("none", 1, None), - "xreview-scout-codex": ("none", 0, None), - "xreview-scout-cursor": ("none", 0, None), + "sei-spec": ("none", 8, "speckit-", None), + "root-cause": ("none", 1, None, None), + "xreview-scout-codex": ("none", 0, None, "high"), + "xreview-scout-cursor": ("none", 0, None, None), } root = Path("/opt/sei-omnigent/agents") dirs = sorted(p for p in root.iterdir() if p.is_dir()) @@ -110,7 +111,7 @@ jobs: bad = False for d in dirs: spec = parse(d) - want_filter, want_skills, want_prefix = EXPECT[d.name] + want_filter, want_skills, want_prefix, want_effort = EXPECT[d.name] names = sorted(s.name for s in spec.skills) n = len(names) if spec.name != d.name: @@ -123,6 +124,9 @@ jobs: stray = [x for x in names if not x.startswith(want_prefix)] if stray: print(f"FAIL {d.name}: non-{want_prefix} skills vendored in: {stray}"); bad = True + effort = getattr(spec.executor, "reasoning_effort", None) + if want_effort is not None and effort != want_effort: + print(f"FAIL {d.name}: reasoning effort {effort!r} != {want_effort!r}"); bad = True if spec.skills_filter == "none" and discover_host_skills(d, spec.skills_filter): print(f"FAIL {d.name}: host scope leaked despite skills: none"); bad = True print(f" ok {d.name} filter={spec.skills_filter!r} skills={n}") diff --git a/.github/workflows/verify-agent-bundles.yml b/.github/workflows/verify-agent-bundles.yml index 1419860b..0f217a6b 100644 --- a/.github/workflows/verify-agent-bundles.yml +++ b/.github/workflows/verify-agent-bundles.yml @@ -37,7 +37,7 @@ jobs: - name: Install the parser # No PyPI release matches the fork build the server runs, so this gate # approximates it. ecr-server.yml parses with what production has. - run: pip install --disable-pip-version-check 'omnigent==0.9.0' + run: pip install --disable-pip-version-check 'omnigent==0.16.0' - name: Parse every bundle and assert its contract run: | @@ -55,7 +55,7 @@ jobs: # A scout carries no method and takes none from the host: the driver's # per-run prompt is its whole instruction, so a discovered skill would # be capability it was never asked to use. - 'xreview-scout-codex': {'filter': 'none', 'skills': 0}, + 'xreview-scout-codex': {'filter': 'none', 'skills': 0, 'effort': 'high'}, # Same contract on a different harness. Listed here because a bundle # with no entry fails below, and the deployment cannot reach its # harness yet: this gate is the only thing asserting it stays parsable @@ -102,6 +102,12 @@ jobs: print(f'FAIL {d.name}: non-{pref} skills vendored in: {stray}') failed = True + # A parser too old to read the field reports None, which fails here too. + effort = getattr(spec.executor, 'reasoning_effort', None) + if 'effort' in exp and effort != exp['effort']: + print(f'FAIL {d.name}: reasoning effort is {effort!r}, expected {exp["effort"]!r}') + failed = True + # With `none`, host discovery must return nothing even from a directory # that has host skills to offer. if spec.skills_filter == 'none': diff --git a/agents/xreview-scout-codex/config.yaml b/agents/xreview-scout-codex/config.yaml index 7151685f..dbdbef91 100644 --- a/agents/xreview-scout-codex/config.yaml +++ b/agents/xreview-scout-codex/config.yaml @@ -28,15 +28,12 @@ executor: # vendor-neutral gateway, read live off a process that was routed at # api.openai.com. Misreading it produced three wrong diagnoses. # - # gpt-5.6-sol is what ai-review's codex pass resolves to: that workflow pins no - # model and takes openai/codex-action's default, measured on a passing run. - # Matching it is the point — a scout on a different model than the reference - # reviewer is not the second opinion we mean to reproduce. - # - # Dotted, not dashed: codex spells it gpt-5.6-sol where the catalog writes - # gpt-5-6-sol. This value goes verbatim to `codex -c model=`, and the dotted - # form is the one the API accepts. - model: gpt-5.6-sol + # gpt-6.1-sol is OpenAI's balanced GPT-6 model: near-Astra on complex work at a + # lower cost, which suits one reading per pull request. Dotted, not dashed: this + # value goes verbatim to `codex -c model=`, and the dotted form is the one the + # API accepts. + model: gpt-6.1-sol + reasoning_effort: high config: # codex, and that is the point. The bundle is what fixes the harness, # so this is the one field that makes this a genuinely different reader