diff --git a/AGENTS.md b/AGENTS.md index e6b30502..64b5c0b6 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -12,3 +12,11 @@ These instructions define repository-specific execution rules and scope limits f - Run `cargo make` from the repository root, and use it whenever an equivalent task exists. - Run standalone commands only when `Makefile.toml` does not cover the capability or cannot produce the required effect for the current task. - When task details are needed, inspect `Makefile.toml` directly or run `cargo make --list-all-steps`. + +## Rust channel policy + +Use the unversioned `stable` channel for Rust builds and tests. Use the +unversioned `nightly` channel for Rust formatting. Do not select numbered +compiler versions or dated nightly versions in local tasks, CI, or container +builds. Rust container images must follow the stable release with a floating +distribution tag. diff --git a/apps/elf-eval/src/bin/real_world_job_benchmark/scoreboard/external/signals.rs b/apps/elf-eval/src/bin/real_world_job_benchmark/scoreboard/external/signals.rs index d08cc642..76b91cd0 100644 --- a/apps/elf-eval/src/bin/real_world_job_benchmark/scoreboard/external/signals.rs +++ b/apps/elf-eval/src/bin/real_world_job_benchmark/scoreboard/external/signals.rs @@ -40,16 +40,23 @@ fn adapter_has_passing_text(adapter: &ExternalAdapterReport, needles: &[&str]) - adapter.result.status, adapter.result.evidence.as_str(), needles, - ) || adapter.capabilities.iter().any(|capability| { - adapter_status_mentions_any(capability.status, capability.capability.as_str(), needles) - || adapter_status_mentions_any(capability.status, capability.evidence.as_str(), needles) - }) || adapter.suites.iter().any(|suite| { - adapter_status_mentions_any(suite.status, suite.suite_id.as_str(), needles) - || adapter_status_mentions_any(suite.status, suite.evidence.as_str(), needles) - }) || adapter.scenarios.iter().any(|scenario| { - adapter_status_mentions_any(scenario.status, scenario.scenario_id.as_str(), needles) - || adapter_status_mentions_any(scenario.status, scenario.evidence.as_str(), needles) - }) + ) + || adapter.capabilities.iter().any(|capability| { + adapter_status_mentions_any(capability.status, capability.capability.as_str(), needles) + || adapter_status_mentions_any( + capability.status, + capability.evidence.as_str(), + needles, + ) + }) + || adapter.suites.iter().any(|suite| { + adapter_status_mentions_any(suite.status, suite.suite_id.as_str(), needles) + || adapter_status_mentions_any(suite.status, suite.evidence.as_str(), needles) + }) + || adapter.scenarios.iter().any(|scenario| { + adapter_status_mentions_any(scenario.status, scenario.scenario_id.as_str(), needles) + || adapter_status_mentions_any(scenario.status, scenario.evidence.as_str(), needles) + }) } fn adapter_has_reported_same_corpus_text( diff --git a/apps/elf-eval/src/bin/real_world_job_benchmark/validation/common.rs b/apps/elf-eval/src/bin/real_world_job_benchmark/validation/common.rs index 4f6c8e73..f5f7bccb 100644 --- a/apps/elf-eval/src/bin/real_world_job_benchmark/validation/common.rs +++ b/apps/elf-eval/src/bin/real_world_job_benchmark/validation/common.rs @@ -95,7 +95,8 @@ pub(super) fn is_memory_summary_category(category: &str) -> bool { category, "top_of_mind" | "background" - | "stale" | "superseded" + | "stale" + | "superseded" | "tombstone" | "derived_project_profile" ) @@ -107,7 +108,8 @@ pub(super) fn is_memory_summary_freshness_status(status: &str) -> bool { "current" | "background" | "historical" - | "stale" | "superseded" + | "stale" + | "superseded" | "tombstoned" | "unsupported" ) diff --git a/apps/elf-eval/tests/real_world_job_benchmark/external_adapters/first_generation.rs b/apps/elf-eval/tests/real_world_job_benchmark/external_adapters/first_generation.rs index 53582147..50667823 100644 --- a/apps/elf-eval/tests/real_world_job_benchmark/external_adapters/first_generation.rs +++ b/apps/elf-eval/tests/real_world_job_benchmark/external_adapters/first_generation.rs @@ -110,7 +110,8 @@ pub(super) fn assert_memsearch_first_generation_records(memsearch: &Value) { |evidence| evidence.contains("fixture-backed retrieval-debug prompt coverage") && evidence.contains( "No live memsearch runtime adapter executes retrieval prompt scoring yet" - ) && evidence.contains("not a suite pass") + ) + && evidence.contains("not a suite pass") )); assert_eq!(memsearch.pointer("/scenarios/1/status").and_then(Value::as_str), Some("pass")); assert_eq!( diff --git a/docker/benchmark/openviking.Dockerfile b/docker/benchmark/openviking.Dockerfile index 4bf21593..562297ac 100644 --- a/docker/benchmark/openviking.Dockerfile +++ b/docker/benchmark/openviking.Dockerfile @@ -1,4 +1,4 @@ -FROM rust:1.91.1-trixie AS rust-toolchain +FROM rust:trixie AS rust-toolchain FROM ghcr.io/astral-sh/uv:python3.13-trixie-slim AS builder diff --git a/docs/runbook/agent-setup.md b/docs/runbook/agent-setup.md index d803e754..20c4dfee 100644 --- a/docs/runbook/agent-setup.md +++ b/docs/runbook/agent-setup.md @@ -124,7 +124,8 @@ Then set `search.expansion.mode = "off"` to avoid LLM-backed query expansion. Th The machine must have: -- Rust toolchain (pinned by `rust-toolchain.toml`). +- The `stable` Rust toolchain selected by `rust-toolchain.toml`. +- The unversioned `nightly` toolchain with Rustfmt for formatting. - Docker Compose for the checked-in local dependency stack, or separately running Postgres and Qdrant. - `psql` available on PATH. - Running Postgres instance with `pgvector` installed/enabled when not using Compose. diff --git a/packages/elf-service/src/context_pack/assembly.rs b/packages/elf-service/src/context_pack/assembly.rs index 180cbd5a..bf3fc5c8 100644 --- a/packages/elf-service/src/context_pack/assembly.rs +++ b/packages/elf-service/src/context_pack/assembly.rs @@ -252,7 +252,8 @@ fn stale_or_non_current(freshness_state: &str) -> bool { "deleted" | "deprecated" | "expired" - | "stale" | "superseded" + | "stale" + | "superseded" | "tombstoned" | "historical" | "future"