From d0cb640860d7da5a39792a385e7f499610c0536f Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Fri, 14 Aug 2026 16:41:35 -0700 Subject: [PATCH 01/52] Add PLAN-123 citation-claim evidence contract and first CT-001 packet Land the citation-driven simulation-trust foundation on main: - PLAN-123 plan, durable contact-trust design, citation-claim corpus with the 2026-08-14 current-state audit, and the dev-task handoff. - Machine-checked claims-manifest.json (20 rows, corpus-consistent) and the fail-closed check-citation-evidence gate wired into check-lint: missing target commit, scene digest, requested/resolved solver identity, command, ensemble, disposition, claim boundary, or typed-unsupported metrics fail validation, and a permanent intentionally incomplete negative-control packet must keep failing. - First complete CT-001 rolling-direction packet: deterministic sphere slide-to-roll launch-angle sweep under SEQUENTIAL_IMPULSE and BOXED_LCP with per-method requested/resolved readback; friction-pyramid lateral drift up to 2.1e-3 m with the 0/45/90-degree symmetry nulls, so the bounded claim is disposition reproduced within its recorded boundary. - 53 pytest cases proving field-by-field fail-closed behavior, registered in test-ai-infra. --- CHANGELOG.md | 9 + docs/design/README.md | 1 + .../design/contact_trust_and_observability.md | 304 ++++++++ .../README.md | 158 ++++ .../RESUME.md | 81 +++ .../decisions.md | 154 ++++ .../verification.md | 95 +++ .../123-citation-driven-simulation-trust.md | 227 ++++++ .../citation-claim-corpus.md | 193 +++++ .../claims-manifest.json | 426 +++++++++++ .../CT-001-dart7-rolling-direction.json | 684 ++++++++++++++++++ ...CT-002-dart7-intentionally-incomplete.json | 57 ++ docs/plans/README.md | 2 + docs/plans/dashboard.md | 23 + pixi.toml | 7 + scripts/check_citation_evidence.py | 645 +++++++++++++++++ ...citation_ct001_rolling_direction_packet.py | 516 +++++++++++++ tests/test_check_citation_evidence.py | 420 +++++++++++ 18 files changed, 4002 insertions(+) create mode 100644 docs/design/contact_trust_and_observability.md create mode 100644 docs/dev_tasks/citation_driven_simulation_trust/README.md create mode 100644 docs/dev_tasks/citation_driven_simulation_trust/RESUME.md create mode 100644 docs/dev_tasks/citation_driven_simulation_trust/decisions.md create mode 100644 docs/dev_tasks/citation_driven_simulation_trust/verification.md create mode 100644 docs/plans/123-citation-driven-simulation-trust.md create mode 100644 docs/plans/123-citation-driven-simulation-trust/citation-claim-corpus.md create mode 100644 docs/plans/123-citation-driven-simulation-trust/claims-manifest.json create mode 100644 docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json create mode 100644 docs/plans/123-citation-driven-simulation-trust/evidence/negative-controls/CT-002-dart7-intentionally-incomplete.json create mode 100644 scripts/check_citation_evidence.py create mode 100644 scripts/write_citation_ct001_rolling_direction_packet.py create mode 100644 tests/test_check_citation_evidence.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 2bf9a797078d7..b18611d8487eb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -607,6 +607,15 @@ compatibility remains on the active DART 6 LTS branch._ #### Tests, Benchmarks, and Quality Gates +- Added the PLAN-123 citation-claim evidence contract: a machine-checked claims + manifest for external/historical DART claims, a fail-closed + `pixi run check-citation-evidence` gate (missing commit, scene digest, + requested/resolved solver identity, command, ensemble, disposition, claim + boundary, or typed-unsupported metrics fail validation, and a permanent + intentionally incomplete negative-control packet must keep failing), and the + first branch-qualified CT-001 rolling-direction evidence packet showing the + friction-pyramid direction dependence on `main` under both Sequential + Impulse and boxed-LCP contact solvers. - Reorganized tests and CI coverage around DART 7 components, with focused unit, integration, benchmark, rendering, CUDA-smoke, collision, and simulation gates replacing broad stale test targets. diff --git a/docs/design/README.md b/docs/design/README.md index ed0cd08e961df..bcfdb36ce4cd6 100644 --- a/docs/design/README.md +++ b/docs/design/README.md @@ -27,6 +27,7 @@ rule itself. | [`ai_spec_kit_assessment.md`](ai_spec_kit_assessment.md) | Assessment of GitHub Spec Kit for DART AI workflows, including the decision to adapt its lifecycle ideas without installing it as the default DART workflow | | [`batched_world_device_residency.md`](batched_world_device_residency.md) | Contract for homogeneous batched `World` semantics, SoA Model/State blocks, and internal device-residency boundaries (PLAN-091 WP-091.33) | | [`compute_backend_research.md`](compute_backend_research.md) | Evidence survey and DART workload-candidate ranking behind the scalable-compute roadmap | +| [`contact_trust_and_observability.md`](contact_trust_and_observability.md) | Durable PLAN-123 contract: citation-claim evidence rows, contact observation/identity and impulse/wrench semantics, requested-vs-resolved solver reporting, and exact-cone promotion gates | | [`cpp23_modernization.md`](cpp23_modernization.md) | C++23 adoption: compiler floor, portability gate (adopt-now/guarded/deferred), phased rollout, and per-phase execution status | | [`dartsim_gui_simulator.md`](dartsim_gui_simulator.md) | Headless editor engine + thin ImGui/Filament GUI architecture for the dartsim simulator | | [`dartsim_gui_toolkit_decisions.md`](dartsim_gui_toolkit_decisions.md) | Toolkit (ImGui vs Qt), app-shell language (C++ vs Python), Filament headless, and deploy decisions for the dartsim application | diff --git a/docs/design/contact_trust_and_observability.md b/docs/design/contact_trust_and_observability.md new file mode 100644 index 0000000000000..a5b1a45aa49be --- /dev/null +++ b/docs/design/contact_trust_and_observability.md @@ -0,0 +1,304 @@ +# Contact Trust, Citation Reproduction, And Solver Observability + +## Status + +Durable DART 7 design and evidence contract. Mutable priority and execution +state live in `docs/plans/123-citation-driven-simulation-trust.md` and +`docs/dev_tasks/citation_driven_simulation_trust/`. + +## Purpose + +DART should be an auditable multi-solver research platform: a paper, benchmark, +application, or user report that makes a claim about DART must be convertible +into a reproducible scene, a branch/version-qualified result, and a permanent +regression or documented limitation. A benchmark must prove which method +actually ran, what physical quantity was measured, whether the solve converged, +and which fallback or substitution occurred. + +The goal is not one universally best solver. The goal is a trustworthy +speed-accuracy-capability surface across sequential impulse, boxed LCP, +exact-cone contact, rigid/barrier contact, VBD/AVBD, differentiable paths, and +future methods while reusing one model, contact, state, diagnostics, and +evidence infrastructure. + +## Existing DART 7 seams to extend + +Do not create parallel replacements for existing DART-owned surfaces: + +- `dart/simulation/body/contact.hpp` owns query contact data. +- `dart/simulation/compute/rigid_body_constraint.*` and + `unified_constraint.*` own current rigid/articulated contact problem assembly. +- `dart/math/lcp/lcp_types.hpp` owns solver status, residual, + complementarity, options, and problem/result vocabulary. +- `dart/simulation/compute/world_step_profile.hpp` owns + `WorldStepProfile`, `ResolvedSolverConfiguration`, and `StepMetrics`. +- `docs/design/simulation_solver_architecture.md` owns solver-family, + model/state/control/contact, coupler, and public-boundary rules. +- PLAN-080, PLAN-082, PLAN-083, PLAN-104, PLAN-110, and PLAN-122 own their + existing solver, differentiability, and allocation work. + +This design promotes shared contracts only where a second consumer proves the +need. Solver-specific mathematical state may remain local behind adapters. + +## Branch boundary + +### DART 7 `main` + +DART 7 may add clean-break internal architecture and carefully promoted public +value types. It may add a solver-neutral contact-observation layer, exact-cone +problem family, contact-aware inverse dynamics, one-model/many-state execution, +and optional companion integrations. Public names remain DART-owned and +backend-neutral. ECS storage, solver registries, device handles, and reference +project names stay private. + +### DART 6 `release-6.20` + +DART 6 is evidence and maintenance, not an architecture backport. It may add +tests, benchmarks, demos, internal/opt-in diagnostics, and compatibility-safe +fixes to confirmed defects. It must preserve C++17, pybind11, installed +headers/components, ABI-sensitive layouts, default FCL behavior, OSG, and +Gazebo/gz-physics compatibility. See +`docs/design/dart6_citation_driven_contact_trust.md` on that branch. + +A defect shared by both lines receives two independently reviewed PRs. DART 7 +code is evidence for DART 6, never a mechanical backport justification. + +## Citation claim contract + +Every external or historical claim is a row with a stable identifier. Prose +alone cannot close a row. The row records: + +- exact source and quoted/paraphrased claim; +- source publication date and DART version/commit when discoverable; +- target branch and current commit; +- scene/model source, license, conversion notes, and content digest; +- requested and resolved detector, solver family, integrator, precision, + timestep, substeps, iterations/tolerances, threads/backend, and fallback; +- initial state, control, seed, perturbation ensemble, and measurement window; +- physical metrics, numerical metrics, timing methodology, and allocations; +- baseline/current commands and raw machine-readable evidence; +- disposition: `missing`, `reproduced`, `fixed`, `not-applicable`, + `invalid-original-setup`, `version-specific`, or `unresolved`; +- claim boundary, limitations, and independent review evidence. + +A row closes only when another person or clean-session agent can reproduce the +disposition from tracked commands and source-bound artifacts. A visually +plausible result never replaces a text/numeric oracle. + +## Canonical contact lifecycle + +```text +collision candidates + -> canonical ordered contact observations + -> persistent manifold/contact identities + -> solver-family problem adapter + -> solve + -> typed solution and convergence report + -> state update + -> evidence/diagnostics snapshot +``` + +### Contact observation + +A solver-neutral observation should eventually carry, where available: + +- canonical ordered body and shape identities; +- shape-local feature identities and local/world witness points; +- normal orientation with a documented A-to-B convention; +- penetration or signed separation; +- stable manifold and point identities across small motions; +- deterministic tangent-frame seed/history; +- material inputs before combination; +- source detector and contact-generation policy. + +Stable identity is required for warm starts, friction history, deterministic +ordering, gradients, replay, and fixed-capacity batches. Identity must be +invalidated explicitly when topology, shape, feature, or policy changes. + +### Solver problem adapters + +The observation layer does not force all methods into one equation. Adapters +may produce: + +- velocity-level pyramidal boxed LCP; +- exact second-order-cone/NCP contact; +- barrier/variational contact; +- AVBD/VBD row systems; +- differentiable snapshots. + +Shared geometry, materials, identity, and evidence stay common; formulation, +iteration state, and factorization remain solver-family-owned. + +## Impulse, force, and wrench semantics + +Impulse-based contact methods authoritatively produce an impulse over an +integration interval. `impulse / dt` is an average force over that named +interval, not an instantaneous continuous force. + +DART should distinguish: + +- raw contact impulse; +- average contact wrench with interval and origin; +- instantaneous force only for formulations that solve a continuous/compliant + force; +- explicitly filtered/derived analysis signals; +- event-integrated impulse across an impact or gait phase. + +Each result states frame, application point or wrench origin, interval, +manifold/contact identity, requested/resolved method, cone approximation, +convergence, and whether the value is raw or derived. Core simulation never +silently smooths or relabels impulses as instantaneous forces. + +## Solver resolution and solve report + +No benchmark, demo, recording, or published packet may identify a method only +from the requested option. Extend `ResolvedSolverConfiguration` rather than +inventing a second resolution system. + +A comparable solve report should include, where meaningful: + +- success/failure/maximum-iteration/numerical/unsupported status; +- termination reason and fallback/substitution; +- iteration and factorization counts; +- primal, dual, residual, complementarity, and cone violations; +- maximum penetration/minimum separation; +- active, separating, sticking, and sliding counts; +- warm-start use, rejection, and reset reason; +- regularization and physical-compliance values kept distinct; +- per-stage time and post-bake allocation counts. + +Unsupported metrics are explicitly absent, not silently zero. Solver-specific +details may be nested, but cross-family fields retain common units and meaning. + +## Evidence packet and Pareto comparison + +A tracked evidence packet binds: + +1. source claim and corpus row; +2. branch, commit, build configuration, dependency resolution, and machine; +3. scene/model digest and exact command; +4. requested/resolved configuration; +5. raw trajectory/metric output; +6. semantic visual evidence when the claim is visible; +7. statistical method, exclusions, and environmental validity checks; +8. result, limitation, and review record. + +Performance comparisons use interleaved same-host runs or an equally justified +method, report distributions rather than one run, and compare at matched +accuracy/residual whenever methods solve different formulations. Changed +contact counts, sleeping outcomes, state hashes, or failure policies are +re-baselines, not speedups. + +## Required initial corpus + +The initial sidecar is +`docs/plans/123-citation-driven-simulation-trust/citation-claim-corpus.md`. +It includes: + +- historical SimBenchmark rolling, dense-contact, energy/momentum, and + articulated-control claims; +- heel-strike/toe-off force artifacts reported by exoskeleton work; +- exact-cone/high-conditioning motivation from compliant-to-rigid contact work; +- Nimble differentiability and performance claims; +- RobotDART reset/concurrency/research-workflow needs; +- PEEL and Fibration Trees planning/collision-query workloads; +- Behavior Policy Learning reset, controller, perception, and randomization + needs; +- MASS/FARMS biomechanics and experiment-data needs; +- Gazebo Plants codimensional rod/plant gap; +- relevant DART issues and existing DART 6/7 benchmark packets. + +The corpus is expanded only with a bounded acceptance row, not an open-ended +request to “support more papers.” + +## Exact-cone and inverse-dynamics direction + +An exact-cone initiative begins only after the common corpus and observability +foundation can measure its value. The missing unit is a DART-owned cone-contact +problem, not merely another class named ADMM. + +The problem should separate physical compliance from numerical proximal +regularization and support at least one reference algorithm before public +promotion. Proximal ADMM and primal-dual interior point are candidate +algorithms over a shared problem; they are not separate public engines. + +Promotion requires: + +- materially lower directional friction error or better conditioning on named + corpus rows; +- deterministic and fail-closed behavior; +- no post-bake allocation for the promoted step shape; +- matched-residual speed/accuracy evidence; +- stable warm starts/manifolds; +- serialization and requested/resolved reporting; +- finite-difference or implicit-gradient evidence before a differentiable + claim. + +Contact-aware inverse dynamics should reuse the same contact/material problem +and return torques, contact wrenches, feasibility, active constraints, and +residual explanations as a pure query. It must not mutate the primary World. + +## One-model/many-state and companion layers + +A baked immutable model with independent State, Control, Contacts, RNG, solver +history, and scratch per lane is the shared prerequisite for high-throughput +planning, RL, and differentiable rollouts. Reset semantics must explicitly +choose whether solver/contact history is preserved. + +Planning adapters, environment/sensor APIs, and biomechanics should be +maintained companions unless a second core consumer proves a smaller shared +primitive. Gym, Torch, JAX, OMPL, dataset formats, neural controllers, muscles, +and clinical interpretation do not become core dependencies. + +Rods and shells belong to existing deformable/contact solver families and reuse +shared distance, CCD, material, coupling, diagnostics, and evidence contracts. +They are sequenced after the trust foundation, not parallel feature expansion. + +## Determinism, allocation, and failure policy + +- Canonical ordering and identities are stable for the same model/state/policy. +- Baked repeated same-capacity steps do not grow World allocators or allocate + through global heap/raw malloc paths. +- A solver failure cannot partially advance a World without an explicit, + documented continuation policy. +- Fallback is recorded per island/group and cumulative across a step. +- Strict research modes fail closed; compatibility/realtime modes may use + explicit configured fallback. +- Cloning, serialization, and reset preserve or intentionally clear all + result-affecting configuration and history. +- Unsupported capability is an error or recorded substitution, never silence. + +## Public API boundary + +The common path remains simple: configure capabilities and step/query the +World. Advanced users receive DART-owned plain value objects and options. +Public APIs do not expose: + +- solver registries or polymorphic implementation types; +- ECS components/storage; +- CUDA/SYCL streams or device pointers; +- reference repository/project names; +- factorization or reverse-pass cache ownership; +- third-party tensor/framework types in C++ core. + +## Verification and promotion + +Every behavioral slice requires: + +- a negative control that failed before the fix or proves the metric is + non-vacuous; +- focused unit/integration tests; +- deterministic repeated and perturbation-ensemble evidence where thresholds + or chaotic outcomes are claimed; +- baseline/current raw packets; +- text-first physics oracle plus claim-tied visual review when visible; +- at least two clean independent or role-separated reviews on the post-fix + state; +- branch-required lint/build/test/downstream gates; +- an explicit changelog decision. + +The plan closes when the citation corpus, contact semantics, and evidence +packet schema have durable code/docs owners; DART 6 dispositions are promoted +to branch design/testing owners; and any remaining exact-cone, adoption, or +domain-extension work has a bounded durable plan rather than an eternal task +folder. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/README.md b/docs/dev_tasks/citation_driven_simulation_trust/README.md new file mode 100644 index 0000000000000..f5fa21be3070d --- /dev/null +++ b/docs/dev_tasks/citation_driven_simulation_trust/README.md @@ -0,0 +1,158 @@ +# Citation-Driven Simulation Trust — Dev Task + +## Current status + +- [x] Phase 0: Reconcile these bootstrap docs with current `main` and + `release-6.20` (2026-08-14 audit recorded in the corpus sidecar; + PLAN-123/PLAN-623 IDs confirmed free; PLAN-091 heritage and open-PR + owners mapped). +- [x] Phase 1: Land the checked citation-claim/evidence manifest and first + negative-control fixtures (`claims-manifest.json`, + `scripts/check_citation_evidence.py`, `pixi run check-citation-evidence` + in `check-lint`, permanent negative control, CT-001 packet). +- [ ] Phase 2: Reproduce and classify the six-family first-wave corpus + (CT-001 `main` packet landed with disposition `reproduced`; remaining + families and all DART 6 lanes open). +- [ ] Phase 3: Land contact identity and impulse/wrench semantics. +- [ ] Phase 4: Extend resolved-method and cross-family solve diagnostics. +- [ ] Phase 5: Record exact-cone GO/NO-GO; implement only after GO. +- [ ] Phase 6: Implement contact-aware inverse dynamics after a stable shared + contact problem. +- [ ] Phase 7: Derive bounded adoption/domain follow-ups and close this task. + +## Goal + +Implement PLAN-123 so DART can convert material external claims into +branch-qualified, reproducible evidence and compare solver families without +silent substitution or ambiguous contact-force semantics. + +## Specification intake + +- **Value:** scientific credibility, regression prevention, solver selection, + reusable research infrastructure, and evidence-backed prioritization. +- **Scope:** evidence schemas/harnesses, tests/benchmarks/demos, contact + observations and result semantics, solver-resolution reports, branch + coordination, and conditional solver/query work. +- **Assumptions:** existing DART 7 model/contact/metrics/evidence surfaces are + extended rather than replaced; exact-cone work is gated by evidence. +- **Traceability:** PLAN-123, its corpus sidecar, the durable design doc, the + DART citation audit, existing solver plans, DART issues/PRs, and primary + papers. +- **Acceptance evidence:** source-bound packets, negative controls, deterministic + or ensemble oracles, matched comparisons, visual review where applicable, + branch gates, and independent post-fix reviews. + +## Required reading + +- `AGENTS.md` +- `docs/ai/principles.md` +- `docs/ai/north-star.md` +- `docs/ai/verification.md` +- `docs/design/contact_trust_and_observability.md` +- `docs/design/simulation_solver_architecture.md` +- `docs/plans/123-citation-driven-simulation-trust.md` +- `docs/plans/123-citation-driven-simulation-trust/citation-claim-corpus.md` +- `docs/plans/solver-family-intake.md` +- current owner plans named by PLAN-123 +- this folder, especially `RESUME.md` + +## Branch/worktree discipline + +Use separate current worktrees and topic branches for: + +- DART 7: latest `origin/main`; +- DART 6: latest `origin/release-6.20`, governed by its branch-local design and + `docs/dev_tasks/dart6_citation_contact_trust/`. + +Do not mix branch changes in one PR, rebase published PR branches, overwrite +uncommitted work, or mechanically port DART 7 architecture to DART 6. Shared +bugs get independently adapted fixes and evidence. + +Before implementation, inspect open PRs, issues, active dev tasks, dashboards, +and current code. Mark already-landed work in the corpus; do not recreate it. + +## Deliverables + +1. Checked claim/evidence manifest and validator. +2. Six capped first-wave fixture families with branch-qualified packets. +3. Canonical contact identity/ordering and tested normal/tangent semantics. +4. Explicit impulse/average-wrench/continuous-force/filtered/event semantics. +5. Requested/resolved method plus comparable solve/fallback diagnostics. +6. Cross-family matched speed-accuracy dashboard or packets. +7. Exact-cone GO/NO-GO record; implementation only after GO. +8. Contact-aware inverse-dynamics query after a stable shared problem. +9. Bounded follow-up plans/companions for planning, environments/sensors, + differentiability, biomechanics, and rods/shells where evidence supports + them. +10. Durable docs/changelog/user examples and task-folder cleanup at completion. + +## Non-goals + +- One universal solver ranking. +- Reimplementing every citing paper. +- New DART 6 public solver/model architecture. +- Silent smoothing of raw contact outputs. +- Public solver registry, ECS/device storage, reference-project, or tensor + framework types. +- Open-ended fixture or paper expansion. + +## Work-package rules + +- Each PR closes a bounded set of corpus rows or one shared contract. +- A feature PR includes the fixture/evidence that justified it. +- No exact-cone public option before the common harness and GO record. +- No companion or domain extension before a named owner and stopping condition. +- Changed physical outcomes are re-baselines, not performance wins. +- Every threshold/robustness claim uses an appropriate perturbation ensemble. +- Strict modes fail closed; continuation/fallback modes are explicit and typed. + +## Acceptance evidence + +- At least one pre-fix or negative-control failure for each behavioral slice. +- Machine-readable packets bound to commit, scene digest, command, and resolved + configuration. +- Text/numeric physics oracle; semantic visual inspection when visible. +- Stable repeated/ensemble results and exact claim boundaries. +- Same-host interleaved or justified performance methodology. +- Post-bake allocation gates for promoted repeated paths. +- Two clean independent or role-separated reviews on the current post-fix + state. +- Changelog decision and branch-required tests. + +## Gates + +Always use current branch-owned tasks; do not invent aliases. + +Typical DART 7 gates: + +- `pixi run lint` +- `pixi run check-docs-policy` +- `pixi run build` +- focused C++/Python/simulation tests +- benchmark/evidence validators +- `pixi run check-api-boundaries` +- `pixi run test-all` +- `pixi run -e cuda test-all` on a visible CUDA host when affected + +DART 6 gates are owned by its task and include `pixi run -e gazebo test-gz` +for downstream-sensitive changes. + +## Open decisions + +- Exact-cone CPU algorithm and problem boundary: decide only after WS1–WS4 + evidence. +- Whether a second cone algorithm is warranted: require a distinct validation + purpose. +- Public contact result surface: promote only fields with multiple stable + consumers. +- Planning/environment/biomechanics/rod ownership: companion or existing plan, + not assumed core. + +## Immediate next steps + +1. Read `RESUME.md` and verify both branch tips/worktrees. +2. Audit current code/docs/PRs against every initial corpus row. +3. Integrate PLAN-123/dashboard/index without clobbering newer state. +4. Implement the smallest checked manifest plus one intentionally incomplete + fixture that proves the validator fails closed. +5. Select the first three corpus rows using reuse and risk, not novelty. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md new file mode 100644 index 0000000000000..2ec52f4f7521a --- /dev/null +++ b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md @@ -0,0 +1,81 @@ +# Resume: Citation-Driven Simulation Trust + +## Current reality + +Reconciled against a live checkout on 2026-08-14. The bootstrap-package +statuses were verified; the corpus sidecar now carries the WS0 audit record. + +## Last session summary + +- WS0 audit done against `main` 20501341226 and `release-6.20` 39ccd52068b: + plan IDs free (PLAN-123/PLAN-623), PLAN-091 heritage mapped, open PRs + (#3432 PLAN-104, #3377 FBF friction, #3431 soft-foot ensembles, #3428 + WP-PG.50) and issue #3056 routed as row owners, existing demos scenes + recorded as fixture seeds. +- PLAN-123 integrated into `docs/plans/dashboard.md` and `docs/plans/README.md`. +- WS1 landed on the DART 7 branch: `claims-manifest.json` (20 rows), + fail-closed `scripts/check_citation_evidence.py` + (`pixi run check-citation-evidence`, wired into `check-lint`), + `tests/test_check_citation_evidence.py`, a permanent intentionally + incomplete negative-control packet, and the first complete CT-001 + rolling-direction packet (disposition `reproduced`: friction-pyramid + lateral drift up to 2.1 mm with zero drift at 0/45/90 deg, identical-run + determinism, per-method requested/resolved readback). + +## Current branches + +- DART 7: `feature/citation-trust-foundation` in + `.claude/worktrees/citation-trust-main`, based on `origin/main` 20501341226. +- DART 6: `feature/dart6-citation-contact-trust` in + `.claude/worktrees/citation-trust-620`, based on `origin/release-6.20` + 39ccd52068b (docs applied; branch-local adoption in progress). + +Both are local-only; nothing pushed. GitHub mutations need maintainer +approval. + +## Immediate next step + +Finish the DART 6 Phase 1 adoption on `release-6.20`: branch-local +`check_citation_evidence.py` + manifest + negative control (no DART 7 API +backports), then run branch gates on both worktrees and record reviews. + +After that, Phase 2 continues with the next first-wave family on `main` +(dense inelastic/elastic contact, CT-002/CT-003) reusing +`rigid_restitution_ladder` and a 6x6x6 grid fixture. + +## Context that would be lost + +- `ResolvedSolverConfiguration` is C++-only (`World::getResolvedConfiguration`); + packets record resolved identity from World property readback plus step + profile stage names, and the Python exposure is the first WS4 slice. +- The CT-001 packet must not be quoted as a solver comparison: SI and boxed + LCP metric summaries coincide to printed precision on the single-contact + scene while trajectory hashes differ per solver. +- `check-citation-evidence` validates committed packets structurally; + `--freshness` (packet commit == HEAD) is a packet-writing aid, not CI, + because squash merges retire topic commits. +- Negative-control packets live under `evidence/negative-controls/` and must + keep failing validation with >= 3 errors; the gate rejects a passing + negative control as vacuous. +- dartpy runs from a built tree via + `PYTHONPATH=build/default/cpp/Release/python pixi run python ...`. + +## How to resume + +```bash +git worktree list +cd .claude/worktrees/citation-trust-main && git status && git log -3 --oneline +cd ../citation-trust-620 && git status && git log -3 --oneline +pixi run check-citation-evidence # in citation-trust-main +``` + +Then continue with the DART 6 adoption or the next first-wave family per the +README status. + +## Verification before ending the next session + +- Run `pixi run lint` for any repository edit. +- Run `pixi run check-citation-evidence`, `pixi run check-docs-policy`, and + `pixi run test-ai-infra` for evidence/docs changes. +- Record commands/results in `verification.md` and decisions in + `decisions.md`. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/decisions.md b/docs/dev_tasks/citation_driven_simulation_trust/decisions.md new file mode 100644 index 0000000000000..c7864f84b5d6d --- /dev/null +++ b/docs/dev_tasks/citation_driven_simulation_trust/decisions.md @@ -0,0 +1,154 @@ +# Decisions: Citation-Driven Simulation Trust + +## 2026-08-14 — Machine manifest and packets are JSON + +**Decision:** `claims-manifest.json` and evidence packets use JSON with schema +tags (`dart.citation_claim_manifest/v1`, `dart.citation_claim_evidence/v1`). +The corpus markdown stays the human authority for claim text/oracles; the +manifest owns live per-lane status/dispositions; the validator enforces exact +claim-ID agreement between the two. + +**Why:** Every existing evidence validator in `scripts/` is JSON with no YAML +dependency anywhere in the tooling environment; a second format would split +the evidence system. + +**Revisit when:** A packet needs human-authored long-form fields that JSON +makes error-prone. + +## 2026-08-14 — Negative controls are committed packets that must fail + +**Decision:** Intentionally incomplete packets live permanently under +`evidence/negative-controls/`; the gate requires each to produce >= 3 +validation errors and fails if one starts passing. Packets do not carry a +self-describing negative-control flag the validator could special-case. + +**Why:** A fail-closed claim needs a permanent executable counterexample; a +flag the validator reads would let the special case rot into a bypass. + +**Revisit when:** A schema change legitimately shrinks a control below three +errors; then strengthen the control, not the threshold. + +## 2026-08-14 — Structural validation in CI, freshness on demand + +**Decision:** `check-citation-evidence` (in `check-lint`) validates structure, +consistency, and cross-links of committed packets. `--freshness` (packet +commit must equal current HEAD) is used when writing/regenerating packets, +not in CI. + +**Why:** Squash merges legitimately retire topic-branch commits, so an +ancestry gate would poison every packet after merge; the commit still binds +the packet to its exact source for reproduction. + +**Revisit when:** Evidence regeneration becomes automated enough to re-stamp +packets on merge. + +## 2026-08-14 — `main` manifest dart6 lanes are routing pointers + +**Decision:** The PLAN-123 manifest on `main` records dart6 lanes as +routing/audit pointers; the `release-6.20` branch manifest +(`docs/design/dart6_citation_driven_contact_trust/claims-manifest.json` on +that branch) owns live DART 6 lane status/dispositions. `main` mirrors +promoted DART 6 results only at explicit sync points. + +**Why:** Branches cannot share one mutable file, and two live owners of the +same lane state would drift; claim identity stays single-owned by the corpus +on `main`. + +**Revisit when:** A release process needs automated cross-branch mirroring. + +## 2026-08-14 — CT-001 disposition thresholds + +**Decision:** The rolling-direction packet calls the claim `reproduced` when +any solver shows max |lateral drift| > 1e-4 m, max |heading error| > 0.1 deg, +or relative travel spread > 1% across the launch-angle sweep; otherwise +`unresolved`. Measured on `main` 20501341226: lateral drift up to 2.1e-3 m +and travel spread 4.8e-3 with the symmetry-consistent sign structure (zero at +0/45/90 deg), under both SEQUENTIAL_IMPULSE and BOXED_LCP. + +**Why:** The bounded claim is rotational-symmetry breaking; the thresholds sit +an order of magnitude above observed numerical noise at the symmetry angles +(heading error ~1e-14 deg) and well below the measured signal. + +**Revisit when:** A wider sweep, other shapes/speeds, or an exact-cone method +needs a shared anisotropy metric; then promote the metric definition to the +durable design doc. + +## 2026-08-14 — One DART 7 durable owner, separate DART 6 adaptation + +**Decision:** DART 7 owns the long-term contact-trust architecture and PLAN-123. +DART 6 owns a branch-local compatibility design and dev task. + +**Why:** DART 7 can evolve clean-break internals and APIs; DART 6 must preserve +ABI, defaults, packages, language/binding/rendering floors, and downstream +behavior. Shared defects still receive separate branch PRs. + +**Revisit when:** DART 6.20 leaves maintenance or a new active LTS branch +changes the compatibility boundary. + +## 2026-08-14 — Claims are stable rows, not prose conclusions + +**Decision:** Every material external claim receives a stable ID, bounded +oracle, branch/version, evidence packet, disposition, and claim boundary. + +**Why:** Citation sentiment and benchmark prose cannot distinguish historical +versions, invalid comparisons, fixed defects, and current limitations. + +**Revisit when:** The evidence schema proves too rigid for a named source type; +extend it without weakening required provenance. + +## 2026-08-14 — Raw impulses remain authoritative + +**Decision:** Impulse-based methods report raw impulse. Average wrench, +continuous force, filtering, and event integration are distinct typed/metadata +semantics. + +**Why:** Labeling `impulse / dt` as instantaneous force creates incorrect +biomechanics/control interpretations and hides timestep dependence. + +**Revisit when:** A shared public result surface is designed from at least two +solver families and two downstream consumers. + +## 2026-08-14 — Extend current DART-owned observability + +**Decision:** Extend `ResolvedSolverConfiguration`, `StepMetrics`, +`WorldStepProfile`, `LcpResult`, and current contact/problem structures before +adding parallel diagnostics. + +**Why:** Method identity and physical metrics already have owners; duplication +would make evidence inconsistent. + +**Revisit when:** A metric has incompatible mathematical meaning across +families; keep the common field absent and add a solver-specific nested report. + +## 2026-08-14 — Exact-cone work is evidence-gated + +**Decision:** Do not begin with a public ADMM/exact-cone solver. First build the +corpus, contact semantics, and matched comparison. Then record a GO/NO-GO. + +**Why:** DART already has multiple algorithms and an ADMM boxed-LCP solver. The +missing value is a true cone-contact problem and a demonstrated Pareto region, +not another algorithm name. + +**Revisit when:** WS1–WS4 evidence identifies named failures that an exact cone +can plausibly solve and defines promotion metrics. + +## 2026-08-14 — Companion layers stay optional + +**Decision:** Planning, Gym/Torch/JAX integration, sensors, datasets, +biomechanics, and application controllers are companion candidates. Core +receives only smaller primitives justified by multiple consumers. + +**Why:** DART should remain a physics/research platform without forcing +third-party application stacks or frameworks into the core dependency graph. + +**Revisit when:** A second core consumer proves a stable smaller abstraction. + +## 2026-08-14 — First-wave corpus is capped + +**Decision:** The first implementation wave has six fixture families. + +**Why:** A bounded set can be completed and reviewed; an open-ended paper +backlog would prevent closeout. + +**Revisit when:** All six have branch-qualified dispositions or the maintainer +explicitly changes the cap. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md new file mode 100644 index 0000000000000..f32d95a0da67c --- /dev/null +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -0,0 +1,95 @@ +# Verification: Citation-Driven Simulation Trust + +## WS0 + WS1 slice — 2026-08-14 + +- Branch: `feature/citation-trust-foundation` from `origin/main` 20501341226 + in `.claude/worktrees/citation-trust-main`; worktree started clean. +- Corpus rows touched: audit section added for all rows; CT-001 `dart7` lane + `in-progress` with the first packet; CT-002 used by the permanent + negative control. +- What changed: PLAN-123 docs integrated (dashboard/README/plan/corpus); + `claims-manifest.json`; `scripts/check_citation_evidence.py`; + `scripts/write_citation_ct001_rolling_direction_packet.py`; + `tests/test_check_citation_evidence.py`; pixi tasks + (`check-citation-evidence` in `check-lint`, test registered in + `test-ai-infra`); CT-001 packet + negative control; CHANGELOG entry. +- Commands and results: + - `pixi run build` — success (Release, dartpy 7.0.0 importable). + - `PYTHONPATH=build/default/cpp/Release/python pixi run python +scripts/write_citation_ct001_rolling_direction_packet.py` — wrote the + packet; deterministic repeats identical; per-method resolved readback + asserted; disposition `reproduced`. + - `pixi run check-citation-evidence` — OK (exit 0). + - Fail-closed demo: copying the negative control into `evidence/` in a + scratch tree produced 15 errors and exit 1. + - `pixi run python -I scripts/run_pytest.py +tests/test_check_citation_evidence.py -q` — 53 passed. + - `--freshness` run — exit 0 (packet commit == HEAD). +- Negative control: permanent `evidence/negative-controls/` packet (missing + resolved identity/digest/commands, single-run ensemble, unsupported-as-zero + metric, no claim boundary) fails with >= 3 errors and the gate enforces + that it keeps failing. +- Determinism evidence: per-cell trajectory SHA-256 equal across 2 repeats + for all 28 cells; cross-invocation reruns reproduced identical summary + metrics. +- Performance/allocation: explicitly typed unsupported in the packet (no + timing/allocation claims made). +- Visual evidence: typed not-applicable (numeric rotational-symmetry oracle); + no visible-behavior claim in this packet. +- Review passes: recorded below once completed for the slice head. +- Known gaps: DART 6 adoption slice pending; remaining first-wave families + pending; `ResolvedSolverConfiguration` not yet Python-exposed. +- Changelog: entry added under "Tests, Benchmarks, and Quality Gates". + +## Bootstrap record — 2026-08-14 + +### What changed + +Documentation package only: + +- durable DART 7 contact-trust design; +- PLAN-123 and citation corpus; +- DART 7 and DART 6 dev-task contracts; +- merge-safe dashboard/index snippets; +- Codex/Claude goal prompt. + +### What is not claimed + +- No repository checkout was modified. +- No branch, commit, PR, issue, or GitHub state was created or changed. +- No current-head build, test, benchmark, visual, or solver result was run. +- Corpus dispositions remain audit-required. +- Plan IDs and dashboard ordering require current checkout verification. + +### Required verification after integration + +At minimum: + +```bash +pixi run lint +pixi run check-lint-md +pixi run check-lint-spell +pixi run check-docs-policy +git diff --check +``` + +Use current branch-owned task names if they differ. + +### Required record for each implementation slice + +Add a dated section with: + +- branch, base, head, and whether the worktree was clean; +- corpus rows and claim boundaries changed; +- source/model/evidence digests; +- code/docs/tests changed at a high level; +- exact commands and results; +- negative control; +- determinism/ensemble evidence; +- performance host validity and raw packet path when applicable; +- visual capture and semantic review when applicable; +- independent review passes; +- known gaps and immediate next step; +- changelog decision. + +Never replace raw evidence with a summary number. diff --git a/docs/plans/123-citation-driven-simulation-trust.md b/docs/plans/123-citation-driven-simulation-trust.md new file mode 100644 index 0000000000000..21ed8e30a1194 --- /dev/null +++ b/docs/plans/123-citation-driven-simulation-trust.md @@ -0,0 +1,227 @@ +# PLAN-123: Citation-Driven Simulation Trust + +- Operating state: `PLAN-123` in [`dashboard.md`](dashboard.md) +- Outcome: every material external or historical claim about DART can be + reproduced, classified, and guarded on the applicable branch; every DART 7 + solver benchmark proves the requested and resolved method, contact/force + semantics, convergence, failure policy, and matched speed-accuracy evidence; + the resulting evidence determines whether exact-cone contact, contact-aware + inverse dynamics, and adoption/domain extensions are promoted. +- Current evidence: + - DART 7 already provides `ResolvedSolverConfiguration`, `StepMetrics`, + `WorldStepProfile`, structured `LcpResult`, rigid/unified contact problems, + multiple solver families, differentiability, and post-bake allocation gates. + - DART 6.20 has active collision/performance and deformable paper-parity work, + a stable backend compatibility contract, and Gazebo/gz-physics downstream + gates. + - The DART citation audit found mostly neutral infrastructure use, clear + positive extension evidence from Nimble/RobotDART, and bounded criticisms + involving contact-force transients, friction/contact formulation, + rigid-only scope, and test-specific failures. + - Current plans own individual methods but no owner currently converts the + cross-paper citation record into one branch-qualified claim corpus and one + solver-neutral trust contract. + +## Owner docs + +- Durable architecture and semantics: + [`../design/contact_trust_and_observability.md`](../design/contact_trust_and_observability.md) +- Citation and claim matrix: + [`123-citation-driven-simulation-trust/citation-claim-corpus.md`](123-citation-driven-simulation-trust/citation-claim-corpus.md) +- Active implementation state: + [`../dev_tasks/citation_driven_simulation_trust/`](../dev_tasks/citation_driven_simulation_trust/) +- Existing solver owners: + PLAN-080, PLAN-082, PLAN-083, PLAN-104, PLAN-110, and PLAN-122. +- Archived heritage: PLAN-091 landed `ResolvedSolverConfiguration` + (WP-091.11 slice 1), `StepMetrics`, `WorldStepProfile`, and + `docs/design/dart7_cross_family_metrics_corpus.json`; PLAN-123 WS4 owns the + recorded silent-substitution follow-up on those seams. +- DART 6 adaptation owner on `release-6.20`: + `docs/design/dart6_citation_driven_contact_trust.md`. + +## Scope + +PLAN-123 is cross-cutting evidence and shared-contract work. It does not absorb +the detailed implementation ownership of existing solver plans. It owns: + +- the stable citation-claim identifiers and dispositions; +- the evidence-packet and matched-comparison contract; +- common contact observation, identity, impulse/wrench semantics, and + cross-family solve reporting; +- promotion decisions that require evidence across solver families; +- branch coordination and stopping conditions. + +Existing plans continue to own their methods, kernels, paper corpora, and public +API slices. A PLAN-123 row links to the solver owner rather than duplicating its +checklist. + +## Dependencies and sequencing + +1. The initial corpus and evidence schema land before new solver architecture. +2. Existing DART 6 and DART 7 scenes/metrics are reused before adding fixtures. +3. Contact semantics and requested/resolved identity land before comparing + methods. +4. Exact-cone work starts only after a recorded GO based on corpus gaps. +5. Contact-aware inverse dynamics follows a stable cone/contact problem. +6. One-model/many-state, planning, environment/sensor, biomechanics, and + rod/shell work proceeds as bounded companion or existing-plan slices after + the common trust foundation. + +## Workstreams + +### WS0 — Current-state and ownership audit + +- Reconcile the corpus with current `main`, `release-6.20`, open PRs, issues, + plans, active dev tasks, and benchmark/evidence infrastructure. +- Mark already-landed work; do not recreate it. +- Route every row to exactly one implementation owner and one durable evidence + owner. +- Verify that PLAN-123 and any DART 6 plan ID do not conflict with newer state. + +Exit: no duplicate owners, stale status, or unbounded “support all papers” +language. + +### WS1 — Claim corpus and machine-readable evidence contract + +- Implement a checked manifest/schema for corpus rows and dispositions. +- Require source/scene/build/configuration/metric/review provenance. +- Add validators for missing resolved-method data, unsupported zero-valued + metrics, stale evidence commits, incomplete ensembles, and unbounded claims. +- Integrate with existing visual/evidence publication rather than inventing a + parallel artifact system. + +Exit: at least three representative rows fail the validator before completion +and pass after complete evidence is supplied. + +### WS2 — Baseline reproduction campaign + +First wave: + +- directional rolling/friction sweep; +- dense inelastic and elastic contact stress; +- articulated energy/momentum and controlled-contact scenes; +- heel-strike/toe-off transition; +- high mass-ratio/ill-conditioned stacks and manipulation; +- reset/concurrency and planning collision-query throughput. + +Each row runs on the relevant DART 6 and DART 7 configurations or records why a +lane is not applicable. Results are branch/version-qualified, never generalized +from one historical version. + +Exit: each first-wave row has a disposition, raw packet, negative control, and +review record. + +### WS3 — Contact observation and physical-result semantics + +- Stabilize canonical body/shape/feature ordering and manifold/point identity. +- Define normal orientation and deterministic tangent-history policy. +- Distinguish raw impulse, interval-average wrench, continuous force, filtered + analysis output, and event-integrated impulse. +- Preserve existing query APIs and add only the smallest value-type/API surface + supported by multiple consumers. +- Add gait/grasping/stack tests for sign, frame, interval, identity, reset, and + clone behavior. + +Exit: a user can determine exactly what a contact value means without reading +the solver implementation. + +### WS4 — Cross-family resolution, diagnostics, and Pareto dashboard + +- Extend existing `ResolvedSolverConfiguration`, `StepMetrics`, + `WorldStepProfile`, and solver results. +- Report fallback/substitution per group/island and cumulatively per step. +- Add comparable residual, complementarity/cone, penetration/separation, + contact-mode, warm-start, iteration, timing, and allocation fields where + mathematically meaningful. +- Publish matched-residual/accuracy comparisons and label re-baselines when + outcomes/contact sets differ. + +Exit: no benchmark or demo can silently claim a requested method that did not +run. + +### WS5 — Exact-cone contact GO/no-GO and implementation + +- Audit existing ADMM, Dojo/IPM, SAP, boxed-LCP, IPC, and AVBD seams. +- Define one DART-owned exact-cone problem separating physical compliance from + numerical regularization. +- Build one CPU reference algorithm and evidence adapters before public API. +- Add a second algorithm only when it tests a real shared contract. +- Promote only if named corpus rows establish a useful Pareto region. + +Exit: recorded GO with promotion evidence, or NO-GO with reusable findings and +no permanent experimental surface. + +### WS6 — Contact-aware inverse dynamics + +- Reuse the stable contact/material problem. +- Provide a pure query for torques, contact wrenches, feasibility, limits, + active constraints, and residual explanations. +- Add whole-body, exoskeleton, and manipulation fixtures. +- Keep mutation, solver internals, and third-party optimization types private. + +Exit: forward/inverse consistency and infeasibility tests pass on named corpus +scenes. + +### WS7 — Execution and adoption layers + +- Finish explicit one-model/many-state capture, restore, partial reset, RNG, + solver-history, and batch semantics. +- Derive bounded planning and environment/sensor companion work. +- Extend differentiable validity diagnostics and batched/checkpointed VJPs + through PLAN-110. +- Start biomechanics or rods/shells only with a named owner, shared-primitive + audit, and bounded corpus. + +Exit: each extension is a separate accepted initiative or companion with no +core dependency leakage. + +## DART 6 coordination + +DART 6 work is implemented on `release-6.20` under its branch-local design and +dev-task owners. PLAN-123 consumes its evidence but does not direct mechanical +backports. DART 6 may reproduce claims and fix confirmed defects while +preserving ABI, defaults, packages, C++17/pybind11/OSG, and gz compatibility. +New solver families, public contact architecture, one-model/many-state APIs, +and domain expansion are DART 7 only. + +## Acceptance criteria + +- Every initial corpus row has a source-bound disposition and reproducible + commands; no row is closed by prose or screenshots alone. +- Requested and resolved detector/solver/integrator/backend/precision are + recorded for every result-affecting packet. +- Contact impulse/wrench/force semantics are explicit and tested. +- Comparable metrics distinguish unsupported from measured zero. +- Failure, fallback, continuation, and partial-state policies are typed and + tested. +- Promoted repeated step shapes satisfy PLAN-122 post-bake allocation gates. +- Performance claims use validated hosts and distributions; solver comparisons + are matched by accuracy/residual or explicitly labeled non-equivalent. +- Behavioral claims use deterministic repeats or perturbation ensembles + appropriate to the system. +- Public APIs remain DART-owned, easy on the common path, serializable when + result-affecting, and backend/storage/framework neutral. +- DART 6 and DART 7 changes remain separate, branch-appropriate PRs. +- Each slice runs branch-required lint/build/tests, visual evidence when + applicable, two clean independent reviews, and a changelog decision. +- Completing dev-task folders are removed in their completing PR after durable + content is promoted. + +## Non-goals + +- Declaring one solver universally superior. +- Porting all citing-paper application stacks into core DART. +- Adding exact-cone, biomechanics, sensors, planning, rods, or shells to DART 6. +- Exposing reference-project names, solver registries, ECS/device storage, or + tensor frameworks through the public C++ core. +- Using citation count or routine library use as evidence of physical accuracy. +- Keeping an unbounded paper-support backlog in this plan. + +## Revision triggers + +- A corpus row shows an existing plan already owns the proposed shared surface. +- Exact-cone evidence produces a GO/NO-GO decision. +- A contact metric proves formulation-specific and cannot retain common meaning. +- DART 6 compatibility or downstream evidence changes the branch boundary. +- PLAN-080/082/083/104/110/122 changes the shared seam. +- A task completes, splits, is parked, or requires a new bounded plan. diff --git a/docs/plans/123-citation-driven-simulation-trust/citation-claim-corpus.md b/docs/plans/123-citation-driven-simulation-trust/citation-claim-corpus.md new file mode 100644 index 0000000000000..edce2fec110d3 --- /dev/null +++ b/docs/plans/123-citation-driven-simulation-trust/citation-claim-corpus.md @@ -0,0 +1,193 @@ +# PLAN-123 Citation And Claim Corpus + +## Purpose + +This sidecar is the stable row-level intake for claims that motivate +PLAN-123. It is not a bibliography, sentiment table, or promise to reproduce +every citation. A row exists only when it has a bounded DART-relevant claim, +scene, metric, and closing condition. + +The machine-readable manifest should be derived from this document once the +current repository evidence system is audited. Until then, this file is the +human authority for row identity and intended claim boundary. + +## Dispositions + +| Disposition | Meaning | +| ------------------------ | ------------------------------------------------------------------------------------------------------------------------------- | +| `missing` | No adequate current-branch evidence exists. | +| `reproduced` | The claimed behavior is observed under a source-faithful or explicitly bounded reconstruction. | +| `fixed` | A negative claim reproduces on the qualified baseline and no longer reproduces on the current target, with regression coverage. | +| `version-specific` | The claim is valid for a named historical version but not generalized to other versions. | +| `not-applicable` | The branch/method cannot represent the claim and the reason is explicit. | +| `invalid-original-setup` | A demonstrated confound or invalid comparison prevents the original conclusion; the corrected experiment is recorded. | +| `unresolved` | Evidence is conflicting, incomplete, or not reproducible. | + +A `fixed` result is not automatically a performance improvement; changed +contact sets, sleeping, state hashes, failure policy, or physical model require +a re-baseline label. + +## Required row fields + +Each machine-readable row must carry: + +- `id`, `title`, `source`, `source_date`, and exact claim text or bounded + paraphrase; +- original DART version/commit and external model/source provenance when known; +- target branch/commit and owner plan/dev task; +- scene/model digest and license disposition; +- requested/resolved detector, contact method, integrator, precision, backend, + timestep/substeps, iterations/tolerances, threads, and fallback policy; +- seed/perturbation ensemble and measurement window; +- physical, numerical, performance, allocation, and determinism metrics; +- baseline/current commands and evidence paths; +- disposition, limitations, claim boundary, and review record. + +## Initial corpus + +Row identity and claim boundaries below are stable. Live per-branch status, +owners, and dispositions are machine-readable in +[`claims-manifest.json`](claims-manifest.json), validated by +`pixi run check-citation-evidence`; this table is not re-edited per status +change. The 2026-08-14 current-state audit (WS0) is recorded after the table. + +| ID | Source / motivation | Bounded DART claim or need | First oracle | DART 6 lane | DART 7 owner | +| ------ | -------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------ | ----------------------------------------------------------------------------- | ------------------------------------------------------------------------ | +| CT-001 | SimBenchmark / historical DART contact comparison | Polyhedral friction can produce direction-dependent rolling/sliding behavior. | Rotate initial tangent direction; stopping distance, lateral drift, energy loss, cone violation. | Reproduce existing DART 6 methods/detectors only. | PLAN-080/123; compare SI, boxed LCP, future exact cone. | +| CT-002 | SimBenchmark dense `6x6x6` inelastic contact | Dense contact may fail, become unstable, or scale poorly for some timestep/solver settings. | Finite state, penetration, residual, energy, iterations, wall time over timestep grid. | Compatibility-safe regression and benchmark. | PLAN-080/122/123. | +| CT-003 | SimBenchmark dense elastic contact | Elastic dense contact may inject energy or expose solver failure. | Restitution outcome, energy envelope, non-finite/failure status. | Existing restitution/contact paths only. | PLAN-080/123; formulation comparison. | +| CT-004 | SimBenchmark articulated momentum/energy | Articulated integration/contact accuracy must be compared at matched cost. | Momentum/energy drift versus timestep and runtime. | Use current Skeleton/World behavior. | PLAN-080/084/123 using `StepMetrics`. | +| CT-005 | SimBenchmark articulated PD control | Controlled robot tracking exposes whole-step speed/accuracy tradeoffs. | Tracking error, constraint error, energy/control work, timing distribution. | License-clean DART 6 reconstruction. | PLAN-080/123. | +| CT-006 | 2026 exoskeleton contact-force criticism | Velocity-level/polyhedral/finite-iteration contact can create heel-strike and toe-off transients that are easy to misinterpret as force. | Raw impulse, interval-average wrench, event-integrated impulse, CoP, mode/cone/residual traces. | Clarify/diagnose current `Contact.force` behavior without ABI/default change. | PLAN-123 contact semantics; method comparison. | +| CT-007 | From Compliant to Rigid Contact Simulation | Exact Coulomb cones and adaptive proximal methods may improve conditioning and remove friction-pyramid anisotropy. | Rolling isotropy, high mass ratio, dense stacks, matched residual/time. | Evidence only; no new solver family. | PLAN-123 exact-cone GO/NO-GO. | +| CT-008 | Same source | Rigid and compliant contact can share a contact problem while keeping physical compliance distinct from numerical regularization. | Compliance-limit sweep and parameter-unit/semantics tests. | Document existing behavior only. | PLAN-123 exact-cone problem design. | +| CT-009 | Same source / robotics need | Contact-aware inverse dynamics should return feasible torques/contact wrenches and explain infeasibility. | Forward/inverse consistency, torque/friction limits, infeasible target. | Not applicable beyond existing APIs. | PLAN-123 WS6. | +| CT-010 | Nimble | Analytic hard-contact derivatives should agree with finite differences within a fixed mode and outperform finite differencing. | FD sweep, active-set margins, VJP runtime/memory. | Not applicable as new architecture. | PLAN-110 plus PLAN-123 validity evidence. | +| CT-011 | RobotDART | Research workflows need fast reset, concurrency, low overhead, and deterministic synchronous stepping around DART. | Restore equivalence, thread/lane isolation, reset cost, allocation. | Existing clone/reset/Recording evidence only. | PLAN-030/122/123 one-model/many-state. | +| CT-012 | PEEL | Long-horizon disassembly needs high-throughput collision checking and reliable execution validation. | Query throughput, determinism, disabled-part updates, continuous validation. | Optional benchmark/adapter evidence. | PLAN-120/123 planning companion intake. | +| CT-013 | Fibration Trees | High-DOF multi-robot planning needs thread-safe state-validity and collision queries. | Independent query contexts, 2–N robot scaling, state mapping correctness. | Optional query benchmark. | PLAN-120/123. | +| CT-014 | Behavior Policy Learning | Multi-stage manipulation needs reset, controller scheduling, observations, randomization, and sim-to-real evidence. | Deterministic episode reset; controller/observation timestamps; randomized replay. | Companion-only evidence. | PLAN-103/123 environment companion intake. | +| CT-015 | MASS / musculoskeletal DART use | Biomechanics needs scalable muscle-actuated models and trustworthy gait/contact outputs. | Muscle/actuator scaling plus gait impulse/wrench semantics. | Existing SoftBody/contact evidence only; no new subsystem. | Future companion after PLAN-123 foundation. | +| CT-016 | FARMS | Reproducible animal/robot experiments need standardized model, sensor, controller, data, and analysis contracts. | Schema round-trip, units, deterministic logging/replay. | Companion-only. | PLAN-103/123 companion intake. | +| CT-017 | Gazebo Plants | Rigid-body engines do not represent flexible plants/codimensional rods naturally. | Rod bending/twist/contact scene and rigid approximation failure. | Not applicable. | Existing deformable families; new bounded intake after trust foundation. | +| CT-018 | DART issue #3056 and DART 6 performance campaign | Large settled worlds must not remain orders of magnitude slower because sleeping/collision paths continue unnecessary work. | Source-faithful 3k-shape and generated scenes, state/rest/contact hashes, RTF distribution. | Audit current DART 6 closure and guard it. | DART 7 scalability evidence where equivalent. | +| CT-019 | Historical contact-normal / wrench issues (#1425, #1073) | Contact normal, object ordering, force sign, frame, and wrench ownership must be unambiguous downstream. | Pair-order swap, static rest, grasp map, Gazebo sensor integration. | Compatibility regression/downstream test. | PLAN-123 observation and wrench contract. | +| CT-020 | Current DART paper-parity work | Single trajectories and saturated/non-monotonic thresholds can create false claims. | Deterministic perturbation ensembles, prefix thresholds, geometry/control comparability. | Reuse PLAN-621/622 evidence patterns. | Apply to all threshold/robustness claims. | + +## Current-state audit (2026-08-14, WS0) + +Reconciled against `main` 20501341226, `release-6.20` 39ccd52068b, open +PRs/issues, and existing plan/dev-task owners. Corrections to the bootstrap +assumptions: + +- PLAN-091 (DART 7 Architecture Hardening) is archived and already landed the + seams this plan extends: `ResolvedSolverConfiguration` (WP-091.11 slice 1; + requested == resolved today, silent-substitution classification is the open + follow-up), `StepMetrics`/`WorldStepProfile`, and the cross-family metrics + corpus at `docs/design/dart7_cross_family_metrics_corpus.json` with + per-row `resolved_solver_identity`. WS4 continues that follow-up under + PLAN-123 rather than a new system. +- `dartpy` already exposes `StepMetrics` (energy, momentum, contact count, max + penetration, iterations, residual) and per-World + `contact_solver_method` (`SEQUENTIAL_IMPULSE`, `BOXED_LCP`) and + `rigid_body_solver` (`SEQUENTIAL_IMPULSE`, `IPC`) selection. + `World::getResolvedConfiguration()` is C++-only as of the audit. +- Existing `python/examples/demos/scenes` seed first-wave fixtures: + `rigid_spin_roll_coupling` (CT-001), `rigid_restitution_ladder` (CT-003), + `atlas_simbicon` (CT-005), `rigid_multibody_solver_family`, + `rigid_contact_inspector`, and the AVBD/IPC scene sets. +- Open PR #3432 (PLAN-104) owns VBD/AVBD paper parity with its own + fail-closed 88-row contracts; PLAN-123 links, never duplicates. +- Open PR #3377 (`release-6.20` research lane) owns opt-in exact-Coulomb FBF + friction, rolling/incline/turntable/Painleve scenes, and schema-v3 visual + evidence on DART 6; CT-001/CT-007 DART 6 lanes reference it instead of + recreating fixtures. +- Open PR #3431 owns the DART 6 soft-foot perturbation-ensemble methodology + (CT-020 pattern); PR #3428 and open issue #3056 keep PLAN-621 the CT-018 + owner. PLAN-622 rows stay with their dev task. +- Plan IDs confirmed free at audit time: PLAN-123 on `main`, PLAN-623 on + `release-6.20`. +- Machine-readable manifest decision: JSON (not YAML) to match every existing + packet/validator in `scripts/` and add no dependency. The YAML skeleton + below documents field intent; the checked schema is + `dart.citation_claim_evidence/v1` JSON. + +## Source anchors + +Use primary sources and repository evidence during implementation: + +- DART JOSS paper: +- SimBenchmark: +- Predictable behavior during contact simulation: + +- From Compliant to Rigid Contact Simulation: + +- Nimble: +- RobotDART: +- PEEL: +- Fibration Trees: +- Gazebo Plants: +- DART repository issues, PRs, plan docs, tests, and benchmark packets are + primary evidence for current implementation state. + +For sources whose final bibliographic record or license is uncertain, record +that uncertainty and do not vendor assets until provenance is resolved. + +## First implementation wave + +The first wave is intentionally capped at six fixture families: + +1. rolling/friction-direction sweep; +2. dense inelastic/elastic contacts; +3. articulated energy/momentum/control; +4. heel-strike/toe-off transition; +5. high mass-ratio stack/manipulation; +6. reset/concurrency/planning collision query. + +Do not add a seventh family until all six have branch-qualified dispositions or +a maintainer explicitly revises the cap. + +## Evidence packet skeleton + +```yaml +schema: dart.citation_claim_evidence/v1 +claim_id: CT-001 +source: + url: ... + claim: ... +target: + branch: main + commit: ... +scene: + id: ... + digest: ... +configuration: + requested: ... + resolved: ... + detector: ... + timestep: ... + substeps: ... + tolerances: ... + fallback_policy: ... +ensemble: + seeds: [...] + perturbations: ... +metrics: + physical: { ... } + numerical: { ... } + performance: { ... } + allocation: { ... } +evidence: + commands: [...] + raw_paths: [...] + visual_paths: [...] +result: + disposition: unresolved + claim_boundary: ... + limitations: [...] +review: + passes: [...] +``` + +The schema must fail closed on missing target commit, scene digest, requested +and resolved method, command, disposition, or claim boundary. diff --git a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json new file mode 100644 index 0000000000000..84b912aecad0c --- /dev/null +++ b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json @@ -0,0 +1,426 @@ +{ + "schema": "dart.citation_claim_manifest/v1", + "corpus_doc": "docs/plans/123-citation-driven-simulation-trust/citation-claim-corpus.md", + "audited": "2026-08-14", + "first_wave_families": [ + "rolling_friction_direction", + "dense_contact", + "articulated_energy_momentum_control", + "heel_strike_toe_off", + "high_mass_ratio_stack_manipulation", + "reset_concurrency_queries" + ], + "claims": [ + { + "id": "CT-001", + "title": "Rolling-direction friction dependence", + "source": "SimBenchmark / historical DART contact comparison", + "first_wave_family": "rolling_friction_direction", + "lanes": { + "dart7": { + "owner": "PLAN-123 with PLAN-080", + "status": "in-progress", + "disposition": null, + "evidence": [ + "evidence/CT-001-dart7-rolling-direction.json" + ], + "notes": "First packet: sphere rolling-direction sweep, SI vs boxed LCP; seeded by python/examples/demos/scenes/rigid_spin_roll_coupling.py." + }, + "dart6": { + "owner": "docs/dev_tasks/dart6_citation_contact_trust (reference PR #3377 scenes)", + "status": "audit-required", + "disposition": null, + "evidence": [], + "notes": "Open PR #3377 owns exact-Coulomb FBF rolling scenes on release-6.20; reuse, do not duplicate." + } + } + }, + { + "id": "CT-002", + "title": "Dense inelastic contact stability and scaling", + "source": "SimBenchmark dense 6x6x6 inelastic contact", + "first_wave_family": "dense_contact", + "lanes": { + "dart7": { + "owner": "PLAN-123 with PLAN-080/122", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": "docs/dev_tasks/dart6_citation_contact_trust", + "status": "audit-required", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-003", + "title": "Dense elastic contact energy behavior", + "source": "SimBenchmark dense elastic contact", + "first_wave_family": "dense_contact", + "lanes": { + "dart7": { + "owner": "PLAN-123 with PLAN-080", + "status": "audit-required", + "disposition": null, + "evidence": [], + "notes": "python/examples/demos/scenes/rigid_restitution_ladder.py is the existing seed scene." + }, + "dart6": { + "owner": "docs/dev_tasks/dart6_citation_contact_trust", + "status": "audit-required", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-004", + "title": "Articulated momentum/energy accuracy at matched cost", + "source": "SimBenchmark articulated momentum/energy", + "first_wave_family": "articulated_energy_momentum_control", + "lanes": { + "dart7": { + "owner": "PLAN-123 with PLAN-080/084 using StepMetrics", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": "docs/dev_tasks/dart6_citation_contact_trust", + "status": "audit-required", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-005", + "title": "Articulated PD-control tracking tradeoffs", + "source": "SimBenchmark articulated PD control", + "first_wave_family": "articulated_energy_momentum_control", + "lanes": { + "dart7": { + "owner": "PLAN-123 with PLAN-080", + "status": "audit-required", + "disposition": null, + "evidence": [], + "notes": "python/examples/demos/scenes/atlas_simbicon.py is the existing controlled-articulated seed." + }, + "dart6": { + "owner": "docs/dev_tasks/dart6_citation_contact_trust", + "status": "audit-required", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-006", + "title": "Heel-strike/toe-off contact-force transients", + "source": "2026 exoskeleton contact-force criticism", + "first_wave_family": "heel_strike_toe_off", + "lanes": { + "dart7": { + "owner": "PLAN-123 contact semantics (WS3)", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": "docs/dev_tasks/dart6_citation_contact_trust", + "status": "audit-required", + "disposition": null, + "evidence": [], + "notes": "Clarify/diagnose legacy Contact.force semantics without ABI/default change." + } + } + }, + { + "id": "CT-007", + "title": "Exact Coulomb cones vs pyramid anisotropy/conditioning", + "source": "From Compliant to Rigid Contact Simulation", + "first_wave_family": "high_mass_ratio_stack_manipulation", + "lanes": { + "dart7": { + "owner": "PLAN-123 exact-cone GO/NO-GO (WS5)", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": "evidence only; reference PR #3377 research lane", + "status": "audit-required", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-008", + "title": "Shared contact problem with compliance vs regularization split", + "source": "From Compliant to Rigid Contact Simulation", + "first_wave_family": null, + "lanes": { + "dart7": { + "owner": "PLAN-123 exact-cone problem design (WS5)", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": "document existing behavior only", + "status": "audit-required", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-009", + "title": "Contact-aware inverse dynamics with feasibility diagnostics", + "source": "From Compliant to Rigid Contact Simulation / robotics need", + "first_wave_family": null, + "lanes": { + "dart7": { + "owner": "PLAN-123 WS6", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": null, + "status": "not-applicable", + "reason": "No new inverse-dynamics architecture on the LTS branch beyond existing APIs.", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-010", + "title": "Analytic hard-contact derivatives agree with finite differences", + "source": "Nimble", + "first_wave_family": null, + "lanes": { + "dart7": { + "owner": "PLAN-110 with PLAN-123 validity evidence", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": null, + "status": "not-applicable", + "reason": "Differentiable architecture is DART 7 only.", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-011", + "title": "Fast reset, concurrency, and deterministic synchronous stepping", + "source": "RobotDART", + "first_wave_family": "reset_concurrency_queries", + "lanes": { + "dart7": { + "owner": "PLAN-030/122/123 one-model/many-state", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": "existing clone/reset/Recording evidence only", + "status": "audit-required", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-012", + "title": "High-throughput collision checking for long-horizon planning", + "source": "PEEL", + "first_wave_family": "reset_concurrency_queries", + "lanes": { + "dart7": { + "owner": "PLAN-120/123 planning companion intake", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": "optional benchmark/adapter evidence", + "status": "audit-required", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-013", + "title": "Thread-safe state-validity and collision queries", + "source": "Fibration Trees", + "first_wave_family": "reset_concurrency_queries", + "lanes": { + "dart7": { + "owner": "PLAN-120/123", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": "optional query benchmark", + "status": "audit-required", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-014", + "title": "Multi-stage manipulation environment contracts", + "source": "Behavior Policy Learning", + "first_wave_family": null, + "lanes": { + "dart7": { + "owner": "PLAN-103/123 environment companion intake", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": "companion-only evidence", + "status": "audit-required", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-015", + "title": "Scalable muscle-actuated models and trustworthy gait outputs", + "source": "MASS / musculoskeletal DART use", + "first_wave_family": null, + "lanes": { + "dart7": { + "owner": "future companion after PLAN-123 foundation", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": "existing SoftBody/contact evidence only", + "status": "audit-required", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-016", + "title": "Standardized experiment model/sensor/data contracts", + "source": "FARMS", + "first_wave_family": null, + "lanes": { + "dart7": { + "owner": "PLAN-103/123 companion intake", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": "companion-only", + "status": "audit-required", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-017", + "title": "Codimensional rods/plants beyond rigid-body approximation", + "source": "Gazebo Plants", + "first_wave_family": null, + "lanes": { + "dart7": { + "owner": "existing deformable families; bounded intake after trust foundation", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": null, + "status": "not-applicable", + "reason": "No codimensional rod/shell architecture on the LTS branch.", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-018", + "title": "Large settled worlds must not stay orders of magnitude slower", + "source": "DART issue #3056 and DART 6 performance campaign", + "first_wave_family": null, + "lanes": { + "dart7": { + "owner": "PLAN-123 scalability evidence where equivalent", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": "PLAN-621 (docs/dev_tasks/dart6_performance_generalization)", + "status": "audit-required", + "disposition": null, + "evidence": [], + "notes": "Issue #3056 open; PR #3428 active; PLAN-621 stays the owner and this row only references its evidence." + } + } + }, + { + "id": "CT-019", + "title": "Unambiguous contact normal/order/sign/frame/wrench ownership", + "source": "Historical contact-normal / wrench issues (#1425, #1073)", + "first_wave_family": null, + "lanes": { + "dart7": { + "owner": "PLAN-123 observation and wrench contract (WS3)", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": "docs/dev_tasks/dart6_citation_contact_trust", + "status": "audit-required", + "disposition": null, + "evidence": [] + } + } + }, + { + "id": "CT-020", + "title": "Single trajectories and saturated thresholds create false claims", + "source": "Current DART paper-parity work", + "first_wave_family": null, + "lanes": { + "dart7": { + "owner": "PLAN-123 methodology; apply to all threshold/robustness claims", + "status": "audit-required", + "disposition": null, + "evidence": [] + }, + "dart6": { + "owner": "PLAN-622 perturbation-ensemble evidence (PR #3431)", + "status": "audit-required", + "disposition": null, + "evidence": [] + } + } + } + ] +} diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json new file mode 100644 index 0000000000000..e1d09b1288082 --- /dev/null +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json @@ -0,0 +1,684 @@ +{ + "claim_id": "CT-001", + "configuration": { + "detector": "DART 7 native World collision pipeline (the World step API exposes no detector selection on main)", + "fallback_policy": "World defaults; no per-island fallback reporting is exposed on main (PLAN-123 WS4 follow-up)", + "iterations": "World defaults; per-step actual iterations recorded via StepMetrics.last_step_iterations", + "requested": { + "backend": "cpu", + "contact_solver_method_sweep": [ + "SEQUENTIAL_IMPULSE", + "BOXED_LCP" + ], + "integrator": "World default semi-implicit stepping", + "precision": "float64", + "rigid_body_solver": "SEQUENTIAL_IMPULSE (World default)", + "threads": "World default sequential step" + }, + "resolved": { + "by_contact_solver_method": { + "BOXED_LCP": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "SEQUENTIAL_IMPULSE": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + } + } + }, + "resolved_provenance": "World property readback (contact_solver_method, rigid_body_solver, gravity, time_step) after enter_simulation_mode plus final-step WorldStepProfile stage names recorded per run; World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", + "substeps": 1, + "timestep": 0.002 + }, + "ensemble": { + "deterministic_repeats": 2, + "deterministic_repeats_identical": true, + "kind": "parameter-sweep-with-deterministic-repeats", + "measurement_window": { + "end_s": 1.0, + "start_s": 0.0, + "steps": 500 + }, + "sweep": [ + { + "angle_deg": 0.0, + "contact_solver_method": "SEQUENTIAL_IMPULSE" + }, + { + "angle_deg": 15.0, + "contact_solver_method": "SEQUENTIAL_IMPULSE" + }, + { + "angle_deg": 30.0, + "contact_solver_method": "SEQUENTIAL_IMPULSE" + }, + { + "angle_deg": 45.0, + "contact_solver_method": "SEQUENTIAL_IMPULSE" + }, + { + "angle_deg": 60.0, + "contact_solver_method": "SEQUENTIAL_IMPULSE" + }, + { + "angle_deg": 75.0, + "contact_solver_method": "SEQUENTIAL_IMPULSE" + }, + { + "angle_deg": 90.0, + "contact_solver_method": "SEQUENTIAL_IMPULSE" + }, + { + "angle_deg": 0.0, + "contact_solver_method": "BOXED_LCP" + }, + { + "angle_deg": 15.0, + "contact_solver_method": "BOXED_LCP" + }, + { + "angle_deg": 30.0, + "contact_solver_method": "BOXED_LCP" + }, + { + "angle_deg": 45.0, + "contact_solver_method": "BOXED_LCP" + }, + { + "angle_deg": 60.0, + "contact_solver_method": "BOXED_LCP" + }, + { + "angle_deg": 75.0, + "contact_solver_method": "BOXED_LCP" + }, + { + "angle_deg": 90.0, + "contact_solver_method": "BOXED_LCP" + } + ] + }, + "evidence": { + "commands": [ + "PYTHONPATH=build/default/cpp/Release/python pixi run python scripts/write_citation_ct001_rolling_direction_packet.py" + ], + "raw_rows": [ + { + "along_travel_m": 0.7258893117142907, + "angle_deg": 0.0, + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_planar_speed_mps": 0.7142857142857155, + "final_step_stage_names": [ + "rigid_body_velocity", + "rigid_body_contact", + "rigid_body_position", + "kinematics" + ], + "heading_error_deg": 0.0, + "lateral_drift_m": 0.0, + "max_active_contacts": 1, + "max_energy_gain_j": 0.0, + "max_penetration_m": 0.0, + "max_solver_iterations": 8, + "max_solver_residual": 0.0, + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "slide_end_time_s": 0.082, + "trajectory_sha256": "a839b7f050262f80e450a25831e5b4144eb74f3f725807eb0e92855f762b96d5" + }, + { + "along_travel_m": 0.7249208585743429, + "angle_deg": 15.0, + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_planar_speed_mps": 0.7142857142857154, + "final_step_stage_names": [ + "rigid_body_velocity", + "rigid_body_contact", + "rigid_body_position", + "kinematics" + ], + "heading_error_deg": -3.1169435878342416e-14, + "lateral_drift_m": -0.0021005566494608774, + "max_active_contacts": 1, + "max_energy_gain_j": 5.551115123125783e-17, + "max_penetration_m": 0.0, + "max_solver_iterations": 8, + "max_solver_residual": 0.0, + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "slide_end_time_s": 0.078, + "trajectory_sha256": "4579581ae2bab0980e689803c5fc8e7296f2b79314c1b9644855360102d16ced" + }, + { + "along_travel_m": 0.7232079593029788, + "angle_deg": 30.0, + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_planar_speed_mps": 0.7142857142857152, + "final_step_stage_names": [ + "rigid_body_velocity", + "rigid_body_contact", + "rigid_body_position", + "kinematics" + ], + "heading_error_deg": -5.343331864858703e-14, + "lateral_drift_m": -0.001883289782513231, + "max_active_contacts": 1, + "max_energy_gain_j": 5.551115123125783e-17, + "max_penetration_m": 0.0, + "max_solver_iterations": 8, + "max_solver_residual": 0.0, + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "slide_end_time_s": 0.07, + "trajectory_sha256": "ace9350655e315351bede1b8474feac16c223921633fbf2a5497af3ae57d941a" + }, + { + "along_travel_m": 0.7224082209135987, + "angle_deg": 45.0, + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_planar_speed_mps": 0.7142857142857155, + "final_step_stage_names": [ + "rigid_body_velocity", + "rigid_body_contact", + "rigid_body_position", + "kinematics" + ], + "heading_error_deg": 0.0, + "lateral_drift_m": 5.551115123125783e-17, + "max_active_contacts": 1, + "max_energy_gain_j": 5.551115123125783e-17, + "max_penetration_m": 0.0, + "max_solver_iterations": 8, + "max_solver_residual": 0.0, + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "slide_end_time_s": 0.058, + "trajectory_sha256": "8446fd6f1793674e853a71724405db019de7f5acd799e9e61af877310443edc7" + }, + { + "along_travel_m": 0.7232079593029789, + "angle_deg": 60.0, + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_planar_speed_mps": 0.7142857142857151, + "final_step_stage_names": [ + "rigid_body_velocity", + "rigid_body_contact", + "rigid_body_position", + "kinematics" + ], + "heading_error_deg": 5.343331864858703e-14, + "lateral_drift_m": 0.0018832897825133976, + "max_active_contacts": 1, + "max_energy_gain_j": 1.1102230246251565e-16, + "max_penetration_m": 0.0, + "max_solver_iterations": 8, + "max_solver_residual": 0.0, + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "slide_end_time_s": 0.07, + "trajectory_sha256": "fce89de8287195fea41ecdbc5771c027b838e91d2c5be2e7255a6cc2abffeb2d" + }, + { + "along_travel_m": 0.7249208585743429, + "angle_deg": 75.0, + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_planar_speed_mps": 0.7142857142857154, + "final_step_stage_names": [ + "rigid_body_velocity", + "rigid_body_contact", + "rigid_body_position", + "kinematics" + ], + "heading_error_deg": 3.1169435878342416e-14, + "lateral_drift_m": 0.0021005566494608774, + "max_active_contacts": 1, + "max_energy_gain_j": 5.551115123125783e-17, + "max_penetration_m": 0.0, + "max_solver_iterations": 8, + "max_solver_residual": 0.0, + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "slide_end_time_s": 0.078, + "trajectory_sha256": "2ca8f5b51e27817c76eedef74e1b9c59813fa9052244f219e3143eb0327f5d7a" + }, + { + "along_travel_m": 0.7258893117142907, + "angle_deg": 90.0, + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_planar_speed_mps": 0.7142857142857155, + "final_step_stage_names": [ + "rigid_body_velocity", + "rigid_body_contact", + "rigid_body_position", + "kinematics" + ], + "heading_error_deg": 6.426647569961019e-30, + "lateral_drift_m": 7.105154224747257e-19, + "max_active_contacts": 1, + "max_energy_gain_j": 0.0, + "max_penetration_m": 0.0, + "max_solver_iterations": 8, + "max_solver_residual": 0.0, + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "slide_end_time_s": 0.082, + "trajectory_sha256": "d7bd5003831ed9c7c1a477345d578fe7f3f5a99706d3a4e9720e11d770a9c5b8" + }, + { + "along_travel_m": 0.7258893117976364, + "angle_deg": 0.0, + "contact_solver_method": "BOXED_LCP", + "final_planar_speed_mps": 0.7142857142857155, + "final_step_stage_names": [ + "rigid_body_velocity", + "rigid_body_contact", + "rigid_body_position", + "kinematics" + ], + "heading_error_deg": 0.0, + "lateral_drift_m": 0.0, + "max_active_contacts": 1, + "max_energy_gain_j": 5.551115123125783e-17, + "max_penetration_m": 0.0, + "max_solver_iterations": 0, + "max_solver_residual": 0.0, + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "slide_end_time_s": 0.082, + "trajectory_sha256": "f92fc6c64c534a96459ce3ac0a970c931618a8533dd3974b565db61b7cacd150" + }, + { + "along_travel_m": 0.7249208586267563, + "angle_deg": 15.0, + "contact_solver_method": "BOXED_LCP", + "final_planar_speed_mps": 0.7142857142857154, + "final_step_stage_names": [ + "rigid_body_velocity", + "rigid_body_contact", + "rigid_body_position", + "kinematics" + ], + "heading_error_deg": -3.1169435878342416e-14, + "lateral_drift_m": -0.002100556554215066, + "max_active_contacts": 1, + "max_energy_gain_j": 5.551115123125783e-17, + "max_penetration_m": 0.0, + "max_solver_iterations": 0, + "max_solver_residual": 0.0, + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "slide_end_time_s": 0.078, + "trajectory_sha256": "fbe57d7b17ff7fdfe97b1cfe5112698ae2ae9c7c413e4c08eb13baba931e0899" + }, + { + "along_travel_m": 0.7232079593620271, + "angle_deg": 30.0, + "contact_solver_method": "BOXED_LCP", + "final_planar_speed_mps": 0.7142857142857152, + "final_step_stage_names": [ + "rigid_body_velocity", + "rigid_body_contact", + "rigid_body_position", + "kinematics" + ], + "heading_error_deg": -5.343331864858703e-14, + "lateral_drift_m": -0.0018832896891917694, + "max_active_contacts": 1, + "max_energy_gain_j": 5.551115123125783e-17, + "max_penetration_m": 0.0, + "max_solver_iterations": 0, + "max_solver_residual": 0.0, + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "slide_end_time_s": 0.07, + "trajectory_sha256": "baed575b592007dcfedeefa6f7b8c17858d47860acb3e6f2c06e925766e575fe" + }, + { + "along_travel_m": 0.7224082209952696, + "angle_deg": 45.0, + "contact_solver_method": "BOXED_LCP", + "final_planar_speed_mps": 0.7142857142857154, + "final_step_stage_names": [ + "rigid_body_velocity", + "rigid_body_contact", + "rigid_body_position", + "kinematics" + ], + "heading_error_deg": -4.4527765540489165e-15, + "lateral_drift_m": 5.551115123125783e-17, + "max_active_contacts": 1, + "max_energy_gain_j": 5.551115123125783e-17, + "max_penetration_m": 0.0, + "max_solver_iterations": 0, + "max_solver_residual": 0.0, + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "slide_end_time_s": 0.058, + "trajectory_sha256": "4f8571c3013779682bcf99ce7dd5f3cc7bb3ccb49182fc64f8f55199fa3b19ef" + }, + { + "along_travel_m": 0.7232079593620271, + "angle_deg": 60.0, + "contact_solver_method": "BOXED_LCP", + "final_planar_speed_mps": 0.7142857142857152, + "final_step_stage_names": [ + "rigid_body_velocity", + "rigid_body_contact", + "rigid_body_position", + "kinematics" + ], + "heading_error_deg": 4.8980542094538094e-14, + "lateral_drift_m": 0.0018832896891918804, + "max_active_contacts": 1, + "max_energy_gain_j": 5.551115123125783e-17, + "max_penetration_m": 0.0, + "max_solver_iterations": 0, + "max_solver_residual": 0.0, + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "slide_end_time_s": 0.07, + "trajectory_sha256": "be2f9b06b52ffb94c3c4c65e2ef2a1ba188e02b852e89c1fcd355fb87e27079d" + }, + { + "along_travel_m": 0.7249208586267563, + "angle_deg": 75.0, + "contact_solver_method": "BOXED_LCP", + "final_planar_speed_mps": 0.7142857142857154, + "final_step_stage_names": [ + "rigid_body_velocity", + "rigid_body_contact", + "rigid_body_position", + "kinematics" + ], + "heading_error_deg": 3.1169435878342416e-14, + "lateral_drift_m": 0.002100556554215066, + "max_active_contacts": 1, + "max_energy_gain_j": 5.551115123125783e-17, + "max_penetration_m": 0.0, + "max_solver_iterations": 0, + "max_solver_residual": 0.0, + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "slide_end_time_s": 0.078, + "trajectory_sha256": "dd250dfabf218e68548f9edf30a342bc1bc595d6675b5f602f4bf608f88921e0" + }, + { + "along_travel_m": 0.7258893117976364, + "angle_deg": 90.0, + "contact_solver_method": "BOXED_LCP", + "final_planar_speed_mps": 0.7142857142857155, + "final_step_stage_names": [ + "rigid_body_velocity", + "rigid_body_contact", + "rigid_body_position", + "kinematics" + ], + "heading_error_deg": 6.426647569961019e-30, + "lateral_drift_m": 7.105150776790931e-19, + "max_active_contacts": 1, + "max_energy_gain_j": 5.551115123125783e-17, + "max_penetration_m": 0.0, + "max_solver_iterations": 0, + "max_solver_residual": 0.0, + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "slide_end_time_s": 0.082, + "trajectory_sha256": "631d89a93d54a08badc23834c755a2a84c88f7d6f6b94a2357bd81a5a3e7d9a5" + } + ], + "visual": { + "reason": "The oracle is numeric rotational symmetry; per-angle trajectories have no visual claim beyond the recorded metrics. Visual capture joins when a corpus row makes a visible-behavior claim.", + "status": "not-applicable" + } + }, + "host": { + "machine": "x86_64", + "note": "Host recorded for provenance only; no timing methodology was applied", + "performance_valid": false, + "platform": "Linux-7.0.0-29-generic-x86_64-with-glibc2.43", + "python": "3.14.6" + }, + "metrics": { + "allocation": { + "reason": "Post-bake allocation gates are owned by PLAN-122 tooling; this packet does not measure allocations", + "status": "unsupported" + }, + "numerical": { + "max_active_contacts": 1, + "max_penetration_m": 0.0, + "max_solver_iterations": 8, + "max_solver_residual": 0.0, + "method": "Max over run of StepMetrics.max_penetration_depth, last_step_iterations, last_step_residual, active_contact_count" + }, + "performance": { + "reason": "This packet makes no timing claim; interleaved same-host methodology is required before any performance row", + "status": "unsupported" + }, + "physical": { + "anisotropic_methods": [ + "BOXED_LCP", + "SEQUENTIAL_IMPULSE" + ], + "isotropy_tolerance": { + "max_abs_heading_error_deg": 0.1, + "max_abs_lateral_drift_m": 0.0001, + "travel_spread_relative": 0.01 + }, + "method": "Rotational-symmetry sweep: per-angle lateral drift from the launch ray, final-velocity heading error, along-ray travel, slide-to-roll transition time, and max single-step kinetic-energy gain from StepMetrics.kinetic_energy", + "per_solver_summary": { + "BOXED_LCP": { + "max_abs_heading_error_deg": 5.343331864858703e-14, + "max_abs_lateral_drift_m": 0.002100556554215066, + "max_energy_gain_j": 5.551115123125783e-17, + "max_penetration_m": 0.0, + "max_solver_residual": 0.0, + "runs": 7, + "travel_mean_m": 0.7243492115097299, + "travel_spread_m": 0.003481090802366804, + "travel_spread_relative": 0.00480581844648014 + }, + "SEQUENTIAL_IMPULSE": { + "max_abs_heading_error_deg": 5.343331864858703e-14, + "max_abs_lateral_drift_m": 0.0021005566494608774, + "max_energy_gain_j": 1.1102230246251565e-16, + "max_penetration_m": 0.0, + "max_solver_residual": 0.0, + "runs": 7, + "travel_mean_m": 0.7243492114424034, + "travel_spread_m": 0.0034810908006920327, + "travel_spread_relative": 0.004805818444614723 + } + } + } + }, + "result": { + "claim_boundary": "DART 7 main, this commit, one sphere sliding to rolling on a static ground box at v0=1 m/s, mu=0.35, dt=2 ms, 1 s horizon, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers, launch angles 0-90 deg in 15 deg steps. Says nothing about other speeds, shapes, stacks, historical DART versions, or DART 6.", + "disposition": "reproduced", + "limitations": [ + "ResolvedSolverConfiguration is not Python-exposed; resolved identity is property readback plus stage names.", + "Friction-cone/complementarity residual per contact is not exposed; solver residual is the aggregate StepMetrics value.", + "The sweep covers 0-90 deg; pyramid orientation with a period other than 90 deg would need a wider sweep.", + "No high-mass-ratio or stacked variant yet; those belong to CT-002/CT-007 fixtures.", + "SEQUENTIAL_IMPULSE and BOXED_LCP produce metric summaries that agree to printed precision on this single-contact scene while their full trajectory hashes differ; this packet is not a solver comparison and must not be quoted as one." + ] + }, + "review": { + "passes": [] + }, + "scene": { + "description": "One 1 kg sphere per run, radius 0.08 m, launched sliding (no spin) at 1.0 m/s along a swept in-plane angle on a static ground box, restitution 0, friction 0.35 on both bodies.", + "digest": "sha256:1c722e156e8b4cd38c8580347b09734dd07d23fa42e6f75fc52ebe633845856d", + "id": "ct001_rolling_direction_sweep", + "parameters": { + "contact_solver_methods": [ + "SEQUENTIAL_IMPULSE", + "BOXED_LCP" + ], + "description": "One 1 kg sphere per run, radius 0.08 m, launched sliding (no spin) at 1.0 m/s along a swept in-plane angle on a static ground box, restitution 0, friction 0.35 on both bodies.", + "deterministic_repeats": 2, + "friction": 0.35, + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "ground_half_extents_m": [ + 2.5, + 2.5, + 0.05 + ], + "launch_angles_deg": [ + 0.0, + 15.0, + 30.0, + 45.0, + 60.0, + 75.0, + 90.0 + ], + "launch_speed_mps": 1.0, + "restitution": 0.0, + "scene_id": "ct001_rolling_direction_sweep", + "slip_ratio_rolling_threshold": 0.02, + "sphere_mass_kg": 1.0, + "sphere_radius_m": 0.08, + "step_count": 500, + "time_step_s": 0.002 + } + }, + "schema": "dart.citation_claim_evidence/v1", + "source": { + "claim": "Polyhedral friction can produce direction-dependent rolling/sliding behavior.", + "url": "https://leggedrobotics.github.io/SimBenchmark/" + }, + "target": { + "branch": "main", + "commit": "20501341226338f42740f1b6ef1ab52a34e5a035" + }, + "title": "Rolling-direction friction dependence (DART 7 first packet)" +} diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/negative-controls/CT-002-dart7-intentionally-incomplete.json b/docs/plans/123-citation-driven-simulation-trust/evidence/negative-controls/CT-002-dart7-intentionally-incomplete.json new file mode 100644 index 0000000000000..16552d10280fe --- /dev/null +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/negative-controls/CT-002-dart7-intentionally-incomplete.json @@ -0,0 +1,57 @@ +{ + "schema": "dart.citation_claim_evidence/v1", + "claim_id": "CT-002", + "title": "Intentionally incomplete negative control: this packet must FAIL check_citation_evidence forever; it proves the validator rejects missing resolved identity, missing scene digest, single-run ensembles, unsupported-as-zero metrics, and prose-only closure", + "source": { + "url": "https://leggedrobotics.github.io/SimBenchmark/", + "claim": "Dense contact may fail, become unstable, or scale poorly for some timestep/solver settings." + }, + "target": { + "branch": "main", + "commit": "not-a-commit" + }, + "scene": { + "id": "ct002_dense_contact_grid", + "description": "A 6x6x6 sphere grid described only in prose, with no digest." + }, + "configuration": { + "requested": { + "contact_solver_method": "BOXED_LCP" + }, + "detector": "dart7 native pipeline", + "timestep": 0.002, + "substeps": 1 + }, + "ensemble": { + "kind": "single-run", + "deterministic_repeats": 1 + }, + "metrics": { + "physical": { + "max_penetration_m": 0.0, + "energy_drift_j": 0.0 + }, + "numerical": { + "status": "unsupported" + }, + "performance": { + "wall_time_ms": 0.0 + }, + "allocation": { + "status": "unsupported", + "reason": "not measured" + } + }, + "evidence": { + "commands": [], + "raw_paths": [], + "visual": [] + }, + "result": { + "disposition": "reproduced", + "limitations": [] + }, + "review": { + "passes": [] + } +} diff --git a/docs/plans/README.md b/docs/plans/README.md index 6186f0bf25df4..38154d6e11115 100644 --- a/docs/plans/README.md +++ b/docs/plans/README.md @@ -111,6 +111,8 @@ they should not own priority, timeline, or active implementation state. | [`120-inverse-kinematics-and-motion.md`](120-inverse-kinematics-and-motion.md) | PLAN-120 IK solver portfolio, whole-body parity, and motion IK | | [`122-simulation-loop-allocation-hardening.md`](122-simulation-loop-allocation-hardening.md) | PLAN-122 DART 7 no-allocation-after-bake closure plan | | [`122-simulation-loop-allocation-hardening/coverage-matrix.md`](122-simulation-loop-allocation-hardening/coverage-matrix.md) | PLAN-122 DART 7 simulation-loop allocation coverage matrix | +| [`123-citation-driven-simulation-trust.md`](123-citation-driven-simulation-trust.md) | PLAN-123 citation reproduction, contact trust, and solver observability | +| [`123-citation-driven-simulation-trust/citation-claim-corpus.md`](123-citation-driven-simulation-trust/citation-claim-corpus.md) | PLAN-123 row-level claim and evidence intake | | Numbered initiative files, such as `010-*.md` | Active plan scope, evidence, open gaps, and acceptance criteria | | [`AGENTS.md`](AGENTS.md) | Local rules for agents editing plan docs | | [`../ai/north-star.md`](../ai/north-star.md) | Mission, current state, missing capabilities, and readiness bar | diff --git a/docs/plans/dashboard.md b/docs/plans/dashboard.md index 385569b223b24..f6027eee7bf33 100644 --- a/docs/plans/dashboard.md +++ b/docs/plans/dashboard.md @@ -44,6 +44,29 @@ its own line so status updates remain git-history friendly. through global heap/raw malloc paths on measured hosts; migrated DART 7 paths must add the gate before promotion. +### PLAN-123: Citation-Driven Simulation Trust + +- Owner doc: + [`123-citation-driven-simulation-trust.md`](123-citation-driven-simulation-trust.md) +- Status: Active +- Horizon: Now +- Dimension: Algorithm extensibility +- Next step: WS0 audit and the WS1 fail-closed claim/evidence manifest landed + with the first CT-001 rolling-direction packet and a permanent + negative-control packet. Next: finish the capped six first-wave fixture + families (dense inelastic/elastic, articulated energy/momentum/control, + heel-strike/toe-off, high mass-ratio, reset/concurrency queries) with + branch-qualified dispositions, reusing existing demos scenes and PLAN-104/621/622 + evidence, before contact-identity/semantics work (WS3) and cross-family + diagnostics (WS4). History and sequencing live in the owner plan and + `docs/dev_tasks/citation_driven_simulation_trust/`. +- Gate: Every claim row is branch/version-qualified and source-bound; + `pixi run check-citation-evidence` fails closed on missing commit, scene + digest, requested/resolved method, command, ensemble, disposition, or claim + boundary and rejects unsupported-as-zero metrics; behavioral slices need + negative controls, deterministic or ensemble evidence, two clean reviews, + and PLAN-122 allocation coverage before promotion. + ### PLAN-012: Cloud Dartpy Tutorials - Owner doc: diff --git a/pixi.toml b/pixi.toml index a949d4d721b10..a969eadf78c60 100644 --- a/pixi.toml +++ b/pixi.toml @@ -391,6 +391,11 @@ check-lint-cmake = { cmd = ["python", "scripts/lint_cmake.py", "--check"] } check-docs-policy = { cmd = ["python", "scripts/check_docs_policy.py"] } +check-citation-evidence = { cmd = [ + "python", + "scripts/check_citation_evidence.py", +] } + check-dart7-clean-break-policy = { cmd = [ "python", "scripts/check_dart7_clean_break_policy.py", @@ -494,6 +499,7 @@ test-ai-infra = { cmd = [ "tests/test_check_agent_hook.py", "tests/test_setup_ai.py", "tests/test_check_docs_policy.py", + "tests/test_check_citation_evidence.py", "-q", ] } @@ -534,6 +540,7 @@ check-lint = { depends-on = [ "check-lcp-solver-roster", "check-plan122-allocation-matrix", "check-avbd-packets", + "check-citation-evidence", "check-architecture-page", "check-ai-commands", "check-ai-infra", diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py new file mode 100644 index 0000000000000..a9d50d5dfddc3 --- /dev/null +++ b/scripts/check_citation_evidence.py @@ -0,0 +1,645 @@ +#!/usr/bin/env python3 +"""Fail-closed validator for the PLAN-123 citation claim/evidence contract. + +Validates, without running any simulation: + +- `claims-manifest.json` against the `dart.citation_claim_manifest/v1` schema, + including exact claim-ID agreement with the human corpus table and the capped + first-wave family list; +- every packet in `evidence/` against the `dart.citation_claim_evidence/v1` + schema (missing target commit, scene digest, requested/resolved method, + command, ensemble, disposition, claim boundary, or review record fails); +- metric groups that must be measured-with-method or explicitly typed + `unsupported` with a reason -- never silently absent, null, or NaN; +- every packet in `evidence/negative-controls/`, which must FAIL validation + (a permanent proof that the validator rejects incomplete evidence); +- manifest lane/evidence cross-links: referenced packets exist, agree on claim + ID and branch, and `closed` lanes carry a disposition plus at least two + review passes. + +`--freshness` additionally requires every non-negative-control packet to +record the current `HEAD` commit; it is a packet-writing aid, not a CI gate, +because squash merges legitimately retire topic-branch commits. +""" + +from __future__ import annotations + +import argparse +import json +import math +import re +import subprocess +import sys +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[1] +PLAN_DIR = REPO_ROOT / "docs" / "plans" / "123-citation-driven-simulation-trust" +MANIFEST_PATH = PLAN_DIR / "claims-manifest.json" +CORPUS_PATH = PLAN_DIR / "citation-claim-corpus.md" +EVIDENCE_DIR = PLAN_DIR / "evidence" +NEGATIVE_DIR = EVIDENCE_DIR / "negative-controls" + +MANIFEST_SCHEMA = "dart.citation_claim_manifest/v1" +PACKET_SCHEMA = "dart.citation_claim_evidence/v1" + +CLAIM_ID_RE = re.compile(r"^CT-\d{3}$") +COMMIT_RE = re.compile(r"^[0-9a-f]{40}$") +SCENE_DIGEST_RE = re.compile(r"^sha256:[0-9a-f]{64}$") +CORPUS_ROW_RE = re.compile(r"^\|\s*(CT-\d{3})\s*\|", re.MULTILINE) + +DISPOSITIONS = ( + "missing", + "reproduced", + "fixed", + "version-specific", + "not-applicable", + "invalid-original-setup", + "unresolved", +) +LANE_STATUSES = ("audit-required", "in-progress", "closed", "not-applicable") +LANE_KEYS = ("dart7", "dart6") +BRANCH_BY_LANE = {"dart7": "main", "dart6": "release-6.20"} +FIRST_WAVE_FAMILY_CAP = 6 + +PACKET_TOP_LEVEL_KEYS = { + "schema", + "claim_id", + "title", + "source", + "target", + "scene", + "configuration", + "ensemble", + "metrics", + "evidence", + "result", + "review", + "host", +} +REQUIRED_PACKET_KEYS = PACKET_TOP_LEVEL_KEYS - {"host"} +METRIC_GROUPS = ("physical", "numerical", "performance", "allocation") + +GIT_QUERY_ERRORS = (OSError, subprocess.CalledProcessError) + + +def _is_nonempty_str(value: object) -> bool: + return isinstance(value, str) and bool(value.strip()) + + +def _is_finite_number(value: object) -> bool: + return ( + isinstance(value, (int, float)) + and not isinstance(value, bool) + and math.isfinite(value) + ) + + +def _metric_leaves(value: object) -> list[object]: + if isinstance(value, dict): + leaves: list[object] = [] + for child in value.values(): + leaves.extend(_metric_leaves(child)) + return leaves + if isinstance(value, list): + leaves = [] + for child in value: + leaves.extend(_metric_leaves(child)) + return leaves + return [value] + + +def corpus_claim_ids(corpus_text: str) -> list[str]: + """Extract the ordered unique claim IDs from the corpus markdown table.""" + seen: dict[str, None] = {} + for match in CORPUS_ROW_RE.finditer(corpus_text): + seen.setdefault(match.group(1)) + return list(seen) + + +def metric_group_errors(name: str, group: object) -> list[str]: + """A metric group is measured-with-method or typed unsupported. Nothing else.""" + errors: list[str] = [] + if not isinstance(group, dict) or not group: + return [f"metrics.{name} must be a non-empty object"] + if group.get("status") == "unsupported": + if not _is_nonempty_str(group.get("reason")): + errors.append(f"metrics.{name} is unsupported but has no non-empty reason") + extra = set(group) - {"status", "reason"} + if extra: + errors.append( + f"metrics.{name} mixes unsupported status with values: " + f"{sorted(extra)}" + ) + return errors + if "status" in group: + errors.append(f"metrics.{name}.status must be 'unsupported' when present") + if not _is_nonempty_str(group.get("method")): + errors.append( + f"metrics.{name} must record a non-empty measurement 'method' " + "(or be typed unsupported with a reason)" + ) + value_keys = [key for key in group if key != "method"] + if not value_keys: + errors.append(f"metrics.{name} has a method but no measured values") + for key in value_keys: + for leaf in _metric_leaves(group[key]): + if leaf is None: + errors.append( + f"metrics.{name}.{key} contains null; unsupported values " + "must be typed, not null" + ) + elif isinstance(leaf, (int, float)) and not _is_finite_number(leaf): + errors.append(f"metrics.{name}.{key} contains a non-finite number") + return errors + + +def packet_errors( + packet: object, + *, + known_claim_ids: set[str] | None = None, + expected_branches: tuple[str, ...] = ("main", "release-6.20"), +) -> list[str]: + """Return every fail-closed violation for one evidence packet.""" + if not isinstance(packet, dict): + return ["packet must be a JSON object"] + errors: list[str] = [] + + if packet.get("schema") != PACKET_SCHEMA: + errors.append(f"schema must be {PACKET_SCHEMA!r}") + unknown = set(packet) - PACKET_TOP_LEVEL_KEYS + if unknown: + errors.append(f"unknown top-level keys: {sorted(unknown)}") + missing = REQUIRED_PACKET_KEYS - set(packet) + if missing: + errors.append(f"missing required top-level keys: {sorted(missing)}") + + claim_id = packet.get("claim_id") + if not (isinstance(claim_id, str) and CLAIM_ID_RE.match(claim_id)): + errors.append("claim_id must match CT-NNN") + elif known_claim_ids is not None and claim_id not in known_claim_ids: + errors.append(f"claim_id {claim_id} is not in the claims manifest") + if not _is_nonempty_str(packet.get("title")): + errors.append("title must be a non-empty string") + + source = packet.get("source") + if not isinstance(source, dict): + errors.append("source must be an object") + else: + if not _is_nonempty_str(source.get("url")): + errors.append("source.url must be a non-empty string") + if not _is_nonempty_str(source.get("claim")): + errors.append("source.claim must be a non-empty string") + + target = packet.get("target") + if not isinstance(target, dict): + errors.append("target must be an object") + else: + branch = target.get("branch") + if branch not in expected_branches: + errors.append(f"target.branch must be one of {list(expected_branches)}") + commit = target.get("commit") + if not (isinstance(commit, str) and COMMIT_RE.match(commit)): + errors.append("target.commit must be a 40-hex commit hash") + + scene = packet.get("scene") + if not isinstance(scene, dict): + errors.append("scene must be an object") + else: + if not _is_nonempty_str(scene.get("id")): + errors.append("scene.id must be a non-empty string") + digest = scene.get("digest") + if not (isinstance(digest, str) and SCENE_DIGEST_RE.match(digest)): + errors.append("scene.digest must match sha256:<64 hex>") + + configuration = packet.get("configuration") + if not isinstance(configuration, dict): + errors.append("configuration must be an object") + else: + for side in ("requested", "resolved"): + value = configuration.get(side) + if not isinstance(value, dict) or not value: + errors.append(f"configuration.{side} must be a non-empty object") + if not _is_nonempty_str(configuration.get("resolved_provenance")): + errors.append( + "configuration.resolved_provenance must name how the resolved " + "method identity was obtained" + ) + if not _is_nonempty_str(configuration.get("detector")): + errors.append("configuration.detector must be a non-empty string") + timestep = configuration.get("timestep") + if not (_is_finite_number(timestep) and timestep > 0.0): + errors.append("configuration.timestep must be a positive number") + if not _is_nonempty_str(configuration.get("fallback_policy")): + errors.append("configuration.fallback_policy must be a non-empty string") + + ensemble = packet.get("ensemble") + if not isinstance(ensemble, dict): + errors.append("ensemble must be an object") + else: + if not _is_nonempty_str(ensemble.get("kind")): + errors.append("ensemble.kind must be a non-empty string") + repeats = ensemble.get("deterministic_repeats") + sweep = ensemble.get("sweep") + seeds = ensemble.get("seeds") + has_repeats = ( + isinstance(repeats, int) and not isinstance(repeats, bool) and repeats >= 2 + ) + has_sweep = isinstance(sweep, list) and len(sweep) >= 2 + has_seeds = isinstance(seeds, list) and len(seeds) >= 2 + if not (has_repeats or has_sweep or has_seeds): + errors.append( + "ensemble must record deterministic_repeats >= 2, a sweep of " + ">= 2 points, or >= 2 seeds; single runs are not evidence" + ) + if "measurement_window" not in ensemble: + errors.append("ensemble.measurement_window is required") + + metrics = packet.get("metrics") + if not isinstance(metrics, dict): + errors.append("metrics must be an object") + else: + for name in METRIC_GROUPS: + if name not in metrics: + errors.append( + f"metrics.{name} is required (measured or typed " "unsupported)" + ) + else: + errors.extend(metric_group_errors(name, metrics[name])) + + evidence = packet.get("evidence") + if not isinstance(evidence, dict): + errors.append("evidence must be an object") + else: + commands = evidence.get("commands") + if not ( + isinstance(commands, list) + and commands + and all(_is_nonempty_str(command) for command in commands) + ): + errors.append("evidence.commands must be a non-empty list of commands") + raw_paths = evidence.get("raw_paths") + raw_rows = evidence.get("raw_rows") + has_paths = isinstance(raw_paths, list) and bool(raw_paths) + has_rows = isinstance(raw_rows, list) and bool(raw_rows) + if not (has_paths or has_rows): + errors.append("evidence must carry raw_rows inline or non-empty raw_paths") + visual = evidence.get("visual") + if isinstance(visual, dict): + if visual.get("status") != "not-applicable" or not _is_nonempty_str( + visual.get("reason") + ): + errors.append( + "evidence.visual object form must be " + "{'status': 'not-applicable', 'reason': ...}" + ) + elif not (isinstance(visual, list) and visual): + errors.append( + "evidence.visual must list visual artifacts or be typed " + "not-applicable with a reason" + ) + + result = packet.get("result") + if not isinstance(result, dict): + errors.append("result must be an object") + else: + if result.get("disposition") not in DISPOSITIONS: + errors.append(f"result.disposition must be one of {list(DISPOSITIONS)}") + if not _is_nonempty_str(result.get("claim_boundary")): + errors.append("result.claim_boundary must be a non-empty string") + limitations = result.get("limitations") + if not ( + isinstance(limitations, list) + and limitations + and all(_is_nonempty_str(item) for item in limitations) + ): + errors.append( + "result.limitations must be a non-empty list; every packet " + "has at least one honest limitation" + ) + + review = packet.get("review") + if not isinstance(review, dict): + errors.append("review must be an object") + else: + passes = review.get("passes") + if not isinstance(passes, list): + errors.append("review.passes must be a list") + else: + for index, entry in enumerate(passes): + if ( + not isinstance(entry, dict) + or not _is_nonempty_str(entry.get("reviewer")) + or not _is_nonempty_str(entry.get("summary")) + ): + errors.append( + f"review.passes[{index}] needs non-empty reviewer " + "and summary" + ) + + return errors + + +def manifest_errors(manifest: object, corpus_ids: list[str]) -> list[str]: + """Validate the claims manifest structure and corpus agreement.""" + if not isinstance(manifest, dict): + return ["manifest must be a JSON object"] + errors: list[str] = [] + if manifest.get("schema") != MANIFEST_SCHEMA: + errors.append(f"manifest schema must be {MANIFEST_SCHEMA!r}") + + families = manifest.get("first_wave_families") + if not ( + isinstance(families, list) + and len(families) == FIRST_WAVE_FAMILY_CAP + and all(_is_nonempty_str(family) for family in families) + and len(set(families)) == FIRST_WAVE_FAMILY_CAP + ): + errors.append( + "first_wave_families must list exactly " + f"{FIRST_WAVE_FAMILY_CAP} unique families (the cap is a " + "maintainer decision, not an editable default)" + ) + families = [] + + claims = manifest.get("claims") + if not isinstance(claims, list) or not claims: + errors.append("claims must be a non-empty list") + return errors + + seen_ids: list[str] = [] + for claim in claims: + if not isinstance(claim, dict): + errors.append("every claim must be an object") + continue + claim_id = claim.get("id", "") + if not (isinstance(claim_id, str) and CLAIM_ID_RE.match(claim_id)): + errors.append(f"claim id {claim_id!r} must match CT-NNN") + continue + seen_ids.append(claim_id) + if not _is_nonempty_str(claim.get("title")): + errors.append(f"{claim_id}: title must be non-empty") + if not _is_nonempty_str(claim.get("source")): + errors.append(f"{claim_id}: source must be non-empty") + family = claim.get("first_wave_family") + if family is not None and family not in families: + errors.append( + f"{claim_id}: first_wave_family {family!r} is not in the " + "capped family list" + ) + lanes = claim.get("lanes") + if not isinstance(lanes, dict) or set(lanes) != set(LANE_KEYS): + errors.append(f"{claim_id}: lanes must define exactly {LANE_KEYS}") + continue + for lane_name, lane in lanes.items(): + prefix = f"{claim_id}.lanes.{lane_name}" + if not isinstance(lane, dict): + errors.append(f"{prefix} must be an object") + continue + status = lane.get("status") + if status not in LANE_STATUSES: + errors.append(f"{prefix}.status must be one of {list(LANE_STATUSES)}") + continue + evidence = lane.get("evidence") + if not isinstance(evidence, list): + errors.append(f"{prefix}.evidence must be a list") + evidence = [] + if status == "not-applicable": + if not _is_nonempty_str(lane.get("reason")): + errors.append( + f"{prefix} is not-applicable and must record a reason" + ) + continue + if not _is_nonempty_str(lane.get("owner")): + errors.append(f"{prefix}.owner must be non-empty") + disposition = lane.get("disposition") + if status == "closed": + if disposition not in DISPOSITIONS: + errors.append(f"{prefix} is closed without a valid disposition") + if not evidence: + errors.append( + f"{prefix} is closed without evidence packets; prose " + "cannot close a row" + ) + elif disposition is not None and disposition not in DISPOSITIONS: + errors.append( + f"{prefix}.disposition must be null or one of " + f"{list(DISPOSITIONS)}" + ) + + duplicates = sorted({cid for cid in seen_ids if seen_ids.count(cid) > 1}) + if duplicates: + errors.append(f"duplicate claim ids: {duplicates}") + if corpus_ids and sorted(seen_ids) != sorted(corpus_ids): + missing = sorted(set(corpus_ids) - set(seen_ids)) + extra = sorted(set(seen_ids) - set(corpus_ids)) + if missing: + errors.append(f"claims missing from manifest: {missing}") + if extra: + errors.append(f"manifest claims not in corpus table: {extra}") + return errors + + +def _load_json(path: Path, errors: list[str]) -> object | None: + try: + return json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as error: + errors.append(f"{path}: unreadable JSON ({error})") + return None + + +def _git_head(repo_root: Path) -> str | None: + try: + return ( + subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=repo_root, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + or None + ) + except GIT_QUERY_ERRORS: + return None + + +def validate_tree( + plan_dir: Path, + *, + freshness_head: str | None = None, +) -> list[str]: + """Validate the manifest, evidence packets, and negative controls.""" + errors: list[str] = [] + manifest_path = plan_dir / "claims-manifest.json" + corpus_path = plan_dir / "citation-claim-corpus.md" + evidence_dir = plan_dir / "evidence" + negative_dir = evidence_dir / "negative-controls" + + corpus_ids: list[str] = [] + if corpus_path.is_file(): + corpus_ids = corpus_claim_ids(corpus_path.read_text(encoding="utf-8")) + if not corpus_ids: + errors.append(f"{corpus_path}: no CT-NNN rows found") + else: + errors.append(f"{corpus_path}: missing corpus document") + + manifest = _load_json(manifest_path, errors) + known_ids: set[str] = set() + lane_evidence: dict[str, tuple[str, str]] = {} + closed_lane_packets: set[str] = set() + if manifest is not None: + manifest_issues = manifest_errors(manifest, corpus_ids) + errors.extend(f"{manifest_path}: {issue}" for issue in manifest_issues) + if isinstance(manifest, dict) and isinstance(manifest.get("claims"), list): + for claim in manifest["claims"]: + if not isinstance(claim, dict): + continue + claim_id = claim.get("id") + if isinstance(claim_id, str): + known_ids.add(claim_id) + lanes = claim.get("lanes") + if not isinstance(lanes, dict): + continue + for lane_name, lane in lanes.items(): + if not isinstance(lane, dict): + continue + for rel in lane.get("evidence") or []: + if isinstance(rel, str): + lane_evidence[rel] = ( + str(claim_id), + str(lane_name), + ) + if lane.get("status") == "closed": + closed_lane_packets.add(rel) + + for rel, (claim_id, lane_name) in sorted(lane_evidence.items()): + packet_path = plan_dir / rel + if not packet_path.is_file(): + errors.append( + f"{manifest_path}: {claim_id}.lanes.{lane_name} references " + f"missing packet {rel}" + ) + + packet_paths = sorted(evidence_dir.glob("*.json")) if evidence_dir.is_dir() else [] + for packet_path in packet_paths: + packet = _load_json(packet_path, errors) + if packet is None: + continue + issues = packet_errors(packet, known_claim_ids=known_ids or None) + errors.extend(f"{packet_path}: {issue}" for issue in issues) + if issues or not isinstance(packet, dict): + continue + rel = packet_path.relative_to(plan_dir).as_posix() + linked = lane_evidence.get(rel) + if linked is None: + errors.append( + f"{packet_path}: not referenced by any manifest lane; every " + "packet needs a claim owner" + ) + else: + claim_id, lane_name = linked + if packet.get("claim_id") != claim_id: + errors.append( + f"{packet_path}: claim_id {packet.get('claim_id')} does " + f"not match manifest lane {claim_id}.{lane_name}" + ) + expected_branch = BRANCH_BY_LANE.get(lane_name) + branch = ( + packet.get("target", {}).get("branch") + if isinstance(packet.get("target"), dict) + else None + ) + if expected_branch is not None and branch != expected_branch: + errors.append( + f"{packet_path}: target.branch {branch!r} does not match " + f"lane {lane_name} branch {expected_branch!r}" + ) + if rel in closed_lane_packets: + passes = ( + packet.get("review", {}).get("passes") + if isinstance(packet.get("review"), dict) + else None + ) + if not isinstance(passes, list) or len(passes) < 2: + errors.append( + f"{packet_path}: a packet closing a lane needs at " + "least two recorded review passes" + ) + if freshness_head is not None: + commit = ( + packet.get("target", {}).get("commit") + if isinstance(packet.get("target"), dict) + else None + ) + if commit != freshness_head: + errors.append( + f"{packet_path}: target.commit {commit} is not the " + f"current HEAD {freshness_head} (--freshness)" + ) + + negative_paths = ( + sorted(negative_dir.glob("*.json")) if negative_dir.is_dir() else [] + ) + if not negative_paths: + errors.append( + f"{negative_dir}: at least one intentionally incomplete " + "negative-control packet is required to prove the validator " + "fails closed" + ) + for packet_path in negative_paths: + packet = _load_json(packet_path, errors) + if packet is None: + continue + issues = packet_errors(packet, known_claim_ids=known_ids or None) + if len(issues) < 3: + errors.append( + f"{packet_path}: negative control produced only " + f"{len(issues)} validation error(s); it must stay clearly " + "incomplete (>= 3) or the fail-closed proof is vacuous" + ) + + return errors + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--plan-dir", + type=Path, + default=PLAN_DIR, + help="Plan sidecar directory holding the manifest and evidence", + ) + parser.add_argument( + "--freshness", + action="store_true", + help="Require every evidence packet to record the current HEAD", + ) + return parser.parse_args() + + +def main() -> int: + args = parse_args() + freshness_head: str | None = None + if args.freshness: + freshness_head = _git_head(REPO_ROOT) + if freshness_head is None: + print( + "check_citation_evidence: --freshness requires a readable " "git HEAD", + file=sys.stderr, + ) + return 1 + errors = validate_tree(args.plan_dir, freshness_head=freshness_head) + if errors: + print( + f"check_citation_evidence: {len(errors)} error(s)", + file=sys.stderr, + ) + for error in errors: + print(f" - {error}", file=sys.stderr) + return 1 + print("check_citation_evidence: OK") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/write_citation_ct001_rolling_direction_packet.py b/scripts/write_citation_ct001_rolling_direction_packet.py new file mode 100644 index 0000000000000..e1c30276a0026 --- /dev/null +++ b/scripts/write_citation_ct001_rolling_direction_packet.py @@ -0,0 +1,516 @@ +#!/usr/bin/env python3 +"""Write the CT-001 rolling-direction evidence packet (PLAN-123 WS2). + +Bounded claim (corpus row CT-001): polyhedral friction can produce +direction-dependent rolling/sliding behavior. Under an isotropic Coulomb +friction model, a sphere launched sliding (no spin) on a horizontal plane +behaves identically for every launch direction; a friction pyramid aligned to +fixed tangent axes breaks that rotational symmetry. + +The fixture launches one sphere per run at speed `v0` along a swept in-plane +angle, without initial spin, on a static ground box, and measures per angle: + +- lateral drift from the launch ray and final-velocity heading error at the + measurement horizon (symmetry-breaking signals); +- slide-to-roll transition time and post-transition speed; +- mechanical-energy trajectory (friction during sliding must dissipate, never + inject, energy); +- `StepMetrics` numerics (max penetration, iterations, residual) and active + contact counts. + +Each (contact solver, angle) cell runs twice and must be bit-identical +(deterministic repeats). The packet records requested and resolved solver +identity; `ResolvedSolverConfiguration` is not yet Python-exposed, so the +resolved identity comes from World property readback plus step-profile stage +names, and that gap is a recorded limitation feeding PLAN-123 WS4. + +Usage (after `pixi run build`): + + PYTHONPATH=build/default/cpp/Release/python pixi run python \ + scripts/write_citation_ct001_rolling_direction_packet.py +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import platform +import subprocess +import sys +from pathlib import Path +from typing import Any + +import dartpy as sx +import numpy as np + +REPO_ROOT = Path(__file__).resolve().parents[1] +DEFAULT_OUTPUT = ( + REPO_ROOT + / "docs" + / "plans" + / "123-citation-driven-simulation-trust" + / "evidence" + / "CT-001-dart7-rolling-direction.json" +) + +SCENE_PARAMETERS: dict[str, Any] = { + "scene_id": "ct001_rolling_direction_sweep", + "description": ( + "One 1 kg sphere per run, radius 0.08 m, launched sliding (no spin) " + "at 1.0 m/s along a swept in-plane angle on a static ground box, " + "restitution 0, friction 0.35 on both bodies." + ), + "gravity_mps2": [0.0, 0.0, -9.81], + "time_step_s": 0.002, + "step_count": 500, + "sphere_radius_m": 0.08, + "sphere_mass_kg": 1.0, + "launch_speed_mps": 1.0, + "friction": 0.35, + "restitution": 0.0, + "ground_half_extents_m": [2.5, 2.5, 0.05], + "launch_angles_deg": [0.0, 15.0, 30.0, 45.0, 60.0, 75.0, 90.0], + "contact_solver_methods": ["SEQUENTIAL_IMPULSE", "BOXED_LCP"], + "deterministic_repeats": 2, + "slip_ratio_rolling_threshold": 0.02, +} + + +def scene_digest(parameters: dict[str, Any]) -> str: + canonical = json.dumps(parameters, sort_keys=True, separators=(",", ":")) + return "sha256:" + hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + +def _sphere_inertia(mass: float, radius: float) -> np.ndarray: + moment = 0.4 * mass * radius * radius + return np.diag([moment, moment, moment]) + + +def _transform_at(position: np.ndarray) -> np.ndarray: + transform = np.eye(4) + transform[:3, 3] = position + return transform + + +def run_single( + method_name: str, angle_deg: float, parameters: dict[str, Any] +) -> dict[str, Any]: + """Run one launch and return raw metrics plus a trajectory hash.""" + radius = float(parameters["sphere_radius_m"]) + mass = float(parameters["sphere_mass_kg"]) + speed = float(parameters["launch_speed_mps"]) + dt = float(parameters["time_step_s"]) + step_count = int(parameters["step_count"]) + ground_half = np.asarray(parameters["ground_half_extents_m"], dtype=float) + angle = math.radians(angle_deg) + direction = np.array([math.cos(angle), math.sin(angle), 0.0]) + + world = sx.World( + time_step=dt, + gravity=np.asarray(parameters["gravity_mps2"], dtype=float), + contact_solver_method=sx.ContactSolverMethod[method_name], + ) + world.step_profiling_enabled = True + + ground = world.add_rigid_body("ct001_ground") + ground.is_static = True + ground.set_collision_shape(sx.CollisionShape.box(ground_half)) + ground.transform = _transform_at(np.array([0.0, 0.0, -ground_half[2]])) + ground.friction = float(parameters["friction"]) + ground.restitution = float(parameters["restitution"]) + + sphere = world.add_rigid_body("ct001_sphere") + sphere.mass = mass + sphere.inertia = _sphere_inertia(mass, radius) + sphere.set_collision_shape(sx.CollisionShape.sphere(radius)) + sphere.friction = float(parameters["friction"]) + sphere.restitution = float(parameters["restitution"]) + start = np.array([0.0, 0.0, radius]) + sphere.transform = _transform_at(start) + sphere.linear_velocity = speed * direction + sphere.angular_velocity = (0.0, 0.0, 0.0) + + world.enter_simulation_mode() + + resolved = { + "contact_solver_method": world.contact_solver_method.name, + "rigid_body_solver": world.rigid_body_solver.name, + "gravity_mps2": [float(v) for v in np.asarray(world.gravity)], + "time_step_s": float(world.time_step), + } + + trajectory = hashlib.sha256() + slide_end_time = None + max_energy_gain = 0.0 + max_penetration = 0.0 + max_iterations = 0 + max_residual = 0.0 + contact_count_max = 0 + previous_energy = None + stage_names: list[str] = [] + + lateral_axis = np.array([-direction[1], direction[0], 0.0]) + for step_index in range(step_count): + world.step() + position = np.asarray(sphere.translation, dtype=float) + velocity = np.asarray(sphere.linear_velocity, dtype=float) + angular = np.asarray(sphere.angular_velocity, dtype=float) + trajectory.update(position.tobytes()) + trajectory.update(velocity.tobytes()) + trajectory.update(angular.tobytes()) + + metrics = world.compute_step_metrics() + kinetic = float(metrics.kinetic_energy) + if previous_energy is not None: + max_energy_gain = max(max_energy_gain, kinetic - previous_energy) + previous_energy = kinetic + max_penetration = max(max_penetration, float(metrics.max_penetration_depth)) + max_iterations = max(max_iterations, int(metrics.last_step_iterations)) + max_residual = max(max_residual, float(metrics.last_step_residual)) + contact_count_max = max(contact_count_max, int(metrics.active_contact_count)) + + if slide_end_time is None: + planar_speed = float(np.linalg.norm(velocity[:2])) + spin_axis = np.array([-direction[1], direction[0], 0.0]) + surface_speed = radius * float(np.dot(angular, spin_axis)) + slip = abs(planar_speed - surface_speed) + denom = planar_speed + abs(surface_speed) + 1.0e-9 + if slip / denom < float(parameters["slip_ratio_rolling_threshold"]): + slide_end_time = (step_index + 1) * dt + + if step_index == step_count - 1: + profile = world.last_step_profile + if not profile.is_empty(): + stage_names = [stage.name for stage in profile.stages] + + final_position = np.asarray(sphere.translation, dtype=float) + final_velocity = np.asarray(sphere.linear_velocity, dtype=float) + displacement = final_position - start + along = float(np.dot(displacement, direction)) + lateral = float(np.dot(displacement, lateral_axis)) + planar_speed = float(np.linalg.norm(final_velocity[:2])) + if planar_speed > 1.0e-9: + heading_error_deg = math.degrees( + math.atan2( + float(np.dot(final_velocity, lateral_axis)), + float(np.dot(final_velocity, direction)), + ) + ) + else: + heading_error_deg = 0.0 + + return { + "angle_deg": angle_deg, + "contact_solver_method": method_name, + "resolved": resolved, + "trajectory_sha256": trajectory.hexdigest(), + "along_travel_m": along, + "lateral_drift_m": lateral, + "heading_error_deg": heading_error_deg, + "final_planar_speed_mps": planar_speed, + "slide_end_time_s": slide_end_time, + "max_energy_gain_j": max_energy_gain, + "max_penetration_m": max_penetration, + "max_solver_iterations": max_iterations, + "max_solver_residual": max_residual, + "max_active_contacts": contact_count_max, + "final_step_stage_names": stage_names, + } + + +def summarize(rows: list[dict[str, Any]]) -> dict[str, Any]: + """Per-solver rotational-symmetry summary across the angle sweep.""" + summary: dict[str, Any] = {} + for method in SCENE_PARAMETERS["contact_solver_methods"]: + method_rows = [row for row in rows if row["contact_solver_method"] == method] + drifts = [abs(row["lateral_drift_m"]) for row in method_rows] + travels = [row["along_travel_m"] for row in method_rows] + headings = [abs(row["heading_error_deg"]) for row in method_rows] + travel_mean = sum(travels) / len(travels) + travel_spread = max(travels) - min(travels) + summary[method] = { + "runs": len(method_rows), + "max_abs_lateral_drift_m": max(drifts), + "max_abs_heading_error_deg": max(headings), + "travel_mean_m": travel_mean, + "travel_spread_m": travel_spread, + "travel_spread_relative": ( + travel_spread / travel_mean if travel_mean > 0.0 else 0.0 + ), + "max_energy_gain_j": max(row["max_energy_gain_j"] for row in method_rows), + "max_penetration_m": max(row["max_penetration_m"] for row in method_rows), + "max_solver_residual": max( + row["max_solver_residual"] for row in method_rows + ), + } + return summary + + +def git_head() -> str: + return subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=REPO_ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + + +def build_packet() -> dict[str, Any]: + parameters = SCENE_PARAMETERS + rows: list[dict[str, Any]] = [] + determinism_failures: list[str] = [] + resolved_by_method: dict[str, dict[str, Any]] = {} + + for method in parameters["contact_solver_methods"]: + for angle in parameters["launch_angles_deg"]: + repeats = [ + run_single(method, angle, parameters) + for _ in range(int(parameters["deterministic_repeats"])) + ] + hashes = {run["trajectory_sha256"] for run in repeats} + if len(hashes) != 1: + determinism_failures.append( + f"{method} angle {angle}: trajectory hashes differ " + f"{sorted(hashes)}" + ) + row = repeats[0] + readback = row["resolved"] + if readback["contact_solver_method"] != method: + raise SystemExit( + f"requested contact solver {method} but World readback " + f"reports {readback['contact_solver_method']}" + ) + resolved_by_method.setdefault(method, readback) + rows.append(row) + + if determinism_failures: + raise SystemExit( + "deterministic repeats failed:\n " + "\n ".join(determinism_failures) + ) + + summary = summarize(rows) + isotropy_tolerance = { + "max_abs_lateral_drift_m": 1.0e-4, + "max_abs_heading_error_deg": 0.1, + "travel_spread_relative": 0.01, + } + anisotropic_methods = sorted( + method + for method, stats in summary.items() + if stats["max_abs_lateral_drift_m"] + > isotropy_tolerance["max_abs_lateral_drift_m"] + or stats["max_abs_heading_error_deg"] + > isotropy_tolerance["max_abs_heading_error_deg"] + or stats["travel_spread_relative"] + > isotropy_tolerance["travel_spread_relative"] + ) + disposition = "reproduced" if anisotropic_methods else "unresolved" + + command = ( + "PYTHONPATH=build/default/cpp/Release/python pixi run python " + "scripts/write_citation_ct001_rolling_direction_packet.py" + ) + packet: dict[str, Any] = { + "schema": "dart.citation_claim_evidence/v1", + "claim_id": "CT-001", + "title": "Rolling-direction friction dependence (DART 7 first packet)", + "source": { + "url": "https://leggedrobotics.github.io/SimBenchmark/", + "claim": ( + "Polyhedral friction can produce direction-dependent " + "rolling/sliding behavior." + ), + }, + "target": { + "branch": "main", + "commit": git_head(), + }, + "scene": { + "id": parameters["scene_id"], + "digest": scene_digest(parameters), + "description": parameters["description"], + "parameters": parameters, + }, + "configuration": { + "requested": { + "contact_solver_method_sweep": parameters["contact_solver_methods"], + "rigid_body_solver": "SEQUENTIAL_IMPULSE (World default)", + "integrator": "World default semi-implicit stepping", + "precision": "float64", + "backend": "cpu", + "threads": "World default sequential step", + }, + "resolved": { + "by_contact_solver_method": resolved_by_method, + }, + "resolved_provenance": ( + "World property readback (contact_solver_method, " + "rigid_body_solver, gravity, time_step) after " + "enter_simulation_mode plus final-step WorldStepProfile " + "stage names recorded per run; " + "World::getResolvedConfiguration() is not yet exposed to " + "Python (PLAN-123 WS4 follow-up)." + ), + "detector": ( + "DART 7 native World collision pipeline (the World step API " + "exposes no detector selection on main)" + ), + "timestep": parameters["time_step_s"], + "substeps": 1, + "iterations": ( + "World defaults; per-step actual iterations recorded via " + "StepMetrics.last_step_iterations" + ), + "fallback_policy": ( + "World defaults; no per-island fallback reporting is exposed " + "on main (PLAN-123 WS4 follow-up)" + ), + }, + "ensemble": { + "kind": "parameter-sweep-with-deterministic-repeats", + "sweep": [ + {"contact_solver_method": method, "angle_deg": angle} + for method in parameters["contact_solver_methods"] + for angle in parameters["launch_angles_deg"] + ], + "deterministic_repeats": int(parameters["deterministic_repeats"]), + "deterministic_repeats_identical": True, + "measurement_window": { + "start_s": 0.0, + "end_s": parameters["time_step_s"] * parameters["step_count"], + "steps": parameters["step_count"], + }, + }, + "metrics": { + "physical": { + "method": ( + "Rotational-symmetry sweep: per-angle lateral drift from " + "the launch ray, final-velocity heading error, along-ray " + "travel, slide-to-roll transition time, and max " + "single-step kinetic-energy gain from " + "StepMetrics.kinetic_energy" + ), + "per_solver_summary": summary, + "isotropy_tolerance": isotropy_tolerance, + "anisotropic_methods": anisotropic_methods, + }, + "numerical": { + "method": ( + "Max over run of StepMetrics.max_penetration_depth, " + "last_step_iterations, last_step_residual, " + "active_contact_count" + ), + "max_penetration_m": max(row["max_penetration_m"] for row in rows), + "max_solver_iterations": max( + row["max_solver_iterations"] for row in rows + ), + "max_solver_residual": max(row["max_solver_residual"] for row in rows), + "max_active_contacts": max(row["max_active_contacts"] for row in rows), + }, + "performance": { + "status": "unsupported", + "reason": ( + "This packet makes no timing claim; interleaved " + "same-host methodology is required before any " + "performance row" + ), + }, + "allocation": { + "status": "unsupported", + "reason": ( + "Post-bake allocation gates are owned by PLAN-122 " + "tooling; this packet does not measure allocations" + ), + }, + }, + "evidence": { + "commands": [command], + "raw_rows": rows, + "visual": { + "status": "not-applicable", + "reason": ( + "The oracle is numeric rotational symmetry; per-angle " + "trajectories have no visual claim beyond the recorded " + "metrics. Visual capture joins when a corpus row makes " + "a visible-behavior claim." + ), + }, + }, + "result": { + "disposition": disposition, + "claim_boundary": ( + "DART 7 main, this commit, one sphere sliding to rolling on " + "a static ground box at v0=1 m/s, mu=0.35, dt=2 ms, 1 s " + "horizon, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers, " + "launch angles 0-90 deg in 15 deg steps. Says nothing about " + "other speeds, shapes, stacks, historical DART versions, or " + "DART 6." + ), + "limitations": [ + "ResolvedSolverConfiguration is not Python-exposed; " + "resolved identity is property readback plus stage names.", + "Friction-cone/complementarity residual per contact is not " + "exposed; solver residual is the aggregate StepMetrics " + "value.", + "The sweep covers 0-90 deg; pyramid orientation with a " + "period other than 90 deg would need a wider sweep.", + "No high-mass-ratio or stacked variant yet; those belong " + "to CT-002/CT-007 fixtures.", + "SEQUENTIAL_IMPULSE and BOXED_LCP produce metric summaries " + "that agree to printed precision on this single-contact " + "scene while their full trajectory hashes differ; this " + "packet is not a solver comparison and must not be quoted " + "as one.", + ], + }, + "review": {"passes": []}, + "host": { + "platform": platform.platform(), + "python": sys.version.split()[0], + "machine": platform.machine(), + "performance_valid": False, + "note": ( + "Host recorded for provenance only; no timing methodology " + "was applied" + ), + }, + } + return packet + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--output", + type=Path, + default=DEFAULT_OUTPUT, + help=f"Packet output path (default: {DEFAULT_OUTPUT})", + ) + return parser.parse_args() + + +def main() -> int: + args = parse_args() + packet = build_packet() + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text( + json.dumps(packet, indent=2, sort_keys=True) + "\n", encoding="utf-8" + ) + summary = packet["metrics"]["physical"]["per_solver_summary"] + print(f"wrote {args.output}") + for method, stats in summary.items(): + print( + f" {method}: max |lateral drift| " + f"{stats['max_abs_lateral_drift_m']:.3e} m, max |heading err| " + f"{stats['max_abs_heading_error_deg']:.3e} deg, travel spread " + f"{stats['travel_spread_relative']:.3e} rel" + ) + print(f" disposition: {packet['result']['disposition']}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py new file mode 100644 index 0000000000000..dbd11cda7a326 --- /dev/null +++ b/tests/test_check_citation_evidence.py @@ -0,0 +1,420 @@ +"""Tests for the PLAN-123 citation evidence validator (fail-closed contract). + +Every required provenance field must fail validation when missing or +degraded; a complete packet must pass; negative-control packets must fail. +""" + +import copy +import importlib.util +import json +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +SCRIPT = ROOT / "scripts" / "check_citation_evidence.py" +PLAN_DIR = ROOT / "docs" / "plans" / "123-citation-driven-simulation-trust" + + +def _load_module(): + spec = importlib.util.spec_from_file_location("check_citation_evidence", SCRIPT) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + sys.modules[spec.name] = module + spec.loader.exec_module(module) + return module + + +MODULE = _load_module() + + +def complete_packet() -> dict: + return { + "schema": "dart.citation_claim_evidence/v1", + "claim_id": "CT-001", + "title": "Test packet", + "source": {"url": "https://example.org/claim", "claim": "A claim."}, + "target": {"branch": "main", "commit": "0" * 40}, + "scene": { + "id": "test_scene", + "digest": "sha256:" + "a" * 64, + "description": "A scene.", + }, + "configuration": { + "requested": {"contact_solver_method": "BOXED_LCP"}, + "resolved": {"contact_solver_method": "BOXED_LCP"}, + "resolved_provenance": "World property readback", + "detector": "dart7 native pipeline", + "timestep": 0.002, + "substeps": 1, + "iterations": "defaults", + "fallback_policy": "World defaults; none observed", + }, + "ensemble": { + "kind": "deterministic-repeats", + "deterministic_repeats": 2, + "measurement_window": {"start_s": 0.0, "end_s": 1.0}, + }, + "metrics": { + "physical": {"method": "sweep", "lateral_drift_m": 0.0}, + "numerical": {"method": "step metrics", "max_penetration_m": 1e-5}, + "performance": { + "status": "unsupported", + "reason": "no timing methodology", + }, + "allocation": { + "status": "unsupported", + "reason": "PLAN-122 tooling owns allocation gates", + }, + }, + "evidence": { + "commands": ["pixi run python scripts/example.py"], + "raw_rows": [{"angle_deg": 0.0, "lateral_drift_m": 0.0}], + "visual": { + "status": "not-applicable", + "reason": "numeric oracle only", + }, + }, + "result": { + "disposition": "unresolved", + "claim_boundary": "This commit, this scene only.", + "limitations": ["Single fixture."], + }, + "review": {"passes": []}, + } + + +def test_complete_packet_passes(): + assert MODULE.packet_errors(complete_packet()) == [] + + +def test_non_object_packet_fails(): + assert MODULE.packet_errors([1, 2, 3]) == ["packet must be a JSON object"] + + +@pytest.mark.parametrize( + "path", + [ + ("target", "commit"), + ("target", "branch"), + ("scene", "digest"), + ("configuration", "requested"), + ("configuration", "resolved"), + ("configuration", "resolved_provenance"), + ("configuration", "detector"), + ("configuration", "timestep"), + ("configuration", "fallback_policy"), + ("ensemble", "measurement_window"), + ("evidence", "commands"), + ("result", "disposition"), + ("result", "claim_boundary"), + ("result", "limitations"), + ], +) +def test_each_required_field_fails_closed(path): + packet = complete_packet() + section, field = path + del packet[section][field] + errors = MODULE.packet_errors(packet) + assert errors, f"deleting {section}.{field} must fail validation" + assert any(field in error for error in errors) + + +@pytest.mark.parametrize( + "section", + [ + "source", + "target", + "scene", + "configuration", + "ensemble", + "metrics", + "evidence", + "result", + "review", + ], +) +def test_each_required_section_fails_closed(section): + packet = complete_packet() + del packet[section] + errors = MODULE.packet_errors(packet) + assert any("missing required top-level keys" in error for error in errors) + + +def test_unknown_top_level_key_fails(): + packet = complete_packet() + packet["extra_notes"] = "sneaky" + errors = MODULE.packet_errors(packet) + assert any("unknown top-level keys" in error for error in errors) + + +def test_short_commit_fails(): + packet = complete_packet() + packet["target"]["commit"] = "abc123" + assert any("40-hex" in error for error in MODULE.packet_errors(packet)) + + +def test_single_run_ensemble_fails(): + packet = complete_packet() + packet["ensemble"]["deterministic_repeats"] = 1 + errors = MODULE.packet_errors(packet) + assert any("single runs are not evidence" in error for error in errors) + + +def test_unsupported_metric_requires_reason(): + packet = complete_packet() + packet["metrics"]["performance"] = {"status": "unsupported"} + errors = MODULE.packet_errors(packet) + assert any("no non-empty reason" in error for error in errors) + + +def test_measured_metric_requires_method(): + packet = complete_packet() + packet["metrics"]["physical"] = {"lateral_drift_m": 0.0} + errors = MODULE.packet_errors(packet) + assert any("measurement 'method'" in error for error in errors) + + +def test_unsupported_metric_cannot_carry_values(): + packet = complete_packet() + packet["metrics"]["performance"] = { + "status": "unsupported", + "reason": "no methodology", + "wall_time_ms": 0.0, + } + errors = MODULE.packet_errors(packet) + assert any("mixes unsupported status" in error for error in errors) + + +def test_null_metric_value_fails(): + packet = complete_packet() + packet["metrics"]["numerical"]["max_penetration_m"] = None + errors = MODULE.packet_errors(packet) + assert any("contains null" in error for error in errors) + + +def test_nan_metric_value_fails(): + packet = complete_packet() + packet["metrics"]["numerical"]["max_penetration_m"] = float("nan") + errors = MODULE.packet_errors(packet) + assert any("non-finite" in error for error in errors) + + +def test_missing_metric_group_fails(): + packet = complete_packet() + del packet["metrics"]["allocation"] + errors = MODULE.packet_errors(packet) + assert any("metrics.allocation is required" in error for error in errors) + + +def test_untyped_visual_fails(): + packet = complete_packet() + packet["evidence"]["visual"] = [] + errors = MODULE.packet_errors(packet) + assert any("visual" in error for error in errors) + + +def test_invalid_disposition_fails(): + packet = complete_packet() + packet["result"]["disposition"] = "looks-fine" + errors = MODULE.packet_errors(packet) + assert any("disposition" in error for error in errors) + + +def test_unknown_claim_id_fails_when_manifest_known(): + packet = complete_packet() + errors = MODULE.packet_errors(packet, known_claim_ids={"CT-999"}) + assert any("not in the claims manifest" in error for error in errors) + + +def test_review_pass_needs_reviewer_and_summary(): + packet = complete_packet() + packet["review"]["passes"] = [{"reviewer": "someone"}] + errors = MODULE.packet_errors(packet) + assert any("review.passes[0]" in error for error in errors) + + +def _minimal_manifest(ids): + return { + "schema": "dart.citation_claim_manifest/v1", + "first_wave_families": [ + "family_a", + "family_b", + "family_c", + "family_d", + "family_e", + "family_f", + ], + "claims": [ + { + "id": claim_id, + "title": f"Claim {claim_id}", + "source": "somewhere", + "first_wave_family": None, + "lanes": { + "dart7": { + "owner": "PLAN-123", + "status": "audit-required", + "disposition": None, + "evidence": [], + }, + "dart6": { + "owner": "dev task", + "status": "audit-required", + "disposition": None, + "evidence": [], + }, + }, + } + for claim_id in ids + ], + } + + +def test_manifest_matches_corpus_ids(): + manifest = _minimal_manifest(["CT-001", "CT-002"]) + assert MODULE.manifest_errors(manifest, ["CT-001", "CT-002"]) == [] + errors = MODULE.manifest_errors(manifest, ["CT-001", "CT-002", "CT-003"]) + assert any("missing from manifest" in error for error in errors) + + +def test_manifest_rejects_wrong_family_count(): + manifest = _minimal_manifest(["CT-001"]) + manifest["first_wave_families"] = ["only_one"] + errors = MODULE.manifest_errors(manifest, ["CT-001"]) + assert any("exactly 6" in error for error in errors) + + +def test_manifest_rejects_unknown_family_reference(): + manifest = _minimal_manifest(["CT-001"]) + manifest["claims"][0]["first_wave_family"] = "family_x" + errors = MODULE.manifest_errors(manifest, ["CT-001"]) + assert any("capped family list" in error for error in errors) + + +def test_closed_lane_requires_disposition_and_evidence(): + manifest = _minimal_manifest(["CT-001"]) + manifest["claims"][0]["lanes"]["dart7"]["status"] = "closed" + errors = MODULE.manifest_errors(manifest, ["CT-001"]) + assert any("without a valid disposition" in error for error in errors) + assert any("prose cannot close a row" in error for error in errors) + + +def test_not_applicable_lane_requires_reason(): + manifest = _minimal_manifest(["CT-001"]) + manifest["claims"][0]["lanes"]["dart6"]["status"] = "not-applicable" + errors = MODULE.manifest_errors(manifest, ["CT-001"]) + assert any("must record a reason" in error for error in errors) + + +def test_corpus_claim_ids_parses_table_rows(): + text = "| CT-001 | x |\n| CT-002 | y |\nno row\n| CT-002 | dup |\n" + assert MODULE.corpus_claim_ids(text) == ["CT-001", "CT-002"] + + +def _write_tree(tmp_path, *, packet=None, negative=None, manifest=None): + plan_dir = tmp_path / "plan" + evidence = plan_dir / "evidence" + negative_dir = evidence / "negative-controls" + negative_dir.mkdir(parents=True) + (plan_dir / "citation-claim-corpus.md").write_text( + "| CT-001 | claim |\n", encoding="utf-8" + ) + if manifest is None: + manifest = _minimal_manifest(["CT-001"]) + if packet is not None: + manifest["claims"][0]["lanes"]["dart7"]["status"] = "in-progress" + manifest["claims"][0]["lanes"]["dart7"]["evidence"] = [ + "evidence/packet.json" + ] + (plan_dir / "claims-manifest.json").write_text( + json.dumps(manifest), encoding="utf-8" + ) + if packet is not None: + (evidence / "packet.json").write_text(json.dumps(packet), encoding="utf-8") + if negative is not None: + (negative_dir / "incomplete.json").write_text( + json.dumps(negative), encoding="utf-8" + ) + return plan_dir + + +def test_validate_tree_accepts_complete_state(tmp_path): + incomplete = {"schema": "dart.citation_claim_evidence/v1"} + plan_dir = _write_tree(tmp_path, packet=complete_packet(), negative=incomplete) + assert MODULE.validate_tree(plan_dir) == [] + + +def test_validate_tree_requires_negative_control(tmp_path): + plan_dir = _write_tree(tmp_path, packet=complete_packet()) + errors = MODULE.validate_tree(plan_dir) + assert any("negative-control" in error for error in errors) + + +def test_validate_tree_rejects_passing_negative_control(tmp_path): + plan_dir = _write_tree( + tmp_path, packet=complete_packet(), negative=complete_packet() + ) + errors = MODULE.validate_tree(plan_dir) + assert any("fail-closed proof is vacuous" in error for error in errors) + + +def test_validate_tree_rejects_unreferenced_packet(tmp_path): + manifest = _minimal_manifest(["CT-001"]) + plan_dir = _write_tree( + tmp_path, + packet=complete_packet(), + negative={"schema": "x"}, + manifest=manifest, + ) + errors = MODULE.validate_tree(plan_dir) + assert any("not referenced by any manifest lane" in error for error in errors) + + +def test_validate_tree_rejects_branch_lane_mismatch(tmp_path): + packet = copy.deepcopy(complete_packet()) + packet["target"]["branch"] = "release-6.20" + plan_dir = _write_tree(tmp_path, packet=packet, negative={"schema": "x"}) + errors = MODULE.validate_tree(plan_dir) + assert any("does not match lane dart7" in error for error in errors) + + +def test_validate_tree_rejects_missing_referenced_packet(tmp_path): + manifest = _minimal_manifest(["CT-001"]) + manifest["claims"][0]["lanes"]["dart7"]["evidence"] = ["evidence/ghost.json"] + plan_dir = _write_tree(tmp_path, negative={"schema": "x"}, manifest=manifest) + errors = MODULE.validate_tree(plan_dir) + assert any("missing packet" in error for error in errors) + + +def test_validate_tree_closed_lane_needs_two_review_passes(tmp_path): + packet = copy.deepcopy(complete_packet()) + packet["review"]["passes"] = [{"reviewer": "first", "summary": "clean"}] + manifest = _minimal_manifest(["CT-001"]) + lane = manifest["claims"][0]["lanes"]["dart7"] + lane["status"] = "closed" + lane["disposition"] = "reproduced" + lane["evidence"] = ["evidence/packet.json"] + plan_dir = _write_tree( + tmp_path, packet=packet, negative={"schema": "x"}, manifest=manifest + ) + errors = MODULE.validate_tree(plan_dir) + assert any("at least two recorded review passes" in error for error in errors) + packet["review"]["passes"].append({"reviewer": "second", "summary": "clean"}) + (plan_dir / "evidence" / "packet.json").write_text( + json.dumps(packet), encoding="utf-8" + ) + assert MODULE.validate_tree(plan_dir) == [] + + +def test_validate_tree_freshness_flags_stale_commit(tmp_path): + plan_dir = _write_tree(tmp_path, packet=complete_packet(), negative={"schema": "x"}) + errors = MODULE.validate_tree(plan_dir, freshness_head="1" * 40) + assert any("--freshness" in error for error in errors) + + +def test_repository_tree_validates(): + """The committed manifest, packets, and negative controls must pass.""" + errors = MODULE.validate_tree(PLAN_DIR) + assert errors == [] From ac0c074d637304fa7c02aa457d123d54e63f738e Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Fri, 14 Aug 2026 18:05:58 -0700 Subject: [PATCH 02/52] Harden the citation evidence gate and add the dense-contact packets Address two independent review passes on d0cb640860d: - Close the bypass where a manifest lane could satisfy the evidence check with a file the packet checks never reached: every lane-referenced path is now validated wherever it sits, and prose, non-JSON files, paths outside evidence/, negative controls, and duplicate packet owners are rejected. - Make "unsupported is never silently zero" an enforced rule rather than a promise: every exact zero inside a measured metric group must be typed {status: unsupported, reason} or declared in measured_zero_fields, and spelled placeholders, nulls, NaNs, dangling raw_paths, empty measurement windows, and empty-container metrics now fail. - Fix the packets that violated it. StepMetrics.last_step_residual is structurally zero on the rigid contact path (recordSolverDiagnostics takes residual = 0.0 and no rigid call site passes one) and the boxed-LCP branch records no iteration count at all, so both are now typed unsupported with their source citations in scripts/citation_packet_utils.py instead of being published as measurements. - Distinguish mechanism from deviation in CT-001: record which isotropy criterion fired, test the antisymmetric friction-pyramid signature, and gate the disposition on physical validity so a degenerate run cannot strengthen a "reproduced" verdict. - Add CT-002 dense inelastic contact (reproduced: sequential impulse at dt=4 ms is still moving at the 2 s horizon) and CT-003 dense elastic contact (unresolved: no energy injection and no solver failure, recorded as an honest negative result rather than a refutation of the original report). Tests grow to 68 cases covering the bypass and each new rule. --- CHANGELOG.md | 8 +- .../README.md | 25 +- .../RESUME.md | 32 +- .../verification.md | 70 ++- .../citation-claim-corpus.md | 110 ++-- .../claims-manifest.json | 15 +- .../CT-001-dart7-rolling-direction.json | 163 +++++- .../CT-002-dart7-dense-inelastic-contact.json | 309 +++++++++++ .../CT-003-dart7-dense-elastic-contact.json | 293 ++++++++++ docs/plans/dashboard.md | 18 +- scripts/check_citation_evidence.py | 204 ++++++- scripts/citation_packet_utils.py | 95 ++++ ...citation_ct001_rolling_direction_packet.py | 216 ++++++-- ...ite_citation_ct002_dense_contact_packet.py | 514 ++++++++++++++++++ ...e_citation_ct003_elastic_contact_packet.py | 492 +++++++++++++++++ tests/test_check_citation_evidence.py | 155 +++++- 16 files changed, 2545 insertions(+), 174 deletions(-) create mode 100644 docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json create mode 100644 docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json create mode 100644 scripts/citation_packet_utils.py create mode 100644 scripts/write_citation_ct002_dense_contact_packet.py create mode 100644 scripts/write_citation_ct003_elastic_contact_packet.py diff --git a/CHANGELOG.md b/CHANGELOG.md index b18611d8487eb..dd7e039d3df77 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -613,9 +613,11 @@ compatibility remains on the active DART 6 LTS branch._ requested/resolved solver identity, command, ensemble, disposition, claim boundary, or typed-unsupported metrics fail validation, and a permanent intentionally incomplete negative-control packet must keep failing), and the - first branch-qualified CT-001 rolling-direction evidence packet showing the - friction-pyramid direction dependence on `main` under both Sequential - Impulse and boxed-LCP contact solvers. + first branch-qualified evidence packets: CT-001 rolling-direction (the + friction-pyramid direction dependence reproduces on `main` under both + Sequential Impulse and boxed-LCP contact solvers), CT-002 dense inelastic + contact, and CT-003 dense elastic contact (no energy injection observed, + recorded as an honest negative result rather than a reproduction). - Reorganized tests and CI coverage around DART 7 components, with focused unit, integration, benchmark, rendering, CUDA-smoke, collision, and simulation gates replacing broad stale test targets. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/README.md b/docs/dev_tasks/citation_driven_simulation_trust/README.md index f5fa21be3070d..485d7fd55fc61 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/README.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/README.md @@ -10,9 +10,14 @@ negative-control fixtures (`claims-manifest.json`, `scripts/check_citation_evidence.py`, `pixi run check-citation-evidence` in `check-lint`, permanent negative control, CT-001 packet). -- [ ] Phase 2: Reproduce and classify the six-family first-wave corpus - (CT-001 `main` packet landed with disposition `reproduced`; remaining - families and all DART 6 lanes open). +- [ ] Phase 2: Reproduce and classify the six-family first-wave corpus. + Landed on `main`: CT-001 rolling/friction-direction (`reproduced`), + CT-002 dense inelastic contact (`reproduced`), CT-003 dense elastic + contact (`unresolved` -- no energy injection observed). Landed on + `release-6.20`: CT-001 detector sweep (`reproduced` on fcl/dart/ode; + bullet excluded for lacking the pyramid signature). Open: articulated + energy/momentum/control, heel-strike/toe-off, high mass-ratio, and + reset/concurrency families. - [ ] Phase 3: Land contact identity and impulse/wrench semantics. - [ ] Phase 4: Extend resolved-method and cross-family solve diagnostics. - [ ] Phase 5: Record exact-cone GO/NO-GO; implement only after GO. @@ -151,8 +156,12 @@ for downstream-sensitive changes. ## Immediate next steps 1. Read `RESUME.md` and verify both branch tips/worktrees. -2. Audit current code/docs/PRs against every initial corpus row. -3. Integrate PLAN-123/dashboard/index without clobbering newer state. -4. Implement the smallest checked manifest plus one intentionally incomplete - fixture that proves the validator fails closed. -5. Select the first three corpus rows using reuse and risk, not novelty. +2. Continue Phase 2 with the articulated energy/momentum family (CT-004), + reusing `StepMetrics` and the PLAN-084 variational integrator, then the + controlled-articulated row (CT-005) seeded by + `python/examples/demos/scenes/atlas_simbicon.py`. +3. Start WS4's first slice in parallel where it unblocks packets: expose + `ResolvedSolverConfiguration` to Python and a comparable per-solve + residual, both currently typed unsupported in every packet. +4. Reuse PR #3377 fixtures for the DART 6 CT-007 lane rather than adding + new scenes; keep PLAN-621/622 rows referenced, not copied. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md index 2ec52f4f7521a..0fe9099c021cf 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md @@ -13,14 +13,18 @@ statuses were verified; the corpus sidecar now carries the WS0 audit record. WP-PG.50) and issue #3056 routed as row owners, existing demos scenes recorded as fixture seeds. - PLAN-123 integrated into `docs/plans/dashboard.md` and `docs/plans/README.md`. -- WS1 landed on the DART 7 branch: `claims-manifest.json` (20 rows), - fail-closed `scripts/check_citation_evidence.py` - (`pixi run check-citation-evidence`, wired into `check-lint`), - `tests/test_check_citation_evidence.py`, a permanent intentionally - incomplete negative-control packet, and the first complete CT-001 - rolling-direction packet (disposition `reproduced`: friction-pyramid - lateral drift up to 2.1 mm with zero drift at 0/45/90 deg, identical-run - determinism, per-method requested/resolved readback). +- WS1 landed on both branches: checked claims manifests, fail-closed + `check-citation-evidence` gates wired into `check-lint`, permanent + intentionally incomplete negative controls, and pytest suites (68 cases on + `main`, 48 on `release-6.20`). +- Two independent role-separated reviews ran per branch. All four returned + "not clean". Every finding was verified against source and fixed; the + substantive ones were a validator bypass (a lane could close a row with a + file the packet checks never reached), unsupported-as-zero metrics in this + program's own packets, and a DART 6 attribution error crediting bullet with + a friction-pyramid signature its data does not show. See `verification.md`. +- First-wave packets landed: CT-001 (both branches), CT-002 and CT-003 + (`main`). ## Current branches @@ -35,13 +39,11 @@ approval. ## Immediate next step -Finish the DART 6 Phase 1 adoption on `release-6.20`: branch-local -`check_citation_evidence.py` + manifest + negative control (no DART 7 API -backports), then run branch gates on both worktrees and record reviews. - -After that, Phase 2 continues with the next first-wave family on `main` -(dense inelastic/elastic contact, CT-002/CT-003) reusing -`rigid_restitution_ladder` and a 6x6x6 grid fixture. +Continue Phase 2 on `main` with the articulated energy/momentum family +(CT-004): a passive multi-link chain swept over timestep, using +`StepMetrics.total_energy`/`angular_momentum` as the oracle and the PLAN-084 +variational integrator as the comparison arm, written with +`scripts/citation_packet_utils.py` so unsupported quantities stay typed. ## Context that would be lost diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index f32d95a0da67c..9a07c56a57fdd 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -23,24 +23,84 @@ scripts/write_citation_ct001_rolling_direction_packet.py` — wrote the - Fail-closed demo: copying the negative control into `evidence/` in a scratch tree produced 15 errors and exit 1. - `pixi run python -I scripts/run_pytest.py -tests/test_check_citation_evidence.py -q` — 53 passed. +tests/test_check_citation_evidence.py -q` — 68 passed. - `--freshness` run — exit 0 (packet commit == HEAD). - Negative control: permanent `evidence/negative-controls/` packet (missing resolved identity/digest/commands, single-run ensemble, unsupported-as-zero metric, no claim boundary) fails with >= 3 errors and the gate enforces that it keeps failing. - Determinism evidence: per-cell trajectory SHA-256 equal across 2 repeats - for all 28 cells; cross-invocation reruns reproduced identical summary + for all 14 sweep cells (2 repeats each, 28 runs); cross-invocation reruns reproduced identical summary metrics. - Performance/allocation: explicitly typed unsupported in the packet (no timing/allocation claims made). - Visual evidence: typed not-applicable (numeric rotational-symmetry oracle); no visible-behavior claim in this packet. -- Review passes: recorded below once completed for the slice head. -- Known gaps: DART 6 adoption slice pending; remaining first-wave families - pending; `ResolvedSolverConfiguration` not yet Python-exposed. +- Review passes (two independent, role-separated, on commit d0cb640860d): + a tooling/correctness pass and a physics/evidence-honesty pass. Both + returned "not clean"; every substantive finding was verified against source + before acting. Confirmed and fixed in the follow-up slice below: + 1. BLOCKER — `evidence_dir.glob("*.json")` was non-recursive, so a lane + could close a row with prose, an empty object, or the negative control + itself one directory deeper. Reproduced in a scratch tree (zero errors + reported for four bypass variants). + 2. MAJOR — "rejects unsupported-as-zero" was asserted in four places and + implemented in none, and the shipped CT-001 packet committed exactly + that: `max_solver_residual: 0.0` is structurally never computed + (`recordSolverDiagnostics(World&, size_t, double residual = 0.0)` with + no rigid call site passing a residual, verified at + `dart/simulation/compute/rigid_body_contact_stage.cpp:165,736,874,926`), + and BOXED_LCP records no iteration count at all (its branch returns + before the diagnostics call, verified at the same file, line 898). + 3. MAJOR — `evidence.raw_paths` was never checked for existence. + 4. MAJOR — `target.commit` is the parent commit (the writer stamps HEAD + before the packet is committed); the measured library state is that + commit, now stated explicitly in `target.commit_role`. + 5. MINOR — packet did not say which isotropy criterion fired; resolved + provenance credited stage names that do not discriminate; repeat hashes + were not recorded; corpus packet skeleton did not match the real schema. +- Known gaps: remaining first-wave families (CT-004 articulated, + CT-005 control, CT-006 heel-strike, CT-011 reset/concurrency); + `ResolvedSolverConfiguration` not yet Python-exposed (WS4). - Changelog: entry added under "Tests, Benchmarks, and Quality Gates". +## Review-fix slice — 2026-08-14 + +- What changed: validator hardened (recursive plus lane-referenced packet + validation, negative-control and non-JSON lane references rejected, + duplicate packet owners rejected, scalar `evidence` guarded, `raw_paths` + existence, non-empty measurement window, spelled-placeholder rejection, + empty-container rejection, and the enforced zero rule: every exact zero in + a measured group must be typed unsupported with a reason or declared in + `measured_zero_fields`); `scripts/citation_packet_utils.py` centralizes the + typed-unsupported markers with their source citations; CT-001 regenerated + with honest metric typing, per-criterion anisotropy findings, a pyramid + antisymmetry-signature test, a physical-validity gate on the disposition, + and recorded per-repeat hashes; CT-002 and CT-003 packets added. +- Commands and results: + - `pixi run check-citation-evidence` — first run after hardening reported + 3 errors, all in this program's own packets (the exact unsupported-as-zero + defect the reviews named); OK after the packets were corrected. + - `pixi run python -I scripts/run_pytest.py +tests/test_check_citation_evidence.py -q` — 68 passed (15 new cases + covering the blocker and each new rule). + - `PYTHONPATH=build/default/cpp/Release/python pixi run python +scripts/write_citation_ct002_dense_contact_packet.py` — all cells finite; + max penetration 2.047e-2 m; settled max speed 8.066e-2 m/s; max + post-settle energy gain 9.594e-10 J; one unstable cell + (SEQUENTIAL_IMPULSE at dt=4 ms still moving at the horizon) → + disposition `reproduced`. + - `PYTHONPATH=... scripts/write_citation_ct003_elastic_contact_packet.py` — + all cells finite, energy-envelope excess exactly 0.0 J against a + 1.801e-4 J tolerance → disposition `unresolved`: the cited energy + injection did NOT occur in this bounded reconstruction, which is recorded + as a negative result rather than a refutation of the original report. +- Negative control: still fails (>= 3 errors) and is now also rejected if a + lane references it as evidence. +- Known gaps after this slice: CT-002/CT-003 are bounded reconstructions, not + source-exact SimBenchmark scenes; per-contact cone/complementarity metrics + remain unavailable until WS4. + ## Bootstrap record — 2026-08-14 ### What changed diff --git a/docs/plans/123-citation-driven-simulation-trust/citation-claim-corpus.md b/docs/plans/123-citation-driven-simulation-trust/citation-claim-corpus.md index edce2fec110d3..c8db987e5cce4 100644 --- a/docs/plans/123-citation-driven-simulation-trust/citation-claim-corpus.md +++ b/docs/plans/123-citation-driven-simulation-trust/citation-claim-corpus.md @@ -7,9 +7,11 @@ PLAN-123. It is not a bibliography, sentiment table, or promise to reproduce every citation. A row exists only when it has a bounded DART-relevant claim, scene, metric, and closing condition. -The machine-readable manifest should be derived from this document once the -current repository evidence system is audited. Until then, this file is the -human authority for row identity and intended claim boundary. +This file is the human authority for row identity, bounded claim text, and +intended oracle. Live per-branch status and dispositions live in the checked +manifest [`claims-manifest.json`](claims-manifest.json), which +`pixi run check-citation-evidence` requires to carry exactly the claim IDs in +the table below. ## Dispositions @@ -149,45 +151,67 @@ a maintainer explicitly revises the cap. ## Evidence packet skeleton -```yaml -schema: dart.citation_claim_evidence/v1 -claim_id: CT-001 -source: - url: ... - claim: ... -target: - branch: main - commit: ... -scene: - id: ... - digest: ... -configuration: - requested: ... - resolved: ... - detector: ... - timestep: ... - substeps: ... - tolerances: ... - fallback_policy: ... -ensemble: - seeds: [...] - perturbations: ... -metrics: - physical: { ... } - numerical: { ... } - performance: { ... } - allocation: { ... } -evidence: - commands: [...] - raw_paths: [...] - visual_paths: [...] -result: - disposition: unresolved - claim_boundary: ... - limitations: [...] -review: - passes: [...] +Packets are JSON (`dart.citation_claim_evidence/v1`) under +`evidence/`, validated by `pixi run check-citation-evidence`. The shape below +matches what the validator actually requires; a packet written from it passes. + +```json +{ + "schema": "dart.citation_claim_evidence/v1", + "claim_id": "CT-001", + "title": "One line naming the fixture and lane", + "source": { "url": "...", "claim": "quoted or bounded paraphrase" }, + "target": { "branch": "main", "commit": "<40 hex>", "commit_role": "..." }, + "scene": { "id": "...", "digest": "sha256:<64 hex>", "parameters": {} }, + "configuration": { + "requested": {}, + "resolved": {}, + "resolved_provenance": "how resolved identity was obtained", + "detector": "...", + "timestep": 0.002, + "substeps": 1, + "iterations": "...", + "fallback_policy": "..." + }, + "ensemble": { + "kind": "parameter-sweep-with-deterministic-repeats", + "deterministic_repeats": 2, + "measurement_window": { "start_s": 0.0, "end_s": 1.0 } + }, + "metrics": { + "physical": { "method": "...", "measured_zero_fields": [] }, + "numerical": { + "method": "...", + "solver_residual": { "status": "unsupported", "reason": "..." } + }, + "performance": { "status": "unsupported", "reason": "..." }, + "allocation": { "status": "unsupported", "reason": "..." } + }, + "evidence": { + "commands": ["..."], + "raw_rows": [], + "visual": { "status": "not-applicable", "reason": "..." } + }, + "result": { + "disposition": "unresolved", + "claim_boundary": "...", + "limitations": ["..."] + }, + "review": { "passes": [] } +} ``` -The schema must fail closed on missing target commit, scene digest, requested -and resolved method, command, disposition, or claim boundary. +The validator fails closed on a missing target commit, scene digest, +requested/resolved method, resolved provenance, command, ensemble (a single +run is not evidence), disposition, claim boundary, or limitation. Inside a +measured metric group it also rejects nulls, NaNs, spelled placeholders such +as `"n/a"`, and any exact zero that is not either typed +`{"status": "unsupported", "reason": ...}` or declared in +`measured_zero_fields` -- that is how "unsupported is never silently zero" +becomes an enforced rule rather than a promise. A lane may not be closed by +prose, a non-JSON file, a path outside `evidence/`, or a negative control, +and a closing packet needs two recorded review passes. + +Negative controls live in `evidence/negative-controls/`. They must keep +failing validation (at least three errors); the gate rejects a negative +control that starts passing as a vacuous proof. diff --git a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json index 84b912aecad0c..4b4f75cfc83ab 100644 --- a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json +++ b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json @@ -43,9 +43,12 @@ "lanes": { "dart7": { "owner": "PLAN-123 with PLAN-080/122", - "status": "audit-required", + "status": "in-progress", "disposition": null, - "evidence": [] + "evidence": [ + "evidence/CT-002-dart7-dense-inelastic-contact.json" + ], + "notes": "Bounded (not source-exact) 6x6x6 sphere-grid drop over a timestep/solver grid; the intentionally incomplete negative control also carries claim CT-002 and is never lane evidence." }, "dart6": { "owner": "docs/dev_tasks/dart6_citation_contact_trust", @@ -63,10 +66,12 @@ "lanes": { "dart7": { "owner": "PLAN-123 with PLAN-080", - "status": "audit-required", + "status": "in-progress", "disposition": null, - "evidence": [], - "notes": "python/examples/demos/scenes/rigid_restitution_ladder.py is the existing seed scene." + "evidence": [ + "evidence/CT-003-dart7-dense-elastic-contact.json" + ], + "notes": "Bounded 6x6x6 elastic (restitution 0.8) grid drop: no energy injection and no solver failure observed on main, so the packet disposition is unresolved rather than reproduced. python/examples/demos/scenes/rigid_restitution_ladder.py remains the seed scene for per-body restitution outcomes." }, "dart6": { "owner": "docs/dev_tasks/dart6_citation_contact_trust", diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json index e1d09b1288082..d40b222e028cd 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json @@ -3,7 +3,7 @@ "configuration": { "detector": "DART 7 native World collision pipeline (the World step API exposes no detector selection on main)", "fallback_policy": "World defaults; no per-island fallback reporting is exposed on main (PLAN-123 WS4 follow-up)", - "iterations": "World defaults; per-step actual iterations recorded via StepMetrics.last_step_iterations", + "iterations": "World defaults. Sequential impulse reports its configured sweep count through StepMetrics.last_step_iterations; the boxed-LCP path records no iteration count at all (see metrics.numerical.solver_iterations_by_method).", "requested": { "backend": "cpu", "contact_solver_method_sweep": [ @@ -39,7 +39,7 @@ } } }, - "resolved_provenance": "World property readback (contact_solver_method, rigid_body_solver, gravity, time_step) after enter_simulation_mode plus final-step WorldStepProfile stage names recorded per run; World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", + "resolved_provenance": "World property readback (contact_solver_method, rigid_body_solver, gravity, time_step) after enter_simulation_mode, asserted equal to the request and stable across every repeat and sweep point. The independent evidence that the selection changed behavior is that the per-angle trajectory hashes differ between the two contact solvers at every angle; the recorded WorldStepProfile stage names are identical for both methods and therefore do not discriminate. World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", "substeps": 1, "timestep": 0.002 }, @@ -133,7 +133,11 @@ "max_energy_gain_j": 0.0, "max_penetration_m": 0.0, "max_solver_iterations": 8, - "max_solver_residual": 0.0, + "raw_last_step_residual_max": 0.0, + "repeat_trajectory_sha256": [ + "a839b7f050262f80e450a25831e5b4144eb74f3f725807eb0e92855f762b96d5", + "a839b7f050262f80e450a25831e5b4144eb74f3f725807eb0e92855f762b96d5" + ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", "gravity_mps2": [ @@ -164,7 +168,11 @@ "max_energy_gain_j": 5.551115123125783e-17, "max_penetration_m": 0.0, "max_solver_iterations": 8, - "max_solver_residual": 0.0, + "raw_last_step_residual_max": 0.0, + "repeat_trajectory_sha256": [ + "4579581ae2bab0980e689803c5fc8e7296f2b79314c1b9644855360102d16ced", + "4579581ae2bab0980e689803c5fc8e7296f2b79314c1b9644855360102d16ced" + ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", "gravity_mps2": [ @@ -175,7 +183,7 @@ "rigid_body_solver": "SEQUENTIAL_IMPULSE", "time_step_s": 0.002 }, - "slide_end_time_s": 0.078, + "slide_end_time_s": 0.08, "trajectory_sha256": "4579581ae2bab0980e689803c5fc8e7296f2b79314c1b9644855360102d16ced" }, { @@ -195,7 +203,11 @@ "max_energy_gain_j": 5.551115123125783e-17, "max_penetration_m": 0.0, "max_solver_iterations": 8, - "max_solver_residual": 0.0, + "raw_last_step_residual_max": 0.0, + "repeat_trajectory_sha256": [ + "ace9350655e315351bede1b8474feac16c223921633fbf2a5497af3ae57d941a", + "ace9350655e315351bede1b8474feac16c223921633fbf2a5497af3ae57d941a" + ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", "gravity_mps2": [ @@ -226,7 +238,11 @@ "max_energy_gain_j": 5.551115123125783e-17, "max_penetration_m": 0.0, "max_solver_iterations": 8, - "max_solver_residual": 0.0, + "raw_last_step_residual_max": 0.0, + "repeat_trajectory_sha256": [ + "8446fd6f1793674e853a71724405db019de7f5acd799e9e61af877310443edc7", + "8446fd6f1793674e853a71724405db019de7f5acd799e9e61af877310443edc7" + ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", "gravity_mps2": [ @@ -257,7 +273,11 @@ "max_energy_gain_j": 1.1102230246251565e-16, "max_penetration_m": 0.0, "max_solver_iterations": 8, - "max_solver_residual": 0.0, + "raw_last_step_residual_max": 0.0, + "repeat_trajectory_sha256": [ + "fce89de8287195fea41ecdbc5771c027b838e91d2c5be2e7255a6cc2abffeb2d", + "fce89de8287195fea41ecdbc5771c027b838e91d2c5be2e7255a6cc2abffeb2d" + ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", "gravity_mps2": [ @@ -288,7 +308,11 @@ "max_energy_gain_j": 5.551115123125783e-17, "max_penetration_m": 0.0, "max_solver_iterations": 8, - "max_solver_residual": 0.0, + "raw_last_step_residual_max": 0.0, + "repeat_trajectory_sha256": [ + "2ca8f5b51e27817c76eedef74e1b9c59813fa9052244f219e3143eb0327f5d7a", + "2ca8f5b51e27817c76eedef74e1b9c59813fa9052244f219e3143eb0327f5d7a" + ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", "gravity_mps2": [ @@ -299,7 +323,7 @@ "rigid_body_solver": "SEQUENTIAL_IMPULSE", "time_step_s": 0.002 }, - "slide_end_time_s": 0.078, + "slide_end_time_s": 0.08, "trajectory_sha256": "2ca8f5b51e27817c76eedef74e1b9c59813fa9052244f219e3143eb0327f5d7a" }, { @@ -319,7 +343,11 @@ "max_energy_gain_j": 0.0, "max_penetration_m": 0.0, "max_solver_iterations": 8, - "max_solver_residual": 0.0, + "raw_last_step_residual_max": 0.0, + "repeat_trajectory_sha256": [ + "d7bd5003831ed9c7c1a477345d578fe7f3f5a99706d3a4e9720e11d770a9c5b8", + "d7bd5003831ed9c7c1a477345d578fe7f3f5a99706d3a4e9720e11d770a9c5b8" + ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", "gravity_mps2": [ @@ -350,7 +378,11 @@ "max_energy_gain_j": 5.551115123125783e-17, "max_penetration_m": 0.0, "max_solver_iterations": 0, - "max_solver_residual": 0.0, + "raw_last_step_residual_max": 0.0, + "repeat_trajectory_sha256": [ + "f92fc6c64c534a96459ce3ac0a970c931618a8533dd3974b565db61b7cacd150", + "f92fc6c64c534a96459ce3ac0a970c931618a8533dd3974b565db61b7cacd150" + ], "resolved": { "contact_solver_method": "BOXED_LCP", "gravity_mps2": [ @@ -381,7 +413,11 @@ "max_energy_gain_j": 5.551115123125783e-17, "max_penetration_m": 0.0, "max_solver_iterations": 0, - "max_solver_residual": 0.0, + "raw_last_step_residual_max": 0.0, + "repeat_trajectory_sha256": [ + "fbe57d7b17ff7fdfe97b1cfe5112698ae2ae9c7c413e4c08eb13baba931e0899", + "fbe57d7b17ff7fdfe97b1cfe5112698ae2ae9c7c413e4c08eb13baba931e0899" + ], "resolved": { "contact_solver_method": "BOXED_LCP", "gravity_mps2": [ @@ -392,7 +428,7 @@ "rigid_body_solver": "SEQUENTIAL_IMPULSE", "time_step_s": 0.002 }, - "slide_end_time_s": 0.078, + "slide_end_time_s": 0.08, "trajectory_sha256": "fbe57d7b17ff7fdfe97b1cfe5112698ae2ae9c7c413e4c08eb13baba931e0899" }, { @@ -412,7 +448,11 @@ "max_energy_gain_j": 5.551115123125783e-17, "max_penetration_m": 0.0, "max_solver_iterations": 0, - "max_solver_residual": 0.0, + "raw_last_step_residual_max": 0.0, + "repeat_trajectory_sha256": [ + "baed575b592007dcfedeefa6f7b8c17858d47860acb3e6f2c06e925766e575fe", + "baed575b592007dcfedeefa6f7b8c17858d47860acb3e6f2c06e925766e575fe" + ], "resolved": { "contact_solver_method": "BOXED_LCP", "gravity_mps2": [ @@ -443,7 +483,11 @@ "max_energy_gain_j": 5.551115123125783e-17, "max_penetration_m": 0.0, "max_solver_iterations": 0, - "max_solver_residual": 0.0, + "raw_last_step_residual_max": 0.0, + "repeat_trajectory_sha256": [ + "4f8571c3013779682bcf99ce7dd5f3cc7bb3ccb49182fc64f8f55199fa3b19ef", + "4f8571c3013779682bcf99ce7dd5f3cc7bb3ccb49182fc64f8f55199fa3b19ef" + ], "resolved": { "contact_solver_method": "BOXED_LCP", "gravity_mps2": [ @@ -474,7 +518,11 @@ "max_energy_gain_j": 5.551115123125783e-17, "max_penetration_m": 0.0, "max_solver_iterations": 0, - "max_solver_residual": 0.0, + "raw_last_step_residual_max": 0.0, + "repeat_trajectory_sha256": [ + "be2f9b06b52ffb94c3c4c65e2ef2a1ba188e02b852e89c1fcd355fb87e27079d", + "be2f9b06b52ffb94c3c4c65e2ef2a1ba188e02b852e89c1fcd355fb87e27079d" + ], "resolved": { "contact_solver_method": "BOXED_LCP", "gravity_mps2": [ @@ -505,7 +553,11 @@ "max_energy_gain_j": 5.551115123125783e-17, "max_penetration_m": 0.0, "max_solver_iterations": 0, - "max_solver_residual": 0.0, + "raw_last_step_residual_max": 0.0, + "repeat_trajectory_sha256": [ + "dd250dfabf218e68548f9edf30a342bc1bc595d6675b5f602f4bf608f88921e0", + "dd250dfabf218e68548f9edf30a342bc1bc595d6675b5f602f4bf608f88921e0" + ], "resolved": { "contact_solver_method": "BOXED_LCP", "gravity_mps2": [ @@ -516,7 +568,7 @@ "rigid_body_solver": "SEQUENTIAL_IMPULSE", "time_step_s": 0.002 }, - "slide_end_time_s": 0.078, + "slide_end_time_s": 0.08, "trajectory_sha256": "dd250dfabf218e68548f9edf30a342bc1bc595d6675b5f602f4bf608f88921e0" }, { @@ -536,7 +588,11 @@ "max_energy_gain_j": 5.551115123125783e-17, "max_penetration_m": 0.0, "max_solver_iterations": 0, - "max_solver_residual": 0.0, + "raw_last_step_residual_max": 0.0, + "repeat_trajectory_sha256": [ + "631d89a93d54a08badc23834c755a2a84c88f7d6f6b94a2357bd81a5a3e7d9a5", + "631d89a93d54a08badc23834c755a2a84c88f7d6f6b94a2357bd81a5a3e7d9a5" + ], "resolved": { "contact_solver_method": "BOXED_LCP", "gravity_mps2": [ @@ -571,9 +627,23 @@ "numerical": { "max_active_contacts": 1, "max_penetration_m": 0.0, - "max_solver_iterations": 8, - "max_solver_residual": 0.0, - "method": "Max over run of StepMetrics.max_penetration_depth, last_step_iterations, last_step_residual, active_contact_count" + "measured_zero_fields": [ + "max_penetration_m" + ], + "method": "Max over run of StepMetrics.max_penetration_depth and active_contact_count; per-method iteration counts where the runtime records them", + "penetration_semantics": "StepMetrics.max_penetration_depth clamps each contact depth with std::max(0.0, depth), so a reported 0.0 means no positive penetration was observed and does not distinguish resting-exactly-tangent from separated contacts.", + "sequential_impulse_iterations_semantics": "Sequential impulse records the configured iteration count (recordSolverDiagnostics(world, m_iterations)); its Gauss-Seidel loop has no convergence exit, so this is the configured sweep count, not an observed iteration-to-convergence.", + "solver_iterations_by_method": { + "BOXED_LCP": { + "reason": "The BoxedLcp branch in dart/simulation/compute/rigid_body_contact_stage.cpp returns before recordSolverDiagnostics() runs, so StepMetrics.last_step_iterations is never written for this method; a recorded 0 would mean 'not reported', not 'zero iterations'.", + "status": "unsupported" + }, + "SEQUENTIAL_IMPULSE": 8 + }, + "solver_residual": { + "reason": "StepMetrics.last_step_residual is structurally zero for rigid contact on this branch: recordSolverDiagnostics() in dart/simulation/compute/rigid_body_contact_stage.cpp takes residual = 0.0 by default and no rigid contact call site passes one, so no residual is computed for either contact solver. Exposing a comparable residual is PLAN-123 WS4 work.", + "status": "unsupported" + } }, "performance": { "reason": "This packet makes no timing claim; interleaved same-host methodology is required before any performance row", @@ -584,44 +654,78 @@ "BOXED_LCP", "SEQUENTIAL_IMPULSE" ], + "anisotropy_findings": { + "BOXED_LCP": { + "criteria_exceeded": [ + "max_abs_lateral_drift_m" + ], + "pyramid_signature": true, + "signature_test": "lateral drift antisymmetric about 45 deg to within 5% of peak drift" + }, + "SEQUENTIAL_IMPULSE": { + "criteria_exceeded": [ + "max_abs_lateral_drift_m" + ], + "pyramid_signature": true, + "signature_test": "lateral drift antisymmetric about 45 deg to within 5% of peak drift" + } + }, "isotropy_tolerance": { "max_abs_heading_error_deg": 0.1, "max_abs_lateral_drift_m": 0.0001, "travel_spread_relative": 0.01 }, + "measured_zero_fields": [ + "per_solver_summary.SEQUENTIAL_IMPULSE.max_penetration_m", + "per_solver_summary.BOXED_LCP.max_penetration_m" + ], "method": "Rotational-symmetry sweep: per-angle lateral drift from the launch ray, final-velocity heading error, along-ray travel, slide-to-roll transition time, and max single-step kinetic-energy gain from StepMetrics.kinetic_energy", "per_solver_summary": { "BOXED_LCP": { + "antisymmetry_residual_m": 1.1102230246251565e-16, + "antisymmetry_residual_over_peak_drift": 5.285375546768002e-14, "max_abs_heading_error_deg": 5.343331864858703e-14, "max_abs_lateral_drift_m": 0.002100556554215066, "max_energy_gain_j": 5.551115123125783e-17, "max_penetration_m": 0.0, - "max_solver_residual": 0.0, "runs": 7, + "solver_residual": { + "reason": "StepMetrics.last_step_residual is structurally zero for rigid contact on this branch: recordSolverDiagnostics() in dart/simulation/compute/rigid_body_contact_stage.cpp takes residual = 0.0 by default and no rigid contact call site passes one, so no residual is computed for either contact solver. Exposing a comparable residual is PLAN-123 WS4 work.", + "status": "unsupported" + }, "travel_mean_m": 0.7243492115097299, "travel_spread_m": 0.003481090802366804, "travel_spread_relative": 0.00480581844648014 }, "SEQUENTIAL_IMPULSE": { + "antisymmetry_residual_m": 1.6653345369377348e-16, + "antisymmetry_residual_over_peak_drift": 7.928062960668804e-14, "max_abs_heading_error_deg": 5.343331864858703e-14, "max_abs_lateral_drift_m": 0.0021005566494608774, "max_energy_gain_j": 1.1102230246251565e-16, "max_penetration_m": 0.0, - "max_solver_residual": 0.0, "runs": 7, + "solver_residual": { + "reason": "StepMetrics.last_step_residual is structurally zero for rigid contact on this branch: recordSolverDiagnostics() in dart/simulation/compute/rigid_body_contact_stage.cpp takes residual = 0.0 by default and no rigid contact call site passes one, so no residual is computed for either contact solver. Exposing a comparable residual is PLAN-123 WS4 work.", + "status": "unsupported" + }, "travel_mean_m": 0.7243492114424034, "travel_spread_m": 0.0034810908006920327, "travel_spread_relative": 0.004805818444614723 } - } + }, + "validity_failures": [] } }, "result": { "claim_boundary": "DART 7 main, this commit, one sphere sliding to rolling on a static ground box at v0=1 m/s, mu=0.35, dt=2 ms, 1 s horizon, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers, launch angles 0-90 deg in 15 deg steps. Says nothing about other speeds, shapes, stacks, historical DART versions, or DART 6.", "disposition": "reproduced", "limitations": [ - "ResolvedSolverConfiguration is not Python-exposed; resolved identity is property readback plus stage names.", - "Friction-cone/complementarity residual per contact is not exposed; solver residual is the aggregate StepMetrics value.", + "ResolvedSolverConfiguration is not Python-exposed; resolved identity is property readback, corroborated by per-solver trajectory-hash differences.", + "No solver residual exists on this path: it is typed unsupported, not reported as zero. Friction-cone and complementarity violations are not exposed either.", + "The boxed-LCP path records no iteration count; only sequential impulse reports one, and that is its configured sweep count rather than an observed convergence measure.", + "Penetration is clamped at zero by the runtime, so 0.0 means no positive penetration was observed and does not distinguish resting-tangent from separated contacts.", + "The corpus row also names stopping distance and cone violation as oracles; neither is measured here (the sphere rolls indefinitely at 5/7 v0, and no cone metric is exposed).", "The sweep covers 0-90 deg; pyramid orientation with a period other than 90 deg would need a wider sweep.", "No high-mass-ratio or stacked variant yet; those belong to CT-002/CT-007 fixtures.", "SEQUENTIAL_IMPULSE and BOXED_LCP produce metric summaries that agree to printed precision on this single-contact scene while their full trajectory hashes differ; this packet is not a solver comparison and must not be quoted as one." @@ -678,7 +782,8 @@ }, "target": { "branch": "main", - "commit": "20501341226338f42740f1b6ef1ab52a34e5a035" + "commit": "d0cb640860d7da5a39792a385e7f499610c0536f", + "commit_role": "Source state measured: the library and fixture were run at this commit, which is HEAD at capture time. The packet and its writer land in a later commit, so re-running the recorded command requires the child commit that adds the writer; the measured behavior belongs to this one." }, "title": "Rolling-direction friction dependence (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json new file mode 100644 index 0000000000000..a974c3a7468bb --- /dev/null +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json @@ -0,0 +1,309 @@ +{ + "claim_id": "CT-002", + "configuration": { + "detector": "DART 7 native World collision pipeline (the World step API exposes no detector selection on main)", + "fallback_policy": "World defaults; no per-island fallback reporting is exposed on main (PLAN-123 WS4 follow-up)", + "iterations": "World defaults; per-step actual iterations recorded via StepMetrics.last_step_iterations", + "requested": { + "backend": "cpu", + "contact_solver_method_sweep": [ + "SEQUENTIAL_IMPULSE", + "BOXED_LCP" + ], + "integrator": "World default semi-implicit stepping", + "precision": "float64", + "rigid_body_solver": "SEQUENTIAL_IMPULSE (World default)", + "threads": "World default sequential step", + "timestep_sweep_s": [ + 0.002, + 0.004 + ] + }, + "resolved": { + "by_contact_solver_method": { + "BOXED_LCP": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "SEQUENTIAL_IMPULSE": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + } + } + }, + "resolved_provenance": "World property readback (contact_solver_method, rigid_body_solver, gravity, time_step) after enter_simulation_mode, asserted equal to the request per cell; World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", + "substeps": 1, + "timestep": 0.002 + }, + "ensemble": { + "deterministic_repeats": 2, + "deterministic_repeats_identical": true, + "kind": "parameter-sweep-with-deterministic-repeats", + "measurement_window": { + "end_s": 2.0, + "settle_window_start_fraction": 0.5, + "start_s": 0.0 + }, + "sweep": [ + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "timestep_s": 0.002 + }, + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "timestep_s": 0.004 + }, + { + "contact_solver_method": "BOXED_LCP", + "timestep_s": 0.002 + }, + { + "contact_solver_method": "BOXED_LCP", + "timestep_s": 0.004 + } + ] + }, + "evidence": { + "commands": [ + "PYTHONPATH=build/default/cpp/Release/python pixi run python scripts/write_citation_ct002_dense_contact_packet.py" + ], + "raw_rows": [ + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_kinetic_energy_j": 0.0073483693432820196, + "final_max_speed_mps": 0.04032772547075842, + "final_min_height_m": 0.048578028173510944, + "final_state_sha256": "d3f6fdd984ba5e6eabaa2841a1bc305f6237a6e224039a079177ec6ca400eb04", + "finite": true, + "initial_total_energy_j": 180.1116000000002, + "max_active_contacts": 216, + "max_energy_gain_after_settle_j": 0.0, + "max_penetration_m": 0.006675685044093327, + "max_solver_iterations": 8, + "raw_last_step_residual_max": 0.0, + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "steps": 1000, + "timestep_s": 0.002 + }, + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_kinetic_energy_j": 0.029393477373128078, + "final_max_speed_mps": 0.08065545094151684, + "final_min_height_m": 0.04459709802896138, + "final_state_sha256": "bed22ef86fbea67a6176944a75e64973b4d0ab7f14c0d2604f61d67823b05f9d", + "finite": true, + "initial_total_energy_j": 180.1116000000002, + "max_active_contacts": 216, + "max_energy_gain_after_settle_j": 0.0, + "max_penetration_m": 0.011362281723803547, + "max_solver_iterations": 8, + "raw_last_step_residual_max": 0.0, + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.004 + }, + "steps": 500, + "timestep_s": 0.004 + }, + { + "contact_solver_method": "BOXED_LCP", + "final_kinetic_energy_j": 1.4819720438078604e-27, + "final_max_speed_mps": 2.3901020052008448e-14, + "final_min_height_m": 0.04989999999999998, + "final_state_sha256": "5260253337e1348e1b0f89954a78b9d01a5715d982d77fe0384818189a8f9e33", + "finite": true, + "initial_total_energy_j": 180.1116000000002, + "max_active_contacts": 216, + "max_energy_gain_after_settle_j": 6.459280144009318e-31, + "max_penetration_m": 0.005411200000000366, + "max_solver_iterations": 0, + "raw_last_step_residual_max": 0.0, + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "steps": 1000, + "timestep_s": 0.002 + }, + { + "contact_solver_method": "BOXED_LCP", + "final_kinetic_energy_j": 3.674601533738957e-28, + "final_max_speed_mps": 1.1914080833008711e-14, + "final_min_height_m": 0.04989999999999998, + "final_state_sha256": "1a94d8a3a56f6cf1088d78188cbc7982f71e7b0287d0b2335c18dfac375ca3b8", + "finite": true, + "initial_total_energy_j": 180.1116000000002, + "max_active_contacts": 216, + "max_energy_gain_after_settle_j": 9.594362860530006e-10, + "max_penetration_m": 0.02046842021148934, + "max_solver_iterations": 0, + "raw_last_step_residual_max": 0.0, + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.004 + }, + "steps": 500, + "timestep_s": 0.004 + } + ], + "visual": { + "reason": "The oracle is numeric (finite state, settle speed, penetration, energy monotonicity); no visible-behavior claim is made by this packet.", + "status": "not-applicable" + } + }, + "host": { + "machine": "x86_64", + "note": "Host recorded for provenance only; no timing methodology was applied", + "performance_valid": false, + "platform": "Linux-7.0.0-29-generic-x86_64-with-glibc2.43", + "python": "3.14.6" + }, + "metrics": { + "allocation": { + "reason": "Post-bake allocation gates are owned by PLAN-122 tooling; this packet does not measure allocations", + "status": "unsupported" + }, + "numerical": { + "max_active_contacts": 216, + "max_penetration_m": 0.02046842021148934, + "method": "Max over run of StepMetrics.max_penetration_depth and active_contact_count; per-method iteration counts where the runtime records them", + "penetration_semantics": "StepMetrics.max_penetration_depth clamps each contact depth with std::max(0.0, depth), so a reported 0.0 means no positive penetration was observed and does not distinguish resting-exactly-tangent from separated contacts.", + "sequential_impulse_iterations_semantics": "Sequential impulse records the configured iteration count (recordSolverDiagnostics(world, m_iterations)); its Gauss-Seidel loop has no convergence exit, so this is the configured sweep count, not an observed iteration-to-convergence.", + "solver_iterations_by_method": { + "BOXED_LCP": { + "reason": "The BoxedLcp branch in dart/simulation/compute/rigid_body_contact_stage.cpp returns before recordSolverDiagnostics() runs, so StepMetrics.last_step_iterations is never written for this method; a recorded 0 would mean 'not reported', not 'zero iterations'.", + "status": "unsupported" + }, + "SEQUENTIAL_IMPULSE": 8 + }, + "solver_residual": { + "reason": "StepMetrics.last_step_residual is structurally zero for rigid contact on this branch: recordSolverDiagnostics() in dart/simulation/compute/rigid_body_contact_stage.cpp takes residual = 0.0 by default and no rigid contact call site passes one, so no residual is computed for either contact solver. Exposing a comparable residual is PLAN-123 WS4 work.", + "status": "unsupported" + } + }, + "performance": { + "reason": "This packet makes no timing or scaling claim; interleaved same-host methodology is required before any performance row", + "status": "unsupported" + }, + "physical": { + "all_cells_finite": true, + "max_energy_gain_after_settle_j": 9.594362860530006e-10, + "max_penetration_m": 0.02046842021148934, + "method": "Per-cell finite-state check over 216 bodies, final max body speed, final min body height, kinetic energy at the horizon, and max single-step kinetic-energy gain after the settle window from StepMetrics", + "settled_final_max_speed_mps": 0.08065545094151684, + "stability_tolerance": { + "final_max_speed_mps": 0.05, + "max_energy_gain_after_settle_j": 0.001, + "max_penetration_m": 0.025 + }, + "unstable_cells": [ + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "reasons": [ + "pile still moving at horizon" + ], + "timestep_s": 0.004 + } + ] + } + }, + "result": { + "claim_boundary": "DART 7 main, this commit, a bounded (not source-exact) 6x6x6 sphere-grid drop with restitution 0 and mu=0.8, 2 s horizon, timesteps 2 and 4 ms, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers. Says nothing about the original SimBenchmark scene parameters, historical DART versions, other densities/materials, or DART 6.", + "disposition": "reproduced", + "limitations": [ + "Bounded reconstruction: the original SimBenchmark asset, material, and timestep grid are not reproduced exactly; sourcing the exact historical setup is future corpus work.", + "ResolvedSolverConfiguration is not Python-exposed; resolved identity is property readback.", + "Two timesteps and two solvers only; a wider grid belongs to a follow-up once per-island diagnostics exist.", + "No per-contact residual/cone reporting is available on main; solver residual is the aggregate StepMetrics value." + ] + }, + "review": { + "passes": [] + }, + "scene": { + "description": "6x6x6 grid of 216 spheres (radius 0.05 m, mass 0.1 kg, spacing 0.12 m) dropped from 0.5 m clearance onto a static ground box, restitution 0, friction 0.8 on all bodies, simulated to a 2 s horizon per timestep/solver cell.", + "digest": "sha256:063ef6bc7a4efd38bdd2c1a7e6f3a04d62475d788d05bef401a726deead1ad94", + "id": "ct002_dense_inelastic_grid", + "parameters": { + "contact_solver_methods": [ + "SEQUENTIAL_IMPULSE", + "BOXED_LCP" + ], + "description": "6x6x6 grid of 216 spheres (radius 0.05 m, mass 0.1 kg, spacing 0.12 m) dropped from 0.5 m clearance onto a static ground box, restitution 0, friction 0.8 on all bodies, simulated to a 2 s horizon per timestep/solver cell.", + "deterministic_repeats": 2, + "drop_clearance_m": 0.5, + "friction": 0.8, + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "grid_dimension": 6, + "grid_spacing_m": 0.12, + "ground_half_extents_m": [ + 2.0, + 2.0, + 0.05 + ], + "horizon_s": 2.0, + "impact_settle_fraction": 0.5, + "restitution": 0.0, + "scene_id": "ct002_dense_inelastic_grid", + "sphere_mass_kg": 0.1, + "sphere_radius_m": 0.05, + "timesteps_s": [ + 0.002, + 0.004 + ] + } + }, + "schema": "dart.citation_claim_evidence/v1", + "source": { + "claim": "Dense contact may fail, become unstable, or scale poorly for some timestep/solver settings.", + "url": "https://leggedrobotics.github.io/SimBenchmark/" + }, + "target": { + "branch": "main", + "commit": "d0cb640860d7da5a39792a385e7f499610c0536f" + }, + "title": "Dense 6x6x6 inelastic contact stability (DART 7 first packet)" +} diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json new file mode 100644 index 0000000000000..db08f8ee9e57f --- /dev/null +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json @@ -0,0 +1,293 @@ +{ + "claim_id": "CT-003", + "configuration": { + "detector": "DART 7 native World collision pipeline (the World step API exposes no detector selection on main)", + "fallback_policy": "World defaults; no per-island fallback reporting is exposed on main (PLAN-123 WS4 follow-up)", + "iterations": "World defaults; per-step actual iterations recorded via StepMetrics.last_step_iterations", + "requested": { + "backend": "cpu", + "contact_solver_method_sweep": [ + "SEQUENTIAL_IMPULSE", + "BOXED_LCP" + ], + "integrator": "World default semi-implicit stepping", + "precision": "float64", + "rigid_body_solver": "SEQUENTIAL_IMPULSE (World default)", + "threads": "World default sequential step", + "timestep_sweep_s": [ + 0.002, + 0.004 + ] + }, + "resolved": { + "by_contact_solver_method": { + "BOXED_LCP": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "SEQUENTIAL_IMPULSE": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + } + } + }, + "resolved_provenance": "World property readback (contact_solver_method, rigid_body_solver, gravity, time_step) after enter_simulation_mode, asserted equal to the request per cell; World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", + "substeps": 1, + "timestep": 0.002 + }, + "ensemble": { + "deterministic_repeats": 2, + "deterministic_repeats_identical": true, + "kind": "parameter-sweep-with-deterministic-repeats", + "measurement_window": { + "end_s": 2.0, + "start_s": 0.0 + }, + "sweep": [ + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "timestep_s": 0.002 + }, + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "timestep_s": 0.004 + }, + { + "contact_solver_method": "BOXED_LCP", + "timestep_s": 0.002 + }, + { + "contact_solver_method": "BOXED_LCP", + "timestep_s": 0.004 + } + ] + }, + "evidence": { + "commands": [ + "PYTHONPATH=build/default/cpp/Release/python pixi run python scripts/write_citation_ct003_elastic_contact_packet.py" + ], + "raw_rows": [ + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "energy_envelope_excess_j": 0.0, + "final_state_sha256": "f1d5fac4ff2da8ee3e862cc55b49ea2362ecefd6a6527464dab5352f784c949c", + "final_total_energy_j": 63.44696118939535, + "finite": true, + "initial_total_energy_j": 180.1116000000002, + "max_active_contacts": 216, + "max_penetration_m": 0.007707330160000275, + "max_solver_iterations": 8, + "max_total_energy_j": 180.1116000000002, + "raw_last_step_residual_max": 0.0, + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "steps": 1000, + "timestep_s": 0.002 + }, + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "energy_envelope_excess_j": 0.0, + "final_state_sha256": "6cc4601838f1c5213104137dabb8e8064e376090d6cb22068335c12c87a8b125", + "final_total_energy_j": 63.25766139362141, + "finite": true, + "initial_total_energy_j": 180.1116000000002, + "max_active_contacts": 216, + "max_penetration_m": 0.01716770175999996, + "max_solver_iterations": 8, + "max_total_energy_j": 180.1116000000002, + "raw_last_step_residual_max": 0.0, + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.004 + }, + "steps": 500, + "timestep_s": 0.004 + }, + { + "contact_solver_method": "BOXED_LCP", + "energy_envelope_excess_j": 0.0, + "final_state_sha256": "31c88c2d34b7e738c731aaa2e3e5a4dc9ebca29de00e901dc7296ebddb3402c1", + "final_total_energy_j": 63.58479412413587, + "finite": true, + "initial_total_energy_j": 180.1116000000002, + "max_active_contacts": 216, + "max_penetration_m": 0.007707330160000275, + "max_solver_iterations": 0, + "max_total_energy_j": 180.1116000000002, + "raw_last_step_residual_max": 0.0, + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "steps": 1000, + "timestep_s": 0.002 + }, + { + "contact_solver_method": "BOXED_LCP", + "energy_envelope_excess_j": 0.0, + "final_state_sha256": "2438f7c534de26f786af0b223b92f0898d8fcc5618698326c5e9f14be6936513", + "final_total_energy_j": 63.746668542792605, + "finite": true, + "initial_total_energy_j": 180.1116000000002, + "max_active_contacts": 216, + "max_penetration_m": 0.01716770175999996, + "max_solver_iterations": 0, + "max_total_energy_j": 180.1116000000002, + "raw_last_step_residual_max": 0.0, + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.004 + }, + "steps": 500, + "timestep_s": 0.004 + } + ], + "visual": { + "reason": "The oracle is the numeric energy envelope and finite state; no visible-behavior claim is made by this packet.", + "status": "not-applicable" + } + }, + "host": { + "machine": "x86_64", + "note": "Host recorded for provenance only; no timing methodology was applied", + "performance_valid": false, + "platform": "Linux-7.0.0-29-generic-x86_64-with-glibc2.43", + "python": "3.14.6" + }, + "metrics": { + "allocation": { + "reason": "Post-bake allocation gates are owned by PLAN-122 tooling; this packet does not measure allocations", + "status": "unsupported" + }, + "numerical": { + "max_active_contacts": 216, + "max_penetration_m": 0.01716770175999996, + "method": "Max over run of StepMetrics.max_penetration_depth and active_contact_count; per-method iteration counts where the runtime records them", + "penetration_semantics": "StepMetrics.max_penetration_depth clamps each contact depth with std::max(0.0, depth), so a reported 0.0 means no positive penetration was observed and does not distinguish resting-exactly-tangent from separated contacts.", + "sequential_impulse_iterations_semantics": "Sequential impulse records the configured iteration count (recordSolverDiagnostics(world, m_iterations)); its Gauss-Seidel loop has no convergence exit, so this is the configured sweep count, not an observed iteration-to-convergence.", + "solver_iterations_by_method": { + "BOXED_LCP": { + "reason": "The BoxedLcp branch in dart/simulation/compute/rigid_body_contact_stage.cpp returns before recordSolverDiagnostics() runs, so StepMetrics.last_step_iterations is never written for this method; a recorded 0 would mean 'not reported', not 'zero iterations'.", + "status": "unsupported" + }, + "SEQUENTIAL_IMPULSE": 8 + }, + "solver_residual": { + "reason": "StepMetrics.last_step_residual is structurally zero for rigid contact on this branch: recordSolverDiagnostics() in dart/simulation/compute/rigid_body_contact_stage.cpp takes residual = 0.0 by default and no rigid contact call site passes one, so no residual is computed for either contact solver. Exposing a comparable residual is PLAN-123 WS4 work.", + "status": "unsupported" + } + }, + "performance": { + "reason": "This packet makes no timing claim; interleaved same-host methodology is required before any performance row", + "status": "unsupported" + }, + "physical": { + "all_cells_finite": true, + "envelope_tolerance_j": 0.0001801116000000002, + "initial_total_energy_j": 180.1116000000002, + "max_energy_envelope_excess_j": 0.0, + "measured_zero_fields": [ + "max_energy_envelope_excess_j" + ], + "method": "Mechanical-energy envelope from StepMetrics.total_energy (max over run vs initial), finite-state check over 216 bodies", + "violating_cells": [] + } + }, + "result": { + "claim_boundary": "DART 7 main, this commit, a bounded (not source-exact) 6x6x6 sphere-grid drop with restitution 0.8 and mu=0.8, 2 s horizon, timesteps 2 and 4 ms, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers, energy envelope from StepMetrics.total_energy. The cited failure mode was NOT observed here: no cell exceeded its initial mechanical energy and no cell went non-finite, so this bounded reconstruction does not reproduce the claim. That is not a refutation of the original report, which concerns a different engine version and scene. Says nothing about the original SimBenchmark scene parameters, historical DART versions, restitution values other than 0.8, or DART 6.", + "disposition": "unresolved", + "limitations": [ + "Bounded reconstruction: the original SimBenchmark elastic test assets and parameters are not reproduced exactly.", + "The envelope uses aggregate world energy; per-body restitution-outcome tracking (bounce-height ratios) is future work for this row.", + "ResolvedSolverConfiguration is not Python-exposed; resolved identity is property readback.", + "Two timesteps and two solvers only." + ] + }, + "review": { + "passes": [] + }, + "scene": { + "description": "6x6x6 grid of 216 spheres (radius 0.05 m, mass 0.1 kg, spacing 0.12 m) dropped from 0.5 m clearance onto a static ground box, restitution 0.8, friction 0.8 on all bodies, simulated to a 2 s horizon per timestep/solver cell.", + "digest": "sha256:56ba1fb073c06dd4e961fe729d03dac79162db39301c4086fd61cbd427e048bb", + "id": "ct003_dense_elastic_grid", + "parameters": { + "contact_solver_methods": [ + "SEQUENTIAL_IMPULSE", + "BOXED_LCP" + ], + "description": "6x6x6 grid of 216 spheres (radius 0.05 m, mass 0.1 kg, spacing 0.12 m) dropped from 0.5 m clearance onto a static ground box, restitution 0.8, friction 0.8 on all bodies, simulated to a 2 s horizon per timestep/solver cell.", + "deterministic_repeats": 2, + "drop_clearance_m": 0.5, + "friction": 0.8, + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "grid_dimension": 6, + "grid_spacing_m": 0.12, + "ground_half_extents_m": [ + 2.0, + 2.0, + 0.05 + ], + "horizon_s": 2.0, + "restitution": 0.8, + "scene_id": "ct003_dense_elastic_grid", + "sphere_mass_kg": 0.1, + "sphere_radius_m": 0.05, + "timesteps_s": [ + 0.002, + 0.004 + ] + } + }, + "schema": "dart.citation_claim_evidence/v1", + "source": { + "claim": "Elastic dense contact may inject energy or expose solver failure.", + "url": "https://leggedrobotics.github.io/SimBenchmark/" + }, + "target": { + "branch": "main", + "commit": "d0cb640860d7da5a39792a385e7f499610c0536f" + }, + "title": "Dense elastic contact energy envelope (DART 7 first packet)" +} diff --git a/docs/plans/dashboard.md b/docs/plans/dashboard.md index f6027eee7bf33..7405e9de46e39 100644 --- a/docs/plans/dashboard.md +++ b/docs/plans/dashboard.md @@ -52,13 +52,17 @@ its own line so status updates remain git-history friendly. - Horizon: Now - Dimension: Algorithm extensibility - Next step: WS0 audit and the WS1 fail-closed claim/evidence manifest landed - with the first CT-001 rolling-direction packet and a permanent - negative-control packet. Next: finish the capped six first-wave fixture - families (dense inelastic/elastic, articulated energy/momentum/control, - heel-strike/toe-off, high mass-ratio, reset/concurrency queries) with - branch-qualified dispositions, reusing existing demos scenes and PLAN-104/621/622 - evidence, before contact-identity/semantics work (WS3) and cross-family - diagnostics (WS4). History and sequencing live in the owner plan and + with a permanent negative control and three first-wave packets: CT-001 + rolling-direction (`reproduced`), CT-002 dense inelastic contact + (`reproduced`), and CT-003 dense elastic contact (`unresolved` -- no energy + injection observed). Next: the remaining first-wave families (articulated + energy/momentum/control, heel-strike/toe-off, high mass-ratio, + reset/concurrency queries) with branch-qualified dispositions, reusing + existing demos scenes and PLAN-104/621/622 evidence, before + contact-identity/semantics work (WS3) and cross-family diagnostics (WS4); + WS4's first slice is exposing `ResolvedSolverConfiguration` to Python and a + comparable residual, both currently typed unsupported in every packet. + History and sequencing live in the owner plan and `docs/dev_tasks/citation_driven_simulation_trust/`. - Gate: Every claim row is branch/version-qualified and source-bound; `pixi run check-citation-evidence` fails closed on missing commit, scene diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index a9d50d5dfddc3..2adae3d39feb2 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -6,11 +6,18 @@ - `claims-manifest.json` against the `dart.citation_claim_manifest/v1` schema, including exact claim-ID agreement with the human corpus table and the capped first-wave family list; -- every packet in `evidence/` against the `dart.citation_claim_evidence/v1` - schema (missing target commit, scene digest, requested/resolved method, - command, ensemble, disposition, claim boundary, or review record fails); +- every packet under `evidence/` (recursively, excluding negative controls) + and every packet a manifest lane references, wherever it sits, against the + `dart.citation_claim_evidence/v1` schema: missing target commit, scene + digest, requested/resolved method, command, ensemble, disposition, claim + boundary, or review record fails, and a lane may not point at prose, a + non-JSON file, a path outside `evidence/`, or a negative control; - metric groups that must be measured-with-method or explicitly typed - `unsupported` with a reason -- never silently absent, null, or NaN; + `unsupported` with a reason -- never silently absent, null, NaN, a spelled + placeholder such as "n/a", or an unacknowledged exact zero (an unmeasurable + quantity is a typed-unsupported marker; a real zero is declared in + `measured_zero_fields`); +- `raw_paths` that resolve to files that exist; - every packet in `evidence/negative-controls/`, which must FAIL validation (a permanent proof that the validator rejects incomplete evidence); - manifest lane/evidence cross-links: referenced packets exist, agree on claim @@ -57,6 +64,9 @@ "unresolved", ) LANE_STATUSES = ("audit-required", "in-progress", "closed", "not-applicable") +UNSUPPORTED_SENTINELS = frozenset( + {"", "-", "--", "n/a", "na", "none", "null", "tbd", "unknown", "unsupported"} +) LANE_KEYS = ("dart7", "dart6") BRANCH_BY_LANE = {"dart7": "main", "dart6": "release-6.20"} FIRST_WAVE_FAMILY_CAP = 6 @@ -94,18 +104,34 @@ def _is_finite_number(value: object) -> bool: ) -def _metric_leaves(value: object) -> list[object]: +def _is_unsupported_leaf(value: object) -> bool: + """True when a nested value is a typed-unsupported marker.""" + return ( + isinstance(value, dict) + and value.get("status") == "unsupported" + and set(value) <= {"status", "reason"} + ) + + +def _metric_leaves(value: object, path: str = "") -> list[tuple[str, object]]: + """Flatten a metric value into (dotted-path, leaf) pairs. + + Typed-unsupported markers are returned whole so callers can validate them + instead of descending into their `status`/`reason` strings. + """ + if _is_unsupported_leaf(value): + return [(path, value)] if isinstance(value, dict): - leaves: list[object] = [] - for child in value.values(): - leaves.extend(_metric_leaves(child)) + leaves: list[tuple[str, object]] = [] + for key, child in value.items(): + leaves.extend(_metric_leaves(child, f"{path}.{key}" if path else str(key))) return leaves if isinstance(value, list): leaves = [] - for child in value: - leaves.extend(_metric_leaves(child)) + for index, child in enumerate(value): + leaves.extend(_metric_leaves(child, f"{path}[{index}]")) return leaves - return [value] + return [(path, value)] def corpus_claim_ids(corpus_text: str) -> list[str]: @@ -117,7 +143,15 @@ def corpus_claim_ids(corpus_text: str) -> list[str]: def metric_group_errors(name: str, group: object) -> list[str]: - """A metric group is measured-with-method or typed unsupported. Nothing else.""" + """A metric group is measured-with-method or typed unsupported. Nothing else. + + Inside a measured group, every exact-zero number must be acknowledged: an + unmeasurable quantity is a typed-unsupported marker + (`{"status": "unsupported", "reason": ...}`), and a genuinely measured zero + is listed in `measured_zero_fields` by its dotted path. That is what makes + "unsupported is never silently zero" an enforced rule instead of a promise: + a zero cannot reach a packet without its author naming which kind it is. + """ errors: list[str] = [] if not isinstance(group, dict) or not group: return [f"metrics.{name} must be a non-empty object"] @@ -138,18 +172,74 @@ def metric_group_errors(name: str, group: object) -> list[str]: f"metrics.{name} must record a non-empty measurement 'method' " "(or be typed unsupported with a reason)" ) - value_keys = [key for key in group if key != "method"] + + declared_zero_fields = group.get("measured_zero_fields", []) + if not isinstance(declared_zero_fields, list) or not all( + isinstance(item, str) for item in declared_zero_fields + ): + errors.append( + f"metrics.{name}.measured_zero_fields must be a list of dotted " + "field paths" + ) + declared_zero_fields = [] + + value_keys = [key for key in group if key not in {"method", "measured_zero_fields"}] if not value_keys: errors.append(f"metrics.{name} has a method but no measured values") + + observed_zero_fields: list[str] = [] + leaf_count = 0 for key in value_keys: - for leaf in _metric_leaves(group[key]): + leaves = _metric_leaves(group[key], key) + leaf_count += len(leaves) + for path, leaf in leaves: + if _is_unsupported_leaf(leaf): + if not _is_nonempty_str(leaf.get("reason")): + errors.append( + f"metrics.{name}.{path} is typed unsupported but has " + "no non-empty reason" + ) + continue if leaf is None: errors.append( - f"metrics.{name}.{key} contains null; unsupported values " + f"metrics.{name}.{path} contains null; unsupported values " "must be typed, not null" ) - elif isinstance(leaf, (int, float)) and not _is_finite_number(leaf): - errors.append(f"metrics.{name}.{key} contains a non-finite number") + elif ( + isinstance(leaf, str) and leaf.strip().lower() in UNSUPPORTED_SENTINELS + ): + errors.append( + f"metrics.{name}.{path} uses the placeholder {leaf!r}; " + "unsupported values must be typed, not spelled" + ) + elif isinstance(leaf, (int, float)) and not isinstance(leaf, bool): + if not math.isfinite(leaf): + errors.append(f"metrics.{name}.{path} contains a non-finite number") + elif leaf == 0: + observed_zero_fields.append(path) + + if value_keys and leaf_count == 0: + errors.append( + f"metrics.{name} has a method but only empty containers; that is " + "not a measurement" + ) + + unacknowledged = [ + path for path in observed_zero_fields if path not in declared_zero_fields + ] + if unacknowledged: + errors.append( + f"metrics.{name} reports exact zero at {sorted(set(unacknowledged))} " + "without acknowledgement; type each unmeasurable value as " + "{'status': 'unsupported', 'reason': ...} or list a genuinely " + "measured zero in measured_zero_fields" + ) + stale = [path for path in declared_zero_fields if path not in observed_zero_fields] + if stale: + errors.append( + f"metrics.{name}.measured_zero_fields lists {sorted(set(stale))} " + "which are not zero in this packet" + ) return errors @@ -158,6 +248,7 @@ def packet_errors( *, known_claim_ids: set[str] | None = None, expected_branches: tuple[str, ...] = ("main", "release-6.20"), + base_dir: Path | None = None, ) -> list[str]: """Return every fail-closed violation for one evidence packet.""" if not isinstance(packet, dict): @@ -251,8 +342,14 @@ def packet_errors( "ensemble must record deterministic_repeats >= 2, a sweep of " ">= 2 points, or >= 2 seeds; single runs are not evidence" ) + window = ensemble.get("measurement_window") if "measurement_window" not in ensemble: errors.append("ensemble.measurement_window is required") + elif not window or (isinstance(window, str) and not window.strip()): + errors.append( + "ensemble.measurement_window must record an actual window, " + "not an empty value" + ) metrics = packet.get("metrics") if not isinstance(metrics, dict): @@ -283,6 +380,19 @@ def packet_errors( has_rows = isinstance(raw_rows, list) and bool(raw_rows) if not (has_paths or has_rows): errors.append("evidence must carry raw_rows inline or non-empty raw_paths") + if has_paths: + for index, raw_path in enumerate(raw_paths): + if not _is_nonempty_str(raw_path): + errors.append( + f"evidence.raw_paths[{index}] must be a non-empty string" + ) + elif base_dir is not None and not any( + (root / raw_path).exists() for root in (base_dir, REPO_ROOT) + ): + errors.append( + f"evidence.raw_paths[{index}] {raw_path!r} does not " + "resolve to an existing file; a dangling path is prose" + ) visual = evidence.get("visual") if isinstance(visual, dict): if visual.get("status") != "not-applicable" or not _is_nonempty_str( @@ -503,8 +613,23 @@ def validate_tree( for lane_name, lane in lanes.items(): if not isinstance(lane, dict): continue - for rel in lane.get("evidence") or []: + lane_paths = lane.get("evidence") + if lane_paths is not None and not isinstance(lane_paths, list): + errors.append( + f"{manifest_path}: {claim_id}.lanes.{lane_name}" + ".evidence must be a list" + ) + lane_paths = [] + for rel in lane_paths or []: if isinstance(rel, str): + if rel in lane_evidence: + owner = lane_evidence[rel] + errors.append( + f"{manifest_path}: packet {rel} is claimed " + f"by both {owner[0]}.{owner[1]} and " + f"{claim_id}.{lane_name}; one packet has " + "one owner" + ) lane_evidence[rel] = ( str(claim_id), str(lane_name), @@ -512,6 +637,10 @@ def validate_tree( if lane.get("status") == "closed": closed_lane_packets.add(rel) + # Every lane-referenced path is validated as a packet, wherever it sits. + # Enumerating only `evidence/*.json` would let a lane close a row with a + # file the packet checks never reach (prose, an empty object, or the + # negative control itself) simply by living one directory deeper. for rel, (claim_id, lane_name) in sorted(lane_evidence.items()): packet_path = plan_dir / rel if not packet_path.is_file(): @@ -519,13 +648,48 @@ def validate_tree( f"{manifest_path}: {claim_id}.lanes.{lane_name} references " f"missing packet {rel}" ) + continue + if packet_path.suffix != ".json": + errors.append( + f"{manifest_path}: {claim_id}.lanes.{lane_name} references " + f"{rel} which is not a .json packet" + ) + try: + relative = packet_path.resolve().relative_to(evidence_dir.resolve()) + except ValueError: + errors.append( + f"{manifest_path}: {claim_id}.lanes.{lane_name} references " + f"{rel} outside evidence/" + ) + continue + if relative.parts and relative.parts[0] == "negative-controls": + errors.append( + f"{manifest_path}: {claim_id}.lanes.{lane_name} references " + f"negative control {rel}; a control proves the validator " + "fails closed and can never be a claim's evidence" + ) - packet_paths = sorted(evidence_dir.glob("*.json")) if evidence_dir.is_dir() else [] + packet_paths = ( + sorted( + path + for path in evidence_dir.rglob("*.json") + if negative_dir.resolve() not in path.resolve().parents + ) + if evidence_dir.is_dir() + else [] + ) + for rel in lane_evidence: + candidate = plan_dir / rel + if candidate.is_file() and candidate not in packet_paths: + packet_paths.append(candidate) + packet_paths = sorted(set(packet_paths)) for packet_path in packet_paths: packet = _load_json(packet_path, errors) if packet is None: continue - issues = packet_errors(packet, known_claim_ids=known_ids or None) + issues = packet_errors( + packet, known_claim_ids=known_ids or None, base_dir=plan_dir + ) errors.extend(f"{packet_path}: {issue}" for issue in issues) if issues or not isinstance(packet, dict): continue diff --git a/scripts/citation_packet_utils.py b/scripts/citation_packet_utils.py new file mode 100644 index 0000000000000..090f75f84cbd3 --- /dev/null +++ b/scripts/citation_packet_utils.py @@ -0,0 +1,95 @@ +"""Shared helpers for PLAN-123 citation evidence packets. + +Centralizes the typed-unsupported markers for quantities DART 7 `main` does +not currently report, so no packet can quietly publish a sentinel zero as if +it were a measurement. Each marker names the exact code path that makes the +quantity unavailable, so a reader can check the claim and a later slice can +delete the marker once WS4 exposes the value. +""" + +from __future__ import annotations + +from typing import Any + +# `recordSolverDiagnostics(World&, std::size_t iterations, double residual = +# 0.0)` in dart/simulation/compute/rigid_body_contact_stage.cpp is called from +# every rigid contact path without a residual argument, so +# StepMetrics.last_step_residual is structurally 0.0 on this path -- "not +# computed", never "converged to zero". +UNSUPPORTED_SOLVER_RESIDUAL: dict[str, str] = { + "status": "unsupported", + "reason": ( + "StepMetrics.last_step_residual is structurally zero for rigid " + "contact on this branch: recordSolverDiagnostics() in " + "dart/simulation/compute/rigid_body_contact_stage.cpp takes " + "residual = 0.0 by default and no rigid contact call site passes " + "one, so no residual is computed for either contact solver. " + "Exposing a comparable residual is PLAN-123 WS4 work." + ), +} + +# The BoxedLcp branch (rigid_body_contact_stage.cpp, `if +# (world.getContactSolverMethod() == ContactSolverMethod::BoxedLcp)`) returns +# before any recordSolverDiagnostics() call, so a 0 iteration count means +# "not recorded", not "solved in zero iterations". +UNSUPPORTED_BOXED_LCP_ITERATIONS: dict[str, str] = { + "status": "unsupported", + "reason": ( + "The BoxedLcp branch in " + "dart/simulation/compute/rigid_body_contact_stage.cpp returns before " + "recordSolverDiagnostics() runs, so StepMetrics.last_step_iterations " + "is never written for this method; a recorded 0 would mean 'not " + "reported', not 'zero iterations'." + ), +} + +# The sequential-impulse path records the *configured* iteration count +# (`recordSolverDiagnostics(world, m_iterations)`), and its Gauss-Seidel loop +# runs a fixed number of sweeps with no convergence exit, so the number is a +# setting echoed back rather than an observed convergence measurement. +SEQUENTIAL_IMPULSE_ITERATIONS_NOTE = ( + "Sequential impulse records the configured iteration count " + "(recordSolverDiagnostics(world, m_iterations)); its Gauss-Seidel loop " + "has no convergence exit, so this is the configured sweep count, not an " + "observed iteration-to-convergence." +) + +# StepMetrics.max_penetration_depth is accumulated as +# `std::max(metrics.maxPenetrationDepth, std::max(0.0, contact.depth))`, so it +# is one-sided: 0.0 means "no positive penetration observed" and cannot be +# distinguished from "contacts reported non-positive depth". +PENETRATION_CLAMP_NOTE = ( + "StepMetrics.max_penetration_depth clamps each contact depth with " + "std::max(0.0, depth), so a reported 0.0 means no positive penetration " + "was observed and does not distinguish resting-exactly-tangent from " + "separated contacts." +) + + +# A ratio against a peak of exactly zero is undefined, not zero: with no +# signal there is nothing for the antisymmetry residual to be relative to. +UNSUPPORTED_ANTISYMMETRY_RATIO: dict[str, str] = { + "status": "unsupported", + "reason": ( + "Peak lateral drift is exactly zero for this method, so the " + "antisymmetry-to-peak ratio is undefined rather than zero." + ), +} + + +def solver_iterations_by_method( + iterations_by_method: dict[str, int], +) -> dict[str, Any]: + """Type per-method iteration counts, marking unreported methods. + + A method that never reaches `recordSolverDiagnostics()` is emitted as a + typed-unsupported marker instead of the sentinel 0 the runtime leaves in + place. + """ + typed: dict[str, Any] = {} + for method, value in sorted(iterations_by_method.items()): + if method == "BOXED_LCP": + typed[method] = dict(UNSUPPORTED_BOXED_LCP_ITERATIONS) + else: + typed[method] = value + return typed diff --git a/scripts/write_citation_ct001_rolling_direction_packet.py b/scripts/write_citation_ct001_rolling_direction_packet.py index e1c30276a0026..b6aa2d61da5ce 100644 --- a/scripts/write_citation_ct001_rolling_direction_packet.py +++ b/scripts/write_citation_ct001_rolling_direction_packet.py @@ -44,6 +44,13 @@ import dartpy as sx import numpy as np +from citation_packet_utils import ( + PENETRATION_CLAMP_NOTE, + SEQUENTIAL_IMPULSE_ITERATIONS_NOTE, + UNSUPPORTED_ANTISYMMETRY_RATIO, + UNSUPPORTED_SOLVER_RESIDUAL, + solver_iterations_by_method, +) REPO_ROOT = Path(__file__).resolve().parents[1] DEFAULT_OUTPUT = ( @@ -172,12 +179,23 @@ def run_single( contact_count_max = max(contact_count_max, int(metrics.active_contact_count)) if slide_end_time is None: + # Contact-point slip as a planar vector, so a transient sign + # cancellation in one component cannot be mistaken for rolling. + # The contact point moves at v + omega x (-R z_hat), whose planar + # part is (v_x - R*w_y, v_y + R*w_x). + slip_vector = np.array( + [ + velocity[0] - radius * angular[1], + velocity[1] + radius * angular[0], + ] + ) + slip = float(np.linalg.norm(slip_vector)) planar_speed = float(np.linalg.norm(velocity[:2])) - spin_axis = np.array([-direction[1], direction[0], 0.0]) - surface_speed = radius * float(np.dot(angular, spin_axis)) - slip = abs(planar_speed - surface_speed) - denom = planar_speed + abs(surface_speed) + 1.0e-9 - if slip / denom < float(parameters["slip_ratio_rolling_threshold"]): + surface_speed = radius * float(np.linalg.norm(angular[:2])) + denom = planar_speed + surface_speed + 1.0e-9 + rolling = slip / denom < float(parameters["slip_ratio_rolling_threshold"]) + # A sphere at rest has no slip but is not rolling; require motion. + if rolling and planar_speed > 1.0e-6: slide_end_time = (step_index + 1) * dt if step_index == step_count - 1: @@ -214,7 +232,7 @@ def run_single( "max_energy_gain_j": max_energy_gain, "max_penetration_m": max_penetration, "max_solver_iterations": max_iterations, - "max_solver_residual": max_residual, + "raw_last_step_residual_max": max_residual, "max_active_contacts": contact_count_max, "final_step_stage_names": stage_names, } @@ -241,13 +259,34 @@ def summarize(rows: list[dict[str, Any]]) -> dict[str, Any]: ), "max_energy_gain_j": max(row["max_energy_gain_j"] for row in method_rows), "max_penetration_m": max(row["max_penetration_m"] for row in method_rows), - "max_solver_residual": max( - row["max_solver_residual"] for row in method_rows + "solver_residual": dict(UNSUPPORTED_SOLVER_RESIDUAL), + "antisymmetry_residual_m": antisymmetry_residual(method_rows), + "antisymmetry_residual_over_peak_drift": ( + antisymmetry_residual(method_rows) / max(drifts) + if max(drifts) > 0.0 + else dict(UNSUPPORTED_ANTISYMMETRY_RATIO) ), } return summary +def antisymmetry_residual(rows: list[dict[str, Any]]) -> float: + """Largest |d(theta) + d(90-theta)| over the sweep. + + A friction pyramid aligned to the tangent axes makes lateral drift + antisymmetric about 45 degrees, so this residual is ~0 for a genuine + pyramid signature and comparable to the peak drift for isotropic scatter. + It is what separates orientation-dependent anisotropy from contact noise. + """ + by_angle = {row["angle_deg"]: row["lateral_drift_m"] for row in rows} + residual = 0.0 + for angle, drift in by_angle.items(): + mirror = by_angle.get(90.0 - angle) + if mirror is not None: + residual = max(residual, abs(drift + mirror)) + return residual + + def git_head() -> str: return subprocess.run( ["git", "rev-parse", "HEAD"], @@ -277,13 +316,28 @@ def build_packet() -> dict[str, Any]: f"{sorted(hashes)}" ) row = repeats[0] - readback = row["resolved"] - if readback["contact_solver_method"] != method: + for repeat in repeats: + readback = repeat["resolved"] + if readback != row["resolved"]: + raise SystemExit( + f"{method} angle {angle}: resolved configuration " + f"differs between repeats: {readback} vs " + f"{row['resolved']}" + ) + if readback["contact_solver_method"] != method: + raise SystemExit( + f"requested contact solver {method} but World readback " + f"reports {readback['contact_solver_method']}" + ) + row["repeat_trajectory_sha256"] = [ + repeat["trajectory_sha256"] for repeat in repeats + ] + previous = resolved_by_method.setdefault(method, row["resolved"]) + if previous != row["resolved"]: raise SystemExit( - f"requested contact solver {method} but World readback " - f"reports {readback['contact_solver_method']}" + f"{method}: resolved configuration drifted across the " + f"sweep: {previous} vs {row['resolved']}" ) - resolved_by_method.setdefault(method, readback) rows.append(row) if determinism_failures: @@ -297,17 +351,64 @@ def build_packet() -> dict[str, Any]: "max_abs_heading_error_deg": 0.1, "travel_spread_relative": 0.01, } + # Record which criterion fired per method, and whether the drift carries + # the antisymmetric pyramid signature rather than orientation-independent + # scatter. Exceeding a tolerance is not by itself evidence of the cited + # mechanism. + anisotropy_findings: dict[str, Any] = {} + for method, stats in summary.items(): + criteria = sorted( + key + for key, tolerance in isotropy_tolerance.items() + if isinstance(stats.get(key), (int, float)) and stats[key] > tolerance + ) + ratio = stats["antisymmetry_residual_over_peak_drift"] + has_signature = isinstance(ratio, (int, float)) and ratio <= 0.05 + anisotropy_findings[method] = { + "criteria_exceeded": criteria, + "pyramid_signature": has_signature, + "signature_test": ( + "lateral drift antisymmetric about 45 deg to within 5% of " "peak drift" + ), + } anisotropic_methods = sorted( method - for method, stats in summary.items() - if stats["max_abs_lateral_drift_m"] - > isotropy_tolerance["max_abs_lateral_drift_m"] - or stats["max_abs_heading_error_deg"] - > isotropy_tolerance["max_abs_heading_error_deg"] - or stats["travel_spread_relative"] - > isotropy_tolerance["travel_spread_relative"] + for method, finding in anisotropy_findings.items() + if finding["criteria_exceeded"] and finding["pyramid_signature"] ) - disposition = "reproduced" if anisotropic_methods else "unresolved" + + # A degenerate run (tunnelling, blow-up) would also break symmetry, so the + # disposition is gated on physical validity, not on deviation alone. + validity_failures = [ + f"{row['contact_solver_method']} angle {row['angle_deg']}: {reason}" + for row in rows + for reason, bad in ( + ( + "final speed departs from the analytic rolling speed 5/7 v0", + abs( + row["final_planar_speed_mps"] + - (5.0 / 7.0) * float(parameters["launch_speed_mps"]) + ) + > 1.0e-3, + ), + ("energy injected", row["max_energy_gain_j"] > 1.0e-6), + ("never reached rolling", row["slide_end_time_s"] is None), + ) + if bad + ] + disposition = ( + "reproduced" if anisotropic_methods and not validity_failures else "unresolved" + ) + + # Zeros this fixture asserts are genuine measurements, not missing data. + # The validator rejects any other exact zero and any stale entry here, so + # an unexpected zero in a future run fails the gate instead of passing as + # a silent sentinel. + measured_zero_physical = [ + f"per_solver_summary.{method}.max_penetration_m" + for method in parameters["contact_solver_methods"] + ] + measured_zero_numerical = ["max_penetration_m"] command = ( "PYTHONPATH=build/default/cpp/Release/python pixi run python " @@ -327,6 +428,13 @@ def build_packet() -> dict[str, Any]: "target": { "branch": "main", "commit": git_head(), + "commit_role": ( + "Source state measured: the library and fixture were run at " + "this commit, which is HEAD at capture time. The packet and " + "its writer land in a later commit, so re-running the " + "recorded command requires the child commit that adds the " + "writer; the measured behavior belongs to this one." + ), }, "scene": { "id": parameters["scene_id"], @@ -349,10 +457,14 @@ def build_packet() -> dict[str, Any]: "resolved_provenance": ( "World property readback (contact_solver_method, " "rigid_body_solver, gravity, time_step) after " - "enter_simulation_mode plus final-step WorldStepProfile " - "stage names recorded per run; " - "World::getResolvedConfiguration() is not yet exposed to " - "Python (PLAN-123 WS4 follow-up)." + "enter_simulation_mode, asserted equal to the request and " + "stable across every repeat and sweep point. The independent " + "evidence that the selection changed behavior is that the " + "per-angle trajectory hashes differ between the two contact " + "solvers at every angle; the recorded WorldStepProfile stage " + "names are identical for both methods and therefore do not " + "discriminate. World::getResolvedConfiguration() is not yet " + "exposed to Python (PLAN-123 WS4 follow-up)." ), "detector": ( "DART 7 native World collision pipeline (the World step API " @@ -361,8 +473,10 @@ def build_packet() -> dict[str, Any]: "timestep": parameters["time_step_s"], "substeps": 1, "iterations": ( - "World defaults; per-step actual iterations recorded via " - "StepMetrics.last_step_iterations" + "World defaults. Sequential impulse reports its configured " + "sweep count through StepMetrics.last_step_iterations; the " + "boxed-LCP path records no iteration count at all (see " + "metrics.numerical.solver_iterations_by_method)." ), "fallback_policy": ( "World defaults; no per-island fallback reporting is exposed " @@ -377,7 +491,7 @@ def build_packet() -> dict[str, Any]: for angle in parameters["launch_angles_deg"] ], "deterministic_repeats": int(parameters["deterministic_repeats"]), - "deterministic_repeats_identical": True, + "deterministic_repeats_identical": not determinism_failures, "measurement_window": { "start_s": 0.0, "end_s": parameters["time_step_s"] * parameters["step_count"], @@ -395,20 +509,35 @@ def build_packet() -> dict[str, Any]: ), "per_solver_summary": summary, "isotropy_tolerance": isotropy_tolerance, + "anisotropy_findings": anisotropy_findings, "anisotropic_methods": anisotropic_methods, + "validity_failures": validity_failures, + "measured_zero_fields": measured_zero_physical, }, "numerical": { "method": ( - "Max over run of StepMetrics.max_penetration_depth, " - "last_step_iterations, last_step_residual, " - "active_contact_count" + "Max over run of StepMetrics.max_penetration_depth and " + "active_contact_count; per-method iteration counts where " + "the runtime records them" ), "max_penetration_m": max(row["max_penetration_m"] for row in rows), - "max_solver_iterations": max( - row["max_solver_iterations"] for row in rows + "penetration_semantics": PENETRATION_CLAMP_NOTE, + "solver_iterations_by_method": solver_iterations_by_method( + { + method: max( + row["max_solver_iterations"] + for row in rows + if row["contact_solver_method"] == method + ) + for method in parameters["contact_solver_methods"] + } + ), + "sequential_impulse_iterations_semantics": ( + SEQUENTIAL_IMPULSE_ITERATIONS_NOTE ), - "max_solver_residual": max(row["max_solver_residual"] for row in rows), + "solver_residual": dict(UNSUPPORTED_SOLVER_RESIDUAL), "max_active_contacts": max(row["max_active_contacts"] for row in rows), + "measured_zero_fields": measured_zero_numerical, }, "performance": { "status": "unsupported", @@ -451,10 +580,21 @@ def build_packet() -> dict[str, Any]: ), "limitations": [ "ResolvedSolverConfiguration is not Python-exposed; " - "resolved identity is property readback plus stage names.", - "Friction-cone/complementarity residual per contact is not " - "exposed; solver residual is the aggregate StepMetrics " - "value.", + "resolved identity is property readback, corroborated by " + "per-solver trajectory-hash differences.", + "No solver residual exists on this path: it is typed " + "unsupported, not reported as zero. Friction-cone and " + "complementarity violations are not exposed either.", + "The boxed-LCP path records no iteration count; only " + "sequential impulse reports one, and that is its configured " + "sweep count rather than an observed convergence measure.", + "Penetration is clamped at zero by the runtime, so 0.0 means " + "no positive penetration was observed and does not " + "distinguish resting-tangent from separated contacts.", + "The corpus row also names stopping distance and cone " + "violation as oracles; neither is measured here (the sphere " + "rolls indefinitely at 5/7 v0, and no cone metric is " + "exposed).", "The sweep covers 0-90 deg; pyramid orientation with a " "period other than 90 deg would need a wider sweep.", "No high-mass-ratio or stacked variant yet; those belong " diff --git a/scripts/write_citation_ct002_dense_contact_packet.py b/scripts/write_citation_ct002_dense_contact_packet.py new file mode 100644 index 0000000000000..9547a819e5ab8 --- /dev/null +++ b/scripts/write_citation_ct002_dense_contact_packet.py @@ -0,0 +1,514 @@ +#!/usr/bin/env python3 +"""Write the CT-002 dense inelastic-contact evidence packet (PLAN-123 WS2). + +Bounded claim (corpus row CT-002, motivated by the historical SimBenchmark +dense `6x6x6` contact test): dense contact may fail, become unstable, or +scale poorly for some timestep/solver settings. + +Bounded reconstruction (explicitly not source-exact): a 6x6x6 grid of 216 +spheres dropped onto a static ground box with zero restitution, run over a +timestep grid for each rigid contact solver method. Per cell it records: + +- finite-state outcome (any NaN/Inf position or velocity fails the cell); +- settle behavior: final max speed, kinetic energy at the horizon; +- max penetration depth, max active contact count, max solver iterations, + and max solver residual from `StepMetrics`; +- max single-step kinetic-energy gain after the first impact window (an + inelastic pile must dissipate, not inject, energy); +- deterministic repeats: each cell runs twice and must be bit-identical. + +The packet records requested and resolved solver identity per cell +(World property readback; `ResolvedSolverConfiguration` is not yet +Python-exposed and that gap is a recorded limitation feeding PLAN-123 WS4). + +Usage (after `pixi run build`): + + PYTHONPATH=build/default/cpp/Release/python pixi run python \ + scripts/write_citation_ct002_dense_contact_packet.py +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import platform +import subprocess +import sys +from pathlib import Path +from typing import Any + +import dartpy as sx +import numpy as np +from citation_packet_utils import ( + PENETRATION_CLAMP_NOTE, + SEQUENTIAL_IMPULSE_ITERATIONS_NOTE, + UNSUPPORTED_SOLVER_RESIDUAL, + solver_iterations_by_method, +) + +REPO_ROOT = Path(__file__).resolve().parents[1] +DEFAULT_OUTPUT = ( + REPO_ROOT + / "docs" + / "plans" + / "123-citation-driven-simulation-trust" + / "evidence" + / "CT-002-dart7-dense-inelastic-contact.json" +) + +SCENE_PARAMETERS: dict[str, Any] = { + "scene_id": "ct002_dense_inelastic_grid", + "description": ( + "6x6x6 grid of 216 spheres (radius 0.05 m, mass 0.1 kg, spacing " + "0.12 m) dropped from 0.5 m clearance onto a static ground box, " + "restitution 0, friction 0.8 on all bodies, simulated to a 2 s " + "horizon per timestep/solver cell." + ), + "gravity_mps2": [0.0, 0.0, -9.81], + "grid_dimension": 6, + "sphere_radius_m": 0.05, + "sphere_mass_kg": 0.1, + "grid_spacing_m": 0.12, + "drop_clearance_m": 0.5, + "friction": 0.8, + "restitution": 0.0, + "ground_half_extents_m": [2.0, 2.0, 0.05], + "horizon_s": 2.0, + "timesteps_s": [0.002, 0.004], + "contact_solver_methods": ["SEQUENTIAL_IMPULSE", "BOXED_LCP"], + "deterministic_repeats": 2, + "impact_settle_fraction": 0.5, +} + + +def scene_digest(parameters: dict[str, Any]) -> str: + canonical = json.dumps(parameters, sort_keys=True, separators=(",", ":")) + return "sha256:" + hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + +def _sphere_inertia(mass: float, radius: float) -> np.ndarray: + moment = 0.4 * mass * radius * radius + return np.diag([moment, moment, moment]) + + +def _transform_at(position: np.ndarray) -> np.ndarray: + transform = np.eye(4) + transform[:3, 3] = position + return transform + + +def run_single( + method_name: str, timestep: float, parameters: dict[str, Any] +) -> dict[str, Any]: + """Run one dense-grid drop and return raw metrics plus a state hash.""" + radius = float(parameters["sphere_radius_m"]) + mass = float(parameters["sphere_mass_kg"]) + dimension = int(parameters["grid_dimension"]) + spacing = float(parameters["grid_spacing_m"]) + clearance = float(parameters["drop_clearance_m"]) + ground_half = np.asarray(parameters["ground_half_extents_m"], dtype=float) + step_count = int(round(float(parameters["horizon_s"]) / timestep)) + + world = sx.World( + time_step=timestep, + gravity=np.asarray(parameters["gravity_mps2"], dtype=float), + contact_solver_method=sx.ContactSolverMethod[method_name], + ) + + ground = world.add_rigid_body("ct002_ground") + ground.is_static = True + ground.set_collision_shape(sx.CollisionShape.box(ground_half)) + ground.transform = _transform_at(np.array([0.0, 0.0, -ground_half[2]])) + ground.friction = float(parameters["friction"]) + ground.restitution = float(parameters["restitution"]) + + bodies = [] + offset = 0.5 * (dimension - 1) * spacing + for ix in range(dimension): + for iy in range(dimension): + for iz in range(dimension): + body = world.add_rigid_body(f"ct002_sphere_{ix}_{iy}_{iz}") + body.mass = mass + body.inertia = _sphere_inertia(mass, radius) + body.set_collision_shape(sx.CollisionShape.sphere(radius)) + body.friction = float(parameters["friction"]) + body.restitution = float(parameters["restitution"]) + body.transform = _transform_at( + np.array( + [ + ix * spacing - offset, + iy * spacing - offset, + clearance + radius + iz * spacing, + ] + ) + ) + bodies.append(body) + + world.enter_simulation_mode() + + resolved = { + "contact_solver_method": world.contact_solver_method.name, + "rigid_body_solver": world.rigid_body_solver.name, + "gravity_mps2": [float(v) for v in np.asarray(world.gravity)], + "time_step_s": float(world.time_step), + } + + settle_step = int(step_count * float(parameters["impact_settle_fraction"])) + state_hash = hashlib.sha256() + non_finite = False + max_penetration = 0.0 + max_iterations = 0 + max_residual = 0.0 + max_contacts = 0 + max_energy_gain_after_settle = 0.0 + previous_kinetic = None + initial_metrics = world.compute_step_metrics() + initial_total_energy = float(initial_metrics.total_energy) + + for step_index in range(step_count): + world.step() + metrics = world.compute_step_metrics() + kinetic = float(metrics.kinetic_energy) + if not math.isfinite(kinetic): + non_finite = True + break + if previous_kinetic is not None and step_index >= settle_step: + max_energy_gain_after_settle = max( + max_energy_gain_after_settle, kinetic - previous_kinetic + ) + previous_kinetic = kinetic + max_penetration = max(max_penetration, float(metrics.max_penetration_depth)) + max_iterations = max(max_iterations, int(metrics.last_step_iterations)) + max_residual = max(max_residual, float(metrics.last_step_residual)) + max_contacts = max(max_contacts, int(metrics.active_contact_count)) + + speeds = [] + min_height = math.inf + if not non_finite: + for body in bodies: + velocity = np.asarray(body.linear_velocity, dtype=float) + position = np.asarray(body.translation, dtype=float) + if not (np.all(np.isfinite(velocity)) and np.all(np.isfinite(position))): + non_finite = True + break + speeds.append(float(np.linalg.norm(velocity))) + min_height = min(min_height, float(position[2])) + state_hash.update(position.tobytes()) + state_hash.update(velocity.tobytes()) + + final_metrics = world.compute_step_metrics() + return { + "contact_solver_method": method_name, + "timestep_s": timestep, + "steps": step_count, + "resolved": resolved, + "finite": not non_finite, + "final_state_sha256": state_hash.hexdigest() if not non_finite else None, + "final_max_speed_mps": max(speeds) if speeds else None, + "final_min_height_m": min_height if speeds else None, + "final_kinetic_energy_j": ( + float(final_metrics.kinetic_energy) if not non_finite else None + ), + "initial_total_energy_j": initial_total_energy, + "max_energy_gain_after_settle_j": max_energy_gain_after_settle, + "max_penetration_m": max_penetration, + "max_solver_iterations": max_iterations, + "raw_last_step_residual_max": max_residual, + "max_active_contacts": max_contacts, + } + + +def build_packet() -> dict[str, Any]: + parameters = SCENE_PARAMETERS + rows: list[dict[str, Any]] = [] + determinism_failures: list[str] = [] + resolved_by_method: dict[str, dict[str, Any]] = {} + + for method in parameters["contact_solver_methods"]: + for timestep in parameters["timesteps_s"]: + repeats = [ + run_single(method, timestep, parameters) + for _ in range(int(parameters["deterministic_repeats"])) + ] + hashes = {run["final_state_sha256"] for run in repeats} + if len(hashes) != 1: + determinism_failures.append( + f"{method} dt {timestep}: final-state hashes differ " + f"{sorted(map(str, hashes))}" + ) + row = repeats[0] + readback = row["resolved"] + if readback["contact_solver_method"] != method: + raise SystemExit( + f"requested contact solver {method} but World readback " + f"reports {readback['contact_solver_method']}" + ) + resolved_by_method.setdefault(method, readback) + rows.append(row) + + if determinism_failures: + raise SystemExit( + "deterministic repeats failed:\n " + "\n ".join(determinism_failures) + ) + + all_finite = all(row["finite"] for row in rows) + max_penetration = max(row["max_penetration_m"] for row in rows) + max_gain = max(row["max_energy_gain_after_settle_j"] for row in rows) + settle_speed = max(row["final_max_speed_mps"] for row in rows if row["finite"]) + stability_tolerance = { + "max_penetration_m": 0.5 * float(parameters["sphere_radius_m"]), + "final_max_speed_mps": 0.05, + "max_energy_gain_after_settle_j": 1.0e-3, + } + unstable_cells = [ + { + "contact_solver_method": row["contact_solver_method"], + "timestep_s": row["timestep_s"], + "reasons": [ + reason + for reason, bad in ( + ("non-finite state", not row["finite"]), + ( + "penetration above half radius", + row["max_penetration_m"] + > stability_tolerance["max_penetration_m"], + ), + ( + "pile still moving at horizon", + row["finite"] + and row["final_max_speed_mps"] + > stability_tolerance["final_max_speed_mps"], + ), + ( + "energy injected after settle window", + row["max_energy_gain_after_settle_j"] + > stability_tolerance["max_energy_gain_after_settle_j"], + ), + ) + if bad + ], + } + for row in rows + ] + unstable_cells = [cell for cell in unstable_cells if cell["reasons"]] + disposition = "reproduced" if unstable_cells else "unresolved" + + command = ( + "PYTHONPATH=build/default/cpp/Release/python pixi run python " + "scripts/write_citation_ct002_dense_contact_packet.py" + ) + packet: dict[str, Any] = { + "schema": "dart.citation_claim_evidence/v1", + "claim_id": "CT-002", + "title": ("Dense 6x6x6 inelastic contact stability (DART 7 first packet)"), + "source": { + "url": "https://leggedrobotics.github.io/SimBenchmark/", + "claim": ( + "Dense contact may fail, become unstable, or scale poorly " + "for some timestep/solver settings." + ), + }, + "target": {"branch": "main", "commit": git_head()}, + "scene": { + "id": parameters["scene_id"], + "digest": scene_digest(parameters), + "description": parameters["description"], + "parameters": parameters, + }, + "configuration": { + "requested": { + "contact_solver_method_sweep": parameters["contact_solver_methods"], + "timestep_sweep_s": parameters["timesteps_s"], + "rigid_body_solver": "SEQUENTIAL_IMPULSE (World default)", + "integrator": "World default semi-implicit stepping", + "precision": "float64", + "backend": "cpu", + "threads": "World default sequential step", + }, + "resolved": {"by_contact_solver_method": resolved_by_method}, + "resolved_provenance": ( + "World property readback (contact_solver_method, " + "rigid_body_solver, gravity, time_step) after " + "enter_simulation_mode, asserted equal to the request per " + "cell; World::getResolvedConfiguration() is not yet exposed " + "to Python (PLAN-123 WS4 follow-up)." + ), + "detector": ( + "DART 7 native World collision pipeline (the World step API " + "exposes no detector selection on main)" + ), + "timestep": min(parameters["timesteps_s"]), + "substeps": 1, + "iterations": ( + "World defaults; per-step actual iterations recorded via " + "StepMetrics.last_step_iterations" + ), + "fallback_policy": ( + "World defaults; no per-island fallback reporting is exposed " + "on main (PLAN-123 WS4 follow-up)" + ), + }, + "ensemble": { + "kind": "parameter-sweep-with-deterministic-repeats", + "sweep": [ + {"contact_solver_method": method, "timestep_s": timestep} + for method in parameters["contact_solver_methods"] + for timestep in parameters["timesteps_s"] + ], + "deterministic_repeats": int(parameters["deterministic_repeats"]), + "deterministic_repeats_identical": True, + "measurement_window": { + "start_s": 0.0, + "end_s": parameters["horizon_s"], + "settle_window_start_fraction": parameters["impact_settle_fraction"], + }, + }, + "metrics": { + "physical": { + "method": ( + "Per-cell finite-state check over 216 bodies, final max " + "body speed, final min body height, kinetic energy at " + "the horizon, and max single-step kinetic-energy gain " + "after the settle window from StepMetrics" + ), + "all_cells_finite": all_finite, + "max_penetration_m": max_penetration, + "settled_final_max_speed_mps": settle_speed, + "max_energy_gain_after_settle_j": max_gain, + "stability_tolerance": stability_tolerance, + "unstable_cells": unstable_cells, + }, + "numerical": { + "method": ( + "Max over run of StepMetrics.max_penetration_depth and " + "active_contact_count; per-method iteration counts where " + "the runtime records them" + ), + "max_penetration_m": max_penetration, + "penetration_semantics": PENETRATION_CLAMP_NOTE, + "solver_iterations_by_method": solver_iterations_by_method( + { + method: max( + row["max_solver_iterations"] + for row in rows + if row["contact_solver_method"] == method + ) + for method in parameters["contact_solver_methods"] + } + ), + "sequential_impulse_iterations_semantics": ( + SEQUENTIAL_IMPULSE_ITERATIONS_NOTE + ), + "solver_residual": dict(UNSUPPORTED_SOLVER_RESIDUAL), + "max_active_contacts": max(row["max_active_contacts"] for row in rows), + }, + "performance": { + "status": "unsupported", + "reason": ( + "This packet makes no timing or scaling claim; " + "interleaved same-host methodology is required before " + "any performance row" + ), + }, + "allocation": { + "status": "unsupported", + "reason": ( + "Post-bake allocation gates are owned by PLAN-122 " + "tooling; this packet does not measure allocations" + ), + }, + }, + "evidence": { + "commands": [command], + "raw_rows": rows, + "visual": { + "status": "not-applicable", + "reason": ( + "The oracle is numeric (finite state, settle speed, " + "penetration, energy monotonicity); no visible-behavior " + "claim is made by this packet." + ), + }, + }, + "result": { + "disposition": disposition, + "claim_boundary": ( + "DART 7 main, this commit, a bounded (not source-exact) " + "6x6x6 sphere-grid drop with restitution 0 and mu=0.8, " + "2 s horizon, timesteps 2 and 4 ms, SEQUENTIAL_IMPULSE and " + "BOXED_LCP contact solvers. Says nothing about the original " + "SimBenchmark scene parameters, historical DART versions, " + "other densities/materials, or DART 6." + ), + "limitations": [ + "Bounded reconstruction: the original SimBenchmark asset, " + "material, and timestep grid are not reproduced exactly; " + "sourcing the exact historical setup is future corpus work.", + "ResolvedSolverConfiguration is not Python-exposed; " + "resolved identity is property readback.", + "Two timesteps and two solvers only; a wider grid belongs " + "to a follow-up once per-island diagnostics exist.", + "No per-contact residual/cone reporting is available on " + "main; solver residual is the aggregate StepMetrics value.", + ], + }, + "review": {"passes": []}, + "host": { + "platform": platform.platform(), + "python": sys.version.split()[0], + "machine": platform.machine(), + "performance_valid": False, + "note": ( + "Host recorded for provenance only; no timing methodology " + "was applied" + ), + }, + } + return packet + + +def git_head() -> str: + return subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=REPO_ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--output", + type=Path, + default=DEFAULT_OUTPUT, + help=f"Packet output path (default: {DEFAULT_OUTPUT})", + ) + return parser.parse_args() + + +def main() -> int: + args = parse_args() + packet = build_packet() + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text( + json.dumps(packet, indent=2, sort_keys=True) + "\n", encoding="utf-8" + ) + physical = packet["metrics"]["physical"] + print(f"wrote {args.output}") + print( + f" all finite: {physical['all_cells_finite']}; max penetration " + f"{physical['max_penetration_m']:.3e} m; settled max speed " + f"{physical['settled_final_max_speed_mps']:.3e} m/s; max post-settle " + f"energy gain {physical['max_energy_gain_after_settle_j']:.3e} J" + ) + print(f" unstable cells: {physical['unstable_cells']}") + print(f" disposition: {packet['result']['disposition']}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/write_citation_ct003_elastic_contact_packet.py b/scripts/write_citation_ct003_elastic_contact_packet.py new file mode 100644 index 0000000000000..a92d2cc0cf738 --- /dev/null +++ b/scripts/write_citation_ct003_elastic_contact_packet.py @@ -0,0 +1,492 @@ +#!/usr/bin/env python3 +"""Write the CT-003 dense elastic-contact evidence packet (PLAN-123 WS2). + +Bounded claim (corpus row CT-003, motivated by the historical SimBenchmark +dense elastic contact test): elastic dense contact may inject energy or +expose solver failure. + +Bounded reconstruction (explicitly not source-exact): the same 6x6x6 sphere +grid as CT-002 but with restitution 0.8 on every body, dropped onto a static +ground box and simulated over a timestep grid for each rigid contact solver +method. Per cell it records: + +- finite-state outcome (any NaN/Inf fails the cell); +- the mechanical-energy envelope: max total energy over the run relative to + the initial total energy (an elastic pile with restitution < 1 must never + exceed its initial mechanical energy); +- max penetration, max active contacts, max solver iterations/residual from + `StepMetrics`; +- deterministic repeats: each cell runs twice and must be bit-identical. + +Energy accounting uses `StepMetrics.total_energy` (kinetic + gravitational +potential with the world-origin zero reference), so the envelope comparison +is solver-independent and self-consistent. + +Usage (after `pixi run build`): + + PYTHONPATH=build/default/cpp/Release/python pixi run python \ + scripts/write_citation_ct003_elastic_contact_packet.py +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import platform +import subprocess +import sys +from pathlib import Path +from typing import Any + +import dartpy as sx +import numpy as np +from citation_packet_utils import ( + PENETRATION_CLAMP_NOTE, + SEQUENTIAL_IMPULSE_ITERATIONS_NOTE, + UNSUPPORTED_SOLVER_RESIDUAL, + solver_iterations_by_method, +) + +REPO_ROOT = Path(__file__).resolve().parents[1] +DEFAULT_OUTPUT = ( + REPO_ROOT + / "docs" + / "plans" + / "123-citation-driven-simulation-trust" + / "evidence" + / "CT-003-dart7-dense-elastic-contact.json" +) + +SCENE_PARAMETERS: dict[str, Any] = { + "scene_id": "ct003_dense_elastic_grid", + "description": ( + "6x6x6 grid of 216 spheres (radius 0.05 m, mass 0.1 kg, spacing " + "0.12 m) dropped from 0.5 m clearance onto a static ground box, " + "restitution 0.8, friction 0.8 on all bodies, simulated to a 2 s " + "horizon per timestep/solver cell." + ), + "gravity_mps2": [0.0, 0.0, -9.81], + "grid_dimension": 6, + "sphere_radius_m": 0.05, + "sphere_mass_kg": 0.1, + "grid_spacing_m": 0.12, + "drop_clearance_m": 0.5, + "friction": 0.8, + "restitution": 0.8, + "ground_half_extents_m": [2.0, 2.0, 0.05], + "horizon_s": 2.0, + "timesteps_s": [0.002, 0.004], + "contact_solver_methods": ["SEQUENTIAL_IMPULSE", "BOXED_LCP"], + "deterministic_repeats": 2, +} + + +def scene_digest(parameters: dict[str, Any]) -> str: + canonical = json.dumps(parameters, sort_keys=True, separators=(",", ":")) + return "sha256:" + hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + +def _sphere_inertia(mass: float, radius: float) -> np.ndarray: + moment = 0.4 * mass * radius * radius + return np.diag([moment, moment, moment]) + + +def _transform_at(position: np.ndarray) -> np.ndarray: + transform = np.eye(4) + transform[:3, 3] = position + return transform + + +def run_single( + method_name: str, timestep: float, parameters: dict[str, Any] +) -> dict[str, Any]: + """Run one elastic-grid drop and return raw metrics plus a state hash.""" + radius = float(parameters["sphere_radius_m"]) + mass = float(parameters["sphere_mass_kg"]) + dimension = int(parameters["grid_dimension"]) + spacing = float(parameters["grid_spacing_m"]) + clearance = float(parameters["drop_clearance_m"]) + ground_half = np.asarray(parameters["ground_half_extents_m"], dtype=float) + step_count = int(round(float(parameters["horizon_s"]) / timestep)) + + world = sx.World( + time_step=timestep, + gravity=np.asarray(parameters["gravity_mps2"], dtype=float), + contact_solver_method=sx.ContactSolverMethod[method_name], + ) + + ground = world.add_rigid_body("ct003_ground") + ground.is_static = True + ground.set_collision_shape(sx.CollisionShape.box(ground_half)) + ground.transform = _transform_at(np.array([0.0, 0.0, -ground_half[2]])) + ground.friction = float(parameters["friction"]) + ground.restitution = float(parameters["restitution"]) + + bodies = [] + offset = 0.5 * (dimension - 1) * spacing + for ix in range(dimension): + for iy in range(dimension): + for iz in range(dimension): + body = world.add_rigid_body(f"ct003_sphere_{ix}_{iy}_{iz}") + body.mass = mass + body.inertia = _sphere_inertia(mass, radius) + body.set_collision_shape(sx.CollisionShape.sphere(radius)) + body.friction = float(parameters["friction"]) + body.restitution = float(parameters["restitution"]) + body.transform = _transform_at( + np.array( + [ + ix * spacing - offset, + iy * spacing - offset, + clearance + radius + iz * spacing, + ] + ) + ) + bodies.append(body) + + world.enter_simulation_mode() + + resolved = { + "contact_solver_method": world.contact_solver_method.name, + "rigid_body_solver": world.rigid_body_solver.name, + "gravity_mps2": [float(v) for v in np.asarray(world.gravity)], + "time_step_s": float(world.time_step), + } + + initial_metrics = world.compute_step_metrics() + initial_total_energy = float(initial_metrics.total_energy) + + state_hash = hashlib.sha256() + non_finite = False + max_total_energy = initial_total_energy + max_penetration = 0.0 + max_iterations = 0 + max_residual = 0.0 + max_contacts = 0 + + for _ in range(step_count): + world.step() + metrics = world.compute_step_metrics() + total_energy = float(metrics.total_energy) + if not math.isfinite(total_energy): + non_finite = True + break + max_total_energy = max(max_total_energy, total_energy) + max_penetration = max(max_penetration, float(metrics.max_penetration_depth)) + max_iterations = max(max_iterations, int(metrics.last_step_iterations)) + max_residual = max(max_residual, float(metrics.last_step_residual)) + max_contacts = max(max_contacts, int(metrics.active_contact_count)) + + if not non_finite: + for body in bodies: + velocity = np.asarray(body.linear_velocity, dtype=float) + position = np.asarray(body.translation, dtype=float) + if not (np.all(np.isfinite(velocity)) and np.all(np.isfinite(position))): + non_finite = True + break + state_hash.update(position.tobytes()) + state_hash.update(velocity.tobytes()) + + final_metrics = world.compute_step_metrics() + return { + "contact_solver_method": method_name, + "timestep_s": timestep, + "steps": step_count, + "resolved": resolved, + "finite": not non_finite, + "final_state_sha256": state_hash.hexdigest() if not non_finite else None, + "initial_total_energy_j": initial_total_energy, + "max_total_energy_j": max_total_energy, + "energy_envelope_excess_j": max_total_energy - initial_total_energy, + "final_total_energy_j": ( + float(final_metrics.total_energy) if not non_finite else None + ), + "max_penetration_m": max_penetration, + "max_solver_iterations": max_iterations, + "raw_last_step_residual_max": max_residual, + "max_active_contacts": max_contacts, + } + + +def git_head() -> str: + return subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=REPO_ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + + +def build_packet() -> dict[str, Any]: + parameters = SCENE_PARAMETERS + rows: list[dict[str, Any]] = [] + determinism_failures: list[str] = [] + resolved_by_method: dict[str, dict[str, Any]] = {} + + for method in parameters["contact_solver_methods"]: + for timestep in parameters["timesteps_s"]: + repeats = [ + run_single(method, timestep, parameters) + for _ in range(int(parameters["deterministic_repeats"])) + ] + hashes = {run["final_state_sha256"] for run in repeats} + if len(hashes) != 1: + determinism_failures.append( + f"{method} dt {timestep}: final-state hashes differ " + f"{sorted(map(str, hashes))}" + ) + row = repeats[0] + readback = row["resolved"] + if readback["contact_solver_method"] != method: + raise SystemExit( + f"requested contact solver {method} but World readback " + f"reports {readback['contact_solver_method']}" + ) + resolved_by_method.setdefault(method, readback) + rows.append(row) + + if determinism_failures: + raise SystemExit( + "deterministic repeats failed:\n " + "\n ".join(determinism_failures) + ) + + all_finite = all(row["finite"] for row in rows) + max_excess = max(row["energy_envelope_excess_j"] for row in rows) + initial_energy = rows[0]["initial_total_energy_j"] + envelope_tolerance_j = 1.0e-6 * max(1.0, abs(initial_energy)) + violating_cells = [ + { + "contact_solver_method": row["contact_solver_method"], + "timestep_s": row["timestep_s"], + "reasons": [ + reason + for reason, bad in ( + ("non-finite state", not row["finite"]), + ( + "energy envelope exceeded initial total energy", + row["energy_envelope_excess_j"] > envelope_tolerance_j, + ), + ) + if bad + ], + } + for row in rows + ] + violating_cells = [cell for cell in violating_cells if cell["reasons"]] + disposition = "reproduced" if violating_cells else "unresolved" + + command = ( + "PYTHONPATH=build/default/cpp/Release/python pixi run python " + "scripts/write_citation_ct003_elastic_contact_packet.py" + ) + packet: dict[str, Any] = { + "schema": "dart.citation_claim_evidence/v1", + "claim_id": "CT-003", + "title": "Dense elastic contact energy envelope (DART 7 first packet)", + "source": { + "url": "https://leggedrobotics.github.io/SimBenchmark/", + "claim": ( + "Elastic dense contact may inject energy or expose solver " "failure." + ), + }, + "target": {"branch": "main", "commit": git_head()}, + "scene": { + "id": parameters["scene_id"], + "digest": scene_digest(parameters), + "description": parameters["description"], + "parameters": parameters, + }, + "configuration": { + "requested": { + "contact_solver_method_sweep": parameters["contact_solver_methods"], + "timestep_sweep_s": parameters["timesteps_s"], + "rigid_body_solver": "SEQUENTIAL_IMPULSE (World default)", + "integrator": "World default semi-implicit stepping", + "precision": "float64", + "backend": "cpu", + "threads": "World default sequential step", + }, + "resolved": {"by_contact_solver_method": resolved_by_method}, + "resolved_provenance": ( + "World property readback (contact_solver_method, " + "rigid_body_solver, gravity, time_step) after " + "enter_simulation_mode, asserted equal to the request per " + "cell; World::getResolvedConfiguration() is not yet exposed " + "to Python (PLAN-123 WS4 follow-up)." + ), + "detector": ( + "DART 7 native World collision pipeline (the World step API " + "exposes no detector selection on main)" + ), + "timestep": min(parameters["timesteps_s"]), + "substeps": 1, + "iterations": ( + "World defaults; per-step actual iterations recorded via " + "StepMetrics.last_step_iterations" + ), + "fallback_policy": ( + "World defaults; no per-island fallback reporting is exposed " + "on main (PLAN-123 WS4 follow-up)" + ), + }, + "ensemble": { + "kind": "parameter-sweep-with-deterministic-repeats", + "sweep": [ + {"contact_solver_method": method, "timestep_s": timestep} + for method in parameters["contact_solver_methods"] + for timestep in parameters["timesteps_s"] + ], + "deterministic_repeats": int(parameters["deterministic_repeats"]), + "deterministic_repeats_identical": True, + "measurement_window": { + "start_s": 0.0, + "end_s": parameters["horizon_s"], + }, + }, + "metrics": { + "physical": { + "method": ( + "Mechanical-energy envelope from " + "StepMetrics.total_energy (max over run vs initial), " + "finite-state check over 216 bodies" + ), + "all_cells_finite": all_finite, + "initial_total_energy_j": initial_energy, + "max_energy_envelope_excess_j": max_excess, + "envelope_tolerance_j": envelope_tolerance_j, + "violating_cells": violating_cells, + # A zero excess is the measured result, not a missing value: + # total energy never rose above its initial value in any cell. + "measured_zero_fields": ( + ["max_energy_envelope_excess_j"] if max_excess == 0 else [] + ), + }, + "numerical": { + "method": ( + "Max over run of StepMetrics.max_penetration_depth and " + "active_contact_count; per-method iteration counts where " + "the runtime records them" + ), + "max_penetration_m": max(row["max_penetration_m"] for row in rows), + "penetration_semantics": PENETRATION_CLAMP_NOTE, + "solver_iterations_by_method": solver_iterations_by_method( + { + method: max( + row["max_solver_iterations"] + for row in rows + if row["contact_solver_method"] == method + ) + for method in parameters["contact_solver_methods"] + } + ), + "sequential_impulse_iterations_semantics": ( + SEQUENTIAL_IMPULSE_ITERATIONS_NOTE + ), + "solver_residual": dict(UNSUPPORTED_SOLVER_RESIDUAL), + "max_active_contacts": max(row["max_active_contacts"] for row in rows), + }, + "performance": { + "status": "unsupported", + "reason": ( + "This packet makes no timing claim; interleaved " + "same-host methodology is required before any " + "performance row" + ), + }, + "allocation": { + "status": "unsupported", + "reason": ( + "Post-bake allocation gates are owned by PLAN-122 " + "tooling; this packet does not measure allocations" + ), + }, + }, + "evidence": { + "commands": [command], + "raw_rows": rows, + "visual": { + "status": "not-applicable", + "reason": ( + "The oracle is the numeric energy envelope and finite " + "state; no visible-behavior claim is made by this " + "packet." + ), + }, + }, + "result": { + "disposition": disposition, + "claim_boundary": ( + "DART 7 main, this commit, a bounded (not source-exact) " + "6x6x6 sphere-grid drop with restitution 0.8 and mu=0.8, " + "2 s horizon, timesteps 2 and 4 ms, SEQUENTIAL_IMPULSE and " + "BOXED_LCP contact solvers, energy envelope from " + "StepMetrics.total_energy. The cited failure mode was NOT " + "observed here: no cell exceeded its initial mechanical " + "energy and no cell went non-finite, so this bounded " + "reconstruction does not reproduce the claim. That is not a " + "refutation of the original report, which concerns a " + "different engine version and scene. Says nothing about the " + "original SimBenchmark scene parameters, historical DART " + "versions, restitution values other than 0.8, or DART 6." + ), + "limitations": [ + "Bounded reconstruction: the original SimBenchmark elastic " + "test assets and parameters are not reproduced exactly.", + "The envelope uses aggregate world energy; per-body " + "restitution-outcome tracking (bounce-height ratios) is " + "future work for this row.", + "ResolvedSolverConfiguration is not Python-exposed; " + "resolved identity is property readback.", + "Two timesteps and two solvers only.", + ], + }, + "review": {"passes": []}, + "host": { + "platform": platform.platform(), + "python": sys.version.split()[0], + "machine": platform.machine(), + "performance_valid": False, + "note": ( + "Host recorded for provenance only; no timing methodology " + "was applied" + ), + }, + } + return packet + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--output", + type=Path, + default=DEFAULT_OUTPUT, + help=f"Packet output path (default: {DEFAULT_OUTPUT})", + ) + return parser.parse_args() + + +def main() -> int: + args = parse_args() + packet = build_packet() + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text( + json.dumps(packet, indent=2, sort_keys=True) + "\n", encoding="utf-8" + ) + physical = packet["metrics"]["physical"] + print(f"wrote {args.output}") + print( + f" all finite: {physical['all_cells_finite']}; envelope excess " + f"{physical['max_energy_envelope_excess_j']:.3e} J (tolerance " + f"{physical['envelope_tolerance_j']:.3e} J)" + ) + print(f" violating cells: {physical['violating_cells']}") + print(f" disposition: {packet['result']['disposition']}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index dbd11cda7a326..1f0787ddda75b 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -57,7 +57,11 @@ def complete_packet() -> dict: "measurement_window": {"start_s": 0.0, "end_s": 1.0}, }, "metrics": { - "physical": {"method": "sweep", "lateral_drift_m": 0.0}, + "physical": { + "method": "sweep", + "lateral_drift_m": 0.0, + "measured_zero_fields": ["lateral_drift_m"], + }, "numerical": {"method": "step metrics", "max_penetration_m": 1e-5}, "performance": { "status": "unsupported", @@ -388,6 +392,155 @@ def test_validate_tree_rejects_missing_referenced_packet(tmp_path): assert any("missing packet" in error for error in errors) +def test_unacknowledged_zero_fails(): + packet = complete_packet() + packet["metrics"]["numerical"]["max_solver_residual"] = 0.0 + errors = MODULE.packet_errors(packet) + assert any("without acknowledgement" in error for error in errors) + + +def test_acknowledged_measured_zero_passes(): + packet = complete_packet() + packet["metrics"]["numerical"]["max_solver_residual"] = 0.0 + packet["metrics"]["numerical"]["measured_zero_fields"] = ["max_solver_residual"] + assert MODULE.packet_errors(packet) == [] + + +def test_typed_unsupported_leaf_passes_and_needs_reason(): + packet = complete_packet() + packet["metrics"]["numerical"]["solver_residual"] = { + "status": "unsupported", + "reason": "never computed on this path", + } + assert MODULE.packet_errors(packet) == [] + packet["metrics"]["numerical"]["solver_residual"] = {"status": "unsupported"} + errors = MODULE.packet_errors(packet) + assert any("typed unsupported" in error for error in errors) + + +def test_stale_measured_zero_declaration_fails(): + packet = complete_packet() + packet["metrics"]["numerical"]["measured_zero_fields"] = ["not_zero_here"] + errors = MODULE.packet_errors(packet) + assert any("which are not zero" in error for error in errors) + + +def test_nested_zero_is_caught_by_path(): + packet = complete_packet() + packet["metrics"]["physical"]["per_solver_summary"] = { + "BOXED_LCP": {"max_penetration_m": 0.0} + } + errors = MODULE.packet_errors(packet) + assert any( + "per_solver_summary.BOXED_LCP.max_penetration_m" in error for error in errors + ) + + +def test_spelled_placeholder_fails(): + packet = complete_packet() + packet["metrics"]["numerical"]["max_penetration_m"] = "n/a" + errors = MODULE.packet_errors(packet) + assert any("placeholder" in error for error in errors) + + +def test_only_empty_containers_fails(): + packet = complete_packet() + packet["metrics"]["numerical"] = {"method": "m", "values": {}} + errors = MODULE.packet_errors(packet) + assert any("only empty containers" in error for error in errors) + + +def test_empty_list_alongside_real_values_passes(): + packet = complete_packet() + packet["metrics"]["numerical"]["violations"] = [] + assert MODULE.packet_errors(packet) == [] + + +def test_dangling_raw_path_fails(tmp_path): + packet = complete_packet() + del packet["evidence"]["raw_rows"] + packet["evidence"]["raw_paths"] = ["does/not/exist.csv"] + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any("does not resolve" in error for error in errors) + (tmp_path / "real.csv").write_text("x", encoding="utf-8") + packet["evidence"]["raw_paths"] = ["real.csv"] + assert MODULE.packet_errors(packet, base_dir=tmp_path) == [] + + +def test_empty_measurement_window_fails(): + packet = complete_packet() + packet["ensemble"]["measurement_window"] = {} + errors = MODULE.packet_errors(packet) + assert any("measurement_window" in error for error in errors) + + +def test_validate_tree_validates_packets_in_subdirectories(tmp_path): + """A lane may not close a row with a file the packet checks never reach.""" + manifest = _minimal_manifest(["CT-001"]) + lane = manifest["claims"][0]["lanes"]["dart7"] + lane["status"] = "closed" + lane["disposition"] = "reproduced" + lane["evidence"] = ["evidence/sub/prose.json"] + plan_dir = _write_tree(tmp_path, negative={"schema": "x"}, manifest=manifest) + sub = plan_dir / "evidence" / "sub" + sub.mkdir(parents=True) + (sub / "prose.json").write_text( + json.dumps({"this is": "not a packet"}), encoding="utf-8" + ) + errors = MODULE.validate_tree(plan_dir) + assert errors, "a nested non-packet must not close a lane" + assert any("missing required top-level keys" in error for error in errors) + + +def test_validate_tree_rejects_negative_control_as_lane_evidence(tmp_path): + manifest = _minimal_manifest(["CT-001"]) + lane = manifest["claims"][0]["lanes"]["dart7"] + lane["status"] = "closed" + lane["disposition"] = "reproduced" + lane["evidence"] = ["evidence/negative-controls/incomplete.json"] + plan_dir = _write_tree(tmp_path, negative={"schema": "x"}, manifest=manifest) + errors = MODULE.validate_tree(plan_dir) + assert any("negative control" in error for error in errors) + + +def test_validate_tree_rejects_non_json_lane_evidence(tmp_path): + manifest = _minimal_manifest(["CT-001"]) + lane = manifest["claims"][0]["lanes"]["dart7"] + lane["status"] = "closed" + lane["disposition"] = "reproduced" + lane["evidence"] = ["evidence/notes.md"] + plan_dir = _write_tree(tmp_path, negative={"schema": "x"}, manifest=manifest) + (plan_dir / "evidence" / "notes.md").write_text("prose", encoding="utf-8") + errors = MODULE.validate_tree(plan_dir) + assert any("not a .json packet" in error for error in errors) + + +def test_validate_tree_rejects_scalar_lane_evidence(tmp_path): + manifest = _minimal_manifest(["CT-001"]) + manifest["claims"][0]["lanes"]["dart7"]["evidence"] = 7 + plan_dir = _write_tree(tmp_path, negative={"schema": "x"}, manifest=manifest) + errors = MODULE.validate_tree(plan_dir) + assert any("evidence must be a list" in error for error in errors) + + +def test_validate_tree_rejects_shared_packet_owner(tmp_path): + manifest = _minimal_manifest(["CT-001", "CT-002"]) + for claim in manifest["claims"]: + claim["lanes"]["dart7"]["status"] = "in-progress" + claim["lanes"]["dart7"]["evidence"] = ["evidence/packet.json"] + plan_dir = _write_tree( + tmp_path, + packet=complete_packet(), + negative={"schema": "x"}, + manifest=manifest, + ) + (plan_dir / "citation-claim-corpus.md").write_text( + "| CT-001 | claim |\n| CT-002 | claim |\n", encoding="utf-8" + ) + errors = MODULE.validate_tree(plan_dir) + assert any("one packet has one owner" in error for error in errors) + + def test_validate_tree_closed_lane_needs_two_review_passes(tmp_path): packet = copy.deepcopy(complete_packet()) packet["review"]["passes"] = [{"reviewer": "first", "summary": "clean"}] From 54bf5884abed8fa0cbb7ca0921c22b4adfb1bd63 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Fri, 14 Aug 2026 20:03:10 -0700 Subject: [PATCH 03/52] Correct the CT-002 verdict and the packet provenance strings Round-2 verification found the first dense-contact packets still overclaimed: - CT-002's verdict rested on one cell crossing a hardcoded settle-speed threshold, but that speed is bit-exactly linear in dt (speed/dt identical to 17 digits across timesteps), which is converging integrator residual rather than instability. The packet now computes the dt-linearity and refuses to count a dt-linear excess as instability. - In its place CT-002 gains a real oracle: total mechanical energy must not rise once the pile has settled. Sequential impulse gains 1.0e-3 J (2 ms) and 1.3e-3 J (4 ms) per step against a 1.8e-4 J tolerance while boxed LCP stays at or near zero, so the reproduced signal is now specific, small, non-divergent, and attributed to a solver rather than to a threshold. - configuration.resolved_provenance claimed per-cell assertion of four fields while only the contact solver was checked, and the recorded timestep was the first cell's for every cell. Resolved identity is now keyed per cell with contact solver, timestep, and gravity each asserted, in all three packets and on release-6.20. - solver_iterations_by_method() keyed "unsupported" off the method name, so a genuinely recorded count would have been laundered as unsupported; it now keys off the observed value. - CT-002 cited an energy-monotonicity oracle it did not have, and CT-003's envelope maximum was its own seed. Both are now transparent. --- .../verification.md | 44 ++- .../CT-001-dart7-rolling-direction.json | 6 +- .../CT-002-dart7-dense-inelastic-contact.json | 155 ++++++++++- .../CT-003-dart7-dense-elastic-contact.json | 58 +++- scripts/citation_packet_utils.py | 35 ++- ...citation_ct001_rolling_direction_packet.py | 13 +- ...ite_citation_ct002_dense_contact_packet.py | 258 ++++++++++++++---- ...e_citation_ct003_elastic_contact_packet.py | 66 ++++- 8 files changed, 535 insertions(+), 100 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 9a07c56a57fdd..a57543de337f0 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -97,9 +97,51 @@ scripts/write_citation_ct002_dense_contact_packet.py` — all cells finite; as a negative result rather than a refutation of the original report. - Negative control: still fails (>= 3 errors) and is now also rejected if a lane references it as evidence. + +## Round-2 verification and second fix pass — 2026-08-14 + +Two independent verifiers re-checked the post-fix state. Both confirmed the +bypass is closed and the zero rule is real, and both found further defects in +the CT-002/CT-003 packets that were then fixed: + +1. CT-002's `reproduced` verdict rested on one cell crossing a hardcoded + 0.05 m/s settle threshold. The verifier showed the settle speed is + bit-exactly linear in dt (`speed/dt` identical to 17 digits across the two + timesteps), which is a converging integrator's quasi-static residual, not + instability. The packet now computes that dt-linearity explicitly and + refuses to count a dt-linear excess as instability. +2. Replacing it, CT-002 gained a real oracle: total mechanical energy must not + increase after the pile settles. Measured per-step gain in the settle + window is 9.98e-4 J (2 ms) and 1.31e-3 J (4 ms) under SEQUENTIAL_IMPULSE + against a 1.80e-4 J tolerance, while BOXED_LCP shows 0.0 and 1.34e-6 J. + That solver-specific, non-divergent energy gain is what the packet now + reports as reproducing the claim, with the scaling limb marked + uninstrumented. The first attempt at this oracle measured the whole run and + fired on all four cells; that was free-fall discretization error, so the + window is gated to the settled phase. +3. `configuration.resolved_provenance` claimed per-cell assertion of four + fields while only the contact solver was checked, and `setdefault` kept the + first timestep so 4 ms cells were published as 2 ms. Resolved identity is + now keyed per cell (`METHOD@dt=...`) with contact solver, timestep, and + gravity each asserted; the same widening was applied to CT-001 and to the + `release-6.20` writer. +4. `solver_iterations_by_method()` keyed "unsupported" off the method name, so + a genuinely recorded AVBD-sourced count would have been laundered into + "unsupported". It now keys off the observed value, and the BOXED_LCP marker + no longer claims the count is _never_ written. +5. CT-002 cited an "energy monotonicity" oracle it did not have (it tracked + kinetic energy only); CT-003's envelope maximum was set by its seed value. + Both are now transparent: CT-002 tracks total energy, and CT-003 records + `max_total_energy_after_step_j` and `envelope_set_by_initial_state` so the + one-sidedness of the test is visible. + +- Commands after the second fix pass: `pixi run check-citation-evidence` — OK + (it caught two further unacknowledged zeros during the rework, both now + declared as genuine measurements); 68 pytest cases pass. - Known gaps after this slice: CT-002/CT-003 are bounded reconstructions, not source-exact SimBenchmark scenes; per-contact cone/complementarity metrics - remain unavailable until WS4. + remain unavailable until WS4; the solver-failure limb of CT-003 is only + instrumented as non-finite state. ## Bootstrap record — 2026-08-14 diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json index d40b222e028cd..3a4f456dca147 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json @@ -39,7 +39,7 @@ } } }, - "resolved_provenance": "World property readback (contact_solver_method, rigid_body_solver, gravity, time_step) after enter_simulation_mode, asserted equal to the request and stable across every repeat and sweep point. The independent evidence that the selection changed behavior is that the per-angle trajectory hashes differ between the two contact solvers at every angle; the recorded WorldStepProfile stage names are identical for both methods and therefore do not discriminate. World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", + "resolved_provenance": "World property readback (contact_solver_method, rigid_body_solver, gravity, time_step) after enter_simulation_mode, with contact solver, timestep, and gravity each asserted equal to the request and stable across every repeat and sweep point. The independent evidence that the selection changed behavior is that the per-angle trajectory hashes differ between the two contact solvers at every angle; the recorded WorldStepProfile stage names are identical for both methods and therefore do not discriminate. World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", "substeps": 1, "timestep": 0.002 }, @@ -635,7 +635,7 @@ "sequential_impulse_iterations_semantics": "Sequential impulse records the configured iteration count (recordSolverDiagnostics(world, m_iterations)); its Gauss-Seidel loop has no convergence exit, so this is the configured sweep count, not an observed iteration-to-convergence.", "solver_iterations_by_method": { "BOXED_LCP": { - "reason": "The BoxedLcp branch in dart/simulation/compute/rigid_body_contact_stage.cpp returns before recordSolverDiagnostics() runs, so StepMetrics.last_step_iterations is never written for this method; a recorded 0 would mean 'not reported', not 'zero iterations'.", + "reason": "The BoxedLcp branch in dart/simulation/compute/rigid_body_contact_stage.cpp returns before recordSolverDiagnostics() runs, so this scene recorded no iteration count for the method: a 0 here means 'not reported', not 'zero iterations'. (An opt-in AVBD stage earlier in the same step can record a count before that branch is reached; this scene enables none, and a nonzero count is published with its provenance instead of this marker.)", "status": "unsupported" }, "SEQUENTIAL_IMPULSE": 8 @@ -782,7 +782,7 @@ }, "target": { "branch": "main", - "commit": "d0cb640860d7da5a39792a385e7f499610c0536f", + "commit": "ac0c074d637304fa7c02aa457d123d54e63f738e", "commit_role": "Source state measured: the library and fixture were run at this commit, which is HEAD at capture time. The packet and its writer land in a later commit, so re-running the recorded command requires the child commit that adds the writer; the measured behavior belongs to this one." }, "title": "Rolling-direction friction dependence (DART 7 first packet)" diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json index a974c3a7468bb..1253f2f636b19 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json @@ -20,8 +20,8 @@ ] }, "resolved": { - "by_contact_solver_method": { - "BOXED_LCP": { + "by_cell": { + "BOXED_LCP@dt=0.002": { "contact_solver_method": "BOXED_LCP", "gravity_mps2": [ 0.0, @@ -31,7 +31,17 @@ "rigid_body_solver": "SEQUENTIAL_IMPULSE", "time_step_s": 0.002 }, - "SEQUENTIAL_IMPULSE": { + "BOXED_LCP@dt=0.004": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.004 + }, + "SEQUENTIAL_IMPULSE@dt=0.002": { "contact_solver_method": "SEQUENTIAL_IMPULSE", "gravity_mps2": [ 0.0, @@ -40,6 +50,16 @@ ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", "time_step_s": 0.002 + }, + "SEQUENTIAL_IMPULSE@dt=0.004": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.004 } } }, @@ -92,7 +112,12 @@ "max_energy_gain_after_settle_j": 0.0, "max_penetration_m": 0.006675685044093327, "max_solver_iterations": 8, + "max_total_energy_gain_after_settle_j": 0.0009981350274301803, "raw_last_step_residual_max": 0.0, + "repeat_state_sha256": [ + "d3f6fdd984ba5e6eabaa2841a1bc305f6237a6e224039a079177ec6ca400eb04", + "d3f6fdd984ba5e6eabaa2841a1bc305f6237a6e224039a079177ec6ca400eb04" + ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", "gravity_mps2": [ @@ -118,7 +143,12 @@ "max_energy_gain_after_settle_j": 0.0, "max_penetration_m": 0.011362281723803547, "max_solver_iterations": 8, + "max_total_energy_gain_after_settle_j": 0.001308211355691924, "raw_last_step_residual_max": 0.0, + "repeat_state_sha256": [ + "bed22ef86fbea67a6176944a75e64973b4d0ab7f14c0d2604f61d67823b05f9d", + "bed22ef86fbea67a6176944a75e64973b4d0ab7f14c0d2604f61d67823b05f9d" + ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", "gravity_mps2": [ @@ -144,7 +174,12 @@ "max_energy_gain_after_settle_j": 6.459280144009318e-31, "max_penetration_m": 0.005411200000000366, "max_solver_iterations": 0, + "max_total_energy_gain_after_settle_j": 0.0, "raw_last_step_residual_max": 0.0, + "repeat_state_sha256": [ + "5260253337e1348e1b0f89954a78b9d01a5715d982d77fe0384818189a8f9e33", + "5260253337e1348e1b0f89954a78b9d01a5715d982d77fe0384818189a8f9e33" + ], "resolved": { "contact_solver_method": "BOXED_LCP", "gravity_mps2": [ @@ -170,7 +205,12 @@ "max_energy_gain_after_settle_j": 9.594362860530006e-10, "max_penetration_m": 0.02046842021148934, "max_solver_iterations": 0, + "max_total_energy_gain_after_settle_j": 1.342411593441284e-06, "raw_last_step_residual_max": 0.0, + "repeat_state_sha256": [ + "1a94d8a3a56f6cf1088d78188cbc7982f71e7b0287d0b2335c18dfac375ca3b8", + "1a94d8a3a56f6cf1088d78188cbc7982f71e7b0287d0b2335c18dfac375ca3b8" + ], "resolved": { "contact_solver_method": "BOXED_LCP", "gravity_mps2": [ @@ -186,7 +226,7 @@ } ], "visual": { - "reason": "The oracle is numeric (finite state, settle speed, penetration, energy monotonicity); no visible-behavior claim is made by this packet.", + "reason": "The oracle is numeric (finite state, final speed and its dt scaling, penetration, and non-increase of total mechanical energy from StepMetrics.total_energy); no visible-behavior claim is made by this packet.", "status": "not-applicable" } }, @@ -210,7 +250,7 @@ "sequential_impulse_iterations_semantics": "Sequential impulse records the configured iteration count (recordSolverDiagnostics(world, m_iterations)); its Gauss-Seidel loop has no convergence exit, so this is the configured sweep count, not an observed iteration-to-convergence.", "solver_iterations_by_method": { "BOXED_LCP": { - "reason": "The BoxedLcp branch in dart/simulation/compute/rigid_body_contact_stage.cpp returns before recordSolverDiagnostics() runs, so StepMetrics.last_step_iterations is never written for this method; a recorded 0 would mean 'not reported', not 'zero iterations'.", + "reason": "The BoxedLcp branch in dart/simulation/compute/rigid_body_contact_stage.cpp returns before recordSolverDiagnostics() runs, so this scene recorded no iteration count for the method: a 0 here means 'not reported', not 'zero iterations'. (An opt-in AVBD stage earlier in the same step can record a count before that branch is reached; this scene enables none, and a nonzero count is published with its provenance instead of this marker.)", "status": "unsupported" }, "SEQUENTIAL_IMPULSE": 8 @@ -226,34 +266,121 @@ }, "physical": { "all_cells_finite": true, + "cell_findings": [ + { + "classified_as_integrator_residual": false, + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_max_speed_mps": 0.04032772547075842, + "instability_reasons": [ + "total mechanical energy increased after settle" + ], + "speed_over_tolerance": false, + "timestep_s": 0.002 + }, + { + "classified_as_integrator_residual": true, + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_max_speed_mps": 0.08065545094151684, + "instability_reasons": [ + "total mechanical energy increased after settle" + ], + "speed_over_tolerance": true, + "timestep_s": 0.004 + }, + { + "classified_as_integrator_residual": false, + "contact_solver_method": "BOXED_LCP", + "final_max_speed_mps": 2.3901020052008448e-14, + "instability_reasons": [], + "speed_over_tolerance": false, + "timestep_s": 0.002 + }, + { + "classified_as_integrator_residual": false, + "contact_solver_method": "BOXED_LCP", + "final_max_speed_mps": 1.1914080833008711e-14, + "instability_reasons": [], + "speed_over_tolerance": false, + "timestep_s": 0.004 + } + ], + "dt_linearity": { + "BOXED_LCP": { + "linear_in_dt": false, + "relative_spread": 1.2019521264234254, + "speed_over_dt": [ + 1.1950510026004224e-11, + 2.9785202082521778e-12 + ] + }, + "SEQUENTIAL_IMPULSE": { + "linear_in_dt": true, + "relative_spread": 0.0, + "speed_over_dt": [ + 20.163862735379208, + 20.163862735379208 + ] + } + }, + "final_max_speed_mps_by_cell": { + "BOXED_LCP@dt=0.002": 2.3901020052008448e-14, + "BOXED_LCP@dt=0.004": 1.1914080833008711e-14, + "SEQUENTIAL_IMPULSE@dt=0.002": 0.04032772547075842, + "SEQUENTIAL_IMPULSE@dt=0.004": 0.08065545094151684 + }, "max_energy_gain_after_settle_j": 9.594362860530006e-10, "max_penetration_m": 0.02046842021148934, + "max_total_energy_gain_after_settle_j": 0.001308211355691924, + "measured_zero_fields": [ + "dt_linearity.SEQUENTIAL_IMPULSE.relative_spread" + ], "method": "Per-cell finite-state check over 216 bodies, final max body speed, final min body height, kinetic energy at the horizon, and max single-step kinetic-energy gain after the settle window from StepMetrics", - "settled_final_max_speed_mps": 0.08065545094151684, + "residual_speed_tolerance": { + "dt_linearity_relative": 0.001, + "note": "A settle speed proportional to dt is the quasi-static residual of a converging semi-implicit integrator, not instability: halving dt halves it. Instability grows super-linearly or diverges. A cell whose speed/dt matches the other cells of its method to this relative tolerance is classified as residual." + }, "stability_tolerance": { "final_max_speed_mps": 0.05, "max_energy_gain_after_settle_j": 0.001, - "max_penetration_m": 0.025 + "max_penetration_m": 0.025, + "max_total_energy_gain_after_settle_j": 0.0001801116000000002 }, "unstable_cells": [ { + "classified_as_integrator_residual": false, + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_max_speed_mps": 0.04032772547075842, + "instability_reasons": [ + "total mechanical energy increased after settle" + ], + "speed_over_tolerance": false, + "timestep_s": 0.002 + }, + { + "classified_as_integrator_residual": true, "contact_solver_method": "SEQUENTIAL_IMPULSE", - "reasons": [ - "pile still moving at horizon" + "final_max_speed_mps": 0.08065545094151684, + "instability_reasons": [ + "total mechanical energy increased after settle" ], + "speed_over_tolerance": true, "timestep_s": 0.004 } ] } }, "result": { - "claim_boundary": "DART 7 main, this commit, a bounded (not source-exact) 6x6x6 sphere-grid drop with restitution 0 and mu=0.8, 2 s horizon, timesteps 2 and 4 ms, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers. Says nothing about the original SimBenchmark scene parameters, historical DART versions, other densities/materials, or DART 6.", + "claim_boundary": "DART 7 main, this commit, a bounded (not source-exact) 6x6x6 sphere-grid drop with restitution 0 and mu=0.8, 2 s horizon, timesteps 2 and 4 ms, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers. Outcome actually observed: no cell failed (4 of 4 finite, no fall-through, penetration within tolerance), and the one cell above the settle-speed tolerance has a final speed exactly proportional to dt (speed/dt identical across timesteps), which is converging integrator residual and is explicitly NOT counted as instability. What does reproduce is narrower and solver-specific: after the pile settles, total mechanical energy increases per step under SEQUENTIAL_IMPULSE at both timesteps (about 1.0e-3 J at 2 ms and 1.3e-3 J at 4 ms against a 1.8e-4 J tolerance, on a 180 J scene) while BOXED_LCP stays at or near zero. That is a small, non-divergent, non-physical energy gain in a resting inelastic pile, not a blow-up. The poor-scaling limb of the cited claim is not instrumented at all (performance is typed unsupported). Says nothing about the original SimBenchmark scene parameters, historical DART versions, other densities/materials, or DART 6.", "disposition": "reproduced", "limitations": [ "Bounded reconstruction: the original SimBenchmark asset, material, and timestep grid are not reproduced exactly; sourcing the exact historical setup is future corpus work.", - "ResolvedSolverConfiguration is not Python-exposed; resolved identity is property readback.", - "Two timesteps and two solvers only; a wider grid belongs to a follow-up once per-island diagnostics exist.", - "No per-contact residual/cone reporting is available on main; solver residual is the aggregate StepMetrics value." + "The reproduced signal is a small per-step energy gain in a settled pile, not divergence or failure. It must not be quoted as 'dense contact fails' or as a solver ranking.", + "The settle window is the second half of the run; a pile that settles later would put free-fall discretization error inside the window and inflate the energy metric.", + "Resolved identity comes from World property readback per cell (contact solver, timestep, and gravity each asserted against the request); ResolvedSolverConfiguration is not Python-exposed.", + "No solver residual exists on this path and the boxed-LCP branch records no iteration count; both are typed unsupported rather than reported as zero, so the corpus row's residual and iteration oracles are not yet covered.", + "The corpus row also names wall time over the timestep grid; this packet makes no timing claim (performance is typed unsupported) because no interleaved same-host methodology was applied.", + "max_active_contacts is 216 in every cell (a fully stacked column geometry), so it carries no discriminating information here.", + "Two timesteps and two solvers only; a wider grid belongs to a follow-up once per-island diagnostics exist." ] }, "review": { @@ -303,7 +430,7 @@ }, "target": { "branch": "main", - "commit": "d0cb640860d7da5a39792a385e7f499610c0536f" + "commit": "ac0c074d637304fa7c02aa457d123d54e63f738e" }, "title": "Dense 6x6x6 inelastic contact stability (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json index db08f8ee9e57f..919479a366d43 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json @@ -20,8 +20,8 @@ ] }, "resolved": { - "by_contact_solver_method": { - "BOXED_LCP": { + "by_cell": { + "BOXED_LCP@dt=0.002": { "contact_solver_method": "BOXED_LCP", "gravity_mps2": [ 0.0, @@ -31,7 +31,17 @@ "rigid_body_solver": "SEQUENTIAL_IMPULSE", "time_step_s": 0.002 }, - "SEQUENTIAL_IMPULSE": { + "BOXED_LCP@dt=0.004": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.004 + }, + "SEQUENTIAL_IMPULSE@dt=0.002": { "contact_solver_method": "SEQUENTIAL_IMPULSE", "gravity_mps2": [ 0.0, @@ -40,6 +50,16 @@ ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", "time_step_s": 0.002 + }, + "SEQUENTIAL_IMPULSE@dt=0.004": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.004 } } }, @@ -82,6 +102,7 @@ { "contact_solver_method": "SEQUENTIAL_IMPULSE", "energy_envelope_excess_j": 0.0, + "envelope_set_by_initial_state": true, "final_state_sha256": "f1d5fac4ff2da8ee3e862cc55b49ea2362ecefd6a6527464dab5352f784c949c", "final_total_energy_j": 63.44696118939535, "finite": true, @@ -89,8 +110,13 @@ "max_active_contacts": 216, "max_penetration_m": 0.007707330160000275, "max_solver_iterations": 8, + "max_total_energy_after_step_j": 180.10744260048014, "max_total_energy_j": 180.1116000000002, "raw_last_step_residual_max": 0.0, + "repeat_state_sha256": [ + "f1d5fac4ff2da8ee3e862cc55b49ea2362ecefd6a6527464dab5352f784c949c", + "f1d5fac4ff2da8ee3e862cc55b49ea2362ecefd6a6527464dab5352f784c949c" + ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", "gravity_mps2": [ @@ -107,6 +133,7 @@ { "contact_solver_method": "SEQUENTIAL_IMPULSE", "energy_envelope_excess_j": 0.0, + "envelope_set_by_initial_state": true, "final_state_sha256": "6cc4601838f1c5213104137dabb8e8064e376090d6cb22068335c12c87a8b125", "final_total_energy_j": 63.25766139362141, "finite": true, @@ -114,8 +141,13 @@ "max_active_contacts": 216, "max_penetration_m": 0.01716770175999996, "max_solver_iterations": 8, + "max_total_energy_after_step_j": 180.09497040192002, "max_total_energy_j": 180.1116000000002, "raw_last_step_residual_max": 0.0, + "repeat_state_sha256": [ + "6cc4601838f1c5213104137dabb8e8064e376090d6cb22068335c12c87a8b125", + "6cc4601838f1c5213104137dabb8e8064e376090d6cb22068335c12c87a8b125" + ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", "gravity_mps2": [ @@ -132,6 +164,7 @@ { "contact_solver_method": "BOXED_LCP", "energy_envelope_excess_j": 0.0, + "envelope_set_by_initial_state": true, "final_state_sha256": "31c88c2d34b7e738c731aaa2e3e5a4dc9ebca29de00e901dc7296ebddb3402c1", "final_total_energy_j": 63.58479412413587, "finite": true, @@ -139,8 +172,13 @@ "max_active_contacts": 216, "max_penetration_m": 0.007707330160000275, "max_solver_iterations": 0, + "max_total_energy_after_step_j": 180.10744260048014, "max_total_energy_j": 180.1116000000002, "raw_last_step_residual_max": 0.0, + "repeat_state_sha256": [ + "31c88c2d34b7e738c731aaa2e3e5a4dc9ebca29de00e901dc7296ebddb3402c1", + "31c88c2d34b7e738c731aaa2e3e5a4dc9ebca29de00e901dc7296ebddb3402c1" + ], "resolved": { "contact_solver_method": "BOXED_LCP", "gravity_mps2": [ @@ -157,6 +195,7 @@ { "contact_solver_method": "BOXED_LCP", "energy_envelope_excess_j": 0.0, + "envelope_set_by_initial_state": true, "final_state_sha256": "2438f7c534de26f786af0b223b92f0898d8fcc5618698326c5e9f14be6936513", "final_total_energy_j": 63.746668542792605, "finite": true, @@ -164,8 +203,13 @@ "max_active_contacts": 216, "max_penetration_m": 0.01716770175999996, "max_solver_iterations": 0, + "max_total_energy_after_step_j": 180.09497040192002, "max_total_energy_j": 180.1116000000002, "raw_last_step_residual_max": 0.0, + "repeat_state_sha256": [ + "2438f7c534de26f786af0b223b92f0898d8fcc5618698326c5e9f14be6936513", + "2438f7c534de26f786af0b223b92f0898d8fcc5618698326c5e9f14be6936513" + ], "resolved": { "contact_solver_method": "BOXED_LCP", "gravity_mps2": [ @@ -205,7 +249,7 @@ "sequential_impulse_iterations_semantics": "Sequential impulse records the configured iteration count (recordSolverDiagnostics(world, m_iterations)); its Gauss-Seidel loop has no convergence exit, so this is the configured sweep count, not an observed iteration-to-convergence.", "solver_iterations_by_method": { "BOXED_LCP": { - "reason": "The BoxedLcp branch in dart/simulation/compute/rigid_body_contact_stage.cpp returns before recordSolverDiagnostics() runs, so StepMetrics.last_step_iterations is never written for this method; a recorded 0 would mean 'not reported', not 'zero iterations'.", + "reason": "The BoxedLcp branch in dart/simulation/compute/rigid_body_contact_stage.cpp returns before recordSolverDiagnostics() runs, so this scene recorded no iteration count for the method: a 0 here means 'not reported', not 'zero iterations'. (An opt-in AVBD stage earlier in the same step can record a count before that branch is reached; this scene enables none, and a nonzero count is published with its provenance instead of this marker.)", "status": "unsupported" }, "SEQUENTIAL_IMPULSE": 8 @@ -237,7 +281,9 @@ "limitations": [ "Bounded reconstruction: the original SimBenchmark elastic test assets and parameters are not reproduced exactly.", "The envelope uses aggregate world energy; per-body restitution-outcome tracking (bounce-height ratios) is future work for this row.", - "ResolvedSolverConfiguration is not Python-exposed; resolved identity is property readback.", + "The envelope maximum is set by the initial state in every cell (energy never rose above its starting value), so the 1.8e-4 J tolerance is never exercised; the test is one-sided by construction and is reported as such in raw_rows.envelope_set_by_initial_state.", + "The cited claim also names solver failure. That limb is only instrumented as non-finite state: no residual exists on this path and the boxed-LCP branch records no iteration count, so a non-diverging solver failure would not be detected here.", + "max_penetration_m is bit-identical across the two solvers at each timestep while the trajectories diverge, so it is set during the shared first impact and carries no solver-discriminating information.", "Two timesteps and two solvers only." ] }, @@ -287,7 +333,7 @@ }, "target": { "branch": "main", - "commit": "d0cb640860d7da5a39792a385e7f499610c0536f" + "commit": "ac0c074d637304fa7c02aa457d123d54e63f738e" }, "title": "Dense elastic contact energy envelope (DART 7 first packet)" } diff --git a/scripts/citation_packet_utils.py b/scripts/citation_packet_utils.py index 090f75f84cbd3..6e9b4ea3890cd 100644 --- a/scripts/citation_packet_utils.py +++ b/scripts/citation_packet_utils.py @@ -37,9 +37,12 @@ "reason": ( "The BoxedLcp branch in " "dart/simulation/compute/rigid_body_contact_stage.cpp returns before " - "recordSolverDiagnostics() runs, so StepMetrics.last_step_iterations " - "is never written for this method; a recorded 0 would mean 'not " - "reported', not 'zero iterations'." + "recordSolverDiagnostics() runs, so this scene recorded no iteration " + "count for the method: a 0 here means 'not reported', not 'zero " + "iterations'. (An opt-in AVBD stage earlier in the same step can " + "record a count before that branch is reached; this scene enables " + "none, and a nonzero count is published with its provenance instead " + "of this marker.)" ), } @@ -80,16 +83,30 @@ def solver_iterations_by_method( iterations_by_method: dict[str, int], ) -> dict[str, Any]: - """Type per-method iteration counts, marking unreported methods. + """Type per-method iteration counts, marking unrecorded ones. - A method that never reaches `recordSolverDiagnostics()` is emitted as a - typed-unsupported marker instead of the sentinel 0 the runtime leaves in - place. + A zero is the runtime's "nothing was recorded" sentinel: the BoxedLcp + branch returns before `recordSolverDiagnostics()`, and the + sequential-impulse call is skipped when a step assembles no constraints. + Either way an absent count is emitted as a typed-unsupported marker. + A genuinely recorded count is published as measured, whichever method + produced it -- keying off the method name instead would launder a real + AVBD-sourced count into "unsupported". """ typed: dict[str, Any] = {} for method, value in sorted(iterations_by_method.items()): - if method == "BOXED_LCP": + if isinstance(value, int) and value > 0: + typed[method] = value + elif method == "BOXED_LCP": typed[method] = dict(UNSUPPORTED_BOXED_LCP_ITERATIONS) else: - typed[method] = value + typed[method] = { + "status": "unsupported", + "reason": ( + "No iteration count was recorded for this method in this " + "scene: StepMetrics.last_step_iterations stayed at its " + "per-step reset value, which means 'not reported' rather " + "than 'zero iterations'." + ), + } return typed diff --git a/scripts/write_citation_ct001_rolling_direction_packet.py b/scripts/write_citation_ct001_rolling_direction_packet.py index b6aa2d61da5ce..a18399f89e94d 100644 --- a/scripts/write_citation_ct001_rolling_direction_packet.py +++ b/scripts/write_citation_ct001_rolling_direction_packet.py @@ -329,6 +329,16 @@ def build_packet() -> dict[str, Any]: f"requested contact solver {method} but World readback " f"reports {readback['contact_solver_method']}" ) + if readback["time_step_s"] != float(parameters["time_step_s"]): + raise SystemExit( + f"requested timestep {parameters['time_step_s']} but " + f"World readback reports {readback['time_step_s']}" + ) + if readback["gravity_mps2"] != list(parameters["gravity_mps2"]): + raise SystemExit( + f"requested gravity {parameters['gravity_mps2']} but " + f"World readback reports {readback['gravity_mps2']}" + ) row["repeat_trajectory_sha256"] = [ repeat["trajectory_sha256"] for repeat in repeats ] @@ -457,7 +467,8 @@ def build_packet() -> dict[str, Any]: "resolved_provenance": ( "World property readback (contact_solver_method, " "rigid_body_solver, gravity, time_step) after " - "enter_simulation_mode, asserted equal to the request and " + "enter_simulation_mode, with contact solver, timestep, and gravity " + "each asserted equal to the request and " "stable across every repeat and sweep point. The independent " "evidence that the selection changed behavior is that the " "per-angle trajectory hashes differ between the two contact " diff --git a/scripts/write_citation_ct002_dense_contact_packet.py b/scripts/write_citation_ct002_dense_contact_packet.py index 9547a819e5ab8..96278ee89b31e 100644 --- a/scripts/write_citation_ct002_dense_contact_packet.py +++ b/scripts/write_citation_ct002_dense_contact_packet.py @@ -163,7 +163,9 @@ def run_single( max_residual = 0.0 max_contacts = 0 max_energy_gain_after_settle = 0.0 + max_total_energy_gain_after_settle = 0.0 previous_kinetic = None + previous_total = None initial_metrics = world.compute_step_metrics() initial_total_energy = float(initial_metrics.total_energy) @@ -171,9 +173,16 @@ def run_single( world.step() metrics = world.compute_step_metrics() kinetic = float(metrics.kinetic_energy) - if not math.isfinite(kinetic): + total_energy = float(metrics.total_energy) + if not (math.isfinite(kinetic) and math.isfinite(total_energy)): non_finite = True break + if previous_total is not None and step_index >= settle_step: + max_total_energy_gain_after_settle = max( + max_total_energy_gain_after_settle, + total_energy - previous_total, + ) + previous_total = total_energy if previous_kinetic is not None and step_index >= settle_step: max_energy_gain_after_settle = max( max_energy_gain_after_settle, kinetic - previous_kinetic @@ -213,6 +222,7 @@ def run_single( ), "initial_total_energy_j": initial_total_energy, "max_energy_gain_after_settle_j": max_energy_gain_after_settle, + "max_total_energy_gain_after_settle_j": (max_total_energy_gain_after_settle), "max_penetration_m": max_penetration, "max_solver_iterations": max_iterations, "raw_last_step_residual_max": max_residual, @@ -224,7 +234,7 @@ def build_packet() -> dict[str, Any]: parameters = SCENE_PARAMETERS rows: list[dict[str, Any]] = [] determinism_failures: list[str] = [] - resolved_by_method: dict[str, dict[str, Any]] = {} + resolved_by_cell: dict[str, dict[str, Any]] = {} for method in parameters["contact_solver_methods"]: for timestep in parameters["timesteps_s"]: @@ -239,13 +249,36 @@ def build_packet() -> dict[str, Any]: f"{sorted(map(str, hashes))}" ) row = repeats[0] - readback = row["resolved"] - if readback["contact_solver_method"] != method: - raise SystemExit( - f"requested contact solver {method} but World readback " - f"reports {readback['contact_solver_method']}" - ) - resolved_by_method.setdefault(method, readback) + for repeat in repeats: + readback = repeat["resolved"] + if readback != row["resolved"]: + raise SystemExit( + f"{method} dt {timestep}: resolved configuration " + f"differs between repeats: {readback} vs " + f"{row['resolved']}" + ) + # Assert every recorded resolved field against what was asked + # for; recording a field without checking it is how a packet + # ends up misreporting the configuration that actually ran. + if readback["contact_solver_method"] != method: + raise SystemExit( + f"requested contact solver {method} but World readback " + f"reports {readback['contact_solver_method']}" + ) + if readback["time_step_s"] != timestep: + raise SystemExit( + f"requested timestep {timestep} but World readback " + f"reports {readback['time_step_s']}" + ) + if readback["gravity_mps2"] != list(parameters["gravity_mps2"]): + raise SystemExit( + f"requested gravity {parameters['gravity_mps2']} but " + f"World readback reports {readback['gravity_mps2']}" + ) + row["repeat_state_sha256"] = [ + repeat["final_state_sha256"] for repeat in repeats + ] + resolved_by_cell[f"{method}@dt={timestep}"] = row["resolved"] rows.append(row) if determinism_failures: @@ -256,43 +289,100 @@ def build_packet() -> dict[str, Any]: all_finite = all(row["finite"] for row in rows) max_penetration = max(row["max_penetration_m"] for row in rows) max_gain = max(row["max_energy_gain_after_settle_j"] for row in rows) - settle_speed = max(row["final_max_speed_mps"] for row in rows if row["finite"]) + max_total_gain = max(row["max_total_energy_gain_after_settle_j"] for row in rows) + final_max_speed_by_cell = { + f"{row['contact_solver_method']}@dt={row['timestep_s']}": row[ + "final_max_speed_mps" + ] + for row in rows + if row["finite"] + } stability_tolerance = { "max_penetration_m": 0.5 * float(parameters["sphere_radius_m"]), "final_max_speed_mps": 0.05, "max_energy_gain_after_settle_j": 1.0e-3, + # Relative to the scene's own initial mechanical energy: after the + # pile settles, total energy must not climb. During free fall a + # semi-implicit integrator legitimately perturbs total energy, so + # this is measured only in the settle window. + "max_total_energy_gain_after_settle_j": 1.0e-6 + * abs(rows[0]["initial_total_energy_j"]), } - unstable_cells = [ - { - "contact_solver_method": row["contact_solver_method"], - "timestep_s": row["timestep_s"], - "reasons": [ - reason - for reason, bad in ( - ("non-finite state", not row["finite"]), - ( - "penetration above half radius", - row["max_penetration_m"] - > stability_tolerance["max_penetration_m"], - ), - ( - "pile still moving at horizon", - row["finite"] - and row["final_max_speed_mps"] - > stability_tolerance["final_max_speed_mps"], - ), - ( - "energy injected after settle window", - row["max_energy_gain_after_settle_j"] - > stability_tolerance["max_energy_gain_after_settle_j"], - ), - ) - if bad - ], + residual_speed_tolerance = { + "dt_linearity_relative": 1.0e-3, + "note": ( + "A settle speed proportional to dt is the quasi-static residual " + "of a converging semi-implicit integrator, not instability: " + "halving dt halves it. Instability grows super-linearly or " + "diverges. A cell whose speed/dt matches the other cells of its " + "method to this relative tolerance is classified as residual." + ), + } + + # Per method, test whether final speed scales linearly with dt. + speed_over_dt: dict[str, list[float]] = {} + for row in rows: + if row["finite"] and row["timestep_s"] > 0.0: + speed_over_dt.setdefault(row["contact_solver_method"], []).append( + row["final_max_speed_mps"] / row["timestep_s"] + ) + dt_linearity: dict[str, Any] = {} + for method, ratios in speed_over_dt.items(): + mean = sum(ratios) / len(ratios) + spread = (max(ratios) - min(ratios)) / mean if mean > 0.0 else 0.0 + dt_linearity[method] = { + "speed_over_dt": ratios, + "relative_spread": spread, + "linear_in_dt": ( + len(ratios) >= 2 + and mean > 0.0 + and spread <= residual_speed_tolerance["dt_linearity_relative"] + ), } - for row in rows - ] - unstable_cells = [cell for cell in unstable_cells if cell["reasons"]] + + cell_findings = [] + for row in rows: + method = row["contact_solver_method"] + reasons = [] + if not row["finite"]: + reasons.append("non-finite state") + if row["max_penetration_m"] > stability_tolerance["max_penetration_m"]: + reasons.append("penetration above half radius") + if ( + row["max_energy_gain_after_settle_j"] + > stability_tolerance["max_energy_gain_after_settle_j"] + ): + reasons.append("kinetic energy injected after settle window") + if ( + row["max_total_energy_gain_after_settle_j"] + > stability_tolerance["max_total_energy_gain_after_settle_j"] + ): + reasons.append("total mechanical energy increased after settle") + speed_over_tolerance = ( + row["finite"] + and row["final_max_speed_mps"] > stability_tolerance["final_max_speed_mps"] + ) + residual_only = speed_over_tolerance and dt_linearity.get(method, {}).get( + "linear_in_dt", False + ) + if speed_over_tolerance and not residual_only: + reasons.append("pile still moving at horizon, not dt-linear") + cell_findings.append( + { + "contact_solver_method": method, + "timestep_s": row["timestep_s"], + "final_max_speed_mps": row["final_max_speed_mps"], + "speed_over_tolerance": speed_over_tolerance, + "classified_as_integrator_residual": residual_only, + "instability_reasons": reasons, + } + ) + + unstable_cells = [cell for cell in cell_findings if cell["instability_reasons"]] + # The cited claim is failure, instability, or poor scaling. Failure and + # instability are instrumented here; scaling is not (performance is typed + # unsupported). A settle speed that is exactly linear in dt is integrator + # residual, so it does not reproduce the claim. disposition = "reproduced" if unstable_cells else "unresolved" command = ( @@ -327,7 +417,7 @@ def build_packet() -> dict[str, Any]: "backend": "cpu", "threads": "World default sequential step", }, - "resolved": {"by_contact_solver_method": resolved_by_method}, + "resolved": {"by_cell": resolved_by_cell}, "resolved_provenance": ( "World property readback (contact_solver_method, " "rigid_body_solver, gravity, time_step) after " @@ -358,7 +448,7 @@ def build_packet() -> dict[str, Any]: for timestep in parameters["timesteps_s"] ], "deterministic_repeats": int(parameters["deterministic_repeats"]), - "deterministic_repeats_identical": True, + "deterministic_repeats_identical": not determinism_failures, "measurement_window": { "start_s": 0.0, "end_s": parameters["horizon_s"], @@ -375,10 +465,31 @@ def build_packet() -> dict[str, Any]: ), "all_cells_finite": all_finite, "max_penetration_m": max_penetration, - "settled_final_max_speed_mps": settle_speed, + "final_max_speed_mps_by_cell": final_max_speed_by_cell, "max_energy_gain_after_settle_j": max_gain, + "max_total_energy_gain_after_settle_j": max_total_gain, "stability_tolerance": stability_tolerance, + "residual_speed_tolerance": residual_speed_tolerance, + "dt_linearity": dt_linearity, + "cell_findings": cell_findings, "unstable_cells": unstable_cells, + # An exactly-zero spread is the finding, not a missing value: + # the speed/dt ratios are bit-identical across timesteps. + "measured_zero_fields": [ + f"dt_linearity.{method}.relative_spread" + for method, linearity in dt_linearity.items() + if linearity["relative_spread"] == 0 + ] + + [ + f"final_max_speed_mps_by_cell.{cell}" + for cell, speed in final_max_speed_by_cell.items() + if speed == 0 + ] + + ( + ["max_total_energy_gain_after_settle_j"] + if max_total_gain == 0 + else [] + ), }, "numerical": { "method": ( @@ -426,9 +537,10 @@ def build_packet() -> dict[str, Any]: "visual": { "status": "not-applicable", "reason": ( - "The oracle is numeric (finite state, settle speed, " - "penetration, energy monotonicity); no visible-behavior " - "claim is made by this packet." + "The oracle is numeric (finite state, final speed and " + "its dt scaling, penetration, and non-increase of total " + "mechanical energy from StepMetrics.total_energy); no " + "visible-behavior claim is made by this packet." ), }, }, @@ -438,7 +550,22 @@ def build_packet() -> dict[str, Any]: "DART 7 main, this commit, a bounded (not source-exact) " "6x6x6 sphere-grid drop with restitution 0 and mu=0.8, " "2 s horizon, timesteps 2 and 4 ms, SEQUENTIAL_IMPULSE and " - "BOXED_LCP contact solvers. Says nothing about the original " + "BOXED_LCP contact solvers. Outcome actually observed: no " + "cell failed (4 of 4 finite, no fall-through, penetration " + "within tolerance), and the one cell above the settle-speed " + "tolerance has a final speed exactly proportional to dt " + "(speed/dt identical across timesteps), which is converging " + "integrator residual and is explicitly NOT counted as " + "instability. What does reproduce is narrower and " + "solver-specific: after the pile settles, total mechanical " + "energy increases per step under SEQUENTIAL_IMPULSE at both " + "timesteps (about 1.0e-3 J at 2 ms and 1.3e-3 J at 4 ms " + "against a 1.8e-4 J tolerance, on a 180 J scene) while " + "BOXED_LCP stays at or near zero. That is a small, " + "non-divergent, non-physical energy gain in a resting " + "inelastic pile, not a blow-up. The poor-scaling limb of the " + "cited claim is not instrumented at all (performance is " + "typed unsupported). Says nothing about the original " "SimBenchmark scene parameters, historical DART versions, " "other densities/materials, or DART 6." ), @@ -446,12 +573,29 @@ def build_packet() -> dict[str, Any]: "Bounded reconstruction: the original SimBenchmark asset, " "material, and timestep grid are not reproduced exactly; " "sourcing the exact historical setup is future corpus work.", - "ResolvedSolverConfiguration is not Python-exposed; " - "resolved identity is property readback.", + "The reproduced signal is a small per-step energy gain in a " + "settled pile, not divergence or failure. It must not be " + "quoted as 'dense contact fails' or as a solver ranking.", + "The settle window is the second half of the run; a pile " + "that settles later would put free-fall discretization error " + "inside the window and inflate the energy metric.", + "Resolved identity comes from World property readback per " + "cell (contact solver, timestep, and gravity each asserted " + "against the request); ResolvedSolverConfiguration is not " + "Python-exposed.", + "No solver residual exists on this path and the boxed-LCP " + "branch records no iteration count; both are typed " + "unsupported rather than reported as zero, so the corpus " + "row's residual and iteration oracles are not yet covered.", + "The corpus row also names wall time over the timestep " + "grid; this packet makes no timing claim (performance is " + "typed unsupported) because no interleaved same-host " + "methodology was applied.", + "max_active_contacts is 216 in every cell (a fully stacked " + "column geometry), so it carries no discriminating " + "information here.", "Two timesteps and two solvers only; a wider grid belongs " "to a follow-up once per-island diagnostics exist.", - "No per-contact residual/cone reporting is available on " - "main; solver residual is the aggregate StepMetrics value.", ], }, "review": {"passes": []}, @@ -501,10 +645,16 @@ def main() -> int: print(f"wrote {args.output}") print( f" all finite: {physical['all_cells_finite']}; max penetration " - f"{physical['max_penetration_m']:.3e} m; settled max speed " - f"{physical['settled_final_max_speed_mps']:.3e} m/s; max post-settle " - f"energy gain {physical['max_energy_gain_after_settle_j']:.3e} J" + f"{physical['max_penetration_m']:.3e} m; max total-energy gain " + f"{physical['max_total_energy_gain_after_settle_j']:.3e} J" ) + for cell, speed in physical["final_max_speed_mps_by_cell"].items(): + print(f" {cell}: final max speed {speed:.4e} m/s") + for method, linearity in physical["dt_linearity"].items(): + print( + f" {method}: speed/dt spread {linearity['relative_spread']:.3e} " + f"-> linear_in_dt={linearity['linear_in_dt']}" + ) print(f" unstable cells: {physical['unstable_cells']}") print(f" disposition: {packet['result']['disposition']}") return 0 diff --git a/scripts/write_citation_ct003_elastic_contact_packet.py b/scripts/write_citation_ct003_elastic_contact_packet.py index a92d2cc0cf738..22be3865f0059 100644 --- a/scripts/write_citation_ct003_elastic_contact_packet.py +++ b/scripts/write_citation_ct003_elastic_contact_packet.py @@ -161,6 +161,7 @@ def run_single( state_hash = hashlib.sha256() non_finite = False max_total_energy = initial_total_energy + max_total_energy_after_step = None max_penetration = 0.0 max_iterations = 0 max_residual = 0.0 @@ -174,6 +175,11 @@ def run_single( non_finite = True break max_total_energy = max(max_total_energy, total_energy) + max_total_energy_after_step = ( + total_energy + if max_total_energy_after_step is None + else max(max_total_energy_after_step, total_energy) + ) max_penetration = max(max_penetration, float(metrics.max_penetration_depth)) max_iterations = max(max_iterations, int(metrics.last_step_iterations)) max_residual = max(max_residual, float(metrics.last_step_residual)) @@ -199,6 +205,11 @@ def run_single( "final_state_sha256": state_hash.hexdigest() if not non_finite else None, "initial_total_energy_j": initial_total_energy, "max_total_energy_j": max_total_energy, + "max_total_energy_after_step_j": max_total_energy_after_step, + "envelope_set_by_initial_state": ( + max_total_energy_after_step is not None + and max_total_energy_after_step <= initial_total_energy + ), "energy_envelope_excess_j": max_total_energy - initial_total_energy, "final_total_energy_j": ( float(final_metrics.total_energy) if not non_finite else None @@ -224,7 +235,7 @@ def build_packet() -> dict[str, Any]: parameters = SCENE_PARAMETERS rows: list[dict[str, Any]] = [] determinism_failures: list[str] = [] - resolved_by_method: dict[str, dict[str, Any]] = {} + resolved_by_cell: dict[str, dict[str, Any]] = {} for method in parameters["contact_solver_methods"]: for timestep in parameters["timesteps_s"]: @@ -239,13 +250,32 @@ def build_packet() -> dict[str, Any]: f"{sorted(map(str, hashes))}" ) row = repeats[0] - readback = row["resolved"] - if readback["contact_solver_method"] != method: - raise SystemExit( - f"requested contact solver {method} but World readback " - f"reports {readback['contact_solver_method']}" - ) - resolved_by_method.setdefault(method, readback) + for repeat in repeats: + readback = repeat["resolved"] + if readback != row["resolved"]: + raise SystemExit( + f"{method} dt {timestep}: resolved configuration " + f"differs between repeats" + ) + if readback["contact_solver_method"] != method: + raise SystemExit( + f"requested contact solver {method} but World readback " + f"reports {readback['contact_solver_method']}" + ) + if readback["time_step_s"] != timestep: + raise SystemExit( + f"requested timestep {timestep} but World readback " + f"reports {readback['time_step_s']}" + ) + if readback["gravity_mps2"] != list(parameters["gravity_mps2"]): + raise SystemExit( + f"requested gravity {parameters['gravity_mps2']} but " + f"World readback reports {readback['gravity_mps2']}" + ) + row["repeat_state_sha256"] = [ + repeat["final_state_sha256"] for repeat in repeats + ] + resolved_by_cell[f"{method}@dt={timestep}"] = row["resolved"] rows.append(row) if determinism_failures: @@ -309,7 +339,7 @@ def build_packet() -> dict[str, Any]: "backend": "cpu", "threads": "World default sequential step", }, - "resolved": {"by_contact_solver_method": resolved_by_method}, + "resolved": {"by_cell": resolved_by_cell}, "resolved_provenance": ( "World property readback (contact_solver_method, " "rigid_body_solver, gravity, time_step) after " @@ -340,7 +370,7 @@ def build_packet() -> dict[str, Any]: for timestep in parameters["timesteps_s"] ], "deterministic_repeats": int(parameters["deterministic_repeats"]), - "deterministic_repeats_identical": True, + "deterministic_repeats_identical": not determinism_failures, "measurement_window": { "start_s": 0.0, "end_s": parameters["horizon_s"], @@ -438,8 +468,20 @@ def build_packet() -> dict[str, Any]: "The envelope uses aggregate world energy; per-body " "restitution-outcome tracking (bounce-height ratios) is " "future work for this row.", - "ResolvedSolverConfiguration is not Python-exposed; " - "resolved identity is property readback.", + "The envelope maximum is set by the initial state in every " + "cell (energy never rose above its starting value), so the " + "1.8e-4 J tolerance is never exercised; the test is " + "one-sided by construction and is reported as such in " + "raw_rows.envelope_set_by_initial_state.", + "The cited claim also names solver failure. That limb is " + "only instrumented as non-finite state: no residual exists " + "on this path and the boxed-LCP branch records no iteration " + "count, so a non-diverging solver failure would not be " + "detected here.", + "max_penetration_m is bit-identical across the two solvers " + "at each timestep while the trajectories diverge, so it is " + "set during the shared first impact and carries no " + "solver-discriminating information.", "Two timesteps and two solvers only.", ], }, From d7281579701e4f009122968c5782291d12f69ac7 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Fri, 14 Aug 2026 20:11:53 -0700 Subject: [PATCH 04/52] Record the current branch state and packet-writing constraints in RESUME --- .../RESUME.md | 44 ++++++++++++------- 1 file changed, 28 insertions(+), 16 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md index 0fe9099c021cf..acfcf24071abb 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md @@ -29,10 +29,13 @@ statuses were verified; the corpus sidecar now carries the WS0 audit record. ## Current branches - DART 7: `feature/citation-trust-foundation` in - `.claude/worktrees/citation-trust-main`, based on `origin/main` 20501341226. + `.claude/worktrees/citation-trust-main`, based on `origin/main` + 20501341226, now at `54bf5884abe` with three commits: + the PLAN-123 contract, the gate hardening plus dense-contact packets, and + the round-2 verdict/provenance corrections. - DART 6: `feature/dart6-citation-contact-trust` in `.claude/worktrees/citation-trust-620`, based on `origin/release-6.20` - 39ccd52068b (docs applied; branch-local adoption in progress). + 39ccd52068b, now at `afc6d7ac3c2` with four commits. Both are local-only; nothing pushed. GitHub mutations need maintainer approval. @@ -47,20 +50,29 @@ variational integrator as the comparison arm, written with ## Context that would be lost -- `ResolvedSolverConfiguration` is C++-only (`World::getResolvedConfiguration`); - packets record resolved identity from World property readback plus step - profile stage names, and the Python exposure is the first WS4 slice. -- The CT-001 packet must not be quoted as a solver comparison: SI and boxed - LCP metric summaries coincide to printed precision on the single-contact - scene while trajectory hashes differ per solver. -- `check-citation-evidence` validates committed packets structurally; - `--freshness` (packet commit == HEAD) is a packet-writing aid, not CI, - because squash merges retire topic commits. -- Negative-control packets live under `evidence/negative-controls/` and must - keep failing validation with >= 3 errors; the gate rejects a passing - negative control as vacuous. -- dartpy runs from a built tree via - `PYTHONPATH=build/default/cpp/Release/python pixi run python ...`. +- The gate is fail-closed in ways worth knowing before writing a packet: + every exact zero in a measured metric group must be typed + `{"status": "unsupported", "reason": ...}` or listed in + `measured_zero_fields`, and a stale declaration (a listed field that is not + zero) also fails. Use `scripts/citation_packet_utils.py` for the shared + markers rather than inventing wording. +- Quantities that do not exist on `main` today, and must stay typed + unsupported until WS4 lands them: per-solve solver residual (rigid contact + never passes one to `recordSolverDiagnostics`), boxed-LCP iteration counts + (its branch returns before the diagnostics call), per-island fallback + reporting, and `ResolvedSolverConfiguration` in Python. +- Packet dispositions must survive an adversarial read. Two rounds of review + killed a verdict that rested on a hardcoded threshold: a settle speed + exactly linear in dt is integrator residual, not instability. Prefer an + oracle a reviewer can recompute from `raw_rows`. +- `target.commit` is the commit the library was measured at; the packet and + its writer necessarily land one commit later. `target.commit_role` says so. +- dartpy runs from a built tree: + `PYTHONPATH=build/default/cpp/Release/python pixi run python ...` on + `main`, and `.../python/dartpy` on `release-6.20` (which also needs + `pixi run build-py-dev`). +- The `main` manifest keeps dart6 lanes as routing pointers only; the + `release-6.20` branch manifest owns that branch's lane state. ## How to resume From 85e5db2c16858e7366ddea955e5badfcae64990d Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Fri, 14 Aug 2026 20:16:05 -0700 Subject: [PATCH 05/52] Add the CT-004 articulated energy-drift packet Passive 4-link pendulum chain with no contact and no control, released from rest and swept over four timesteps for both multibody integration families. A passive chain conserves total mechanical energy exactly, so the oracle is solver-neutral: relative energy drift versus timestep, summarized by the least-squares log-log slope. Measured: SEMI_IMPLICIT drift falls 5.68e-1 -> 6.47e-2 and VARIATIONAL 8.70e-2 -> 1.05e-2 as dt goes 4 ms -> 0.5 ms, observed slopes 1.04 and 1.02. Both families stay finite and converge, so the convergence half of the row is reproduced. The packet explicitly does not cover the row's 'at matched cost' half: no timing methodology was applied, performance is typed unsupported, and no ranking between the families is claimed. Angular momentum is reported as an observed envelope, not a conservation oracle, because gravity exerts a torque about the world origin. --- CHANGELOG.md | 4 +- .../README.md | 11 +- .../claims-manifest.json | 7 +- ...004-dart7-articulated-energy-momentum.json | 494 ++++++++++++++++ docs/plans/dashboard.md | 9 +- ...itation_ct004_articulated_energy_packet.py | 540 ++++++++++++++++++ 6 files changed, 1053 insertions(+), 12 deletions(-) create mode 100644 docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json create mode 100644 scripts/write_citation_ct004_articulated_energy_packet.py diff --git a/CHANGELOG.md b/CHANGELOG.md index dd7e039d3df77..88f63f54d8d37 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -617,7 +617,9 @@ compatibility remains on the active DART 6 LTS branch._ friction-pyramid direction dependence reproduces on `main` under both Sequential Impulse and boxed-LCP contact solvers), CT-002 dense inelastic contact, and CT-003 dense elastic contact (no energy injection observed, - recorded as an honest negative result rather than a reproduction). + recorded as an honest negative result rather than a reproduction), and + CT-004 articulated energy drift versus timestep across both multibody + integration families. - Reorganized tests and CI coverage around DART 7 components, with focused unit, integration, benchmark, rendering, CUDA-smoke, collision, and simulation gates replacing broad stale test targets. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/README.md b/docs/dev_tasks/citation_driven_simulation_trust/README.md index 485d7fd55fc61..d7e46ef7285c7 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/README.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/README.md @@ -13,7 +13,9 @@ - [ ] Phase 2: Reproduce and classify the six-family first-wave corpus. Landed on `main`: CT-001 rolling/friction-direction (`reproduced`), CT-002 dense inelastic contact (`reproduced`), CT-003 dense elastic - contact (`unresolved` -- no energy injection observed). Landed on + contact (`unresolved` -- no energy injection observed), CT-004 articulated + energy drift versus timestep (`reproduced` for the convergence half; + matched cost explicitly not measured). Landed on `release-6.20`: CT-001 detector sweep (`reproduced` on fcl/dart/ode; bullet excluded for lacking the pyramid signature). Open: articulated energy/momentum/control, heel-strike/toe-off, high mass-ratio, and @@ -156,10 +158,9 @@ for downstream-sensitive changes. ## Immediate next steps 1. Read `RESUME.md` and verify both branch tips/worktrees. -2. Continue Phase 2 with the articulated energy/momentum family (CT-004), - reusing `StepMetrics` and the PLAN-084 variational integrator, then the - controlled-articulated row (CT-005) seeded by - `python/examples/demos/scenes/atlas_simbicon.py`. +2. Continue Phase 2 with the controlled-articulated row (CT-005), seeded by + `python/examples/demos/scenes/atlas_simbicon.py`, then heel-strike/toe-off + (CT-006), which depends on the WS3 contact semantics. 3. Start WS4's first slice in parallel where it unblocks packets: expose `ResolvedSolverConfiguration` to Python and a comparable per-solve residual, both currently typed unsupported in every packet. diff --git a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json index 4b4f75cfc83ab..f0f322ab4e6b7 100644 --- a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json +++ b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json @@ -89,9 +89,12 @@ "lanes": { "dart7": { "owner": "PLAN-123 with PLAN-080/084 using StepMetrics", - "status": "audit-required", + "status": "in-progress", "disposition": null, - "evidence": [] + "evidence": [ + "evidence/CT-004-dart7-articulated-energy-momentum.json" + ], + "notes": "Passive 4-link pendulum chain swept over timestep for both multibody integration families: relative energy drift shrinks with timestep in both (observed log-log slopes ~1.0). The matched-cost half of the row is NOT covered: no timing methodology was applied, so performance is typed unsupported and no family ranking is claimed." }, "dart6": { "owner": "docs/dev_tasks/dart6_citation_contact_trust", diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json new file mode 100644 index 0000000000000..0011548cd64e5 --- /dev/null +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json @@ -0,0 +1,494 @@ +{ + "claim_id": "CT-004", + "configuration": { + "detector": "not applicable: the scene has no collision shapes and runs no narrow phase", + "fallback_policy": "World defaults; no per-island fallback reporting is exposed on main (PLAN-123 WS4 follow-up)", + "iterations": "World defaults; the variational family uses MultibodyOptions.variational_max_iterations/tolerance defaults, which this packet does not vary", + "requested": { + "backend": "cpu", + "contact": "none (passive chain, no collision shapes)", + "control": "none (released from rest)", + "integration_family_sweep": [ + "SEMI_IMPLICIT", + "VARIATIONAL" + ], + "precision": "float64", + "threads": "World default sequential step", + "timestep_sweep_s": [ + 0.004, + 0.002, + 0.001, + 0.0005 + ] + }, + "resolved": { + "by_cell": { + "SEMI_IMPLICIT@dt=0.0005": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.0005 + }, + "SEMI_IMPLICIT@dt=0.001": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.001 + }, + "SEMI_IMPLICIT@dt=0.002": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "SEMI_IMPLICIT@dt=0.004": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.004 + }, + "VARIATIONAL@dt=0.0005": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.0005 + }, + "VARIATIONAL@dt=0.001": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.001 + }, + "VARIATIONAL@dt=0.002": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "VARIATIONAL@dt=0.004": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.004 + } + } + }, + "resolved_provenance": "World property readback (multibody_options.integration_family, rigid_body_solver, gravity, time_step) after enter_simulation_mode, with integration family, timestep, and gravity each asserted equal to the request per repeat and per cell. World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", + "substeps": 1, + "timestep": 0.0005 + }, + "ensemble": { + "deterministic_repeats": 2, + "deterministic_repeats_identical": true, + "kind": "parameter-sweep-with-deterministic-repeats", + "measurement_window": { + "end_s": 2.0, + "start_s": 0.0 + }, + "sweep": [ + { + "integration_family": "SEMI_IMPLICIT", + "timestep_s": 0.004 + }, + { + "integration_family": "SEMI_IMPLICIT", + "timestep_s": 0.002 + }, + { + "integration_family": "SEMI_IMPLICIT", + "timestep_s": 0.001 + }, + { + "integration_family": "SEMI_IMPLICIT", + "timestep_s": 0.0005 + }, + { + "integration_family": "VARIATIONAL", + "timestep_s": 0.004 + }, + { + "integration_family": "VARIATIONAL", + "timestep_s": 0.002 + }, + { + "integration_family": "VARIATIONAL", + "timestep_s": 0.001 + }, + { + "integration_family": "VARIATIONAL", + "timestep_s": 0.0005 + } + ] + }, + "evidence": { + "commands": [ + "PYTHONPATH=build/default/cpp/Release/python pixi run python scripts/write_citation_ct004_articulated_energy_packet.py" + ], + "raw_rows": [ + { + "final_total_energy_j": -1.0538611186027076, + "finite": true, + "initial_total_energy_j": -2.0182964946827866, + "integration_family": "SEMI_IMPLICIT", + "max_abs_angular_momentum": 5.941837277739446, + "max_abs_energy_drift_j": 1.1453730886021916, + "max_abs_relative_energy_drift": 0.5674949600416408, + "repeat_trajectory_sha256": [ + "33a3b41b64f4546027954c752949346e3288deb713e3993639cc54b00f1e6b5b", + "33a3b41b64f4546027954c752949346e3288deb713e3993639cc54b00f1e6b5b" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.004 + }, + "steps": 500, + "timestep_s": 0.004, + "trajectory_sha256": "33a3b41b64f4546027954c752949346e3288deb713e3993639cc54b00f1e6b5b" + }, + { + "final_total_energy_j": -1.5674156373231396, + "finite": true, + "initial_total_energy_j": -2.0182964946827866, + "integration_family": "SEMI_IMPLICIT", + "max_abs_angular_momentum": 5.923917480629072, + "max_abs_energy_drift_j": 0.5434502873429725, + "max_abs_relative_energy_drift": 0.2692618694897877, + "repeat_trajectory_sha256": [ + "ca6f6cab3e3f43b60e5cf1f5d108239cd10e4a78a6287aa0d46b994c5cd9ad03", + "ca6f6cab3e3f43b60e5cf1f5d108239cd10e4a78a6287aa0d46b994c5cd9ad03" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "steps": 1000, + "timestep_s": 0.002, + "trajectory_sha256": "ca6f6cab3e3f43b60e5cf1f5d108239cd10e4a78a6287aa0d46b994c5cd9ad03" + }, + { + "final_total_energy_j": -1.8004734230407382, + "finite": true, + "initial_total_energy_j": -2.0182964946827866, + "integration_family": "SEMI_IMPLICIT", + "max_abs_angular_momentum": 5.915049158062866, + "max_abs_energy_drift_j": 0.26467289477035383, + "max_abs_relative_energy_drift": 0.13113677572528915, + "repeat_trajectory_sha256": [ + "a8ab82d9906be5879e174ba1b6b38f64ce0714df75d4fa426b87c70897f39f5a", + "a8ab82d9906be5879e174ba1b6b38f64ce0714df75d4fa426b87c70897f39f5a" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.001 + }, + "steps": 2000, + "timestep_s": 0.001, + "trajectory_sha256": "a8ab82d9906be5879e174ba1b6b38f64ce0714df75d4fa426b87c70897f39f5a" + }, + { + "final_total_energy_j": -1.911266312486568, + "finite": true, + "initial_total_energy_j": -2.0182964946827866, + "integration_family": "SEMI_IMPLICIT", + "max_abs_angular_momentum": 5.910616829568312, + "max_abs_energy_drift_j": 0.1306014805077389, + "max_abs_relative_energy_drift": 0.0647087684350685, + "repeat_trajectory_sha256": [ + "828539234f8a9e0c7a1abc644374e751a5542dee391886e1708b8a1b82120c73", + "828539234f8a9e0c7a1abc644374e751a5542dee391886e1708b8a1b82120c73" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.0005 + }, + "steps": 4000, + "timestep_s": 0.0005, + "trajectory_sha256": "828539234f8a9e0c7a1abc644374e751a5542dee391886e1708b8a1b82120c73" + }, + { + "final_total_energy_j": -2.0437843185234734, + "finite": true, + "initial_total_energy_j": -2.0182964946827866, + "integration_family": "VARIATIONAL", + "max_abs_angular_momentum": 5.90501861618165, + "max_abs_energy_drift_j": 0.1755340480261065, + "max_abs_relative_energy_drift": 0.08697138824179299, + "repeat_trajectory_sha256": [ + "3b260b2aad289ad0ee2c5d0bb9a88fbc0887ed7ba0bed29b988078cfb9f6ec36", + "3b260b2aad289ad0ee2c5d0bb9a88fbc0887ed7ba0bed29b988078cfb9f6ec36" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.004 + }, + "steps": 500, + "timestep_s": 0.004, + "trajectory_sha256": "3b260b2aad289ad0ee2c5d0bb9a88fbc0887ed7ba0bed29b988078cfb9f6ec36" + }, + { + "final_total_energy_j": -2.0311100411781546, + "finite": true, + "initial_total_energy_j": -2.0182964946827866, + "integration_family": "VARIATIONAL", + "max_abs_angular_momentum": 5.905652977912613, + "max_abs_energy_drift_j": 0.08610771893666502, + "max_abs_relative_energy_drift": 0.042663562644792916, + "repeat_trajectory_sha256": [ + "c320c585b91fb94fdffb008502a85970585970f4d7b6db6e591339c6ece79582", + "c320c585b91fb94fdffb008502a85970585970f4d7b6db6e591339c6ece79582" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002 + }, + "steps": 1000, + "timestep_s": 0.002, + "trajectory_sha256": "c320c585b91fb94fdffb008502a85970585970f4d7b6db6e591339c6ece79582" + }, + { + "final_total_energy_j": -2.02471920214893, + "finite": true, + "initial_total_energy_j": -2.0182964946827866, + "integration_family": "VARIATIONAL", + "max_abs_angular_momentum": 5.905947739122709, + "max_abs_energy_drift_j": 0.04264872954602339, + "max_abs_relative_energy_drift": 0.02113105267654267, + "repeat_trajectory_sha256": [ + "74505b657816a4f2795e8817e1a4e448bf75e939797992abf2d878a4d8ddf8ae", + "74505b657816a4f2795e8817e1a4e448bf75e939797992abf2d878a4d8ddf8ae" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.001 + }, + "steps": 2000, + "timestep_s": 0.001, + "trajectory_sha256": "74505b657816a4f2795e8817e1a4e448bf75e939797992abf2d878a4d8ddf8ae" + }, + { + "final_total_energy_j": -2.0215116473513888, + "finite": true, + "initial_total_energy_j": -2.0182964946827866, + "integration_family": "VARIATIONAL", + "max_abs_angular_momentum": 5.906075287504381, + "max_abs_energy_drift_j": 0.02121937787327255, + "max_abs_relative_energy_drift": 0.01051350875809135, + "repeat_trajectory_sha256": [ + "dac8263daea9125b41a51bb9fed6ef1e4e965f79e4256844dfe826ea243a92dd", + "dac8263daea9125b41a51bb9fed6ef1e4e965f79e4256844dfe826ea243a92dd" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "rigid_body_solver": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.0005 + }, + "steps": 4000, + "timestep_s": 0.0005, + "trajectory_sha256": "dac8263daea9125b41a51bb9fed6ef1e4e965f79e4256844dfe826ea243a92dd" + } + ], + "visual": { + "reason": "The oracle is the numeric energy-drift trend against timestep; no visible-behavior claim is made.", + "status": "not-applicable" + } + }, + "host": { + "machine": "x86_64", + "note": "Host recorded for provenance only; no timing methodology was applied", + "performance_valid": false, + "platform": "Linux-7.0.0-29-generic-x86_64-with-glibc2.43", + "python": "3.14.6" + }, + "metrics": { + "allocation": { + "reason": "Post-bake allocation gates are owned by PLAN-122 tooling; this packet does not measure allocations", + "status": "unsupported" + }, + "numerical": { + "angular_momentum_semantics": "Gravity exerts a torque about the world origin, so angular momentum is NOT conserved in this scene. The envelope is recorded as an observed trajectory bound, not as a conservation oracle.", + "max_abs_angular_momentum_over_cells": 5.941837277739446, + "method": "Angular momentum magnitude about the world origin from StepMetrics.angular_momentum, reported as an observed envelope", + "solver_residual": { + "reason": "StepMetrics.last_step_residual is structurally zero for rigid contact on this branch: recordSolverDiagnostics() in dart/simulation/compute/rigid_body_contact_stage.cpp takes residual = 0.0 by default and no rigid contact call site passes one, so no residual is computed for either contact solver. Exposing a comparable residual is PLAN-123 WS4 work.", + "status": "unsupported" + } + }, + "performance": { + "reason": "The corpus row asks for a matched-cost comparison, but no interleaved same-host timing methodology was applied here, so no cost is measured and no accuracy-per-cost claim is made", + "status": "unsupported" + }, + "physical": { + "all_cells_finite": true, + "families_with_shrinking_drift": [ + "SEMI_IMPLICIT", + "VARIATIONAL" + ], + "method": "Passive chain: total mechanical energy is conserved analytically, so the oracle is max |total_energy - initial| over the run from StepMetrics.total_energy, normalized by the initial energy, swept over timestep. The convergence order is the least-squares slope of log(drift) versus log(dt).", + "per_family_summary": { + "SEMI_IMPLICIT": { + "cells": 4, + "drift_at_largest_timestep": 0.5674949600416408, + "drift_at_smallest_timestep": 0.0647087684350685, + "drift_shrinks_with_timestep": true, + "max_abs_angular_momentum": 5.941837277739446, + "observed_convergence_order": 1.0435660979320294, + "relative_energy_drift_by_timestep": { + "0.0005": 0.0647087684350685, + "0.001": 0.13113677572528915, + "0.002": 0.2692618694897877, + "0.004": 0.5674949600416408 + } + }, + "VARIATIONAL": { + "cells": 4, + "drift_at_largest_timestep": 0.08697138824179299, + "drift_at_smallest_timestep": 0.01051350875809135, + "drift_shrinks_with_timestep": true, + "max_abs_angular_momentum": 5.906075287504381, + "observed_convergence_order": 1.0158529696870273, + "relative_energy_drift_by_timestep": { + "0.0005": 0.01051350875809135, + "0.001": 0.02113105267654267, + "0.002": 0.042663562644792916, + "0.004": 0.08697138824179299 + } + } + } + } + }, + "result": { + "claim_boundary": "DART 7 main, this commit, a passive 4-link planar pendulum chain with no contact and no control, 2 s horizon, timesteps 0.5-4 ms, SEMI_IMPLICIT and VARIATIONAL multibody integration families. What is established: both families stay finite and their relative energy drift shrinks as the timestep shrinks, with the observed log-log slopes recorded per family. What is NOT established: the 'at matched cost' half of the corpus row, because no timing was measured; any contact-accuracy claim, because the scene has no contact; and any ranking between the two families. Says nothing about historical DART versions, controlled or contacting articulated scenes, or DART 6.", + "disposition": "reproduced", + "limitations": [ + "No contact and no control: this isolates integration accuracy and therefore covers only part of the corpus row, whose contact and matched-cost halves need separate work.", + "Cost is recorded as step count only; performance is typed unsupported, so nothing here supports an accuracy-per-cost comparison.", + "Angular momentum is not conserved under gravity about the world origin, so it is reported as an envelope rather than an invariant.", + "Variational iteration limits and tolerance are left at World defaults and not swept; a tighter tolerance would change the drift figures.", + "Four timesteps over one decade; the observed slope is a trend summary, not a certified order of accuracy." + ] + }, + "review": { + "passes": [] + }, + "scene": { + "description": "Passive 4-link planar pendulum chain (revolute hinges about y, link length 0.3 m, mass 1.0 kg) released from rest at alternating joint angles under gravity, no contact and no control, simulated to a 2 s horizon per timestep/integration-family cell.", + "digest": "sha256:26b58c168fd28b84eaad15411bccede42f35661b1f76198dd34d6daa6b238b76", + "id": "ct004_passive_pendulum_chain", + "parameters": { + "description": "Passive 4-link planar pendulum chain (revolute hinges about y, link length 0.3 m, mass 1.0 kg) released from rest at alternating joint angles under gravity, no contact and no control, simulated to a 2 s horizon per timestep/integration-family cell.", + "deterministic_repeats": 2, + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "horizon_s": 2.0, + "initial_joint_angle_rad": 0.35, + "integration_families": [ + "SEMI_IMPLICIT", + "VARIATIONAL" + ], + "link_count": 4, + "link_length_m": 0.3, + "link_mass_kg": 1.0, + "link_radius_m": 0.05, + "scene_id": "ct004_passive_pendulum_chain", + "timesteps_s": [ + 0.004, + 0.002, + 0.001, + 0.0005 + ] + } + }, + "schema": "dart.citation_claim_evidence/v1", + "source": { + "claim": "Articulated integration/contact accuracy must be compared at matched cost.", + "url": "https://leggedrobotics.github.io/SimBenchmark/" + }, + "target": { + "branch": "main", + "commit": "d7281579701e4f009122968c5782291d12f69ac7", + "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit." + }, + "title": "Articulated energy drift versus timestep across integration families (DART 7 first packet)" +} diff --git a/docs/plans/dashboard.md b/docs/plans/dashboard.md index 7405e9de46e39..210a4742c21d0 100644 --- a/docs/plans/dashboard.md +++ b/docs/plans/dashboard.md @@ -54,10 +54,11 @@ its own line so status updates remain git-history friendly. - Next step: WS0 audit and the WS1 fail-closed claim/evidence manifest landed with a permanent negative control and three first-wave packets: CT-001 rolling-direction (`reproduced`), CT-002 dense inelastic contact - (`reproduced`), and CT-003 dense elastic contact (`unresolved` -- no energy - injection observed). Next: the remaining first-wave families (articulated - energy/momentum/control, heel-strike/toe-off, high mass-ratio, - reset/concurrency queries) with branch-qualified dispositions, reusing + (`reproduced`), CT-003 dense elastic contact (`unresolved` -- no energy + injection observed), and CT-004 articulated energy drift versus timestep + (`reproduced`; the matched-cost half is explicitly uncovered). Next: the + remaining first-wave families (articulated control, heel-strike/toe-off, + high mass-ratio, reset/concurrency queries) with branch-qualified dispositions, reusing existing demos scenes and PLAN-104/621/622 evidence, before contact-identity/semantics work (WS3) and cross-family diagnostics (WS4); WS4's first slice is exposing `ResolvedSolverConfiguration` to Python and a diff --git a/scripts/write_citation_ct004_articulated_energy_packet.py b/scripts/write_citation_ct004_articulated_energy_packet.py new file mode 100644 index 0000000000000..8daccf071636e --- /dev/null +++ b/scripts/write_citation_ct004_articulated_energy_packet.py @@ -0,0 +1,540 @@ +#!/usr/bin/env python3 +"""Write the CT-004 articulated energy/momentum evidence packet (PLAN-123 WS2). + +Bounded claim (corpus row CT-004): articulated integration accuracy must be +comparable at matched cost, with energy/momentum drift measured against +timestep. + +Bounded reconstruction (explicitly not source-exact): a passive planar +pendulum chain released from rest under gravity, with no contact and no +control, swept over a timestep grid for each multibody integration family +(`SEMI_IMPLICIT` and `VARIATIONAL`). A passive chain conserves total +mechanical energy exactly, so the oracle is unambiguous and solver-neutral: + +- energy drift relative to the initial total energy, over the run; +- whether that drift shrinks with timestep, and at what observed order; +- angular-momentum behavior about the world origin (gravity exerts a torque + about the origin, so this is reported as a trajectory, not a conservation + claim -- the packet does not pretend it is conserved); +- deterministic repeats: each cell runs twice and must be bit-identical. + +Cost is recorded as step count only. This packet makes NO timing claim: a +matched-cost comparison needs interleaved same-host methodology, which is not +applied here, so `metrics.performance` is typed unsupported. The corpus row's +"at matched cost" half is therefore explicitly not covered yet. + +Usage (after `pixi run build`): + + PYTHONPATH=build/default/cpp/Release/python pixi run python \ + scripts/write_citation_ct004_articulated_energy_packet.py +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import platform +import subprocess +import sys +from pathlib import Path +from typing import Any + +import dartpy as sx +import numpy as np +from citation_packet_utils import UNSUPPORTED_SOLVER_RESIDUAL + +REPO_ROOT = Path(__file__).resolve().parents[1] +DEFAULT_OUTPUT = ( + REPO_ROOT + / "docs" + / "plans" + / "123-citation-driven-simulation-trust" + / "evidence" + / "CT-004-dart7-articulated-energy-momentum.json" +) + +SCENE_PARAMETERS: dict[str, Any] = { + "scene_id": "ct004_passive_pendulum_chain", + "description": ( + "Passive 4-link planar pendulum chain (revolute hinges about y, link " + "length 0.3 m, mass 1.0 kg) released from rest at alternating joint " + "angles under gravity, no contact and no control, simulated to a 2 s " + "horizon per timestep/integration-family cell." + ), + "gravity_mps2": [0.0, 0.0, -9.81], + "link_count": 4, + "link_length_m": 0.3, + "link_mass_kg": 1.0, + "link_radius_m": 0.05, + "initial_joint_angle_rad": 0.35, + "horizon_s": 2.0, + "timesteps_s": [0.004, 0.002, 0.001, 0.0005], + "integration_families": ["SEMI_IMPLICIT", "VARIATIONAL"], + "deterministic_repeats": 2, +} + + +def scene_digest(parameters: dict[str, Any]) -> str: + canonical = json.dumps(parameters, sort_keys=True, separators=(",", ":")) + return "sha256:" + hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + +def _translation(x: float, y: float, z: float) -> np.ndarray: + transform = np.eye(4) + transform[:3, 3] = (x, y, z) + return transform + + +def run_single( + family_name: str, timestep: float, parameters: dict[str, Any] +) -> dict[str, Any]: + """Run one passive-chain release and return raw metrics plus a hash.""" + link_count = int(parameters["link_count"]) + length = float(parameters["link_length_m"]) + mass = float(parameters["link_mass_kg"]) + radius = float(parameters["link_radius_m"]) + angle = float(parameters["initial_joint_angle_rad"]) + step_count = int(round(float(parameters["horizon_s"]) / timestep)) + + world = sx.World( + time_step=timestep, + gravity=np.asarray(parameters["gravity_mps2"], dtype=float), + multibody_options=sx.MultibodyOptions( + integration_family=sx.MultibodyIntegrationFamily[family_name] + ), + ) + + chain = world.add_multibody("ct004_chain") + parent = chain.add_link("base") + ixx = 0.5 * mass * radius * radius + itrans = mass * length * length / 12.0 + for index in range(link_count): + offset = 0.0 if index == 0 else length + link = chain.add_link( + f"link{index}", + parent=parent, + joint=sx.JointSpec( + name=f"hinge{index}", + type=sx.JointType.REVOLUTE, + axis=(0.0, 1.0, 0.0), + transform_from_parent=_translation(offset, 0.0, 0.0), + ), + ) + link.mass = mass + link.inertia = ((ixx, 0.0, 0.0), (0.0, itrans, 0.0), (0.0, 0.0, itrans)) + link.parent_joint.position = [angle * (-1.0 if index % 2 else 1.0)] + parent = link + + world.enter_simulation_mode() + + resolved = { + "integration_family": world.multibody_options.integration_family.name, + "rigid_body_solver": world.rigid_body_solver.name, + "gravity_mps2": [float(v) for v in np.asarray(world.gravity)], + "time_step_s": float(world.time_step), + } + + initial = world.compute_step_metrics() + initial_total_energy = float(initial.total_energy) + initial_angular_momentum = np.asarray(initial.angular_momentum, dtype=float) + + trajectory = hashlib.sha256() + max_abs_energy_drift = 0.0 + max_abs_angular_momentum = float(np.linalg.norm(initial_angular_momentum)) + non_finite = False + + for _ in range(step_count): + world.step() + metrics = world.compute_step_metrics() + total_energy = float(metrics.total_energy) + angular_momentum = np.asarray(metrics.angular_momentum, dtype=float) + if not (math.isfinite(total_energy) and np.all(np.isfinite(angular_momentum))): + non_finite = True + break + max_abs_energy_drift = max( + max_abs_energy_drift, abs(total_energy - initial_total_energy) + ) + max_abs_angular_momentum = max( + max_abs_angular_momentum, float(np.linalg.norm(angular_momentum)) + ) + trajectory.update(np.array([total_energy]).tobytes()) + trajectory.update(angular_momentum.tobytes()) + + final = world.compute_step_metrics() + return { + "integration_family": family_name, + "timestep_s": timestep, + "steps": step_count, + "resolved": resolved, + "finite": not non_finite, + "trajectory_sha256": trajectory.hexdigest(), + "initial_total_energy_j": initial_total_energy, + "final_total_energy_j": float(final.total_energy) if not non_finite else None, + "max_abs_energy_drift_j": max_abs_energy_drift, + "max_abs_relative_energy_drift": ( + max_abs_energy_drift / abs(initial_total_energy) + if initial_total_energy != 0.0 + else None + ), + "max_abs_angular_momentum": max_abs_angular_momentum, + } + + +def git_head() -> str: + return subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=REPO_ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + + +def observed_order(drifts_by_timestep: dict[float, float]) -> Any: + """Least-squares slope of log(drift) vs log(dt) over the sweep. + + A converging integrator shows a positive order: halving the timestep + reduces the energy drift. The slope is the honest summary of that; the + packet does not assert a nominal order the method is "supposed" to have. + """ + points = [ + (math.log(dt), math.log(drift)) + for dt, drift in sorted(drifts_by_timestep.items()) + if dt > 0.0 and drift > 0.0 + ] + if len(points) < 2: + return { + "status": "unsupported", + "reason": ( + "Fewer than two cells have a strictly positive energy drift, " + "so a log-log convergence slope is undefined rather than " + "zero." + ), + } + mean_x = sum(x for x, _ in points) / len(points) + mean_y = sum(y for _, y in points) / len(points) + denominator = sum((x - mean_x) ** 2 for x, _ in points) + if denominator == 0.0: + return { + "status": "unsupported", + "reason": "All sampled timesteps are equal; slope is undefined.", + } + numerator = sum((x - mean_x) * (y - mean_y) for x, y in points) + return numerator / denominator + + +def build_packet() -> dict[str, Any]: + parameters = SCENE_PARAMETERS + rows: list[dict[str, Any]] = [] + determinism_failures: list[str] = [] + resolved_by_cell: dict[str, dict[str, Any]] = {} + + for family in parameters["integration_families"]: + for timestep in parameters["timesteps_s"]: + repeats = [ + run_single(family, timestep, parameters) + for _ in range(int(parameters["deterministic_repeats"])) + ] + hashes = {run["trajectory_sha256"] for run in repeats} + if len(hashes) != 1: + determinism_failures.append( + f"{family} dt {timestep}: trajectory hashes differ " + f"{sorted(hashes)}" + ) + row = repeats[0] + for repeat in repeats: + readback = repeat["resolved"] + if readback != row["resolved"]: + raise SystemExit( + f"{family} dt {timestep}: resolved configuration " + "differs between repeats" + ) + if readback["integration_family"] != family: + raise SystemExit( + f"requested integration family {family} but World " + f"readback reports {readback['integration_family']}" + ) + if readback["time_step_s"] != timestep: + raise SystemExit( + f"requested timestep {timestep} but World readback " + f"reports {readback['time_step_s']}" + ) + if readback["gravity_mps2"] != list(parameters["gravity_mps2"]): + raise SystemExit( + f"requested gravity {parameters['gravity_mps2']} but " + f"World readback reports {readback['gravity_mps2']}" + ) + row["repeat_trajectory_sha256"] = [ + repeat["trajectory_sha256"] for repeat in repeats + ] + resolved_by_cell[f"{family}@dt={timestep}"] = row["resolved"] + rows.append(row) + + if determinism_failures: + raise SystemExit( + "deterministic repeats failed:\n " + "\n ".join(determinism_failures) + ) + + family_summary: dict[str, Any] = {} + for family in parameters["integration_families"]: + family_rows = [row for row in rows if row["integration_family"] == family] + drifts = { + row["timestep_s"]: row["max_abs_relative_energy_drift"] + for row in family_rows + if row["max_abs_relative_energy_drift"] is not None + } + smallest_dt = min(drifts) if drifts else None + largest_dt = max(drifts) if drifts else None + family_summary[family] = { + "cells": len(family_rows), + "relative_energy_drift_by_timestep": { + str(dt): value for dt, value in sorted(drifts.items()) + }, + "observed_convergence_order": observed_order(drifts), + "drift_at_smallest_timestep": drifts[smallest_dt], + "drift_at_largest_timestep": drifts[largest_dt], + "drift_shrinks_with_timestep": drifts[smallest_dt] < drifts[largest_dt], + "max_abs_angular_momentum": max( + row["max_abs_angular_momentum"] for row in family_rows + ), + } + + all_finite = all(row["finite"] for row in rows) + converging = sorted( + family + for family, stats in family_summary.items() + if stats["drift_shrinks_with_timestep"] + ) + disposition = ( + "reproduced" + if all_finite and len(converging) == len(parameters["integration_families"]) + else "unresolved" + ) + + command = ( + "PYTHONPATH=build/default/cpp/Release/python pixi run python " + "scripts/write_citation_ct004_articulated_energy_packet.py" + ) + packet: dict[str, Any] = { + "schema": "dart.citation_claim_evidence/v1", + "claim_id": "CT-004", + "title": ( + "Articulated energy drift versus timestep across integration " + "families (DART 7 first packet)" + ), + "source": { + "url": "https://leggedrobotics.github.io/SimBenchmark/", + "claim": ( + "Articulated integration/contact accuracy must be compared " + "at matched cost." + ), + }, + "target": { + "branch": "main", + "commit": git_head(), + "commit_role": ( + "Source state measured: the library and fixture ran at this " + "commit, which is HEAD at capture time. The packet and its " + "writer land in a later commit." + ), + }, + "scene": { + "id": parameters["scene_id"], + "digest": scene_digest(parameters), + "description": parameters["description"], + "parameters": parameters, + }, + "configuration": { + "requested": { + "integration_family_sweep": parameters["integration_families"], + "timestep_sweep_s": parameters["timesteps_s"], + "contact": "none (passive chain, no collision shapes)", + "control": "none (released from rest)", + "precision": "float64", + "backend": "cpu", + "threads": "World default sequential step", + }, + "resolved": {"by_cell": resolved_by_cell}, + "resolved_provenance": ( + "World property readback (multibody_options.integration_" + "family, rigid_body_solver, gravity, time_step) after " + "enter_simulation_mode, with integration family, timestep, " + "and gravity each asserted equal to the request per repeat " + "and per cell. World::getResolvedConfiguration() is not yet " + "exposed to Python (PLAN-123 WS4 follow-up)." + ), + "detector": ( + "not applicable: the scene has no collision shapes and runs " + "no narrow phase" + ), + "timestep": min(parameters["timesteps_s"]), + "substeps": 1, + "iterations": ( + "World defaults; the variational family uses " + "MultibodyOptions.variational_max_iterations/tolerance " + "defaults, which this packet does not vary" + ), + "fallback_policy": ( + "World defaults; no per-island fallback reporting is exposed " + "on main (PLAN-123 WS4 follow-up)" + ), + }, + "ensemble": { + "kind": "parameter-sweep-with-deterministic-repeats", + "sweep": [ + {"integration_family": family, "timestep_s": timestep} + for family in parameters["integration_families"] + for timestep in parameters["timesteps_s"] + ], + "deterministic_repeats": int(parameters["deterministic_repeats"]), + "deterministic_repeats_identical": not determinism_failures, + "measurement_window": { + "start_s": 0.0, + "end_s": parameters["horizon_s"], + }, + }, + "metrics": { + "physical": { + "method": ( + "Passive chain: total mechanical energy is conserved " + "analytically, so the oracle is max |total_energy - " + "initial| over the run from StepMetrics.total_energy, " + "normalized by the initial energy, swept over timestep. " + "The convergence order is the least-squares slope of " + "log(drift) versus log(dt)." + ), + "all_cells_finite": all_finite, + "per_family_summary": family_summary, + "families_with_shrinking_drift": converging, + }, + "numerical": { + "method": ( + "Angular momentum magnitude about the world origin from " + "StepMetrics.angular_momentum, reported as an observed " + "envelope" + ), + "max_abs_angular_momentum_over_cells": max( + row["max_abs_angular_momentum"] for row in rows + ), + "angular_momentum_semantics": ( + "Gravity exerts a torque about the world origin, so " + "angular momentum is NOT conserved in this scene. The " + "envelope is recorded as an observed trajectory bound, " + "not as a conservation oracle." + ), + "solver_residual": dict(UNSUPPORTED_SOLVER_RESIDUAL), + }, + "performance": { + "status": "unsupported", + "reason": ( + "The corpus row asks for a matched-cost comparison, but " + "no interleaved same-host timing methodology was applied " + "here, so no cost is measured and no accuracy-per-cost " + "claim is made" + ), + }, + "allocation": { + "status": "unsupported", + "reason": ( + "Post-bake allocation gates are owned by PLAN-122 " + "tooling; this packet does not measure allocations" + ), + }, + }, + "evidence": { + "commands": [command], + "raw_rows": rows, + "visual": { + "status": "not-applicable", + "reason": ( + "The oracle is the numeric energy-drift trend against " + "timestep; no visible-behavior claim is made." + ), + }, + }, + "result": { + "disposition": disposition, + "claim_boundary": ( + "DART 7 main, this commit, a passive 4-link planar pendulum " + "chain with no contact and no control, 2 s horizon, " + "timesteps 0.5-4 ms, SEMI_IMPLICIT and VARIATIONAL multibody " + "integration families. What is established: both families " + "stay finite and their relative energy drift shrinks as the " + "timestep shrinks, with the observed log-log slopes recorded " + "per family. What is NOT established: the 'at matched cost' " + "half of the corpus row, because no timing was measured; " + "any contact-accuracy claim, because the scene has no " + "contact; and any ranking between the two families. Says " + "nothing about historical DART versions, controlled or " + "contacting articulated scenes, or DART 6." + ), + "limitations": [ + "No contact and no control: this isolates integration " + "accuracy and therefore covers only part of the corpus row, " + "whose contact and matched-cost halves need separate work.", + "Cost is recorded as step count only; performance is typed " + "unsupported, so nothing here supports an accuracy-per-cost " + "comparison.", + "Angular momentum is not conserved under gravity about the " + "world origin, so it is reported as an envelope rather than " + "an invariant.", + "Variational iteration limits and tolerance are left at " + "World defaults and not swept; a tighter tolerance would " + "change the drift figures.", + "Four timesteps over one decade; the observed slope is a " + "trend summary, not a certified order of accuracy.", + ], + }, + "review": {"passes": []}, + "host": { + "platform": platform.platform(), + "python": sys.version.split()[0], + "machine": platform.machine(), + "performance_valid": False, + "note": ( + "Host recorded for provenance only; no timing methodology " + "was applied" + ), + }, + } + return packet + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--output", + type=Path, + default=DEFAULT_OUTPUT, + help=f"Packet output path (default: {DEFAULT_OUTPUT})", + ) + return parser.parse_args() + + +def main() -> int: + args = parse_args() + packet = build_packet() + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text( + json.dumps(packet, indent=2, sort_keys=True) + "\n", encoding="utf-8" + ) + physical = packet["metrics"]["physical"] + print(f"wrote {args.output}") + for family, stats in physical["per_family_summary"].items(): + order = stats["observed_convergence_order"] + order_text = ( + f"{order:.2f}" if isinstance(order, (int, float)) else "unsupported" + ) + print( + f" {family}: drift {stats['drift_at_largest_timestep']:.3e} -> " + f"{stats['drift_at_smallest_timestep']:.3e} (relative), observed " + f"order {order_text}" + ) + print(f" disposition: {packet['result']['disposition']}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) From 9101d917fdb33998abd0a48d705fe42a632d1493 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Fri, 14 Aug 2026 20:18:18 -0700 Subject: [PATCH 06/52] Record the CT-004 slice and refresh the resume state --- .../RESUME.md | 26 +++++++++++------- .../verification.md | 27 +++++++++++++++++++ 2 files changed, 43 insertions(+), 10 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md index acfcf24071abb..0bd5d92f82142 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md @@ -23,16 +23,18 @@ statuses were verified; the corpus sidecar now carries the WS0 audit record. file the packet checks never reached), unsupported-as-zero metrics in this program's own packets, and a DART 6 attribution error crediting bullet with a friction-pyramid signature its data does not show. See `verification.md`. -- First-wave packets landed: CT-001 (both branches), CT-002 and CT-003 - (`main`). +- First-wave packets landed: CT-001 (both branches), CT-002, CT-003, and + CT-004 (`main`). Four of the six capped families now have at least one + branch-qualified packet. ## Current branches - DART 7: `feature/citation-trust-foundation` in `.claude/worktrees/citation-trust-main`, based on `origin/main` - 20501341226, now at `54bf5884abe` with three commits: - the PLAN-123 contract, the gate hardening plus dense-contact packets, and - the round-2 verdict/provenance corrections. + 20501341226, now at `85e5db2c168` with five commits: the PLAN-123 + contract, the gate hardening plus dense-contact packets, the round-2 + verdict/provenance corrections, the RESUME state record, and the CT-004 + articulated energy-drift packet. - DART 6: `feature/dart6-citation-contact-trust` in `.claude/worktrees/citation-trust-620`, based on `origin/release-6.20` 39ccd52068b, now at `afc6d7ac3c2` with four commits. @@ -42,11 +44,15 @@ approval. ## Immediate next step -Continue Phase 2 on `main` with the articulated energy/momentum family -(CT-004): a passive multi-link chain swept over timestep, using -`StepMetrics.total_energy`/`angular_momentum` as the oracle and the PLAN-084 -variational integrator as the comparison arm, written with -`scripts/citation_packet_utils.py` so unsupported quantities stay typed. +Continue Phase 2 on `main` with the controlled-articulated row (CT-005), +seeded by `python/examples/demos/scenes/atlas_simbicon.py`: tracking error and +constraint error under a PD controller, swept over timestep, with control work +recorded. Keep the same discipline as CT-004 -- no timing claim unless an +interleaved same-host methodology is actually applied. + +Remaining first-wave families after that: heel-strike/toe-off (CT-006, which +depends on the WS3 contact semantics), high mass-ratio stacks (CT-007), and +reset/concurrency queries (CT-011..CT-013). ## Context that would be lost diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index a57543de337f0..a7ba35466a752 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -98,6 +98,33 @@ scripts/write_citation_ct002_dense_contact_packet.py` — all cells finite; - Negative control: still fails (>= 3 errors) and is now also rejected if a lane references it as evidence. +## CT-004 articulated energy slice — 2026-08-14 + +- What changed: `scripts/write_citation_ct004_articulated_energy_packet.py` + and its packet; manifest lane, dashboard, changelog, and task status. +- Scene: passive 4-link planar pendulum chain (revolute hinges about y, 0.3 m + links, 1 kg each) released from rest at alternating 0.35 rad joint angles, + no contact and no control, 2 s horizon. A passive chain conserves total + mechanical energy exactly, which makes the oracle solver-neutral. +- Commands and results: + - `PYTHONPATH=build/default/cpp/Release/python pixi run python +scripts/write_citation_ct004_articulated_energy_packet.py` — 8 cells + (2 families x 4 timesteps), 2 deterministic repeats each, all + bit-identical; integration family, timestep, and gravity each asserted + against the request per repeat and per cell. + - Measured relative energy drift as dt goes 4 ms -> 0.5 ms: + SEMI_IMPLICIT 5.675e-1 -> 6.471e-2, VARIATIONAL 8.697e-2 -> 1.051e-2. + Observed least-squares log-log slopes 1.04 and 1.02. + - `pixi run check-citation-evidence` — OK; 68 pytest cases pass. +- Disposition `reproduced` covers only the convergence half of the corpus + row. The "at matched cost" half is explicitly uncovered: no timing + methodology was applied, `metrics.performance` is typed unsupported, and no + ranking between the two families is claimed even though their drift figures + differ. +- Angular momentum is recorded as an observed envelope, not a conservation + oracle, because gravity exerts a torque about the world origin. +- Negative control: unchanged and still failing. + ## Round-2 verification and second fix pass — 2026-08-14 Two independent verifiers re-checked the post-fix state. Both confirmed the From 2c33043f8ca9616dad99df65593894138e1c5201 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Fri, 14 Aug 2026 20:48:55 -0700 Subject: [PATCH 07/52] Close the non-string evidence bypass and correct the CT-004 disposition Final independent review found two majors: - A genuine gate bypass: a non-string entry in a lane's evidence list was silently dropped, so evidence: [{"path": ...}] on a closed lane validated with zero errors, skipping the missing-packet, not-JSON, outside-evidence, negative-control, and two-review checks. This is the same shape as the subdirectory bypass an earlier round fixed. Non-string entries now fail. - CT-004's disposition was reproduced on an oracle unrelated to the cited claim. The corpus defines reproduced as "the claimed behavior is observed"; CT-004's claim is a methodological prescription about comparison at matched cost, and this packet measures no cost, so none of its content was observed. The row is now unresolved, following CT-003, and the convergence result it did establish is published as a finding that explicitly does not promote it. Also from that review: raw_paths accepted a directory, negative controls in a subdirectory were enumerated by neither loop, and scene.digest was format-checked but never recomputed from the parameters the packet publishes (a hand-edited scene used to pass). CT-004's relative drift is disclosed as gauge-dependent, CT-001's claim boundary now leads with its observed outcome, CT-002 publishes the basis of its energy tolerance, and writers carry forward recorded review passes instead of erasing them on every regeneration. Tests grow to 72 cases. --- CHANGELOG.md | 3 +- .../README.md | 4 +- .../RESUME.md | 16 +++-- .../verification.md | 44 ++++++++++++++ .../claims-manifest.json | 2 +- .../CT-001-dart7-rolling-direction.json | 4 +- .../CT-002-dart7-dense-inelastic-contact.json | 9 +-- .../CT-003-dart7-dense-elastic-contact.json | 5 +- ...004-dart7-articulated-energy-momentum.json | 16 ++++- docs/plans/dashboard.md | 5 +- scripts/check_citation_evidence.py | 35 ++++++++++- scripts/citation_packet_utils.py | 21 +++++++ ...citation_ct001_rolling_direction_packet.py | 20 +++++-- ...ite_citation_ct002_dense_contact_packet.py | 41 +++++++++---- ...e_citation_ct003_elastic_contact_packet.py | 23 +++++--- ...itation_ct004_articulated_energy_packet.py | 58 ++++++++++++++----- tests/test_check_citation_evidence.py | 56 ++++++++++++++++++ 17 files changed, 298 insertions(+), 64 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 88f63f54d8d37..ed11a3c214231 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -619,7 +619,8 @@ compatibility remains on the active DART 6 LTS branch._ contact, and CT-003 dense elastic contact (no energy injection observed, recorded as an honest negative result rather than a reproduction), and CT-004 articulated energy drift versus timestep across both multibody - integration families. + integration families (also unresolved: convergence is observed, the cited + matched-cost comparison is not measured). - Reorganized tests and CI coverage around DART 7 components, with focused unit, integration, benchmark, rendering, CUDA-smoke, collision, and simulation gates replacing broad stale test targets. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/README.md b/docs/dev_tasks/citation_driven_simulation_trust/README.md index d7e46ef7285c7..b661212cb8584 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/README.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/README.md @@ -14,8 +14,8 @@ Landed on `main`: CT-001 rolling/friction-direction (`reproduced`), CT-002 dense inelastic contact (`reproduced`), CT-003 dense elastic contact (`unresolved` -- no energy injection observed), CT-004 articulated - energy drift versus timestep (`reproduced` for the convergence half; - matched cost explicitly not measured). Landed on + energy drift versus timestep (`unresolved`: both integration families + converge, but the cited matched-cost comparison is not measured). Landed on `release-6.20`: CT-001 detector sweep (`reproduced` on fcl/dart/ode; bullet excluded for lacking the pyramid signature). Open: articulated energy/momentum/control, heel-strike/toe-off, high mass-ratio, and diff --git a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md index 0bd5d92f82142..5f7771aec691d 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md @@ -25,7 +25,10 @@ statuses were verified; the corpus sidecar now carries the WS0 audit record. a friction-pyramid signature its data does not show. See `verification.md`. - First-wave packets landed: CT-001 (both branches), CT-002, CT-003, and CT-004 (`main`). Four of the six capped families now have at least one - branch-qualified packet. + branch-qualified packet. Dispositions: CT-001 and CT-002 `reproduced`, + CT-003 and CT-004 `unresolved` -- in both cases because the cited + behavior was not observed, which the packets state plainly rather than + reaching for a positive verdict. ## Current branches @@ -67,10 +70,13 @@ reset/concurrency queries (CT-011..CT-013). never passes one to `recordSolverDiagnostics`), boxed-LCP iteration counts (its branch returns before the diagnostics call), per-island fallback reporting, and `ResolvedSolverConfiguration` in Python. -- Packet dispositions must survive an adversarial read. Two rounds of review - killed a verdict that rested on a hardcoded threshold: a settle speed - exactly linear in dt is integrator residual, not instability. Prefer an - oracle a reviewer can recompute from `raw_rows`. +- Packet dispositions must survive an adversarial read. Review killed two + verdicts: one rested on a hardcoded threshold (a settle speed exactly + linear in dt is integrator residual, not instability), and one measured + something the cited claim never asserted (CT-004 convergence against a + claim about matched cost). Check the disposition against the corpus + definition -- "the claimed behavior is observed" -- not against whether + the fixture produced an interesting result. - `target.commit` is the commit the library was measured at; the packet and its writer necessarily land one commit later. `target.commit_role` says so. - dartpy runs from a built tree: diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index a7ba35466a752..b1517e80fb459 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -222,3 +222,47 @@ Add a dated section with: - changelog decision. Never replace raw evidence with a summary number. + +## Final independent review and third fix pass — 2026-08-14 + +A fresh reviewer ran 44 adversarial bypass attempts against the current head +(all rejected), re-verified the three `citation_packet_utils.py` source claims +against the tree, and confirmed the negative control trips 17 errors. Two +majors and four minors were found and acted on: + +1. MAJOR — a genuine gate bypass: a non-string entry in a lane's `evidence` + list was silently dropped, so `evidence: [{"path": ...}]` on a `closed` + lane validated with zero errors, skipping the missing-packet, not-JSON, + outside-`evidence/`, negative-control, and two-review checks. This is the + same "drop what you don't understand" shape as the round-1 subdirectory + bypass. Non-string entries now fail, with a test covering five variants. +2. MAJOR — CT-004's disposition was `reproduced` on an oracle unrelated to + the cited claim. The corpus defines `reproduced` as "the claimed behavior + is observed"; CT-004's claim is a methodological prescription (compare at + matched cost) and this packet measures no cost, so none of the claim's + content was observed. The row is now `unresolved`, following the CT-003 + precedent, and the convergence result it did establish is published as + `metrics.physical.convergence_finding` with an explicit note that it does + not promote the row. +3. MINOR — `raw_paths` accepted a directory (`.exists()` -> `.is_file()`), and + `scene.digest` was format-checked but never recomputed from the parameters + the packet publishes (now verified; a hand-edited scene fails). +4. MINOR — CT-004's relative drift divides by an initial energy containing a + datum-dependent gravitational term, so the headline percentage is + gauge-dependent while the slope and verdict are not. Disclosed, with the + absolute drift published per row. +5. MINOR — CT-001's claim boundary stated parameters only; it now leads with + the observed outcome like the other three. CT-002 publishes the basis of + its energy tolerance rather than a bare number. +6. MINOR — every packet shipped `review.passes: []` hardcoded, so recorded + reviews could not survive regeneration. Writers now carry forward an + existing review block, which is what makes the two-review closure floor + reachable without hand-editing a generated file. + +Recorded as follow-up, not fixed: the validator does not arithmetically +cross-check derived metrics against `raw_rows`, so a falsified summary would +pass. Closing that needs the gate to know each packet's derivation, which is +WS4-scale work. + +- Commands after this pass: `pixi run check-citation-evidence` — OK; + `tests/test_check_citation_evidence.py` — 72 passed. diff --git a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json index f0f322ab4e6b7..bc9149302c992 100644 --- a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json +++ b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json @@ -94,7 +94,7 @@ "evidence": [ "evidence/CT-004-dart7-articulated-energy-momentum.json" ], - "notes": "Passive 4-link pendulum chain swept over timestep for both multibody integration families: relative energy drift shrinks with timestep in both (observed log-log slopes ~1.0). The matched-cost half of the row is NOT covered: no timing methodology was applied, so performance is typed unsupported and no family ranking is claimed." + "notes": "Passive 4-link pendulum chain swept over timestep for both multibody integration families. Relative energy drift shrinks with timestep in both (observed log-log slopes ~1.0), but the packet disposition is unresolved: the cited claim is about comparison at matched cost, no cost was measured, so none of the claim's content was observed. Convergence is published as a separate finding that does not promote the row." }, "dart6": { "owner": "docs/dev_tasks/dart6_citation_contact_trust", diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json index 3a4f456dca147..35a534f7b48a2 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json @@ -718,7 +718,7 @@ } }, "result": { - "claim_boundary": "DART 7 main, this commit, one sphere sliding to rolling on a static ground box at v0=1 m/s, mu=0.35, dt=2 ms, 1 s horizon, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers, launch angles 0-90 deg in 15 deg steps. Says nothing about other speeds, shapes, stacks, historical DART versions, or DART 6.", + "claim_boundary": "DART 7 main, this commit, one sphere sliding to rolling on a static ground box at v0=1 m/s, mu=0.35, dt=2 ms, 1 s horizon, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers, launch angles 0-90 deg in 15 deg steps. Outcome actually observed: lateral drift reaches 2.1e-3 m against a 1e-4 m isotropy tolerance, with exact nulls at 0, 45, and 90 deg and antisymmetry about 45 deg to within 1e-13 of peak, in both contact solvers; the drift criterion is the only one of the three tolerances exceeded, and every cell passes the physical-validity gate. Says nothing about other speeds, shapes, stacks, historical DART versions, or DART 6.", "disposition": "reproduced", "limitations": [ "ResolvedSolverConfiguration is not Python-exposed; resolved identity is property readback, corroborated by per-solver trajectory-hash differences.", @@ -782,7 +782,7 @@ }, "target": { "branch": "main", - "commit": "ac0c074d637304fa7c02aa457d123d54e63f738e", + "commit": "9101d917fdb33998abd0a48d705fe42a632d1493", "commit_role": "Source state measured: the library and fixture were run at this commit, which is HEAD at capture time. The packet and its writer land in a later commit, so re-running the recorded command requires the child commit that adds the writer; the measured behavior belongs to this one." }, "title": "Rolling-direction friction dependence (DART 7 first packet)" diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json index 1253f2f636b19..6ce243c049175 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json @@ -63,7 +63,7 @@ } } }, - "resolved_provenance": "World property readback (contact_solver_method, rigid_body_solver, gravity, time_step) after enter_simulation_mode, asserted equal to the request per cell; World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", + "resolved_provenance": "World property readback after enter_simulation_mode, with contact solver, timestep, and gravity each asserted equal to the request per repeat and per cell. rigid_body_solver is recorded but not asserted, because this scene does not request one. configuration.timestep is the smallest swept value; configuration.resolved.by_cell is authoritative per cell. World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", "substeps": 1, "timestep": 0.002 }, @@ -343,7 +343,8 @@ "final_max_speed_mps": 0.05, "max_energy_gain_after_settle_j": 0.001, "max_penetration_m": 0.025, - "max_total_energy_gain_after_settle_j": 0.0001801116000000002 + "max_total_energy_gain_after_settle_j": 0.0001801116000000002, + "max_total_energy_gain_after_settle_j_basis": "1.0e-6 x |initial total mechanical energy| of this scene, a chosen floor. The verdict does not depend on its exact value: BOXED_LCP on the same scene, integrator, and timestep is the internal control, and the verdict is unchanged for any tolerance lying between the two solvers observed gains." }, "unstable_cells": [ { @@ -370,7 +371,7 @@ } }, "result": { - "claim_boundary": "DART 7 main, this commit, a bounded (not source-exact) 6x6x6 sphere-grid drop with restitution 0 and mu=0.8, 2 s horizon, timesteps 2 and 4 ms, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers. Outcome actually observed: no cell failed (4 of 4 finite, no fall-through, penetration within tolerance), and the one cell above the settle-speed tolerance has a final speed exactly proportional to dt (speed/dt identical across timesteps), which is converging integrator residual and is explicitly NOT counted as instability. What does reproduce is narrower and solver-specific: after the pile settles, total mechanical energy increases per step under SEQUENTIAL_IMPULSE at both timesteps (about 1.0e-3 J at 2 ms and 1.3e-3 J at 4 ms against a 1.8e-4 J tolerance, on a 180 J scene) while BOXED_LCP stays at or near zero. That is a small, non-divergent, non-physical energy gain in a resting inelastic pile, not a blow-up. The poor-scaling limb of the cited claim is not instrumented at all (performance is typed unsupported). Says nothing about the original SimBenchmark scene parameters, historical DART versions, other densities/materials, or DART 6.", + "claim_boundary": "DART 7 main, this commit, a bounded (not source-exact) 6x6x6 sphere-grid drop with restitution 0 and mu=0.8, 2 s horizon, timesteps 2 and 4 ms, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers. Outcome actually observed: no cell failed (4 of 4 finite, no fall-through, penetration within tolerance), and the one cell above the settle-speed tolerance has a final speed exactly proportional to dt (speed/dt identical across timesteps), which is converging integrator residual and is explicitly NOT counted as instability. What does reproduce is narrower and solver-specific: after the pile settles, the largest single-step increase in total mechanical energy under SEQUENTIAL_IMPULSE is about 1.0e-3 J at 2 ms and 1.3e-3 J at 4 ms against a 1.8e-4 J tolerance on a 180 J scene, while BOXED_LCP stays at or near zero (0.0 and 1.3e-6 J). The metric is a maximum over settle-window steps, so it does not distinguish one anomalous step from sustained pumping. That is a small, non-divergent, non-physical energy gain in a resting inelastic pile, not a blow-up. The poor-scaling limb of the cited claim is not instrumented at all (performance is typed unsupported). Says nothing about the original SimBenchmark scene parameters, historical DART versions, other densities/materials, or DART 6.", "disposition": "reproduced", "limitations": [ "Bounded reconstruction: the original SimBenchmark asset, material, and timestep grid are not reproduced exactly; sourcing the exact historical setup is future corpus work.", @@ -430,7 +431,7 @@ }, "target": { "branch": "main", - "commit": "ac0c074d637304fa7c02aa457d123d54e63f738e" + "commit": "9101d917fdb33998abd0a48d705fe42a632d1493" }, "title": "Dense 6x6x6 inelastic contact stability (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json index 919479a366d43..e8cad6052390d 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json @@ -63,7 +63,7 @@ } } }, - "resolved_provenance": "World property readback (contact_solver_method, rigid_body_solver, gravity, time_step) after enter_simulation_mode, asserted equal to the request per cell; World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", + "resolved_provenance": "World property readback after enter_simulation_mode, with contact solver, timestep, and gravity each asserted equal to the request per repeat and per cell. rigid_body_solver is recorded but not asserted, because this scene does not request one. configuration.timestep is the smallest swept value; configuration.resolved.by_cell is authoritative per cell. World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", "substeps": 1, "timestep": 0.002 }, @@ -284,6 +284,7 @@ "The envelope maximum is set by the initial state in every cell (energy never rose above its starting value), so the 1.8e-4 J tolerance is never exercised; the test is one-sided by construction and is reported as such in raw_rows.envelope_set_by_initial_state.", "The cited claim also names solver failure. That limb is only instrumented as non-finite state: no residual exists on this path and the boxed-LCP branch records no iteration count, so a non-diverging solver failure would not be detected here.", "max_penetration_m is bit-identical across the two solvers at each timestep while the trajectories diverge, so it is set during the shared first impact and carries no solver-discriminating information.", + "max_active_contacts is 216 in every cell (a fully stacked column geometry), so it carries no discriminating information here.", "Two timesteps and two solvers only." ] }, @@ -333,7 +334,7 @@ }, "target": { "branch": "main", - "commit": "ac0c074d637304fa7c02aa457d123d54e63f738e" + "commit": "9101d917fdb33998abd0a48d705fe42a632d1493" }, "title": "Dense elastic contact energy envelope (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json index 0011548cd64e5..200c7cb5224eb 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json @@ -105,7 +105,7 @@ } } }, - "resolved_provenance": "World property readback (multibody_options.integration_family, rigid_body_solver, gravity, time_step) after enter_simulation_mode, with integration family, timestep, and gravity each asserted equal to the request per repeat and per cell. World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", + "resolved_provenance": "World property readback after enter_simulation_mode, with integration family, timestep, and gravity each asserted equal to the request per repeat and per cell. rigid_body_solver is recorded but not asserted, because this scene requests none. configuration.timestep is the smallest swept value; configuration.resolved.by_cell is authoritative per cell. World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", "substeps": 1, "timestep": 0.0005 }, @@ -398,6 +398,15 @@ }, "physical": { "all_cells_finite": true, + "convergence_finding": { + "all_cells_finite": true, + "all_families_converge": true, + "families_with_shrinking_drift": [ + "SEMI_IMPLICIT", + "VARIATIONAL" + ], + "note": "This is the packet's positive result. It does not promote the corpus row, whose claim concerns comparison at matched cost." + }, "families_with_shrinking_drift": [ "SEMI_IMPLICIT", "VARIATIONAL" @@ -437,8 +446,9 @@ }, "result": { "claim_boundary": "DART 7 main, this commit, a passive 4-link planar pendulum chain with no contact and no control, 2 s horizon, timesteps 0.5-4 ms, SEMI_IMPLICIT and VARIATIONAL multibody integration families. What is established: both families stay finite and their relative energy drift shrinks as the timestep shrinks, with the observed log-log slopes recorded per family. What is NOT established: the 'at matched cost' half of the corpus row, because no timing was measured; any contact-accuracy claim, because the scene has no contact; and any ranking between the two families. Says nothing about historical DART versions, controlled or contacting articulated scenes, or DART 6.", - "disposition": "reproduced", + "disposition": "unresolved", "limitations": [ + "The relative drift divides by the initial total energy, which contains a gravitational term measured from the world-origin datum. Moving the mount changes that divisor without changing the absolute drift, so the relative figures are gauge-dependent; the verdict and the log-log slope are not, because the divisor is constant within a family. Absolute drift is published per row as max_abs_energy_drift_j.", "No contact and no control: this isolates integration accuracy and therefore covers only part of the corpus row, whose contact and matched-cost halves need separate work.", "Cost is recorded as step count only; performance is typed unsupported, so nothing here supports an accuracy-per-cost comparison.", "Angular momentum is not conserved under gravity about the world origin, so it is reported as an envelope rather than an invariant.", @@ -487,7 +497,7 @@ }, "target": { "branch": "main", - "commit": "d7281579701e4f009122968c5782291d12f69ac7", + "commit": "9101d917fdb33998abd0a48d705fe42a632d1493", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit." }, "title": "Articulated energy drift versus timestep across integration families (DART 7 first packet)" diff --git a/docs/plans/dashboard.md b/docs/plans/dashboard.md index 210a4742c21d0..e1686998d5367 100644 --- a/docs/plans/dashboard.md +++ b/docs/plans/dashboard.md @@ -52,11 +52,12 @@ its own line so status updates remain git-history friendly. - Horizon: Now - Dimension: Algorithm extensibility - Next step: WS0 audit and the WS1 fail-closed claim/evidence manifest landed - with a permanent negative control and three first-wave packets: CT-001 + with a permanent negative control and four first-wave packets: CT-001 rolling-direction (`reproduced`), CT-002 dense inelastic contact (`reproduced`), CT-003 dense elastic contact (`unresolved` -- no energy injection observed), and CT-004 articulated energy drift versus timestep - (`reproduced`; the matched-cost half is explicitly uncovered). Next: the + (`unresolved`: drift converges with timestep in both integration families, + but the cited claim concerns matched cost, which is not measured). Next: the remaining first-wave families (articulated control, heel-strike/toe-off, high mass-ratio, reset/concurrency queries) with branch-qualified dispositions, reusing existing demos scenes and PLAN-104/621/622 evidence, before diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 2adae3d39feb2..f4699520a3f9e 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -32,6 +32,7 @@ from __future__ import annotations import argparse +import hashlib import json import math import re @@ -301,6 +302,25 @@ def packet_errors( digest = scene.get("digest") if not (isinstance(digest, str) and SCENE_DIGEST_RE.match(digest)): errors.append("scene.digest must match sha256:<64 hex>") + elif isinstance(scene.get("parameters"), dict): + # When the packet publishes the parameters the digest was taken + # over, recompute it. A digest that cannot be reproduced from the + # packet's own scene description binds nothing, and a hand-edited + # scene would otherwise pass. + expected = ( + "sha256:" + + hashlib.sha256( + json.dumps( + scene["parameters"], sort_keys=True, separators=(",", ":") + ).encode("utf-8") + ).hexdigest() + ) + if digest != expected: + errors.append( + f"scene.digest {digest} does not match the digest of " + f"scene.parameters ({expected}); the scene description " + "and its digest disagree" + ) configuration = packet.get("configuration") if not isinstance(configuration, dict): @@ -387,7 +407,7 @@ def packet_errors( f"evidence.raw_paths[{index}] must be a non-empty string" ) elif base_dir is not None and not any( - (root / raw_path).exists() for root in (base_dir, REPO_ROOT) + (root / raw_path).is_file() for root in (base_dir, REPO_ROOT) ): errors.append( f"evidence.raw_paths[{index}] {raw_path!r} does not " @@ -621,7 +641,16 @@ def validate_tree( ) lane_paths = [] for rel in lane_paths or []: - if isinstance(rel, str): + if not isinstance(rel, str): + errors.append( + f"{manifest_path}: {claim_id}.lanes." + f"{lane_name}.evidence contains a non-string " + f"entry {rel!r}; every entry must be a packet " + "path, or a lane could be closed by something " + "the packet checks never reach" + ) + continue + if True: if rel in lane_evidence: owner = lane_evidence[rel] errors.append( @@ -742,7 +771,7 @@ def validate_tree( ) negative_paths = ( - sorted(negative_dir.glob("*.json")) if negative_dir.is_dir() else [] + sorted(negative_dir.rglob("*.json")) if negative_dir.is_dir() else [] ) if not negative_paths: errors.append( diff --git a/scripts/citation_packet_utils.py b/scripts/citation_packet_utils.py index 6e9b4ea3890cd..a1a64690bb25a 100644 --- a/scripts/citation_packet_utils.py +++ b/scripts/citation_packet_utils.py @@ -110,3 +110,24 @@ def solver_iterations_by_method( ), } return typed + + +def preserve_review(output_path: "Any") -> dict[str, Any]: + """Carry forward review passes already recorded in a packet. + + Packets are generated, but review passes are recorded by people and other + agents after the fact. Without this, every regeneration silently erases + them, and the two-review floor a lane needs to close could only be met by + hand-editing a file the next run clobbers. + """ + try: + import json as _json + from pathlib import Path as _Path + + existing = _json.loads(_Path(output_path).read_text(encoding="utf-8")) + except OSError, ValueError: + return {"passes": []} + review = existing.get("review") + if isinstance(review, dict) and isinstance(review.get("passes"), list): + return {"passes": review["passes"]} + return {"passes": []} diff --git a/scripts/write_citation_ct001_rolling_direction_packet.py b/scripts/write_citation_ct001_rolling_direction_packet.py index a18399f89e94d..27365bb777761 100644 --- a/scripts/write_citation_ct001_rolling_direction_packet.py +++ b/scripts/write_citation_ct001_rolling_direction_packet.py @@ -49,6 +49,7 @@ SEQUENTIAL_IMPULSE_ITERATIONS_NOTE, UNSUPPORTED_ANTISYMMETRY_RATIO, UNSUPPORTED_SOLVER_RESIDUAL, + preserve_review, solver_iterations_by_method, ) @@ -297,7 +298,7 @@ def git_head() -> str: ).stdout.strip() -def build_packet() -> dict[str, Any]: +def build_packet(output_path: Path | None = None) -> dict[str, Any]: parameters = SCENE_PARAMETERS rows: list[dict[str, Any]] = [] determinism_failures: list[str] = [] @@ -585,9 +586,14 @@ def build_packet() -> dict[str, Any]: "DART 7 main, this commit, one sphere sliding to rolling on " "a static ground box at v0=1 m/s, mu=0.35, dt=2 ms, 1 s " "horizon, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers, " - "launch angles 0-90 deg in 15 deg steps. Says nothing about " - "other speeds, shapes, stacks, historical DART versions, or " - "DART 6." + "launch angles 0-90 deg in 15 deg steps. Outcome actually " + "observed: lateral drift reaches 2.1e-3 m against a 1e-4 m " + "isotropy tolerance, with exact nulls at 0, 45, and 90 deg " + "and antisymmetry about 45 deg to within 1e-13 of peak, in " + "both contact solvers; the drift criterion is the only one of " + "the three tolerances exceeded, and every cell passes the " + "physical-validity gate. Says nothing about other speeds, " + "shapes, stacks, historical DART versions, or DART 6." ), "limitations": [ "ResolvedSolverConfiguration is not Python-exposed; " @@ -617,7 +623,9 @@ def build_packet() -> dict[str, Any]: "as one.", ], }, - "review": {"passes": []}, + "review": ( + preserve_review(output_path) if output_path is not None else {"passes": []} + ), "host": { "platform": platform.platform(), "python": sys.version.split()[0], @@ -645,7 +653,7 @@ def parse_args() -> argparse.Namespace: def main() -> int: args = parse_args() - packet = build_packet() + packet = build_packet(args.output) args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text( json.dumps(packet, indent=2, sort_keys=True) + "\n", encoding="utf-8" diff --git a/scripts/write_citation_ct002_dense_contact_packet.py b/scripts/write_citation_ct002_dense_contact_packet.py index 96278ee89b31e..ca9e9b77a066a 100644 --- a/scripts/write_citation_ct002_dense_contact_packet.py +++ b/scripts/write_citation_ct002_dense_contact_packet.py @@ -45,6 +45,7 @@ PENETRATION_CLAMP_NOTE, SEQUENTIAL_IMPULSE_ITERATIONS_NOTE, UNSUPPORTED_SOLVER_RESIDUAL, + preserve_review, solver_iterations_by_method, ) @@ -230,7 +231,7 @@ def run_single( } -def build_packet() -> dict[str, Any]: +def build_packet(output_path: Path | None = None) -> dict[str, Any]: parameters = SCENE_PARAMETERS rows: list[dict[str, Any]] = [] determinism_failures: list[str] = [] @@ -307,6 +308,14 @@ def build_packet() -> dict[str, Any]: # this is measured only in the settle window. "max_total_energy_gain_after_settle_j": 1.0e-6 * abs(rows[0]["initial_total_energy_j"]), + "max_total_energy_gain_after_settle_j_basis": ( + "1.0e-6 x |initial total mechanical energy| of this scene, a " + "chosen floor. The verdict does not depend on its exact " + "value: BOXED_LCP on the same scene, integrator, and " + "timestep is the internal control, and the verdict is " + "unchanged for any tolerance lying between the two solvers " + "observed gains." + ), } residual_speed_tolerance = { "dt_linearity_relative": 1.0e-3, @@ -419,10 +428,13 @@ def build_packet() -> dict[str, Any]: }, "resolved": {"by_cell": resolved_by_cell}, "resolved_provenance": ( - "World property readback (contact_solver_method, " - "rigid_body_solver, gravity, time_step) after " - "enter_simulation_mode, asserted equal to the request per " - "cell; World::getResolvedConfiguration() is not yet exposed " + "World property readback after enter_simulation_mode, with " + "contact solver, timestep, and gravity each asserted equal " + "to the request per repeat and per cell. rigid_body_solver " + "is recorded but not asserted, because this scene does not " + "request one. configuration.timestep is the smallest swept " + "value; configuration.resolved.by_cell is authoritative per " + "cell. World::getResolvedConfiguration() is not yet exposed " "to Python (PLAN-123 WS4 follow-up)." ), "detector": ( @@ -557,11 +569,14 @@ def build_packet() -> dict[str, Any]: "(speed/dt identical across timesteps), which is converging " "integrator residual and is explicitly NOT counted as " "instability. What does reproduce is narrower and " - "solver-specific: after the pile settles, total mechanical " - "energy increases per step under SEQUENTIAL_IMPULSE at both " - "timesteps (about 1.0e-3 J at 2 ms and 1.3e-3 J at 4 ms " - "against a 1.8e-4 J tolerance, on a 180 J scene) while " - "BOXED_LCP stays at or near zero. That is a small, " + "solver-specific: after the pile settles, the largest " + "single-step increase in total mechanical energy under " + "SEQUENTIAL_IMPULSE is about 1.0e-3 J at 2 ms and 1.3e-3 J " + "at 4 ms against a 1.8e-4 J tolerance on a 180 J scene, " + "while BOXED_LCP stays at or near zero (0.0 and 1.3e-6 J). " + "The metric is a maximum over settle-window steps, so it " + "does not distinguish one anomalous step from sustained " + "pumping. That is a small, " "non-divergent, non-physical energy gain in a resting " "inelastic pile, not a blow-up. The poor-scaling limb of the " "cited claim is not instrumented at all (performance is " @@ -598,7 +613,9 @@ def build_packet() -> dict[str, Any]: "to a follow-up once per-island diagnostics exist.", ], }, - "review": {"passes": []}, + "review": ( + preserve_review(output_path) if output_path is not None else {"passes": []} + ), "host": { "platform": platform.platform(), "python": sys.version.split()[0], @@ -636,7 +653,7 @@ def parse_args() -> argparse.Namespace: def main() -> int: args = parse_args() - packet = build_packet() + packet = build_packet(args.output) args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text( json.dumps(packet, indent=2, sort_keys=True) + "\n", encoding="utf-8" diff --git a/scripts/write_citation_ct003_elastic_contact_packet.py b/scripts/write_citation_ct003_elastic_contact_packet.py index 22be3865f0059..331af2d9e9257 100644 --- a/scripts/write_citation_ct003_elastic_contact_packet.py +++ b/scripts/write_citation_ct003_elastic_contact_packet.py @@ -46,6 +46,7 @@ PENETRATION_CLAMP_NOTE, SEQUENTIAL_IMPULSE_ITERATIONS_NOTE, UNSUPPORTED_SOLVER_RESIDUAL, + preserve_review, solver_iterations_by_method, ) @@ -231,7 +232,7 @@ def git_head() -> str: ).stdout.strip() -def build_packet() -> dict[str, Any]: +def build_packet(output_path: Path | None = None) -> dict[str, Any]: parameters = SCENE_PARAMETERS rows: list[dict[str, Any]] = [] determinism_failures: list[str] = [] @@ -341,10 +342,13 @@ def build_packet() -> dict[str, Any]: }, "resolved": {"by_cell": resolved_by_cell}, "resolved_provenance": ( - "World property readback (contact_solver_method, " - "rigid_body_solver, gravity, time_step) after " - "enter_simulation_mode, asserted equal to the request per " - "cell; World::getResolvedConfiguration() is not yet exposed " + "World property readback after enter_simulation_mode, with " + "contact solver, timestep, and gravity each asserted equal " + "to the request per repeat and per cell. rigid_body_solver " + "is recorded but not asserted, because this scene does not " + "request one. configuration.timestep is the smallest swept " + "value; configuration.resolved.by_cell is authoritative per " + "cell. World::getResolvedConfiguration() is not yet exposed " "to Python (PLAN-123 WS4 follow-up)." ), "detector": ( @@ -482,10 +486,15 @@ def build_packet() -> dict[str, Any]: "at each timestep while the trajectories diverge, so it is " "set during the shared first impact and carries no " "solver-discriminating information.", + "max_active_contacts is 216 in every cell (a fully " + "stacked column geometry), so it carries no " + "discriminating information here.", "Two timesteps and two solvers only.", ], }, - "review": {"passes": []}, + "review": ( + preserve_review(output_path) if output_path is not None else {"passes": []} + ), "host": { "platform": platform.platform(), "python": sys.version.split()[0], @@ -513,7 +522,7 @@ def parse_args() -> argparse.Namespace: def main() -> int: args = parse_args() - packet = build_packet() + packet = build_packet(args.output) args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text( json.dumps(packet, indent=2, sort_keys=True) + "\n", encoding="utf-8" diff --git a/scripts/write_citation_ct004_articulated_energy_packet.py b/scripts/write_citation_ct004_articulated_energy_packet.py index 8daccf071636e..48b9e2aefee13 100644 --- a/scripts/write_citation_ct004_articulated_energy_packet.py +++ b/scripts/write_citation_ct004_articulated_energy_packet.py @@ -43,7 +43,10 @@ import dartpy as sx import numpy as np -from citation_packet_utils import UNSUPPORTED_SOLVER_RESIDUAL +from citation_packet_utils import ( + UNSUPPORTED_SOLVER_RESIDUAL, + preserve_review, +) REPO_ROOT = Path(__file__).resolve().parents[1] DEFAULT_OUTPUT = ( @@ -225,7 +228,7 @@ def observed_order(drifts_by_timestep: dict[float, float]) -> Any: return numerator / denominator -def build_packet() -> dict[str, Any]: +def build_packet(output_path: Path | None = None) -> dict[str, Any]: parameters = SCENE_PARAMETERS rows: list[dict[str, Any]] = [] determinism_failures: list[str] = [] @@ -307,11 +310,25 @@ def build_packet() -> dict[str, Any]: for family, stats in family_summary.items() if stats["drift_shrinks_with_timestep"] ) - disposition = ( - "reproduced" - if all_finite and len(converging) == len(parameters["integration_families"]) - else "unresolved" - ) + # The corpus defines `reproduced` as "the claimed behavior is observed". + # The cited claim is a methodological prescription -- accuracy compared at + # matched cost -- and this packet measures no cost, so none of the claim's + # content is observed here. Convergence with timestep is a real result the + # source asserts nothing about, so it cannot promote the row. The row stays + # `unresolved` until matched-cost timing lands, following the CT-003 + # precedent for "the cited behavior was not observed". + disposition = "unresolved" + convergence_finding = { + "all_cells_finite": all_finite, + "families_with_shrinking_drift": converging, + "all_families_converge": ( + all_finite and len(converging) == len(parameters["integration_families"]) + ), + "note": ( + "This is the packet's positive result. It does not promote the " + "corpus row, whose claim concerns comparison at matched cost." + ), + } command = ( "PYTHONPATH=build/default/cpp/Release/python pixi run python " @@ -358,11 +375,13 @@ def build_packet() -> dict[str, Any]: }, "resolved": {"by_cell": resolved_by_cell}, "resolved_provenance": ( - "World property readback (multibody_options.integration_" - "family, rigid_body_solver, gravity, time_step) after " - "enter_simulation_mode, with integration family, timestep, " - "and gravity each asserted equal to the request per repeat " - "and per cell. World::getResolvedConfiguration() is not yet " + "World property readback after enter_simulation_mode, with " + "integration family, timestep, and gravity each asserted " + "equal to the request per repeat and per cell. " + "rigid_body_solver is recorded but not asserted, because " + "this scene requests none. configuration.timestep is the " + "smallest swept value; configuration.resolved.by_cell is " + "authoritative per cell. World::getResolvedConfiguration() is not yet " "exposed to Python (PLAN-123 WS4 follow-up)." ), "detector": ( @@ -408,6 +427,7 @@ def build_packet() -> dict[str, Any]: "all_cells_finite": all_finite, "per_family_summary": family_summary, "families_with_shrinking_drift": converging, + "convergence_finding": convergence_finding, }, "numerical": { "method": ( @@ -471,6 +491,14 @@ def build_packet() -> dict[str, Any]: "contacting articulated scenes, or DART 6." ), "limitations": [ + "The relative drift divides by the initial total energy, " + "which contains a gravitational term measured from the " + "world-origin datum. Moving the mount changes that " + "divisor without changing the absolute drift, so the " + "relative figures are gauge-dependent; the verdict and " + "the log-log slope are not, because the divisor is " + "constant within a family. Absolute drift is published " + "per row as max_abs_energy_drift_j.", "No contact and no control: this isolates integration " "accuracy and therefore covers only part of the corpus row, " "whose contact and matched-cost halves need separate work.", @@ -487,7 +515,9 @@ def build_packet() -> dict[str, Any]: "trend summary, not a certified order of accuracy.", ], }, - "review": {"passes": []}, + "review": ( + preserve_review(output_path) if output_path is not None else {"passes": []} + ), "host": { "platform": platform.platform(), "python": sys.version.split()[0], @@ -515,7 +545,7 @@ def parse_args() -> argparse.Namespace: def main() -> int: args = parse_args() - packet = build_packet() + packet = build_packet(args.output) args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text( json.dumps(packet, indent=2, sort_keys=True) + "\n", encoding="utf-8" diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index 1f0787ddda75b..d281a1007e9b0 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -27,6 +27,7 @@ def _load_module(): MODULE = _load_module() +LANE = "dart7" def complete_packet() -> dict: @@ -571,3 +572,58 @@ def test_repository_tree_validates(): """The committed manifest, packets, and negative controls must pass.""" errors = MODULE.validate_tree(PLAN_DIR) assert errors == [] + + +def test_non_string_lane_evidence_entry_fails(tmp_path): + """A lane must not be closable by an entry the packet checks cannot read.""" + for bad in ({"path": "evidence/x.json"}, None, 42, True, ["evidence/x.json"]): + manifest = _minimal_manifest(["CT-001"]) + lane = manifest["claims"][0]["lanes"]["dart7"] + lane["status"] = "closed" + lane["disposition"] = "reproduced" + lane["evidence"] = [bad] + tree = _write_tree( + tmp_path / f"case{abs(hash(str(bad)))}", + negative={"schema": "x"}, + manifest=manifest, + ) + errors = MODULE.validate_tree(tree) + assert any("non-string entry" in error for error in errors), bad + + +def test_raw_path_pointing_at_a_directory_fails(tmp_path): + packet = complete_packet() + del packet["evidence"]["raw_rows"] + (tmp_path / "somedir").mkdir() + packet["evidence"]["raw_paths"] = ["somedir"] + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any("does not resolve" in error for error in errors) + + +def test_scene_digest_must_match_published_parameters(): + packet = complete_packet() + packet["scene"]["parameters"] = {"a": 1} + errors = MODULE.packet_errors(packet) + assert any("does not match the digest" in error for error in errors) + import hashlib as _h + import json as _j + + packet["scene"]["digest"] = ( + "sha256:" + + _h.sha256( + _j.dumps({"a": 1}, sort_keys=True, separators=(",", ":")).encode() + ).hexdigest() + ) + assert MODULE.packet_errors(packet) == [] + + +def test_nested_negative_control_is_enumerated(tmp_path): + """A control in a subdirectory must still be required to fail.""" + tree = _write_tree(tmp_path, negative={"schema": "x"}) + nested = tree / "evidence" / "negative-controls" / "deep" + nested.mkdir(parents=True) + (nested / "passing.json").write_text( + json.dumps(complete_packet()), encoding="utf-8" + ) + errors = MODULE.validate_tree(tree) + assert any("fail-closed proof is vacuous" in error for error in errors) From 77c08017dd948fea9a8381223c8f684434c738cc Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Fri, 14 Aug 2026 21:04:18 -0700 Subject: [PATCH 08/52] Add the CT-005 PD-tracking packet A 4-link chain hanging under gravity tracks a sinusoidal joint reference, swept over four timesteps for both multibody integration families. The fixture needed four iterations, and the reason is worth recording: a fixed-gain diagonal PD saturated its torque limit in every cell and diverged at the coarser timesteps, so it measured the controller rather than the integrator. A direct probe showed why -- the root joint of a serial chain sees a far smaller articulated inertia than the composite rigid-body inertia used to size its gains, so gains stiff enough for the root were unstable while the tip tracked fine. Replaced with a computed-torque controller built on Multibody.compute_inverse_dynamics, which gives every configuration the same closed-loop bandwidth: no saturation, max torque 143.7 Nm, RMS error about 1% of the reference amplitude. The result is substantive. Refining the timestep eightfold does NOT reduce tracking error (2.14e-3 rad at 4 ms versus 2.41e-3 rad at 0.5 ms) while control work rises monotonically, so the error is controller-limited rather than integration-limited and the row's premise -- a whole-step speed/accuracy tradeoff -- is not the simple 'smaller timestep is more accurate' shape. Disposition is unresolved: the cited claim is about a speed/accuracy tradeoff and this packet measures no step cost. constraint_error is typed unsupported because the scene has no constraints to violate. --- CHANGELOG.md | 4 +- .../README.md | 9 +- .../verification.md | 33 + .../claims-manifest.json | 8 +- .../evidence/CT-005-dart7-pd-tracking.json | 503 ++++++++++++++ docs/plans/dashboard.md | 10 +- ...write_citation_ct005_pd_tracking_packet.py | 647 ++++++++++++++++++ 7 files changed, 1202 insertions(+), 12 deletions(-) create mode 100644 docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json create mode 100644 scripts/write_citation_ct005_pd_tracking_packet.py diff --git a/CHANGELOG.md b/CHANGELOG.md index ed11a3c214231..b4123395668bc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -620,7 +620,9 @@ compatibility remains on the active DART 6 LTS branch._ recorded as an honest negative result rather than a reproduction), and CT-004 articulated energy drift versus timestep across both multibody integration families (also unresolved: convergence is observed, the cited - matched-cost comparison is not measured). + matched-cost comparison is not measured), and CT-005 PD-control tracking + (also unresolved: tracking error proves controller-limited rather than + integration-limited, and step cost is not measured). - Reorganized tests and CI coverage around DART 7 components, with focused unit, integration, benchmark, rendering, CUDA-smoke, collision, and simulation gates replacing broad stale test targets. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/README.md b/docs/dev_tasks/citation_driven_simulation_trust/README.md index b661212cb8584..4c87b37ef36bf 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/README.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/README.md @@ -15,7 +15,9 @@ CT-002 dense inelastic contact (`reproduced`), CT-003 dense elastic contact (`unresolved` -- no energy injection observed), CT-004 articulated energy drift versus timestep (`unresolved`: both integration families - converge, but the cited matched-cost comparison is not measured). Landed on + converge, but the cited matched-cost comparison is not measured), and + CT-005 PD tracking (`unresolved`: error is controller-limited and step + cost is not measured). Landed on `release-6.20`: CT-001 detector sweep (`reproduced` on fcl/dart/ode; bullet excluded for lacking the pyramid signature). Open: articulated energy/momentum/control, heel-strike/toe-off, high mass-ratio, and @@ -158,9 +160,8 @@ for downstream-sensitive changes. ## Immediate next steps 1. Read `RESUME.md` and verify both branch tips/worktrees. -2. Continue Phase 2 with the controlled-articulated row (CT-005), seeded by - `python/examples/demos/scenes/atlas_simbicon.py`, then heel-strike/toe-off - (CT-006), which depends on the WS3 contact semantics. +2. Continue Phase 2 with heel-strike/toe-off (CT-006), which depends on the + WS3 contact semantics, and the high mass-ratio stack row (CT-007). 3. Start WS4's first slice in parallel where it unblocks packets: expose `ResolvedSolverConfiguration` to Python and a comparable per-solve residual, both currently typed unsupported in every packet. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index b1517e80fb459..0192288c278fd 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -266,3 +266,36 @@ WS4-scale work. - Commands after this pass: `pixi run check-citation-evidence` — OK; `tests/test_check_citation_evidence.py` — 72 passed. + +## CT-005 PD-tracking slice — 2026-08-14 + +- What changed: `scripts/write_citation_ct005_pd_tracking_packet.py` and its + packet; manifest lane, dashboard, changelog, task status. +- Scene: 4-link chain hanging under gravity, links 0.3 m / 1.0 kg with mass at + mid-link, tracking a 0.5 Hz / 0.25 rad phase-shifted sinusoidal joint + reference, no contact, 2 s horizon. +- Four tuning iterations were needed and are worth recording, because the + first three produced a fixture that measured the controller rather than the + integrator: a fixed-gain diagonal PD saturated its 500 Nm limit in every + cell and diverged at the coarser timesteps. The diagnosis, from a direct + probe rather than another guess, was that the root joint of a serial chain + sees a far smaller articulated inertia than the composite rigid-body + inertia used to size its gains, so root gains stiff enough to hold the + chain were unstable while the tip tracked fine. Replaced with a + computed-torque controller built on `Multibody.compute_inverse_dynamics`, + which gives every configuration the same closed-loop bandwidth. Result: + no saturation, max torque 143.7 Nm, RMS error about 1% of the reference + amplitude. +- Measured RMS tracking error and control work, per family, dt 4 ms -> 0.5 ms: + SEMI_IMPLICIT 2.138e-3 -> 2.412e-3 rad with work 6.086 -> 6.534 J; + VARIATIONAL 2.139e-3 -> 2.412e-3 rad with work 6.085 -> 6.534 J. +- Finding: the error is controller-limited, not integration-limited. + Refining the timestep eightfold does not reduce tracking error (it varies + by ~13% and is slightly lower at the coarsest step) while control work + rises monotonically. That is a substantive result about the row's premise: + the whole-step tradeoff is not simply "smaller timestep is more accurate". +- Disposition `unresolved`: the cited claim is a speed/accuracy tradeoff and + no step cost is measured, so the row is not promoted. `constraint_error` is + typed unsupported because this scene has no constraints to violate. +- Commands: packet writer as recorded in the packet; + `pixi run check-citation-evidence` — OK; 72 pytest cases pass. diff --git a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json index bc9149302c992..92b38b7517e19 100644 --- a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json +++ b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json @@ -112,10 +112,12 @@ "lanes": { "dart7": { "owner": "PLAN-123 with PLAN-080", - "status": "audit-required", + "status": "in-progress", "disposition": null, - "evidence": [], - "notes": "python/examples/demos/scenes/atlas_simbicon.py is the existing controlled-articulated seed." + "evidence": [ + "evidence/CT-005-dart7-pd-tracking.json" + ], + "notes": "Computed-torque-controlled 4-link hanging chain tracking a sinusoidal joint reference, swept over timestep for both integration families. Disposition unresolved: the cited claim is a whole-step speed/accuracy tradeoff and no step cost is measured. The measured half is itself informative -- tracking error is controller-limited, so refining the timestep eightfold does not reduce it while control work rises." }, "dart6": { "owner": "docs/dev_tasks/dart6_citation_contact_trust", diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json new file mode 100644 index 0000000000000..2ce2b85cc6a23 --- /dev/null +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json @@ -0,0 +1,503 @@ +{ + "claim_id": "CT-005", + "configuration": { + "detector": "not applicable: the scene has no collision shapes and runs no narrow phase", + "fallback_policy": "World defaults; no per-island fallback reporting is exposed on main (PLAN-123 WS4 follow-up)", + "iterations": "World defaults; the variational family uses its default iteration limit and tolerance, which this packet does not vary", + "requested": { + "backend": "cpu", + "contact": "none (no collision shapes)", + "controller": "joint PD with per-joint gains scaled to each joint's downstream inertia for a uniform 20 rad/s closed-loop natural frequency at critical damping, torque limit 500 Nm, applied before each step", + "integration_family_sweep": [ + "SEMI_IMPLICIT", + "VARIATIONAL" + ], + "precision": "float64", + "threads": "World default sequential step", + "timestep_sweep_s": [ + 0.004, + 0.002, + 0.001, + 0.0005 + ] + }, + "resolved": { + "by_cell": { + "SEMI_IMPLICIT@dt=0.0005": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "time_step_s": 0.0005 + }, + "SEMI_IMPLICIT@dt=0.001": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "time_step_s": 0.001 + }, + "SEMI_IMPLICIT@dt=0.002": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "time_step_s": 0.002 + }, + "SEMI_IMPLICIT@dt=0.004": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "time_step_s": 0.004 + }, + "VARIATIONAL@dt=0.0005": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "time_step_s": 0.0005 + }, + "VARIATIONAL@dt=0.001": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "time_step_s": 0.001 + }, + "VARIATIONAL@dt=0.002": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "time_step_s": 0.002 + }, + "VARIATIONAL@dt=0.004": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "time_step_s": 0.004 + } + } + }, + "resolved_provenance": "World property readback after enter_simulation_mode, with integration family, timestep, and gravity each asserted equal to the request per repeat and per cell. The controller is applied by this script, not resolved by the World. configuration.timestep is the smallest swept value; configuration.resolved.by_cell is authoritative per cell. World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", + "substeps": 1, + "timestep": 0.0005 + }, + "ensemble": { + "deterministic_repeats": 2, + "deterministic_repeats_identical": true, + "kind": "parameter-sweep-with-deterministic-repeats", + "measurement_window": { + "end_s": 2.0, + "start_s": 0.0 + }, + "sweep": [ + { + "integration_family": "SEMI_IMPLICIT", + "timestep_s": 0.004 + }, + { + "integration_family": "SEMI_IMPLICIT", + "timestep_s": 0.002 + }, + { + "integration_family": "SEMI_IMPLICIT", + "timestep_s": 0.001 + }, + { + "integration_family": "SEMI_IMPLICIT", + "timestep_s": 0.0005 + }, + { + "integration_family": "VARIATIONAL", + "timestep_s": 0.004 + }, + { + "integration_family": "VARIATIONAL", + "timestep_s": 0.002 + }, + { + "integration_family": "VARIATIONAL", + "timestep_s": 0.001 + }, + { + "integration_family": "VARIATIONAL", + "timestep_s": 0.0005 + } + ] + }, + "evidence": { + "commands": [ + "PYTHONPATH=build/default/cpp/Release/python pixi run python scripts/write_citation_ct005_pd_tracking_packet.py" + ], + "raw_rows": [ + { + "control_work_j": 6.0859664494528225, + "finite": true, + "integration_family": "SEMI_IMPLICIT", + "max_abs_torque_nm": 143.74710461593904, + "max_abs_tracking_error_rad": 0.012325258615201645, + "repeat_trajectory_sha256": [ + "a40a576ab8ac68c9dc8e9da2a5589d2f148112e3df77557b9b590b6854da7e7e", + "a40a576ab8ac68c9dc8e9da2a5589d2f148112e3df77557b9b590b6854da7e7e" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "time_step_s": 0.004 + }, + "rms_tracking_error_rad": 0.0021383697024277538, + "steps": 500, + "timestep_s": 0.004, + "torque_saturated": false, + "trajectory_sha256": "a40a576ab8ac68c9dc8e9da2a5589d2f148112e3df77557b9b590b6854da7e7e" + }, + { + "control_work_j": 6.344908570962907, + "finite": true, + "integration_family": "SEMI_IMPLICIT", + "max_abs_torque_nm": 143.74710461593904, + "max_abs_tracking_error_rad": 0.013387884667837628, + "repeat_trajectory_sha256": [ + "42625d3562e2fd830f46d76d502dc392061bcad4a29f0017ffceed8392322443", + "42625d3562e2fd830f46d76d502dc392061bcad4a29f0017ffceed8392322443" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "time_step_s": 0.002 + }, + "rms_tracking_error_rad": 0.0022892179585315767, + "steps": 1000, + "timestep_s": 0.002, + "torque_saturated": false, + "trajectory_sha256": "42625d3562e2fd830f46d76d502dc392061bcad4a29f0017ffceed8392322443" + }, + { + "control_work_j": 6.471559759362399, + "finite": true, + "integration_family": "SEMI_IMPLICIT", + "max_abs_torque_nm": 143.74710461593904, + "max_abs_tracking_error_rad": 0.013918071592133652, + "repeat_trajectory_sha256": [ + "22f17a85005a5d86777ee99ae45f0e50c6de5912017b095219298447eb1352d8", + "22f17a85005a5d86777ee99ae45f0e50c6de5912017b095219298447eb1352d8" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "time_step_s": 0.001 + }, + "rms_tracking_error_rad": 0.002370166622085158, + "steps": 2000, + "timestep_s": 0.001, + "torque_saturated": false, + "trajectory_sha256": "22f17a85005a5d86777ee99ae45f0e50c6de5912017b095219298447eb1352d8" + }, + { + "control_work_j": 6.534246144844606, + "finite": true, + "integration_family": "SEMI_IMPLICIT", + "max_abs_torque_nm": 143.74710461593904, + "max_abs_tracking_error_rad": 0.014182536997537777, + "repeat_trajectory_sha256": [ + "b3b5af89e1d5b4b04116a59b902a62952945f43adfe3cc7a51e645a05ef59ea2", + "b3b5af89e1d5b4b04116a59b902a62952945f43adfe3cc7a51e645a05ef59ea2" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "SEMI_IMPLICIT", + "time_step_s": 0.0005 + }, + "rms_tracking_error_rad": 0.0024118796589271295, + "steps": 4000, + "timestep_s": 0.0005, + "torque_saturated": false, + "trajectory_sha256": "b3b5af89e1d5b4b04116a59b902a62952945f43adfe3cc7a51e645a05ef59ea2" + }, + { + "control_work_j": 6.084679193001343, + "finite": true, + "integration_family": "VARIATIONAL", + "max_abs_torque_nm": 143.74710461593904, + "max_abs_tracking_error_rad": 0.012325258615201746, + "repeat_trajectory_sha256": [ + "3069b57dec65a0b082ffadea1480ab0838485aac0b4cf9d7ee0bfe8a3f595466", + "3069b57dec65a0b082ffadea1480ab0838485aac0b4cf9d7ee0bfe8a3f595466" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "time_step_s": 0.004 + }, + "rms_tracking_error_rad": 0.0021389444846853533, + "steps": 500, + "timestep_s": 0.004, + "torque_saturated": false, + "trajectory_sha256": "3069b57dec65a0b082ffadea1480ab0838485aac0b4cf9d7ee0bfe8a3f595466" + }, + { + "control_work_j": 6.344268146975802, + "finite": true, + "integration_family": "VARIATIONAL", + "max_abs_torque_nm": 143.74710461593904, + "max_abs_tracking_error_rad": 0.013387884667837479, + "repeat_trajectory_sha256": [ + "a8806d11b3d0d34394958b98e598e40c3a2b6a96d60a6733b72163868b1c52b9", + "a8806d11b3d0d34394958b98e598e40c3a2b6a96d60a6733b72163868b1c52b9" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "time_step_s": 0.002 + }, + "rms_tracking_error_rad": 0.0022896056362997587, + "steps": 1000, + "timestep_s": 0.002, + "torque_saturated": false, + "trajectory_sha256": "a8806d11b3d0d34394958b98e598e40c3a2b6a96d60a6733b72163868b1c52b9" + }, + { + "control_work_j": 6.471245047928654, + "finite": true, + "integration_family": "VARIATIONAL", + "max_abs_torque_nm": 143.74710461593904, + "max_abs_tracking_error_rad": 0.013918071592133655, + "repeat_trajectory_sha256": [ + "0475697e5a6b4918d09ab414085c2814520d0661221cd88bd437236420654861", + "0475697e5a6b4918d09ab414085c2814520d0661221cd88bd437236420654861" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "time_step_s": 0.001 + }, + "rms_tracking_error_rad": 0.002370382161031873, + "steps": 2000, + "timestep_s": 0.001, + "torque_saturated": false, + "trajectory_sha256": "0475697e5a6b4918d09ab414085c2814520d0661221cd88bd437236420654861" + }, + { + "control_work_j": 6.534089744691069, + "finite": true, + "integration_family": "VARIATIONAL", + "max_abs_torque_nm": 143.74710461593904, + "max_abs_tracking_error_rad": 0.014182536997537746, + "repeat_trajectory_sha256": [ + "d3aab588bca5baf1be2eafc3b3d56ffb472c41e0e26cbab3e284764b9fb15ac8", + "d3aab588bca5baf1be2eafc3b3d56ffb472c41e0e26cbab3e284764b9fb15ac8" + ], + "resolved": { + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "integration_family": "VARIATIONAL", + "time_step_s": 0.0005 + }, + "rms_tracking_error_rad": 0.002411992465209295, + "steps": 4000, + "timestep_s": 0.0005, + "torque_saturated": false, + "trajectory_sha256": "d3aab588bca5baf1be2eafc3b3d56ffb472c41e0e26cbab3e284764b9fb15ac8" + } + ], + "visual": { + "reason": "The oracle is numeric tracking error and control work; no visible-behavior claim is made.", + "status": "not-applicable" + } + }, + "host": { + "machine": "x86_64", + "note": "Host recorded for provenance only; no timing methodology was applied", + "performance_valid": false, + "platform": "Linux-7.0.0-29-generic-x86_64-with-glibc2.43", + "python": "3.14.6" + }, + "metrics": { + "allocation": { + "reason": "Post-bake allocation gates are owned by PLAN-122 tooling; this packet does not measure allocations", + "status": "unsupported" + }, + "numerical": { + "any_torque_saturated": false, + "constraint_error": { + "reason": "The corpus row names constraint error as an oracle. This chain has no loop closures or contacts, so there is no constraint residual to measure; the quantity is absent rather than zero.", + "status": "unsupported" + }, + "max_abs_torque_nm_over_cells": 143.74710461593904, + "method": "Maximum applied joint torque and whether the configured torque limit was reached in any cell", + "solver_residual": { + "reason": "StepMetrics.last_step_residual is structurally zero for rigid contact on this branch: recordSolverDiagnostics() in dart/simulation/compute/rigid_body_contact_stage.cpp takes residual = 0.0 by default and no rigid contact call site passes one, so no residual is computed for either contact solver. Exposing a comparable residual is PLAN-123 WS4 work.", + "status": "unsupported" + } + }, + "performance": { + "reason": "The corpus row asks for a whole-step speed/accuracy tradeoff, but no interleaved same-host timing methodology was applied, so step cost is not measured and no tradeoff is established", + "status": "unsupported" + }, + "physical": { + "accuracy_finding": { + "all_cells_finite": true, + "families_with_improving_tracking": [], + "note": "Under computed-torque control the tracking error is set by the closed-loop bandwidth, not by integration error: refining the timestep eightfold does not reduce RMS error (it varies by about 13% and is in fact slightly lower at the coarsest timestep), while control work rises monotonically as the timestep shrinks. So for this system the whole-step tradeoff is not the simple 'smaller timestep is more accurate' shape. Step cost is still not measured, so no speed/accuracy tradeoff is established and this finding does not promote the corpus row.", + "regime": "controller-limited", + "rms_error_ratio_smallest_over_largest_timestep": { + "SEMI_IMPLICIT": 1.1279058322743967, + "VARIATIONAL": 1.127655477960714 + } + }, + "method": "Per-step joint error against the sinusoidal reference (RMS and maximum over the window), and control work as the integral of |tau . qdot| dt accumulated per step", + "per_family_summary": { + "SEMI_IMPLICIT": { + "any_torque_saturated": false, + "cells": 4, + "control_work_by_timestep": { + "0.0005": 6.534246144844606, + "0.001": 6.471559759362399, + "0.002": 6.344908570962907, + "0.004": 6.0859664494528225 + }, + "max_abs_torque_nm": 143.74710461593904, + "rms_error_at_largest_timestep": 0.0021383697024277538, + "rms_error_at_smallest_timestep": 0.0024118796589271295, + "rms_tracking_error_by_timestep": { + "0.0005": 0.0024118796589271295, + "0.001": 0.002370166622085158, + "0.002": 0.0022892179585315767, + "0.004": 0.0021383697024277538 + }, + "tracking_improves_with_timestep": false + }, + "VARIATIONAL": { + "any_torque_saturated": false, + "cells": 4, + "control_work_by_timestep": { + "0.0005": 6.534089744691069, + "0.001": 6.471245047928654, + "0.002": 6.344268146975802, + "0.004": 6.084679193001343 + }, + "max_abs_torque_nm": 143.74710461593904, + "rms_error_at_largest_timestep": 0.0021389444846853533, + "rms_error_at_smallest_timestep": 0.002411992465209295, + "rms_tracking_error_by_timestep": { + "0.0005": 0.002411992465209295, + "0.001": 0.002370382161031873, + "0.002": 0.0022896056362997587, + "0.004": 0.0021389444846853533 + }, + "tracking_improves_with_timestep": false + } + } + } + }, + "result": { + "claim_boundary": "DART 7 main, this commit, a 4-link PD-tracked chain with no contact, fixed gains, a 2 s horizon, timesteps 0.5-4 ms, and both multibody integration families. What is established: the controller stays stable and unsaturated in every cell, and RMS tracking error does NOT improve as the timestep is refined eightfold (about 2.1e-3 rad at 4 ms versus 2.4e-3 rad at 0.5 ms, against a 0.25 rad reference amplitude), while control work rises monotonically as the timestep shrinks -- the error here is controller-limited, not integration-limited. What is NOT established: the cited speed/accuracy tradeoff, because step cost is not measured at all; constraint error, because this scene has no constraints to violate; and any ranking between the integration families. Says nothing about contacting or legged control, gain tuning, historical DART versions, or DART 6.", + "disposition": "unresolved", + "limitations": [ + "No contact: a controlled robot's hard cases are contact transitions, which this fixture deliberately excludes so the tracking signal is unambiguous.", + "Step cost is not measured, so the row's speed half is absent; performance is typed unsupported rather than estimated.", + "One gain policy (uniform 20 rad/s, critically damped) and one reference trajectory; a gain sweep would change the error/effort balance and is not attempted.", + "Control work uses the commanded torque and the joint velocity sampled before the step, so it is a first-order estimate of actuation effort rather than an exact integral.", + "The torque limit is recorded and its saturation flagged; a saturated cell would make the comparison between cells unequal, which is why the flag is published per family." + ] + }, + "review": { + "passes": [] + }, + "scene": { + "description": "4-link chain hanging downward under gravity (revolute hinges about y, 0.3 m links, 1.0 kg each, mass centered at mid-link), driven by an inertia-scaled joint PD controller toward a smooth sinusoidal reference about the hanging equilibrium, no contact, simulated to a 2 s horizon per timestep/integration-family cell.", + "digest": "sha256:36ab2a98ddd1a3b11e3ad60001339f5107442b8f372f9ceca853c35ec46917b9", + "id": "ct005_pd_tracked_chain", + "parameters": { + "description": "4-link chain hanging downward under gravity (revolute hinges about y, 0.3 m links, 1.0 kg each, mass centered at mid-link), driven by an inertia-scaled joint PD controller toward a smooth sinusoidal reference about the hanging equilibrium, no contact, simulated to a 2 s horizon per timestep/integration-family cell.", + "deterministic_repeats": 2, + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "horizon_s": 2.0, + "integration_families": [ + "SEMI_IMPLICIT", + "VARIATIONAL" + ], + "link_count": 4, + "link_length_m": 0.3, + "link_mass_kg": 1.0, + "link_radius_m": 0.05, + "pd_damping_ratio": 1.0, + "pd_natural_frequency_rad_per_s": 20.0, + "reference_amplitude_rad": 0.25, + "reference_frequency_hz": 0.5, + "scene_id": "ct005_pd_tracked_chain", + "timesteps_s": [ + 0.004, + 0.002, + 0.001, + 0.0005 + ], + "torque_limit_nm": 500.0 + } + }, + "schema": "dart.citation_claim_evidence/v1", + "source": { + "claim": "Controlled robot tracking exposes whole-step speed/accuracy tradeoffs.", + "url": "https://leggedrobotics.github.io/SimBenchmark/" + }, + "target": { + "branch": "main", + "commit": "2c33043f8ca9616dad99df65593894138e1c5201", + "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit." + }, + "title": "PD-control tracking error and control work versus timestep (DART 7 first packet)" +} diff --git a/docs/plans/dashboard.md b/docs/plans/dashboard.md index e1686998d5367..3887439d71d0c 100644 --- a/docs/plans/dashboard.md +++ b/docs/plans/dashboard.md @@ -52,14 +52,16 @@ its own line so status updates remain git-history friendly. - Horizon: Now - Dimension: Algorithm extensibility - Next step: WS0 audit and the WS1 fail-closed claim/evidence manifest landed - with a permanent negative control and four first-wave packets: CT-001 + with a permanent negative control and five first-wave packets: CT-001 rolling-direction (`reproduced`), CT-002 dense inelastic contact (`reproduced`), CT-003 dense elastic contact (`unresolved` -- no energy injection observed), and CT-004 articulated energy drift versus timestep (`unresolved`: drift converges with timestep in both integration families, - but the cited claim concerns matched cost, which is not measured). Next: the - remaining first-wave families (articulated control, heel-strike/toe-off, - high mass-ratio, reset/concurrency queries) with branch-qualified dispositions, reusing + but the cited claim concerns matched cost, which is not measured), and + CT-005 PD tracking (`unresolved`: tracking error is controller-limited, so + refining the timestep does not improve it, and step cost is not measured). + Next: the remaining first-wave families (heel-strike/toe-off, high + mass-ratio, reset/concurrency queries) with branch-qualified dispositions, reusing existing demos scenes and PLAN-104/621/622 evidence, before contact-identity/semantics work (WS3) and cross-family diagnostics (WS4); WS4's first slice is exposing `ResolvedSolverConfiguration` to Python and a diff --git a/scripts/write_citation_ct005_pd_tracking_packet.py b/scripts/write_citation_ct005_pd_tracking_packet.py new file mode 100644 index 0000000000000..130708b2fd7ea --- /dev/null +++ b/scripts/write_citation_ct005_pd_tracking_packet.py @@ -0,0 +1,647 @@ +#!/usr/bin/env python3 +"""Write the CT-005 PD-control tracking evidence packet (PLAN-123 WS2). + +Bounded claim (corpus row CT-005): controlled robot tracking exposes +whole-step speed/accuracy tradeoffs. + +Bounded reconstruction (explicitly not source-exact): a 4-link articulated +chain driven by a fixed-gain PD controller toward a smooth sinusoidal joint +reference, swept over a timestep grid for each multibody integration family. +No contact. Per cell it records: + +- tracking error against the reference (RMS and maximum over the window); +- control work, the integral of |tau . qdot| dt, so accuracy can be read + against actuation effort rather than against nothing; +- whether the controller stayed stable (finite state, bounded torque); +- deterministic repeats: each cell runs twice and must be bit-identical. + +The corpus row's oracle also names constraint error and a timing +distribution. This fixture has no constraints beyond the joint structure, and +it measures no time: `metrics.performance` is typed unsupported because no +interleaved same-host methodology was applied. The "speed" half of the +speed/accuracy tradeoff is therefore NOT established here, and the packet +disposition reflects that rather than promoting the row on the accuracy half +alone. + +Usage (after `pixi run build`): + + PYTHONPATH=build/default/cpp/Release/python pixi run python \ + scripts/write_citation_ct005_pd_tracking_packet.py +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import platform +import subprocess +import sys +from pathlib import Path +from typing import Any + +import dartpy as sx +import numpy as np +from citation_packet_utils import ( + UNSUPPORTED_SOLVER_RESIDUAL, + preserve_review, +) + +REPO_ROOT = Path(__file__).resolve().parents[1] +DEFAULT_OUTPUT = ( + REPO_ROOT + / "docs" + / "plans" + / "123-citation-driven-simulation-trust" + / "evidence" + / "CT-005-dart7-pd-tracking.json" +) + +SCENE_PARAMETERS: dict[str, Any] = { + "scene_id": "ct005_pd_tracked_chain", + "description": ( + "4-link chain hanging downward under gravity (revolute hinges " + "about y, 0.3 m links, 1.0 kg each, mass centered at mid-link), " + "driven by an inertia-scaled joint PD controller toward a smooth " + "sinusoidal reference about the hanging equilibrium, no contact, " + "simulated to a 2 s horizon per timestep/integration-family cell." + ), + "gravity_mps2": [0.0, 0.0, -9.81], + "link_count": 4, + "link_length_m": 0.3, + "link_mass_kg": 1.0, + "link_radius_m": 0.05, + "reference_amplitude_rad": 0.25, + "reference_frequency_hz": 0.5, + "pd_natural_frequency_rad_per_s": 20.0, + "pd_damping_ratio": 1.0, + "torque_limit_nm": 500.0, + "horizon_s": 2.0, + "timesteps_s": [0.004, 0.002, 0.001, 0.0005], + "integration_families": ["SEMI_IMPLICIT", "VARIATIONAL"], + "deterministic_repeats": 2, +} + + +def scene_digest(parameters: dict[str, Any]) -> str: + canonical = json.dumps(parameters, sort_keys=True, separators=(",", ":")) + return "sha256:" + hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + +def _translation(x: float, y: float, z: float) -> np.ndarray: + transform = np.eye(4) + transform[:3, 3] = (x, y, z) + return transform + + +def downstream_inertia(index: int, parameters: dict[str, Any]) -> float: + """Inertia the chain below hinge `index` presents about that hinge. + + Each joint of a serial chain sees a different load -- here they differ by + roughly 60x between root and tip -- so a single gain pair cannot serve + both ends: gains stable at the tip are far too soft at the root, and + gains stiff enough for the root make the tip's explicit damping term + unstable at the coarser timesteps. Scaling each joint's gains by its own + downstream inertia gives every joint the same closed-loop natural + frequency and damping ratio, which is what makes the timestep sweep a + measurement of integration rather than of gain tuning. + """ + count = int(parameters["link_count"]) + length = float(parameters["link_length_m"]) + mass = float(parameters["link_mass_kg"]) + own = mass * length * length / 12.0 + total = 0.0 + for j in range(index, count): + lever = (j - index) * length + 0.5 * length + total += mass * lever * lever + own + return total + + +def reference_angle(time: float, index: int, parameters: dict[str, Any]) -> float: + """Smooth per-joint sinusoidal reference, phase-shifted along the chain.""" + amplitude = float(parameters["reference_amplitude_rad"]) + frequency = float(parameters["reference_frequency_hz"]) + phase = 0.5 * math.pi * index / max(1, int(parameters["link_count"])) + return amplitude * math.sin(2.0 * math.pi * frequency * time + phase) + + +def reference_rate(time: float, index: int, parameters: dict[str, Any]) -> float: + """Analytic derivative of the reference, so the controller is not + differentiating a signal numerically.""" + amplitude = float(parameters["reference_amplitude_rad"]) + frequency = float(parameters["reference_frequency_hz"]) + omega = 2.0 * math.pi * frequency + phase = 0.5 * math.pi * index / max(1, int(parameters["link_count"])) + return amplitude * omega * math.cos(omega * time + phase) + + +def reference_acceleration( + time: float, index: int, parameters: dict[str, Any] +) -> float: + """Analytic second derivative of the reference.""" + amplitude = float(parameters["reference_amplitude_rad"]) + frequency = float(parameters["reference_frequency_hz"]) + omega = 2.0 * math.pi * frequency + phase = 0.5 * math.pi * index / max(1, int(parameters["link_count"])) + return -amplitude * omega * omega * math.sin(omega * time + phase) + + +def run_single( + family_name: str, timestep: float, parameters: dict[str, Any] +) -> dict[str, Any]: + """Run one PD-tracked chain and return raw metrics plus a hash.""" + link_count = int(parameters["link_count"]) + length = float(parameters["link_length_m"]) + mass = float(parameters["link_mass_kg"]) + radius = float(parameters["link_radius_m"]) + omega = float(parameters["pd_natural_frequency_rad_per_s"]) + zeta = float(parameters["pd_damping_ratio"]) + torque_limit = float(parameters["torque_limit_nm"]) + step_count = int(round(float(parameters["horizon_s"]) / timestep)) + + world = sx.World( + time_step=timestep, + gravity=np.asarray(parameters["gravity_mps2"], dtype=float), + multibody_options=sx.MultibodyOptions( + integration_family=sx.MultibodyIntegrationFamily[family_name] + ), + ) + + chain = world.add_multibody("ct005_chain") + parent = chain.add_link("base") + ixx = 0.5 * mass * radius * radius + itrans = mass * length * length / 12.0 + joints = [] + for index in range(link_count): + offset = 0.0 if index == 0 else length + link = chain.add_link( + f"link{index}", + parent=parent, + joint=sx.JointSpec( + name=f"hinge{index}", + type=sx.JointType.REVOLUTE, + axis=(0.0, 1.0, 0.0), + transform_from_parent=_translation(0.0, 0.0, -offset), + ), + ) + link.mass = mass + link.inertia = ((ixx, 0.0, 0.0), (0.0, itrans, 0.0), (0.0, 0.0, itrans)) + # Rod-like link: mass centered halfway along its own length, so each + # hinge sees a real gravity lever instead of a point mass sitting on + # the axis. + link.center_of_mass = (0.0, 0.0, -0.5 * length) + link.parent_joint.position = [reference_angle(0.0, index, parameters)] + joints.append(link.parent_joint) + parent = link + + world.enter_simulation_mode() + + resolved = { + "integration_family": world.multibody_options.integration_family.name, + "gravity_mps2": [float(v) for v in np.asarray(world.gravity)], + "time_step_s": float(world.time_step), + } + + trajectory = hashlib.sha256() + squared_error_sum = 0.0 + error_samples = 0 + max_abs_error = 0.0 + control_work = 0.0 + max_abs_torque = 0.0 + torque_saturated = False + non_finite = False + + for step_index in range(step_count): + time = step_index * timestep + positions = np.array( + [float(np.asarray(j.position, dtype=float)[0]) for j in joints] + ) + velocities = np.array( + [float(np.asarray(j.velocity, dtype=float)[0]) for j in joints] + ) + targets = np.array( + [reference_angle(time, i, parameters) for i in range(link_count)] + ) + target_rates = np.array( + [reference_rate(time, i, parameters) for i in range(link_count)] + ) + target_accels = np.array( + [reference_acceleration(time, i, parameters) for i in range(link_count)] + ) + # Computed torque: ask the model what torque realizes the commanded + # acceleration, so the closed loop has the specified natural frequency + # and damping in every configuration. A hand-tuned diagonal PD cannot: + # a serial chain's root joint sees a far smaller articulated inertia + # than its composite rigid-body inertia, so gains sized from geometry + # destabilize the root while the tip tracks fine. + commanded = ( + target_accels + + 2.0 * zeta * omega * (target_rates - velocities) + + omega * omega * (targets - positions) + ) + torques = np.asarray(chain.compute_inverse_dynamics(commanded), dtype=float) + for index, joint in enumerate(joints): + torque = float(torques[index]) + if abs(torque) > torque_limit: + torque_saturated = True + torque = math.copysign(torque_limit, torque) + joint.force = [torque] + max_abs_torque = max(max_abs_torque, abs(torque)) + control_work += abs(torque * velocities[index]) * timestep + + world.step() + + for index, joint in enumerate(joints): + target = reference_angle(time + timestep, index, parameters) + position = float(np.asarray(joint.position, dtype=float)[0]) + velocity = float(np.asarray(joint.velocity, dtype=float)[0]) + if not (math.isfinite(position) and math.isfinite(velocity)): + non_finite = True + break + error = position - target + squared_error_sum += error * error + error_samples += 1 + max_abs_error = max(max_abs_error, abs(error)) + trajectory.update(np.array([position, velocity]).tobytes()) + if non_finite: + break + + rms_error = math.sqrt(squared_error_sum / error_samples) if error_samples else None + return { + "integration_family": family_name, + "timestep_s": timestep, + "steps": step_count, + "resolved": resolved, + "finite": not non_finite, + "trajectory_sha256": trajectory.hexdigest(), + "rms_tracking_error_rad": rms_error, + "max_abs_tracking_error_rad": max_abs_error, + "control_work_j": control_work, + "max_abs_torque_nm": max_abs_torque, + "torque_saturated": torque_saturated, + } + + +def git_head() -> str: + return subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=REPO_ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + + +def build_packet(output_path: Path | None = None) -> dict[str, Any]: + parameters = SCENE_PARAMETERS + rows: list[dict[str, Any]] = [] + determinism_failures: list[str] = [] + resolved_by_cell: dict[str, dict[str, Any]] = {} + + for family in parameters["integration_families"]: + for timestep in parameters["timesteps_s"]: + repeats = [ + run_single(family, timestep, parameters) + for _ in range(int(parameters["deterministic_repeats"])) + ] + hashes = {run["trajectory_sha256"] for run in repeats} + if len(hashes) != 1: + determinism_failures.append( + f"{family} dt {timestep}: trajectory hashes differ " + f"{sorted(hashes)}" + ) + row = repeats[0] + for repeat in repeats: + readback = repeat["resolved"] + if readback != row["resolved"]: + raise SystemExit( + f"{family} dt {timestep}: resolved configuration " + "differs between repeats" + ) + if readback["integration_family"] != family: + raise SystemExit( + f"requested integration family {family} but World " + f"readback reports {readback['integration_family']}" + ) + if readback["time_step_s"] != timestep: + raise SystemExit( + f"requested timestep {timestep} but World readback " + f"reports {readback['time_step_s']}" + ) + if readback["gravity_mps2"] != list(parameters["gravity_mps2"]): + raise SystemExit( + f"requested gravity {parameters['gravity_mps2']} but " + f"World readback reports {readback['gravity_mps2']}" + ) + row["repeat_trajectory_sha256"] = [ + repeat["trajectory_sha256"] for repeat in repeats + ] + resolved_by_cell[f"{family}@dt={timestep}"] = row["resolved"] + rows.append(row) + + if determinism_failures: + raise SystemExit( + "deterministic repeats failed:\n " + "\n ".join(determinism_failures) + ) + + family_summary: dict[str, Any] = {} + for family in parameters["integration_families"]: + family_rows = [row for row in rows if row["integration_family"] == family] + errors_by_dt = { + row["timestep_s"]: row["rms_tracking_error_rad"] + for row in family_rows + if row["rms_tracking_error_rad"] is not None + } + work_by_dt = {row["timestep_s"]: row["control_work_j"] for row in family_rows} + smallest = min(errors_by_dt) if errors_by_dt else None + largest = max(errors_by_dt) if errors_by_dt else None + family_summary[family] = { + "cells": len(family_rows), + "rms_tracking_error_by_timestep": { + str(dt): value for dt, value in sorted(errors_by_dt.items()) + }, + "control_work_by_timestep": { + str(dt): value for dt, value in sorted(work_by_dt.items()) + }, + "rms_error_at_smallest_timestep": errors_by_dt[smallest], + "rms_error_at_largest_timestep": errors_by_dt[largest], + "tracking_improves_with_timestep": ( + errors_by_dt[smallest] < errors_by_dt[largest] + ), + "max_abs_torque_nm": max(row["max_abs_torque_nm"] for row in family_rows), + "any_torque_saturated": any(row["torque_saturated"] for row in family_rows), + } + + all_finite = all(row["finite"] for row in rows) + improving = sorted( + family + for family, stats in family_summary.items() + if stats["tracking_improves_with_timestep"] + ) + + # The cited claim is about a speed/accuracy tradeoff. This packet measures + # accuracy and actuation effort but no speed, so half of the tradeoff is + # absent and the row cannot be promoted -- the same reasoning that keeps + # CT-004 unresolved. The accuracy result is published as its own finding. + disposition = "unresolved" + error_spread = { + family: ( + stats["rms_error_at_smallest_timestep"] + / stats["rms_error_at_largest_timestep"] + ) + for family, stats in family_summary.items() + } + accuracy_finding = { + "all_cells_finite": all_finite, + "families_with_improving_tracking": improving, + "rms_error_ratio_smallest_over_largest_timestep": error_spread, + "regime": ("controller-limited" if not improving else "integration-limited"), + "note": ( + "Under computed-torque control the tracking error is set by the " + "closed-loop bandwidth, not by integration error: refining the " + "timestep eightfold does not reduce RMS error (it varies by " + "about 13% and is in fact slightly lower at the coarsest " + "timestep), while control work rises monotonically as the " + "timestep shrinks. So for this system the whole-step tradeoff is " + "not the simple 'smaller timestep is more accurate' shape. Step " + "cost is still not measured, so no speed/accuracy tradeoff is " + "established and this finding does not promote the corpus row." + ), + } + + command = ( + "PYTHONPATH=build/default/cpp/Release/python pixi run python " + "scripts/write_citation_ct005_pd_tracking_packet.py" + ) + packet: dict[str, Any] = { + "schema": "dart.citation_claim_evidence/v1", + "claim_id": "CT-005", + "title": ( + "PD-control tracking error and control work versus timestep " + "(DART 7 first packet)" + ), + "source": { + "url": "https://leggedrobotics.github.io/SimBenchmark/", + "claim": ( + "Controlled robot tracking exposes whole-step speed/accuracy " + "tradeoffs." + ), + }, + "target": { + "branch": "main", + "commit": git_head(), + "commit_role": ( + "Source state measured: the library and fixture ran at this " + "commit, which is HEAD at capture time. The packet and its " + "writer land in a later commit." + ), + }, + "scene": { + "id": parameters["scene_id"], + "digest": scene_digest(parameters), + "description": parameters["description"], + "parameters": parameters, + }, + "configuration": { + "requested": { + "integration_family_sweep": parameters["integration_families"], + "timestep_sweep_s": parameters["timesteps_s"], + "controller": ( + "joint PD with per-joint gains scaled to each joint's downstream " + "inertia for a uniform 20 rad/s closed-loop natural " + "frequency at critical damping, " + "torque limit 500 Nm, applied before each step" + ), + "contact": "none (no collision shapes)", + "precision": "float64", + "backend": "cpu", + "threads": "World default sequential step", + }, + "resolved": {"by_cell": resolved_by_cell}, + "resolved_provenance": ( + "World property readback after enter_simulation_mode, with " + "integration family, timestep, and gravity each asserted " + "equal to the request per repeat and per cell. The controller " + "is applied by this script, not resolved by the World. " + "configuration.timestep is the smallest swept value; " + "configuration.resolved.by_cell is authoritative per cell. " + "World::getResolvedConfiguration() is not yet exposed to " + "Python (PLAN-123 WS4 follow-up)." + ), + "detector": ( + "not applicable: the scene has no collision shapes and runs " + "no narrow phase" + ), + "timestep": min(parameters["timesteps_s"]), + "substeps": 1, + "iterations": ( + "World defaults; the variational family uses its default " + "iteration limit and tolerance, which this packet does not " + "vary" + ), + "fallback_policy": ( + "World defaults; no per-island fallback reporting is exposed " + "on main (PLAN-123 WS4 follow-up)" + ), + }, + "ensemble": { + "kind": "parameter-sweep-with-deterministic-repeats", + "sweep": [ + {"integration_family": family, "timestep_s": timestep} + for family in parameters["integration_families"] + for timestep in parameters["timesteps_s"] + ], + "deterministic_repeats": int(parameters["deterministic_repeats"]), + "deterministic_repeats_identical": not determinism_failures, + "measurement_window": { + "start_s": 0.0, + "end_s": parameters["horizon_s"], + }, + }, + "metrics": { + "physical": { + "method": ( + "Per-step joint error against the sinusoidal reference " + "(RMS and maximum over the window), and control work as " + "the integral of |tau . qdot| dt accumulated per step" + ), + "per_family_summary": family_summary, + "accuracy_finding": accuracy_finding, + }, + "numerical": { + "method": ( + "Maximum applied joint torque and whether the configured " + "torque limit was reached in any cell" + ), + "max_abs_torque_nm_over_cells": max( + row["max_abs_torque_nm"] for row in rows + ), + "any_torque_saturated": any(row["torque_saturated"] for row in rows), + "constraint_error": { + "status": "unsupported", + "reason": ( + "The corpus row names constraint error as an oracle. " + "This chain has no loop closures or contacts, so " + "there is no constraint residual to measure; the " + "quantity is absent rather than zero." + ), + }, + "solver_residual": dict(UNSUPPORTED_SOLVER_RESIDUAL), + }, + "performance": { + "status": "unsupported", + "reason": ( + "The corpus row asks for a whole-step speed/accuracy " + "tradeoff, but no interleaved same-host timing " + "methodology was applied, so step cost is not measured " + "and no tradeoff is established" + ), + }, + "allocation": { + "status": "unsupported", + "reason": ( + "Post-bake allocation gates are owned by PLAN-122 " + "tooling; this packet does not measure allocations" + ), + }, + }, + "evidence": { + "commands": [command], + "raw_rows": rows, + "visual": { + "status": "not-applicable", + "reason": ( + "The oracle is numeric tracking error and control work; " + "no visible-behavior claim is made." + ), + }, + }, + "result": { + "disposition": disposition, + "claim_boundary": ( + "DART 7 main, this commit, a 4-link PD-tracked chain with no " + "contact, fixed gains, a 2 s horizon, timesteps 0.5-4 ms, and " + "both multibody integration families. What is established: " + "the controller stays stable and unsaturated in every " + "cell, and RMS tracking error does NOT improve as the " + "timestep is refined eightfold (about 2.1e-3 rad at 4 ms " + "versus 2.4e-3 rad at 0.5 ms, against a 0.25 rad reference " + "amplitude), while control work rises monotonically as the " + "timestep shrinks -- the error here is controller-limited, " + "not integration-limited. What is " + "NOT established: the cited speed/accuracy tradeoff, because " + "step cost is not measured at all; constraint error, because " + "this scene has no constraints to violate; and any ranking " + "between the integration families. Says nothing about " + "contacting or legged control, gain tuning, historical DART " + "versions, or DART 6." + ), + "limitations": [ + "No contact: a controlled robot's hard cases are contact " + "transitions, which this fixture deliberately excludes so " + "the tracking signal is unambiguous.", + "Step cost is not measured, so the row's speed half is " + "absent; performance is typed unsupported rather than " + "estimated.", + "One gain policy (uniform 20 rad/s, critically damped) and one " + "reference trajectory; a gain sweep would " + "change the error/effort balance and is not attempted.", + "Control work uses the commanded torque and the joint " + "velocity sampled before the step, so it is a first-order " + "estimate of actuation effort rather than an exact integral.", + "The torque limit is recorded and its saturation flagged; a " + "saturated cell would make the comparison between cells " + "unequal, which is why the flag is published per family.", + ], + }, + "review": ( + preserve_review(output_path) if output_path is not None else {"passes": []} + ), + "host": { + "platform": platform.platform(), + "python": sys.version.split()[0], + "machine": platform.machine(), + "performance_valid": False, + "note": ( + "Host recorded for provenance only; no timing methodology " + "was applied" + ), + }, + } + return packet + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--output", + type=Path, + default=DEFAULT_OUTPUT, + help=f"Packet output path (default: {DEFAULT_OUTPUT})", + ) + return parser.parse_args() + + +def main() -> int: + args = parse_args() + packet = build_packet(args.output) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text( + json.dumps(packet, indent=2, sort_keys=True) + "\n", encoding="utf-8" + ) + physical = packet["metrics"]["physical"] + print(f"wrote {args.output}") + for family, stats in physical["per_family_summary"].items(): + print( + f" {family}: RMS error {stats['rms_error_at_largest_timestep']:.4e}" + f" -> {stats['rms_error_at_smallest_timestep']:.4e} rad, " + f"max torque {stats['max_abs_torque_nm']:.1f} Nm, saturated=" + f"{stats['any_torque_saturated']}" + ) + print(f" disposition: {packet['result']['disposition']}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) From ca78b0980e88617b98bdbcb84cfeb01931238d05 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Fri, 14 Aug 2026 21:07:01 -0700 Subject: [PATCH 09/52] Move the PLAN-123 evidence narrative into the owner plan progress log The docs-policy gate caught the dashboard entry growing to 17 lines against its own 15-line Next step budget, which is exactly the drift that budget exists to prevent: each landed packet had been appended to the operating view instead of to a history owner. Per the dashboard's own rule, the accumulated WS0/WS1/WS2 narrative and the review record now live in a Progress log section of the owner plan, and the dashboard keeps the current action plus a History pointer. --- .../123-citation-driven-simulation-trust.md | 33 +++++++++++++++++++ docs/plans/dashboard.md | 24 +++++--------- 2 files changed, 41 insertions(+), 16 deletions(-) diff --git a/docs/plans/123-citation-driven-simulation-trust.md b/docs/plans/123-citation-driven-simulation-trust.md index 21ed8e30a1194..7a19db0e186ee 100644 --- a/docs/plans/123-citation-driven-simulation-trust.md +++ b/docs/plans/123-citation-driven-simulation-trust.md @@ -217,6 +217,39 @@ and domain expansion are DART 7 only. - Using citation count or routine library use as evidence of physical accuracy. - Keeping an unbounded paper-support backlog in this plan. +## Progress log + +### 2026-08-14 — WS0/WS1 foundation and first-wave packets + +- WS0 audit reconciled the corpus against `main` 20501341226 and + `release-6.20` 39ccd52068b: plan IDs confirmed free, PLAN-091 heritage + mapped, open PRs (#3432, #3377, #3431, #3428) and issue #3056 routed to + existing owners, and existing demos scenes recorded as fixture seeds. +- WS1 landed the checked claim manifest and the fail-closed + `pixi run check-citation-evidence` gate on both branches, wired into + `check-lint`, with permanent intentionally incomplete negative controls. + The gate rejects prose/non-JSON/outside-`evidence/`/negative-control/ + non-string lane references, dangling or directory `raw_paths`, digests that + disagree with the published scene, single-run ensembles, spelled + placeholders, and any exact zero that is not declared a real measurement or + typed unsupported with a reason. +- WS2 first-wave packets on `main`: CT-001 rolling direction (`reproduced`), + CT-002 dense inelastic contact (`reproduced`), CT-003 dense elastic contact + (`unresolved`), CT-004 articulated energy drift (`unresolved`), CT-005 PD + tracking (`unresolved`). On `release-6.20`: CT-001 detector sweep + (`reproduced` on fcl/dart/ode; bullet excluded for lacking the pyramid + signature). Three of six dispositions are negative because the cited + behavior was not observed or the cited quantity was not measured. +- Review record: three independent rounds per branch found a validator + bypass twice (nested paths, then non-string lane entries), + unsupported-as-zero metrics, false provenance strings, a verification block + imported from the wrong branch, and two dispositions that were not earned. + All fixed in-branch; details live in + `docs/dev_tasks/citation_driven_simulation_trust/verification.md`. +- Known gap: the validator does not arithmetically cross-check derived + metrics against `raw_rows`, so a falsified summary would pass. Closing that + needs the gate to know each packet's derivation (WS4-scale). + ## Revision triggers - A corpus row shows an existing plan already owns the proposed shared surface. diff --git a/docs/plans/dashboard.md b/docs/plans/dashboard.md index 3887439d71d0c..6d4f944abfdd9 100644 --- a/docs/plans/dashboard.md +++ b/docs/plans/dashboard.md @@ -51,22 +51,14 @@ its own line so status updates remain git-history friendly. - Status: Active - Horizon: Now - Dimension: Algorithm extensibility -- Next step: WS0 audit and the WS1 fail-closed claim/evidence manifest landed - with a permanent negative control and five first-wave packets: CT-001 - rolling-direction (`reproduced`), CT-002 dense inelastic contact - (`reproduced`), CT-003 dense elastic contact (`unresolved` -- no energy - injection observed), and CT-004 articulated energy drift versus timestep - (`unresolved`: drift converges with timestep in both integration families, - but the cited claim concerns matched cost, which is not measured), and - CT-005 PD tracking (`unresolved`: tracking error is controller-limited, so - refining the timestep does not improve it, and step cost is not measured). - Next: the remaining first-wave families (heel-strike/toe-off, high - mass-ratio, reset/concurrency queries) with branch-qualified dispositions, reusing - existing demos scenes and PLAN-104/621/622 evidence, before - contact-identity/semantics work (WS3) and cross-family diagnostics (WS4); - WS4's first slice is exposing `ResolvedSolverConfiguration` to Python and a - comparable residual, both currently typed unsupported in every packet. - History and sequencing live in the owner plan and +- Next step: WS0/WS1 and five first-wave packets have landed on `main`, one on + `release-6.20`. Continue WS2 with the remaining first-wave families: + heel-strike/toe-off (CT-006, gated on WS3 contact semantics), high + mass-ratio stacks (CT-007), and reset/concurrency queries (CT-011..013). + In parallel, start WS4's first slice, which several packets are waiting on: + expose `ResolvedSolverConfiguration` to Python and a comparable per-solve + residual, both currently typed unsupported everywhere. History: + `## Progress log` in the owner plan and `docs/dev_tasks/citation_driven_simulation_trust/`. - Gate: Every claim row is branch/version-qualified and source-bound; `pixi run check-citation-evidence` fails closed on missing commit, scene From ebbbb853a49cf59bafb590cf111c110acbd0531b Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Fri, 14 Aug 2026 21:33:25 -0700 Subject: [PATCH 10/52] Expose the World's resolved solver configuration to Python (WS4) Every packet had been recording resolved solver identity by reading back the option it had just set, which cannot distinguish a method that ran from one the World silently substituted. Three review rounds named this as the gap behind the typed-unsupported markers. World::getResolvedConfiguration() already existed in C++ but was unreachable from Python. This binds ResolvedConfigurationNote and ResolvedSolverConfiguration, adds the World.resolved_configuration property, and threads it through every packet: each run now records, per domain, what was requested, what actually resolved, why, and whether it was a substitution. Verified on a live World -- empty before bake, four domains after, and the recorded rigid-contact resolution follows the requested contact solver. Stub scope: pixi run generate-stubs rewrote 1990 lines across eight files because the committed stubs were already stale relative to the build. That churn is reverted and only the 37 lines this change needs were added by hand; the pre-existing drift is left for its owner rather than swept in. The remaining WS4 gap is a comparable per-solve residual, which still does not exist and stays typed unsupported in every packet. --- .../verification.md | 30 + .../123-citation-driven-simulation-trust.md | 7 + .../CT-001-dart7-rolling-direction.json | 516 +++++++++++++++++- .../CT-002-dart7-dense-inelastic-contact.json | 260 ++++++++- .../CT-003-dart7-dense-elastic-contact.json | 258 ++++++++- ...004-dart7-articulated-energy-momentum.json | 514 ++++++++++++++++- .../evidence/CT-005-dart7-pd-tracking.json | 516 +++++++++++++++++- docs/plans/dashboard.md | 7 +- python/dartpy/simulation/module_compute.cpp | 47 ++ python/dartpy/simulation/module_world.cpp | 10 + python/stubs/dartpy/simulation.pyi | 37 ++ python/tests/unit/simulation/test_world.py | 59 ++ scripts/citation_packet_utils.py | 24 + ...citation_ct001_rolling_direction_packet.py | 8 +- ...ite_citation_ct002_dense_contact_packet.py | 9 +- ...e_citation_ct003_elastic_contact_packet.py | 2 + ...itation_ct004_articulated_energy_packet.py | 2 + ...write_citation_ct005_pd_tracking_packet.py | 5 +- 18 files changed, 2227 insertions(+), 84 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 0192288c278fd..4756d86eff7e2 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -299,3 +299,33 @@ WS4-scale work. typed unsupported because this scene has no constraints to violate. - Commands: packet writer as recorded in the packet; `pixi run check-citation-evidence` — OK; 72 pytest cases pass. + +## WS4 first slice: resolved configuration in Python — 2026-08-14 + +- What changed: nanobind bindings for `ResolvedConfigurationNote` and + `ResolvedSolverConfiguration` (`python/dartpy/simulation/module_compute.cpp`), + the `World.resolved_configuration` property + (`python/dartpy/simulation/module_world.cpp`), surgical stub entries, two + Python tests, a `world_resolved_configuration()` helper, and all five + packets regenerated to record it. +- Why it mattered: every packet had been recording resolved identity by + reading back the option it had just set, which cannot distinguish a method + that ran from one that was silently substituted. Three review rounds + flagged this as the blocking gap behind the typed-unsupported markers. + Packets now carry, per domain, the World's own bake-time decision: + requested, resolved, reason, and a substitution flag. +- Verified against a live World: before `enter_simulation_mode` the + configuration is empty; after bake it reports four domains + (rigid-body, rigid-contact, multibody, deformable-psd), and requesting + BOXED_LCP versus SEQUENTIAL_IMPULSE changes the recorded `rigid-contact` + resolution accordingly. +- Scope discipline: `pixi run generate-stubs` rewrote 1990 lines across eight + stub files, because the committed stubs were already stale relative to the + build. That churn was reverted and the 37 lines corresponding to this + change were added by hand, keeping the diff surgical; the pre-existing stub + drift is left for whoever owns it rather than being swept in here. +- Commands: `pixi run build` — success; the two new tests pass; + `pixi run check-dartpy-import-layout` — passed; + `pixi run check-docs-policy` — passed. +- Remaining WS4 gap: no comparable per-solve residual exists, so + `metrics.numerical.solver_residual` stays typed unsupported in every packet. diff --git a/docs/plans/123-citation-driven-simulation-trust.md b/docs/plans/123-citation-driven-simulation-trust.md index 7a19db0e186ee..2fb10a17854f3 100644 --- a/docs/plans/123-citation-driven-simulation-trust.md +++ b/docs/plans/123-citation-driven-simulation-trust.md @@ -246,6 +246,13 @@ and domain expansion are DART 7 only. imported from the wrong branch, and two dispositions that were not earned. All fixed in-branch; details live in `docs/dev_tasks/citation_driven_simulation_trust/verification.md`. +- WS4 first slice: `World.resolved_configuration` is now exposed to Python + (`ResolvedSolverConfiguration` and `ResolvedConfigurationNote` bindings, + stubs, and tests), so every packet records the World's own bake-time + resolution per domain -- requested, resolved, reason, substitution flag -- + instead of echoing back the requested option. A benchmark can no longer + name a method that did not run. A comparable per-solve residual remains + unexposed and is still typed unsupported everywhere. - Known gap: the validator does not arithmetically cross-check derived metrics against `raw_rows`, so a falsified summary would pass. Closing that needs the gate to know each packet's derivation (WS4-scale). diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json index 35a534f7b48a2..939d0a9ba0bb6 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json @@ -25,7 +25,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "SEQUENTIAL_IMPULSE": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -35,7 +65,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } } } }, @@ -146,7 +206,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "slide_end_time_s": 0.082, "trajectory_sha256": "a839b7f050262f80e450a25831e5b4144eb74f3f725807eb0e92855f762b96d5" @@ -181,7 +271,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "slide_end_time_s": 0.08, "trajectory_sha256": "4579581ae2bab0980e689803c5fc8e7296f2b79314c1b9644855360102d16ced" @@ -216,7 +336,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "slide_end_time_s": 0.07, "trajectory_sha256": "ace9350655e315351bede1b8474feac16c223921633fbf2a5497af3ae57d941a" @@ -251,7 +401,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "slide_end_time_s": 0.058, "trajectory_sha256": "8446fd6f1793674e853a71724405db019de7f5acd799e9e61af877310443edc7" @@ -286,7 +466,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "slide_end_time_s": 0.07, "trajectory_sha256": "fce89de8287195fea41ecdbc5771c027b838e91d2c5be2e7255a6cc2abffeb2d" @@ -321,7 +531,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "slide_end_time_s": 0.08, "trajectory_sha256": "2ca8f5b51e27817c76eedef74e1b9c59813fa9052244f219e3143eb0327f5d7a" @@ -356,7 +596,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "slide_end_time_s": 0.082, "trajectory_sha256": "d7bd5003831ed9c7c1a477345d578fe7f3f5a99706d3a4e9720e11d770a9c5b8" @@ -391,7 +661,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "slide_end_time_s": 0.082, "trajectory_sha256": "f92fc6c64c534a96459ce3ac0a970c931618a8533dd3974b565db61b7cacd150" @@ -426,7 +726,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "slide_end_time_s": 0.08, "trajectory_sha256": "fbe57d7b17ff7fdfe97b1cfe5112698ae2ae9c7c413e4c08eb13baba931e0899" @@ -461,7 +791,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "slide_end_time_s": 0.07, "trajectory_sha256": "baed575b592007dcfedeefa6f7b8c17858d47860acb3e6f2c06e925766e575fe" @@ -496,7 +856,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "slide_end_time_s": 0.058, "trajectory_sha256": "4f8571c3013779682bcf99ce7dd5f3cc7bb3ccb49182fc64f8f55199fa3b19ef" @@ -531,7 +921,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "slide_end_time_s": 0.07, "trajectory_sha256": "be2f9b06b52ffb94c3c4c65e2ef2a1ba188e02b852e89c1fcd355fb87e27079d" @@ -566,7 +986,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "slide_end_time_s": 0.08, "trajectory_sha256": "dd250dfabf218e68548f9edf30a342bc1bc595d6675b5f602f4bf608f88921e0" @@ -601,7 +1051,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "slide_end_time_s": 0.082, "trajectory_sha256": "631d89a93d54a08badc23834c755a2a84c88f7d6f6b94a2357bd81a5a3e7d9a5" @@ -721,7 +1201,7 @@ "claim_boundary": "DART 7 main, this commit, one sphere sliding to rolling on a static ground box at v0=1 m/s, mu=0.35, dt=2 ms, 1 s horizon, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers, launch angles 0-90 deg in 15 deg steps. Outcome actually observed: lateral drift reaches 2.1e-3 m against a 1e-4 m isotropy tolerance, with exact nulls at 0, 45, and 90 deg and antisymmetry about 45 deg to within 1e-13 of peak, in both contact solvers; the drift criterion is the only one of the three tolerances exceeded, and every cell passes the physical-validity gate. Says nothing about other speeds, shapes, stacks, historical DART versions, or DART 6.", "disposition": "reproduced", "limitations": [ - "ResolvedSolverConfiguration is not Python-exposed; resolved identity is property readback, corroborated by per-solver trajectory-hash differences.", + "Resolved identity now comes from the World's own bake-time resolution, corroborated by per-solver trajectory-hash differences.", "No solver residual exists on this path: it is typed unsupported, not reported as zero. Friction-cone and complementarity violations are not exposed either.", "The boxed-LCP path records no iteration count; only sequential impulse reports one, and that is its configured sweep count rather than an observed convergence measure.", "Penetration is clamped at zero by the runtime, so 0.0 means no positive penetration was observed and does not distinguish resting-tangent from separated contacts.", @@ -782,7 +1262,7 @@ }, "target": { "branch": "main", - "commit": "9101d917fdb33998abd0a48d705fe42a632d1493", + "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05", "commit_role": "Source state measured: the library and fixture were run at this commit, which is HEAD at capture time. The packet and its writer land in a later commit, so re-running the recorded command requires the child commit that adds the writer; the measured behavior belongs to this one." }, "title": "Rolling-direction friction dependence (DART 7 first packet)" diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json index 6ce243c049175..7bcee488904f0 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json @@ -29,7 +29,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "BOXED_LCP@dt=0.004": { "contact_solver_method": "BOXED_LCP", @@ -39,7 +69,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "SEQUENTIAL_IMPULSE@dt=0.002": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -49,7 +109,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "SEQUENTIAL_IMPULSE@dt=0.004": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -59,7 +149,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } } } }, @@ -126,7 +246,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 1000, "timestep_s": 0.002 @@ -157,7 +307,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 500, "timestep_s": 0.004 @@ -188,7 +368,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 1000, "timestep_s": 0.002 @@ -219,7 +429,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 500, "timestep_s": 0.004 @@ -377,7 +617,7 @@ "Bounded reconstruction: the original SimBenchmark asset, material, and timestep grid are not reproduced exactly; sourcing the exact historical setup is future corpus work.", "The reproduced signal is a small per-step energy gain in a settled pile, not divergence or failure. It must not be quoted as 'dense contact fails' or as a solver ranking.", "The settle window is the second half of the run; a pile that settles later would put free-fall discretization error inside the window and inflate the energy metric.", - "Resolved identity comes from World property readback per cell (contact solver, timestep, and gravity each asserted against the request); ResolvedSolverConfiguration is not Python-exposed.", + "Resolved identity comes from the World's own bake-time resolution, with contact solver, timestep, and gravity each additionally asserted against the request per cell.", "No solver residual exists on this path and the boxed-LCP branch records no iteration count; both are typed unsupported rather than reported as zero, so the corpus row's residual and iteration oracles are not yet covered.", "The corpus row also names wall time over the timestep grid; this packet makes no timing claim (performance is typed unsupported) because no interleaved same-host methodology was applied.", "max_active_contacts is 216 in every cell (a fully stacked column geometry), so it carries no discriminating information here.", @@ -431,7 +671,7 @@ }, "target": { "branch": "main", - "commit": "9101d917fdb33998abd0a48d705fe42a632d1493" + "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05" }, "title": "Dense 6x6x6 inelastic contact stability (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json index e8cad6052390d..5a513b3d117ba 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json @@ -29,7 +29,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "BOXED_LCP@dt=0.004": { "contact_solver_method": "BOXED_LCP", @@ -39,7 +69,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "SEQUENTIAL_IMPULSE@dt=0.002": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -49,7 +109,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "SEQUENTIAL_IMPULSE@dt=0.004": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -59,7 +149,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } } } }, @@ -125,7 +245,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 1000, "timestep_s": 0.002 @@ -156,7 +306,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 500, "timestep_s": 0.004 @@ -187,7 +367,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 1000, "timestep_s": 0.002 @@ -218,7 +428,37 @@ -9.81 ], "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 500, "timestep_s": 0.004 @@ -334,7 +574,7 @@ }, "target": { "branch": "main", - "commit": "9101d917fdb33998abd0a48d705fe42a632d1493" + "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05" }, "title": "Dense elastic contact energy envelope (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json index 200c7cb5224eb..1da9f7509325b 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json @@ -31,7 +31,37 @@ ], "integration_family": "SEMI_IMPLICIT", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.0005 + "time_step_s": 0.0005, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "SEMI_IMPLICIT@dt=0.001": { "gravity_mps2": [ @@ -41,7 +71,37 @@ ], "integration_family": "SEMI_IMPLICIT", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.001 + "time_step_s": 0.001, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "SEMI_IMPLICIT@dt=0.002": { "gravity_mps2": [ @@ -51,7 +111,37 @@ ], "integration_family": "SEMI_IMPLICIT", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "SEMI_IMPLICIT@dt=0.004": { "gravity_mps2": [ @@ -61,7 +151,37 @@ ], "integration_family": "SEMI_IMPLICIT", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "VARIATIONAL@dt=0.0005": { "gravity_mps2": [ @@ -71,7 +191,37 @@ ], "integration_family": "VARIATIONAL", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.0005 + "time_step_s": 0.0005, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "VARIATIONAL@dt=0.001": { "gravity_mps2": [ @@ -81,7 +231,37 @@ ], "integration_family": "VARIATIONAL", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.001 + "time_step_s": 0.001, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "VARIATIONAL@dt=0.002": { "gravity_mps2": [ @@ -91,7 +271,37 @@ ], "integration_family": "VARIATIONAL", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "VARIATIONAL@dt=0.004": { "gravity_mps2": [ @@ -101,7 +311,37 @@ ], "integration_family": "VARIATIONAL", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } } } }, @@ -177,7 +417,37 @@ ], "integration_family": "SEMI_IMPLICIT", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 500, "timestep_s": 0.004, @@ -203,7 +473,37 @@ ], "integration_family": "SEMI_IMPLICIT", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 1000, "timestep_s": 0.002, @@ -229,7 +529,37 @@ ], "integration_family": "SEMI_IMPLICIT", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.001 + "time_step_s": 0.001, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 2000, "timestep_s": 0.001, @@ -255,7 +585,37 @@ ], "integration_family": "SEMI_IMPLICIT", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.0005 + "time_step_s": 0.0005, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 4000, "timestep_s": 0.0005, @@ -281,7 +641,37 @@ ], "integration_family": "VARIATIONAL", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 500, "timestep_s": 0.004, @@ -307,7 +697,37 @@ ], "integration_family": "VARIATIONAL", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 1000, "timestep_s": 0.002, @@ -333,7 +753,37 @@ ], "integration_family": "VARIATIONAL", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.001 + "time_step_s": 0.001, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 2000, "timestep_s": 0.001, @@ -359,7 +809,37 @@ ], "integration_family": "VARIATIONAL", "rigid_body_solver": "SEQUENTIAL_IMPULSE", - "time_step_s": 0.0005 + "time_step_s": 0.0005, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "steps": 4000, "timestep_s": 0.0005, @@ -497,7 +977,7 @@ }, "target": { "branch": "main", - "commit": "9101d917fdb33998abd0a48d705fe42a632d1493", + "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit." }, "title": "Articulated energy drift versus timestep across integration families (DART 7 first packet)" diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json index 2ce2b85cc6a23..f723726914060 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json @@ -30,7 +30,37 @@ -9.81 ], "integration_family": "SEMI_IMPLICIT", - "time_step_s": 0.0005 + "time_step_s": 0.0005, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "SEMI_IMPLICIT@dt=0.001": { "gravity_mps2": [ @@ -39,7 +69,37 @@ -9.81 ], "integration_family": "SEMI_IMPLICIT", - "time_step_s": 0.001 + "time_step_s": 0.001, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "SEMI_IMPLICIT@dt=0.002": { "gravity_mps2": [ @@ -48,7 +108,37 @@ -9.81 ], "integration_family": "SEMI_IMPLICIT", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "SEMI_IMPLICIT@dt=0.004": { "gravity_mps2": [ @@ -57,7 +147,37 @@ -9.81 ], "integration_family": "SEMI_IMPLICIT", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "VARIATIONAL@dt=0.0005": { "gravity_mps2": [ @@ -66,7 +186,37 @@ -9.81 ], "integration_family": "VARIATIONAL", - "time_step_s": 0.0005 + "time_step_s": 0.0005, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "VARIATIONAL@dt=0.001": { "gravity_mps2": [ @@ -75,7 +225,37 @@ -9.81 ], "integration_family": "VARIATIONAL", - "time_step_s": 0.001 + "time_step_s": 0.001, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "VARIATIONAL@dt=0.002": { "gravity_mps2": [ @@ -84,7 +264,37 @@ -9.81 ], "integration_family": "VARIATIONAL", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "VARIATIONAL@dt=0.004": { "gravity_mps2": [ @@ -93,11 +303,41 @@ -9.81 ], "integration_family": "VARIATIONAL", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } } } }, - "resolved_provenance": "World property readback after enter_simulation_mode, with integration family, timestep, and gravity each asserted equal to the request per repeat and per cell. The controller is applied by this script, not resolved by the World. configuration.timestep is the smallest swept value; configuration.resolved.by_cell is authoritative per cell. World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", + "resolved_provenance": "World property readback after enter_simulation_mode, with integration family, timestep, and gravity each asserted equal to the request per repeat and per cell. The controller is applied by this script, not resolved by the World. configuration.timestep is the smallest swept value; configuration.resolved.by_cell is authoritative per cell. The resolved identity is the World's own bake-time resolution (World.resolved_configuration), recorded per domain with requested, resolved, reason, and a substitution flag, so a method that did not run cannot be reported as if it had. Requested options are additionally asserted against readback per repeat and per cell.", "substeps": 1, "timestep": 0.0005 }, @@ -166,7 +406,37 @@ -9.81 ], "integration_family": "SEMI_IMPLICIT", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "rms_tracking_error_rad": 0.0021383697024277538, "steps": 500, @@ -191,7 +461,37 @@ -9.81 ], "integration_family": "SEMI_IMPLICIT", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "rms_tracking_error_rad": 0.0022892179585315767, "steps": 1000, @@ -216,7 +516,37 @@ -9.81 ], "integration_family": "SEMI_IMPLICIT", - "time_step_s": 0.001 + "time_step_s": 0.001, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "rms_tracking_error_rad": 0.002370166622085158, "steps": 2000, @@ -241,7 +571,37 @@ -9.81 ], "integration_family": "SEMI_IMPLICIT", - "time_step_s": 0.0005 + "time_step_s": 0.0005, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "rms_tracking_error_rad": 0.0024118796589271295, "steps": 4000, @@ -266,7 +626,37 @@ -9.81 ], "integration_family": "VARIATIONAL", - "time_step_s": 0.004 + "time_step_s": 0.004, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "rms_tracking_error_rad": 0.0021389444846853533, "steps": 500, @@ -291,7 +681,37 @@ -9.81 ], "integration_family": "VARIATIONAL", - "time_step_s": 0.002 + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "rms_tracking_error_rad": 0.0022896056362997587, "steps": 1000, @@ -316,7 +736,37 @@ -9.81 ], "integration_family": "VARIATIONAL", - "time_step_s": 0.001 + "time_step_s": 0.001, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "rms_tracking_error_rad": 0.002370382161031873, "steps": 2000, @@ -341,7 +791,37 @@ -9.81 ], "integration_family": "VARIATIONAL", - "time_step_s": 0.0005 + "time_step_s": 0.0005, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "variational", + "resolved": "variational" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } }, "rms_tracking_error_rad": 0.002411992465209295, "steps": 4000, @@ -496,7 +976,7 @@ }, "target": { "branch": "main", - "commit": "2c33043f8ca9616dad99df65593894138e1c5201", + "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit." }, "title": "PD-control tracking error and control work versus timestep (DART 7 first packet)" diff --git a/docs/plans/dashboard.md b/docs/plans/dashboard.md index 6d4f944abfdd9..d25cb55e75d8f 100644 --- a/docs/plans/dashboard.md +++ b/docs/plans/dashboard.md @@ -55,9 +55,10 @@ its own line so status updates remain git-history friendly. `release-6.20`. Continue WS2 with the remaining first-wave families: heel-strike/toe-off (CT-006, gated on WS3 contact semantics), high mass-ratio stacks (CT-007), and reset/concurrency queries (CT-011..013). - In parallel, start WS4's first slice, which several packets are waiting on: - expose `ResolvedSolverConfiguration` to Python and a comparable per-solve - residual, both currently typed unsupported everywhere. History: + WS4's first slice has landed: `World.resolved_configuration` is exposed to + Python and every packet now records the World's own bake-time resolution. + The remaining WS4 gap is a comparable per-solve residual, still unexposed + and typed unsupported everywhere. History: `## Progress log` in the owner plan and `docs/dev_tasks/citation_driven_simulation_trust/`. - Gate: Every claim row is branch/version-qualified and source-bound; diff --git a/python/dartpy/simulation/module_compute.cpp b/python/dartpy/simulation/module_compute.cpp index 2eb3568a60f84..9d5388df31811 100644 --- a/python/dartpy/simulation/module_compute.cpp +++ b/python/dartpy/simulation/module_compute.cpp @@ -301,6 +301,53 @@ void defSimPartCompute(nb::module_& m) .def("__repr__", &sim::compute::WorldStepProfile::toSummaryText) .def("__str__", &sim::compute::WorldStepProfile::toSummaryText); + nb::class_( + m, "ResolvedConfigurationNote") + .def_ro( + "domain", + &sim::compute::ResolvedConfigurationNote::domain, + "Domain the decision applies to, e.g. 'rigid-contact'.") + .def_ro( + "requested", + &sim::compute::ResolvedConfigurationNote::requested, + "Method family the World was asked for.") + .def_ro( + "resolved", + &sim::compute::ResolvedConfigurationNote::resolved, + "Method family the step actually ran.") + .def_ro( + "reason", + &sim::compute::ResolvedConfigurationNote::reason, + "Why the resolution differs from the request, or 'as requested'.") + .def_prop_ro( + "is_substitution", + &sim::compute::ResolvedConfigurationNote::isSubstitution, + "True when the resolved family differs from the requested one.") + .def("__repr__", [](const sim::compute::ResolvedConfigurationNote& self) { + return self.domain + ": " + self.requested + " -> " + self.resolved + + " (" + self.reason + ")"; + }); + + nb::class_( + m, "ResolvedSolverConfiguration") + .def_ro( + "notes", + &sim::compute::ResolvedSolverConfiguration::notes, + "Per-domain requested -> resolved decisions recorded at bake time.") + .def("is_empty", &sim::compute::ResolvedSolverConfiguration::isEmpty) + .def( + "has_substitution", + &sim::compute::ResolvedSolverConfiguration::hasSubstitution, + "True if any domain resolved to a different family than requested.") + .def( + "summary", + &sim::compute::ResolvedSolverConfiguration::toSummaryText, + "Compact 'domain: requested -> resolved (reason)' table.") + .def( + "__repr__", &sim::compute::ResolvedSolverConfiguration::toSummaryText) + .def( + "__str__", &sim::compute::ResolvedSolverConfiguration::toSummaryText); + nb::class_(m, "StepMetrics") .def_ro( "kinetic_energy", diff --git a/python/dartpy/simulation/module_world.cpp b/python/dartpy/simulation/module_world.cpp index 02cc5291ab7ff..a7169873c3568 100644 --- a/python/dartpy/simulation/module_world.cpp +++ b/python/dartpy/simulation/module_world.cpp @@ -571,6 +571,16 @@ void defSimPartWorld(nb::module_& m) "Per-stage wall-clock profile of the most recent step taken while " "step profiling was enabled (empty otherwise). Call .summary() for a " "compact text breakdown of where the step spent its time.") + .def_prop_ro( + "resolved_configuration", + &sim::World::getResolvedConfiguration, + nb::rv_policy::reference_internal, + "Per-domain solver families this World resolved at " + "enter_simulation_mode: what was requested, what actually ran, and " + "why. Empty until the World enters simulation mode, and re-recorded " + "on every rebake. Use this rather than echoing back the requested " + "option, so a benchmark or evidence packet cannot report a method " + "that did not run.") .def( "compute_step_metrics", &sim::World::computeStepMetrics, diff --git a/python/stubs/dartpy/simulation.pyi b/python/stubs/dartpy/simulation.pyi index babd8d8a1732b..404453d301b4c 100644 --- a/python/stubs/dartpy/simulation.pyi +++ b/python/stubs/dartpy/simulation.pyi @@ -65,6 +65,8 @@ __all__: list[str] = [ "StateVariable", "StepDerivatives", "StepGradient", + "ResolvedConfigurationNote", + "ResolvedSolverConfiguration", "StepMetrics", "World", "WorldEcsDiagnostics", @@ -1625,6 +1627,38 @@ class WorldStepProfile: def __str__(self) -> str: ... +class ResolvedConfigurationNote: + @property + def domain(self) -> str: ... + + @property + def requested(self) -> str: ... + + @property + def resolved(self) -> str: ... + + @property + def reason(self) -> str: ... + + @property + def is_substitution(self) -> bool: ... + + def __repr__(self) -> str: ... + +class ResolvedSolverConfiguration: + @property + def notes(self) -> list[ResolvedConfigurationNote]: ... + + def is_empty(self) -> bool: ... + + def has_substitution(self) -> bool: ... + + def summary(self) -> str: ... + + def __repr__(self) -> str: ... + + def __str__(self) -> str: ... + class StepMetrics: @property def kinetic_energy(self) -> float: ... @@ -2543,6 +2577,9 @@ class World: @property def last_step_profile(self) -> WorldStepProfile: ... + @property + def resolved_configuration(self) -> ResolvedSolverConfiguration: ... + def compute_step_metrics(self) -> StepMetrics: ... @property diff --git a/python/tests/unit/simulation/test_world.py b/python/tests/unit/simulation/test_world.py index 8cecab175cf5d..42b9f377920a9 100644 --- a/python/tests/unit/simulation/test_world.py +++ b/python/tests/unit/simulation/test_world.py @@ -546,6 +546,7 @@ def test_simulation_stub_tracks_public_runtime_symbols(): "is_valid", "step_profiling_enabled", "last_step_profile", + "resolved_configuration", "worker_count", "inline_threshold", "graph_profiles", @@ -9505,3 +9506,61 @@ def test_simulation_deformable_scene_loader_python_api(tmp_path): assert diagnostics.tetrahedron_count == 1 # Unit corner tetrahedron volume 1/6 at density 6 has total mass 1. assert diagnostics.total_mass == pytest.approx(1.0) + + +def test_simulation_world_resolved_configuration_reports_actual_methods(): + """Resolved solver identity must come from the World, not the request. + + A benchmark or evidence packet that echoes back the requested option + cannot tell whether the method it names actually ran. This exposes the + World's own bake-time resolution so a packet can record what really + executed, per domain, with substitutions flagged. + """ + world = dart.simulation.World( + time_step=0.002, + contact_solver_method=dart.simulation.ContactSolverMethod.BOXED_LCP, + ) + assert world.resolved_configuration.is_empty() + + body = world.add_rigid_body("body") + body.mass = 1.0 + body.set_collision_shape(dart.simulation.CollisionShape.sphere(0.1)) + world.enter_simulation_mode() + + resolved = world.resolved_configuration + assert not resolved.is_empty() + assert not resolved.has_substitution() + + by_domain = {note.domain: note for note in resolved.notes} + assert "rigid-contact" in by_domain + contact = by_domain["rigid-contact"] + assert contact.requested == "boxed-lcp" + assert contact.resolved == "boxed-lcp" + assert contact.is_substitution is False + assert contact.reason + assert "boxed-lcp" in resolved.summary() + + for note in resolved.notes: + assert note.is_substitution == (note.requested != note.resolved) + + +def test_simulation_world_resolved_configuration_tracks_requested_method(): + """Changing the requested contact solver changes what is reported.""" + resolutions = {} + for method in ("SEQUENTIAL_IMPULSE", "BOXED_LCP"): + world = dart.simulation.World( + time_step=0.002, + contact_solver_method=getattr( + dart.simulation.ContactSolverMethod, method + ), + ) + body = world.add_rigid_body("body") + body.mass = 1.0 + body.set_collision_shape(dart.simulation.CollisionShape.sphere(0.1)) + world.enter_simulation_mode() + by_domain = { + note.domain: note for note in world.resolved_configuration.notes + } + resolutions[method] = by_domain["rigid-contact"].resolved + + assert resolutions["SEQUENTIAL_IMPULSE"] != resolutions["BOXED_LCP"] diff --git a/scripts/citation_packet_utils.py b/scripts/citation_packet_utils.py index a1a64690bb25a..9e20c3745a34c 100644 --- a/scripts/citation_packet_utils.py +++ b/scripts/citation_packet_utils.py @@ -131,3 +131,27 @@ def preserve_review(output_path: "Any") -> dict[str, Any]: if isinstance(review, dict) and isinstance(review.get("passes"), list): return {"passes": review["passes"]} return {"passes": []} + + +def world_resolved_configuration(world: Any) -> dict[str, Any]: + """Record the World's own bake-time solver resolution. + + This is the authoritative answer to "which method actually ran": the + World reports, per domain, what was requested, what it resolved to, and + why. Echoing back the requested option cannot distinguish a method that + ran from one that was silently substituted. + """ + resolved = world.resolved_configuration + return { + "source": "World.resolved_configuration (bake-time resolution)", + "has_substitution": bool(resolved.has_substitution()), + "by_domain": { + note.domain: { + "requested": note.requested, + "resolved": note.resolved, + "reason": note.reason, + "is_substitution": bool(note.is_substitution), + } + for note in resolved.notes + }, + } diff --git a/scripts/write_citation_ct001_rolling_direction_packet.py b/scripts/write_citation_ct001_rolling_direction_packet.py index 27365bb777761..c6810d3178b16 100644 --- a/scripts/write_citation_ct001_rolling_direction_packet.py +++ b/scripts/write_citation_ct001_rolling_direction_packet.py @@ -51,6 +51,7 @@ UNSUPPORTED_SOLVER_RESIDUAL, preserve_review, solver_iterations_by_method, + world_resolved_configuration, ) REPO_ROOT = Path(__file__).resolve().parents[1] @@ -143,6 +144,7 @@ def run_single( world.enter_simulation_mode() resolved = { + "world_resolution": world_resolved_configuration(world), "contact_solver_method": world.contact_solver_method.name, "rigid_body_solver": world.rigid_body_solver.name, "gravity_mps2": [float(v) for v in np.asarray(world.gravity)], @@ -596,9 +598,9 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "shapes, stacks, historical DART versions, or DART 6." ), "limitations": [ - "ResolvedSolverConfiguration is not Python-exposed; " - "resolved identity is property readback, corroborated by " - "per-solver trajectory-hash differences.", + "Resolved identity now comes from the World's own bake-time " + "resolution, corroborated by per-solver trajectory-hash " + "differences.", "No solver residual exists on this path: it is typed " "unsupported, not reported as zero. Friction-cone and " "complementarity violations are not exposed either.", diff --git a/scripts/write_citation_ct002_dense_contact_packet.py b/scripts/write_citation_ct002_dense_contact_packet.py index ca9e9b77a066a..07c15a8534566 100644 --- a/scripts/write_citation_ct002_dense_contact_packet.py +++ b/scripts/write_citation_ct002_dense_contact_packet.py @@ -47,6 +47,7 @@ UNSUPPORTED_SOLVER_RESIDUAL, preserve_review, solver_iterations_by_method, + world_resolved_configuration, ) REPO_ROOT = Path(__file__).resolve().parents[1] @@ -150,6 +151,7 @@ def run_single( world.enter_simulation_mode() resolved = { + "world_resolution": world_resolved_configuration(world), "contact_solver_method": world.contact_solver_method.name, "rigid_body_solver": world.rigid_body_solver.name, "gravity_mps2": [float(v) for v in np.asarray(world.gravity)], @@ -594,10 +596,9 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "The settle window is the second half of the run; a pile " "that settles later would put free-fall discretization error " "inside the window and inflate the energy metric.", - "Resolved identity comes from World property readback per " - "cell (contact solver, timestep, and gravity each asserted " - "against the request); ResolvedSolverConfiguration is not " - "Python-exposed.", + "Resolved identity comes from the World's own bake-time " + "resolution, with contact solver, timestep, and gravity " + "each additionally asserted against the request per cell.", "No solver residual exists on this path and the boxed-LCP " "branch records no iteration count; both are typed " "unsupported rather than reported as zero, so the corpus " diff --git a/scripts/write_citation_ct003_elastic_contact_packet.py b/scripts/write_citation_ct003_elastic_contact_packet.py index 331af2d9e9257..52e92041b1c25 100644 --- a/scripts/write_citation_ct003_elastic_contact_packet.py +++ b/scripts/write_citation_ct003_elastic_contact_packet.py @@ -48,6 +48,7 @@ UNSUPPORTED_SOLVER_RESIDUAL, preserve_review, solver_iterations_by_method, + world_resolved_configuration, ) REPO_ROOT = Path(__file__).resolve().parents[1] @@ -150,6 +151,7 @@ def run_single( world.enter_simulation_mode() resolved = { + "world_resolution": world_resolved_configuration(world), "contact_solver_method": world.contact_solver_method.name, "rigid_body_solver": world.rigid_body_solver.name, "gravity_mps2": [float(v) for v in np.asarray(world.gravity)], diff --git a/scripts/write_citation_ct004_articulated_energy_packet.py b/scripts/write_citation_ct004_articulated_energy_packet.py index 48b9e2aefee13..84b7c4e9d2d7d 100644 --- a/scripts/write_citation_ct004_articulated_energy_packet.py +++ b/scripts/write_citation_ct004_articulated_energy_packet.py @@ -46,6 +46,7 @@ from citation_packet_utils import ( UNSUPPORTED_SOLVER_RESIDUAL, preserve_review, + world_resolved_configuration, ) REPO_ROOT = Path(__file__).resolve().parents[1] @@ -133,6 +134,7 @@ def run_single( world.enter_simulation_mode() resolved = { + "world_resolution": world_resolved_configuration(world), "integration_family": world.multibody_options.integration_family.name, "rigid_body_solver": world.rigid_body_solver.name, "gravity_mps2": [float(v) for v in np.asarray(world.gravity)], diff --git a/scripts/write_citation_ct005_pd_tracking_packet.py b/scripts/write_citation_ct005_pd_tracking_packet.py index 130708b2fd7ea..d71b4e873f266 100644 --- a/scripts/write_citation_ct005_pd_tracking_packet.py +++ b/scripts/write_citation_ct005_pd_tracking_packet.py @@ -46,6 +46,7 @@ from citation_packet_utils import ( UNSUPPORTED_SOLVER_RESIDUAL, preserve_review, + world_resolved_configuration, ) REPO_ROOT = Path(__file__).resolve().parents[1] @@ -198,6 +199,7 @@ def run_single( world.enter_simulation_mode() resolved = { + "world_resolution": world_resolved_configuration(world), "integration_family": world.multibody_options.integration_family.name, "gravity_mps2": [float(v) for v in np.asarray(world.gravity)], "time_step_s": float(world.time_step), @@ -466,8 +468,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "is applied by this script, not resolved by the World. " "configuration.timestep is the smallest swept value; " "configuration.resolved.by_cell is authoritative per cell. " - "World::getResolvedConfiguration() is not yet exposed to " - "Python (PLAN-123 WS4 follow-up)." + "The resolved identity is the World's own bake-time resolution (World.resolved_configuration), recorded per domain with requested, resolved, reason, and a substitution flag, so a method that did not run cannot be reported as if it had. Requested options are additionally asserted against readback per repeat and per cell." ), "detector": ( "not applicable: the scene has no collision shapes and runs " From 12de896b8b33598e455c8af8bfde2c8782e502c4 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Fri, 14 Aug 2026 21:46:51 -0700 Subject: [PATCH 11/52] Add the CT-007 high-mass-ratio packet and record a default-solver failure Establishes the baseline arm the WS5 exact-cone GO/NO-GO needs: a two-box stack at rest, upper mass swept over four decades, both contact solvers, two deterministic repeats per cell. The row stays unresolved -- the cited claim is comparative and no exact-cone arm exists on this branch -- but the baseline is not a null result. SEQUENTIAL_IMPULSE, the World's default contact solver, holds the stack at mass ratios 1 and 10 (relative closure 1.4e-4 and 1.0e-3) and fails completely at 100 and 1000: the heavy box descends 0.20002 m, exactly one box height, while the light box beneath it moves by microns, and the pair comes to rest fully interpenetrated at near-zero velocity. BOXED_LCP holds the same stack across all four decades (3.3e-5 to 3.0e-4). Both outcomes are bit-identical across repeats, and the scene starts with exactly zero overlap, so the closure is produced by the solve. This is consistent with iterative Gauss-Seidel contact under a fixed iteration budget, which is what makes it credible rather than a fixture artifact, but the packet does not claim the mechanism: confirming it needs the per-solve residual WS4 has not exposed. It needs a maintainer decision on the default path and is the strongest WS5 GO input so far. The gate rejected an untyped null for the 'no failure onset' case on the first attempt; that is now a typed unsupported marker. --- CHANGELOG.md | 5 +- .../README.md | 15 +- .../verification.md | 41 + .../123-citation-driven-simulation-trust.md | 7 + .../claims-manifest.json | 7 +- .../CT-007-dart7-high-mass-ratio.json | 1014 +++++++++++++++++ docs/plans/dashboard.md | 10 +- ...e_citation_ct007_high_mass_ratio_packet.py | 606 ++++++++++ 8 files changed, 1693 insertions(+), 12 deletions(-) create mode 100644 docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json create mode 100644 scripts/write_citation_ct007_high_mass_ratio_packet.py diff --git a/CHANGELOG.md b/CHANGELOG.md index b4123395668bc..265c7397ced56 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -622,7 +622,10 @@ compatibility remains on the active DART 6 LTS branch._ integration families (also unresolved: convergence is observed, the cited matched-cost comparison is not measured), and CT-005 PD-control tracking (also unresolved: tracking error proves controller-limited rather than - integration-limited, and step cost is not measured). + integration-limited, and step cost is not measured), and CT-007 + high-mass-ratio stack conditioning, which records that the default + sequential-impulse contact solver fails to hold a two-box stack at mass + ratios of 100 and above while the boxed-LCP path holds across four decades. - Reorganized tests and CI coverage around DART 7 components, with focused unit, integration, benchmark, rendering, CUDA-smoke, collision, and simulation gates replacing broad stale test targets. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/README.md b/docs/dev_tasks/citation_driven_simulation_trust/README.md index 4c87b37ef36bf..416422ae38639 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/README.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/README.md @@ -17,7 +17,9 @@ energy drift versus timestep (`unresolved`: both integration families converge, but the cited matched-cost comparison is not measured), and CT-005 PD tracking (`unresolved`: error is controller-limited and step - cost is not measured). Landed on + cost is not measured), and CT-007 high mass-ratio stack (`unresolved` + for the comparative claim, but it found that the default contact solver + fails completely at mass ratios 100 and 1000 where boxed LCP holds). Landed on `release-6.20`: CT-001 detector sweep (`reproduced` on fcl/dart/ode; bullet excluded for lacking the pyramid signature). Open: articulated energy/momentum/control, heel-strike/toe-off, high mass-ratio, and @@ -160,10 +162,13 @@ for downstream-sensitive changes. ## Immediate next steps 1. Read `RESUME.md` and verify both branch tips/worktrees. -2. Continue Phase 2 with heel-strike/toe-off (CT-006), which depends on the - WS3 contact semantics, and the high mass-ratio stack row (CT-007). -3. Start WS4's first slice in parallel where it unblocks packets: expose +2. Bring the CT-007 sequential-impulse failure to the maintainer: it is a + default-path defect, the strongest WS5 GO input so far, and confirming its + mechanism needs the per-solve residual WS4 has not exposed. +3. Continue Phase 2 with heel-strike/toe-off (CT-006), which depends on the + WS3 contact semantics, and reset/concurrency queries (CT-011..013). +4. Start WS4's first slice in parallel where it unblocks packets: expose `ResolvedSolverConfiguration` to Python and a comparable per-solve residual, both currently typed unsupported in every packet. -4. Reuse PR #3377 fixtures for the DART 6 CT-007 lane rather than adding +5. Reuse PR #3377 fixtures for the DART 6 CT-007 lane rather than adding new scenes; keep PLAN-621/622 rows referenced, not copied. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 4756d86eff7e2..c127a6d67fa0d 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -329,3 +329,44 @@ WS4-scale work. `pixi run check-docs-policy` — passed. - Remaining WS4 gap: no comparable per-solve residual exists, so `metrics.numerical.solver_residual` stays typed unsupported in every packet. + +## CT-007 high-mass-ratio slice — 2026-08-14 + +- Scene: two-box stack at rest (0.2 m cubes, 1 kg lower box), upper mass swept + over four decades so the loaded contact sees ratios 1:1 to 1000:1, both + contact solvers, dt 2 ms, 2 s horizon, two deterministic repeats per cell. +- Disposition `unresolved`: the cited claim compares against exact-cone + methods and no exact-cone arm exists on this branch, so the row cannot be + reproduced. The packet establishes the baseline arm the WS5 GO/NO-GO needs. + +### Finding worth a maintainer decision + +`SEQUENTIAL_IMPULSE`, the World's default contact solver, fails completely at +mass ratios 100 and 1000: + +| solver | ratio 1 | ratio 10 | ratio 100 | ratio 1000 | +| ------------------ | ------- | -------- | ---------- | ---------- | +| SEQUENTIAL_IMPULSE | 1.41e-4 | 1.02e-3 | **1.0000** | **1.0000** | +| BOXED_LCP | 1.96e-4 | 1.98e-4 | 3.26e-5 | -3.01e-4 | + +(relative closure of the loaded contact: 1.0 means the two box centers have +closed by a full box height.) + +At ratios 100 and 1000 the heavy box descends 0.20002 m -- exactly one box +height -- while the lower box moves by microns (2.4e-6 and 5.3e-6 m), and the +pair comes to rest fully interpenetrated at near-zero velocity (4.8e-7 and +3.5e-6 m/s). The upper box passes through the lower one rather than the pair +sinking into the ground. Both outcomes are bit-identical across repeats. +`BOXED_LCP` holds the same stack across all four decades. + +This is consistent with the known behavior of iterative Gauss-Seidel contact +at high mass ratio under a fixed iteration budget, which is what makes it +credible rather than a fixture artifact. The packet does not claim the +mechanism: confirming it needs the per-solve residual WS4 has not exposed. +The scene rests each box exactly on the one below, so the initial overlap is +zero and the observed closure is produced by the solve. + +- Commands: packet writer as recorded in the packet; + `pixi run check-citation-evidence` — OK (it rejected an untyped `None` for + the "no failure onset" case on the first attempt, which is now a typed + unsupported marker). diff --git a/docs/plans/123-citation-driven-simulation-trust.md b/docs/plans/123-citation-driven-simulation-trust.md index 2fb10a17854f3..b138dd87a931a 100644 --- a/docs/plans/123-citation-driven-simulation-trust.md +++ b/docs/plans/123-citation-driven-simulation-trust.md @@ -253,6 +253,13 @@ and domain expansion are DART 7 only. instead of echoing back the requested option. A benchmark can no longer name a method that did not run. A comparable per-solve residual remains unexposed and is still typed unsupported everywhere. +- CT-007 established the WS5 baseline arm and produced the session's most + consequential finding: the World's default contact solver + (`SEQUENTIAL_IMPULSE`) fails completely on a two-box stack at mass ratios + 100 and 1000 -- the heavy box descends a full box height into the light one + and rests there -- while `BOXED_LCP` holds across all four swept decades. + Deterministic across repeats. This needs a maintainer decision on the + default path, and it is the strongest evidence so far toward a WS5 GO. - Known gap: the validator does not arithmetically cross-check derived metrics against `raw_rows`, so a falsified summary would pass. Closing that needs the gate to know each packet's derivation (WS4-scale). diff --git a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json index 92b38b7517e19..593623fd2d61a 100644 --- a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json +++ b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json @@ -156,9 +156,12 @@ "lanes": { "dart7": { "owner": "PLAN-123 exact-cone GO/NO-GO (WS5)", - "status": "audit-required", + "status": "in-progress", "disposition": null, - "evidence": [] + "evidence": [ + "evidence/CT-007-dart7-high-mass-ratio.json" + ], + "notes": "Two-box stack swept over mass ratios 1..1000 for both contact solvers. Disposition unresolved: the cited claim is comparative and no exact-cone arm exists on this branch. The baseline arm is established and is not a null result -- SEQUENTIAL_IMPULSE (the World default) fails completely at ratios 100 and 1000, the heavy box descending a full box height into the light one, while BOXED_LCP holds across all four decades. This is the WS5 GO/NO-GO input and is worth a maintainer decision on its own." }, "dart6": { "owner": "evidence only; reference PR #3377 research lane", diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json new file mode 100644 index 0000000000000..c424cd970189e --- /dev/null +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json @@ -0,0 +1,1014 @@ +{ + "claim_id": "CT-007", + "configuration": { + "detector": "DART 7 native World collision pipeline (the World step API exposes no detector selection on main)", + "fallback_policy": "World defaults; no per-island fallback reporting is exposed on main", + "iterations": "World defaults; sequential impulse reports its configured sweep count and the boxed-LCP path records none", + "requested": { + "backend": "cpu", + "contact_solver_method_sweep": [ + "SEQUENTIAL_IMPULSE", + "BOXED_LCP" + ], + "exact_cone_solver": "not available on this branch; the comparison arm the cited claim needs does not exist yet", + "mass_ratio_sweep": [ + 1.0, + 10.0, + 100.0, + 1000.0 + ], + "precision": "float64", + "threads": "World default sequential step" + }, + "resolved": { + "by_cell": { + "BOXED_LCP@ratio=1.0": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "BOXED_LCP@ratio=10.0": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "BOXED_LCP@ratio=100.0": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "BOXED_LCP@ratio=1000.0": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "SEQUENTIAL_IMPULSE@ratio=1.0": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "SEQUENTIAL_IMPULSE@ratio=10.0": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "SEQUENTIAL_IMPULSE@ratio=100.0": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "SEQUENTIAL_IMPULSE@ratio=1000.0": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + } + } + }, + "resolved_provenance": "The resolved identity is the World's own bake-time resolution (World.resolved_configuration), recorded per domain with requested, resolved, reason, and a substitution flag. Requested contact solver and timestep are additionally asserted against readback per repeat and per cell.", + "substeps": 1, + "timestep": 0.002 + }, + "ensemble": { + "deterministic_repeats": 2, + "deterministic_repeats_identical": true, + "kind": "parameter-sweep-with-deterministic-repeats", + "measurement_window": { + "end_s": 2.0, + "start_s": 0.0 + }, + "sweep": [ + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "mass_ratio": 1.0 + }, + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "mass_ratio": 10.0 + }, + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "mass_ratio": 100.0 + }, + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "mass_ratio": 1000.0 + }, + { + "contact_solver_method": "BOXED_LCP", + "mass_ratio": 1.0 + }, + { + "contact_solver_method": "BOXED_LCP", + "mass_ratio": 10.0 + }, + { + "contact_solver_method": "BOXED_LCP", + "mass_ratio": 100.0 + }, + { + "contact_solver_method": "BOXED_LCP", + "mass_ratio": 1000.0 + } + ] + }, + "evidence": { + "commands": [ + "PYTHONPATH=build/default/cpp/Release/python pixi run python scripts/write_citation_ct007_high_mass_ratio_packet.py" + ], + "raw_rows": [ + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_max_speed_mps": 0.00045895830534830895, + "final_state_sha256": "0d8e9e8806ccf8792179d1d7a7ab1803a91339b44302827bf36e481d64e3de1c", + "finite": true, + "gap_closure_m": 2.8186383662115455e-05, + "lower_sink_m": 3.8875436905949634e-05, + "mass_ratio": 1.0, + "max_active_contacts": 8, + "max_penetration_m": 0.00010552432357324726, + "max_solver_iterations": 8, + "relative_gap_closure": 0.00014093191831057728, + "repeat_state_sha256": [ + "0d8e9e8806ccf8792179d1d7a7ab1803a91339b44302827bf36e481d64e3de1c", + "0d8e9e8806ccf8792179d1d7a7ab1803a91339b44302827bf36e481d64e3de1c" + ], + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "upper_sink_m": 6.706182056809284e-05 + }, + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_max_speed_mps": 0.029482304187420837, + "final_state_sha256": "546e71ddc989dab2389455d59f9735021eac98f9b8a54fccaf980777d01b8345", + "finite": true, + "gap_closure_m": 0.00020330330581502798, + "lower_sink_m": 0.0003313631340894213, + "mass_ratio": 10.0, + "max_active_contacts": 8, + "max_penetration_m": 0.000993297316667946, + "max_solver_iterations": 8, + "relative_gap_closure": 0.00101651652907514, + "repeat_state_sha256": [ + "546e71ddc989dab2389455d59f9735021eac98f9b8a54fccaf980777d01b8345", + "546e71ddc989dab2389455d59f9735021eac98f9b8a54fccaf980777d01b8345" + ], + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "upper_sink_m": 0.0005346664399044632 + }, + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_max_speed_mps": 4.830545337362943e-07, + "final_state_sha256": "f6872ff128b5b9d6bd9bbd1fc3adee6e270f78b5f69db09dc2f296a07c391ce7", + "finite": true, + "gap_closure_m": 0.2000161894973165, + "lower_sink_m": 2.403341138421111e-06, + "mass_ratio": 100.0, + "max_active_contacts": 12, + "max_penetration_m": 0.03329680014905512, + "max_solver_iterations": 8, + "relative_gap_closure": 1.0000809474865824, + "repeat_state_sha256": [ + "f6872ff128b5b9d6bd9bbd1fc3adee6e270f78b5f69db09dc2f296a07c391ce7", + "f6872ff128b5b9d6bd9bbd1fc3adee6e270f78b5f69db09dc2f296a07c391ce7" + ], + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "upper_sink_m": 0.20001859283845497 + }, + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "final_max_speed_mps": 3.542993305133921e-06, + "final_state_sha256": "d0c911c7ae66e79c3b76279726147f29ab73c6e1b261e2661ce52ea22ee2567e", + "finite": true, + "gap_closure_m": 0.2000041526819762, + "lower_sink_m": 5.348353172895948e-06, + "mass_ratio": 1000.0, + "max_active_contacts": 9, + "max_penetration_m": 0.09336215318424317, + "max_solver_iterations": 8, + "relative_gap_closure": 1.0000207634098808, + "repeat_state_sha256": [ + "d0c911c7ae66e79c3b76279726147f29ab73c6e1b261e2661ce52ea22ee2567e", + "d0c911c7ae66e79c3b76279726147f29ab73c6e1b261e2661ce52ea22ee2567e" + ], + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "upper_sink_m": 0.20000950103514914 + }, + { + "contact_solver_method": "BOXED_LCP", + "final_max_speed_mps": 1.4938402596549068e-11, + "final_state_sha256": "d84ea1f3527a71440ea5adec7ba0f7b7e6441595a2ed019131cd0065904fefbb", + "finite": true, + "gap_closure_m": 3.925327088483144e-05, + "lower_sink_m": 1.3457496714358586e-06, + "mass_ratio": 1.0, + "max_active_contacts": 8, + "max_penetration_m": 3.947738840376358e-05, + "max_solver_iterations": 0, + "relative_gap_closure": 0.00019626635442415719, + "repeat_state_sha256": [ + "d84ea1f3527a71440ea5adec7ba0f7b7e6441595a2ed019131cd0065904fefbb", + "d84ea1f3527a71440ea5adec7ba0f7b7e6441595a2ed019131cd0065904fefbb" + ], + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "upper_sink_m": 4.059902055630893e-05 + }, + { + "contact_solver_method": "BOXED_LCP", + "final_max_speed_mps": 8.665822654413277e-06, + "final_state_sha256": "ac4dc0f23c4de5fbbdb30f19506cb6eb798962e06bc9d429593a7f324e95deaf", + "finite": true, + "gap_closure_m": 3.957920874031462e-05, + "lower_sink_m": 3.093141870136318e-07, + "mass_ratio": 10.0, + "max_active_contacts": 8, + "max_penetration_m": 5.523249223463034e-05, + "max_solver_iterations": 0, + "relative_gap_closure": 0.0001978960437015731, + "repeat_state_sha256": [ + "ac4dc0f23c4de5fbbdb30f19506cb6eb798962e06bc9d429593a7f324e95deaf", + "ac4dc0f23c4de5fbbdb30f19506cb6eb798962e06bc9d429593a7f324e95deaf" + ], + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "upper_sink_m": 3.988852292735601e-05 + }, + { + "contact_solver_method": "BOXED_LCP", + "final_max_speed_mps": 0.0008169729485216048, + "final_state_sha256": "43ac32aeec725c269d2b042c4ebae272c40d5f5bbf817e010c1f0353740e385f", + "finite": true, + "gap_closure_m": 6.527479483597887e-06, + "lower_sink_m": 3.401292682585211e-05, + "mass_ratio": 100.0, + "max_active_contacts": 8, + "max_penetration_m": 0.00017047603952824453, + "max_solver_iterations": 0, + "relative_gap_closure": 3.263739741798943e-05, + "repeat_state_sha256": [ + "43ac32aeec725c269d2b042c4ebae272c40d5f5bbf817e010c1f0353740e385f", + "43ac32aeec725c269d2b042c4ebae272c40d5f5bbf817e010c1f0353740e385f" + ], + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "upper_sink_m": 4.054040630946387e-05 + }, + { + "contact_solver_method": "BOXED_LCP", + "final_max_speed_mps": 0.041688097943890993, + "final_state_sha256": "2ced2859d56f20af8c566c3c5192580f8e5d768e9acff317f5097839f7829fa1", + "finite": true, + "gap_closure_m": -6.027789048593246e-05, + "lower_sink_m": 0.00011855439148550362, + "mass_ratio": 1000.0, + "max_active_contacts": 8, + "max_penetration_m": 0.0009128958700206913, + "max_solver_iterations": 0, + "relative_gap_closure": -0.0003013894524296623, + "repeat_state_sha256": [ + "2ced2859d56f20af8c566c3c5192580f8e5d768e9acff317f5097839f7829fa1", + "2ced2859d56f20af8c566c3c5192580f8e5d768e9acff317f5097839f7829fa1" + ], + "resolved": { + "contact_solver_method": "BOXED_LCP", + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "upper_sink_m": 5.827650099959891e-05 + } + ], + "visual": { + "reason": "The oracle is the numeric steady-state closure of the loaded contact; no visible-behavior claim is made.", + "status": "not-applicable" + } + }, + "host": { + "machine": "x86_64", + "note": "Host recorded for provenance only; no timing methodology was applied", + "performance_valid": false, + "platform": "Linux-7.0.0-29-generic-x86_64-with-glibc2.43", + "python": "3.14.6" + }, + "metrics": { + "allocation": { + "reason": "Post-bake allocation gates are owned by PLAN-122 tooling; this packet does not measure allocations", + "status": "unsupported" + }, + "numerical": { + "max_active_contacts": 12, + "max_penetration_m": 0.09336215318424317, + "method": "Max over run of StepMetrics.max_penetration_depth and active_contact_count; per-method iteration counts where the runtime records them", + "penetration_semantics": "StepMetrics.max_penetration_depth clamps each contact depth with std::max(0.0, depth), so a reported 0.0 means no positive penetration was observed and does not distinguish resting-exactly-tangent from separated contacts.", + "solver_iterations_by_method": { + "BOXED_LCP": { + "reason": "The BoxedLcp branch in dart/simulation/compute/rigid_body_contact_stage.cpp returns before recordSolverDiagnostics() runs, so this scene recorded no iteration count for the method: a 0 here means 'not reported', not 'zero iterations'. (An opt-in AVBD stage earlier in the same step can record a count before that branch is reached; this scene enables none, and a nonzero count is published with its provenance instead of this marker.)", + "status": "unsupported" + }, + "SEQUENTIAL_IMPULSE": 8 + }, + "solver_residual": { + "reason": "StepMetrics.last_step_residual is structurally zero for rigid contact on this branch: recordSolverDiagnostics() in dart/simulation/compute/rigid_body_contact_stage.cpp takes residual = 0.0 by default and no rigid contact call site passes one, so no residual is computed for either contact solver. Exposing a comparable residual is PLAN-123 WS4 work.", + "status": "unsupported" + } + }, + "performance": { + "reason": "This packet makes no timing claim; the matched-residual timing a Pareto comparison needs is WS5 work and requires the exact-cone arm to exist first", + "status": "unsupported" + }, + "physical": { + "all_cells_finite": true, + "baseline_finding": { + "BOXED_LCP": { + "fails_at_mass_ratios": [], + "failure_onset_ratio": { + "reason": "This method held the stack at every swept ratio, so there is no onset within the swept range; the onset is unmeasured, not zero or infinite.", + "status": "unsupported" + }, + "holds_across_swept_range": true + }, + "SEQUENTIAL_IMPULSE": { + "fails_at_mass_ratios": [ + 100.0, + 1000.0 + ], + "failure_onset_ratio": 100.0, + "holds_across_swept_range": false + } + }, + "collapse_threshold_relative": 0.1, + "degraded_cells": [ + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "finite": true, + "mass_ratio": 100.0, + "relative_gap_closure": 1.0000809474865824 + }, + { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "finite": true, + "mass_ratio": 1000.0, + "relative_gap_closure": 1.0000207634098808 + } + ], + "method": "Steady-state closure of the loaded contact: how far the two box centers approach each other by the horizon, relative to a box height. Reported per mass ratio and solver, with survival and residual speed.", + "per_method_summary": { + "BOXED_LCP": { + "all_finite": true, + "cells": 4, + "closure_growth_factor": -1.5356144628758956, + "max_final_speed_mps": 0.041688097943890993, + "max_relative_gap_closure": 0.0001978960437015731, + "relative_gap_closure_by_ratio": { + "1.0": 0.00019626635442415719, + "10.0": 0.0001978960437015731, + "100.0": 3.263739741798943e-05, + "1000.0": -0.0003013894524296623 + } + }, + "SEQUENTIAL_IMPULSE": { + "all_finite": true, + "cells": 4, + "closure_growth_factor": 7095.772025227779, + "max_final_speed_mps": 0.029482304187420837, + "max_relative_gap_closure": 1.0000809474865824, + "relative_gap_closure_by_ratio": { + "1.0": 0.00014093191831057728, + "10.0": 0.00101651652907514, + "100.0": 1.0000809474865824, + "1000.0": 1.0000207634098808 + } + } + } + } + }, + "result": { + "claim_boundary": "DART 7 main, this commit, a two-box stack with a 1 kg lower box, mass ratios 1 to 1000, 2 ms timestep, 2 s horizon, SEQUENTIAL_IMPULSE and BOXED_LCP. The cited claim is comparative and its exact-cone arm does not exist on this branch, so the row is NOT reproduced and nothing here says whether an exact cone would help. What this establishes is the baseline arm the WS5 GO/NO-GO needs, and it is not a null result: SEQUENTIAL_IMPULSE -- the World's default contact solver -- holds the stack at mass ratios 1 and 10 (relative closure 1.4e-4 and 1.0e-3) but fails completely at 100 and 1000, where the heavy box descends a full box height (0.20002 m) while the light box beneath it moves by microns, coming to rest fully interpenetrated at near-zero velocity. BOXED_LCP holds across all four decades (closure 3.3e-5 to 3.0e-4). Both outcomes are bit-identical across repeats. Says nothing about deeper stacks, other shapes or materials, other timesteps, historical DART versions, or DART 6.", + "disposition": "unresolved", + "limitations": [ + "The sequential-impulse failure is a geometric observation at one timestep. This packet does not establish the mechanism; a fixed iteration budget is the obvious suspect, and confirming it needs the per-solve residual WS4 has not exposed yet.", + "Two boxes only. Conditioning problems are usually worse in deeper stacks, so this is a floor on the effect, not a characterization of it.", + "One timestep. Contact stiffness interacts with the step size, so a ratio sweep at a single dt cannot separate the two.", + "No exact-cone arm exists, so the packet cannot test the cited improvement; it only records what the current solvers do.", + "Closure is measured at the horizon, not tracked as a time series, so a stack that closes and recovers is not distinguished from one that never closed.", + "No solver residual exists on this path and boxed LCP records no iteration count; both are typed unsupported rather than reported as zero, so the conditioning signal here is geometric rather than algebraic." + ] + }, + "review": { + "passes": [] + }, + "scene": { + "description": "Two-box stack at rest on a static ground: a 1 kg lower box supporting an upper box whose mass is swept over four decades, so the loaded contact sees mass ratios from 1:1 to 1000:1. Boxes are 0.2 m cubes, friction 0.8, restitution 0, simulated to a 2 s horizon per ratio/solver cell.", + "digest": "sha256:e4abfffd26d2eb9655286a23ad72f626144f0ef8d16fc45abc9e47f87d541c83", + "id": "ct007_high_mass_ratio_stack", + "parameters": { + "box_half_extent_m": 0.1, + "contact_solver_methods": [ + "SEQUENTIAL_IMPULSE", + "BOXED_LCP" + ], + "description": "Two-box stack at rest on a static ground: a 1 kg lower box supporting an upper box whose mass is swept over four decades, so the loaded contact sees mass ratios from 1:1 to 1000:1. Boxes are 0.2 m cubes, friction 0.8, restitution 0, simulated to a 2 s horizon per ratio/solver cell.", + "deterministic_repeats": 2, + "friction": 0.8, + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "ground_half_extents_m": [ + 1.0, + 1.0, + 0.05 + ], + "horizon_s": 2.0, + "lower_mass_kg": 1.0, + "mass_ratios": [ + 1.0, + 10.0, + 100.0, + 1000.0 + ], + "restitution": 0.0, + "scene_id": "ct007_high_mass_ratio_stack", + "timestep_s": 0.002 + } + }, + "schema": "dart.citation_claim_evidence/v1", + "source": { + "claim": "Exact Coulomb cones and adaptive proximal methods may improve conditioning and remove friction-pyramid anisotropy.", + "url": "https://arxiv.org/abs/2405.17020" + }, + "target": { + "branch": "main", + "commit": "ebbbb853a49cf59bafb590cf111c110acbd0531b", + "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit." + }, + "title": "High-mass-ratio stack conditioning baseline for the exact-cone GO/NO-GO (DART 7)" +} diff --git a/docs/plans/dashboard.md b/docs/plans/dashboard.md index d25cb55e75d8f..7e0aa94a3c849 100644 --- a/docs/plans/dashboard.md +++ b/docs/plans/dashboard.md @@ -51,10 +51,12 @@ its own line so status updates remain git-history friendly. - Status: Active - Horizon: Now - Dimension: Algorithm extensibility -- Next step: WS0/WS1 and five first-wave packets have landed on `main`, one on - `release-6.20`. Continue WS2 with the remaining first-wave families: - heel-strike/toe-off (CT-006, gated on WS3 contact semantics), high - mass-ratio stacks (CT-007), and reset/concurrency queries (CT-011..013). +- Next step: WS0/WS1 and six first-wave packets have landed on `main`, one on + `release-6.20`. CT-007 found that the default contact solver + (SEQUENTIAL_IMPULSE) fails completely at mass ratios 100 and 1000 where + BOXED_LCP holds; that needs a maintainer decision and is the strongest WS5 + GO input so far. Continue WS2 with heel-strike/toe-off (CT-006, gated on + WS3 contact semantics) and reset/concurrency queries (CT-011..013). WS4's first slice has landed: `World.resolved_configuration` is exposed to Python and every packet now records the World's own bake-time resolution. The remaining WS4 gap is a comparable per-solve residual, still unexposed diff --git a/scripts/write_citation_ct007_high_mass_ratio_packet.py b/scripts/write_citation_ct007_high_mass_ratio_packet.py new file mode 100644 index 0000000000000..0f7d98ab71db6 --- /dev/null +++ b/scripts/write_citation_ct007_high_mass_ratio_packet.py @@ -0,0 +1,606 @@ +#!/usr/bin/env python3 +"""Write the CT-007 high-mass-ratio conditioning packet (PLAN-123 WS2/WS5). + +Bounded claim (corpus row CT-007): exact Coulomb cones and adaptive proximal +methods may improve conditioning and remove friction-pyramid anisotropy. + +That claim is comparative, and DART `main` has no exact-cone contact solver, +so one arm of the comparison does not exist and the row cannot be +`reproduced` here. What this packet does establish is the other arm: how the +solvers DART actually ships degrade as the mass ratio in a resting stack +grows. That baseline is the input the WS5 exact-cone GO/NO-GO decision needs +-- without it, "an exact cone would help" is untestable. + +Bounded reconstruction: a two-box stack at rest, a light box supporting a +heavy one, swept over mass ratio for each rigid contact solver. Per cell it +records: + +- steady-state penetration of the loaded contact, relative to box size, which + is the direct conditioning signal: a well-conditioned solve holds the stack + apart regardless of the ratio; +- whether the stack survives (finite state, no fall-through, boxes still + stacked at the horizon); +- settling behavior and residual motion at the horizon; +- deterministic repeats: each cell runs twice and must be bit-identical. + +Usage (after `pixi run build`): + + PYTHONPATH=build/default/cpp/Release/python pixi run python \ + scripts/write_citation_ct007_high_mass_ratio_packet.py +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import platform +import subprocess +import sys +from pathlib import Path +from typing import Any + +import dartpy as sx +import numpy as np +from citation_packet_utils import ( + PENETRATION_CLAMP_NOTE, + UNSUPPORTED_SOLVER_RESIDUAL, + preserve_review, + solver_iterations_by_method, + world_resolved_configuration, +) + +REPO_ROOT = Path(__file__).resolve().parents[1] +DEFAULT_OUTPUT = ( + REPO_ROOT + / "docs" + / "plans" + / "123-citation-driven-simulation-trust" + / "evidence" + / "CT-007-dart7-high-mass-ratio.json" +) + +SCENE_PARAMETERS: dict[str, Any] = { + "scene_id": "ct007_high_mass_ratio_stack", + "description": ( + "Two-box stack at rest on a static ground: a 1 kg lower box supporting " + "an upper box whose mass is swept over four decades, so the loaded " + "contact sees mass ratios from 1:1 to 1000:1. Boxes are 0.2 m cubes, " + "friction 0.8, restitution 0, simulated to a 2 s horizon per " + "ratio/solver cell." + ), + "gravity_mps2": [0.0, 0.0, -9.81], + "box_half_extent_m": 0.1, + "lower_mass_kg": 1.0, + "mass_ratios": [1.0, 10.0, 100.0, 1000.0], + "friction": 0.8, + "restitution": 0.0, + "ground_half_extents_m": [1.0, 1.0, 0.05], + "timestep_s": 0.002, + "horizon_s": 2.0, + "contact_solver_methods": ["SEQUENTIAL_IMPULSE", "BOXED_LCP"], + "deterministic_repeats": 2, +} + + +def scene_digest(parameters: dict[str, Any]) -> str: + canonical = json.dumps(parameters, sort_keys=True, separators=(",", ":")) + return "sha256:" + hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + +def _box_inertia(mass: float, half: float) -> np.ndarray: + full = 2.0 * half + moment = mass * (full * full + full * full) / 12.0 + return np.diag([moment, moment, moment]) + + +def _transform_at(position: np.ndarray) -> np.ndarray: + transform = np.eye(4) + transform[:3, 3] = position + return transform + + +def run_single( + method_name: str, ratio: float, parameters: dict[str, Any] +) -> dict[str, Any]: + """Run one stack at a given mass ratio and return raw metrics.""" + half = float(parameters["box_half_extent_m"]) + lower_mass = float(parameters["lower_mass_kg"]) + upper_mass = lower_mass * ratio + dt = float(parameters["timestep_s"]) + ground_half = np.asarray(parameters["ground_half_extents_m"], dtype=float) + step_count = int(round(float(parameters["horizon_s"]) / dt)) + + world = sx.World( + time_step=dt, + gravity=np.asarray(parameters["gravity_mps2"], dtype=float), + contact_solver_method=sx.ContactSolverMethod[method_name], + ) + + ground = world.add_rigid_body("ct007_ground") + ground.is_static = True + ground.set_collision_shape(sx.CollisionShape.box(ground_half)) + ground.transform = _transform_at(np.array([0.0, 0.0, -ground_half[2]])) + ground.friction = float(parameters["friction"]) + ground.restitution = float(parameters["restitution"]) + + boxes = [] + for index, mass in enumerate((lower_mass, upper_mass)): + body = world.add_rigid_body(f"ct007_box{index}") + body.mass = mass + body.inertia = _box_inertia(mass, half) + body.set_collision_shape(sx.CollisionShape.box(np.array([half, half, half]))) + body.friction = float(parameters["friction"]) + body.restitution = float(parameters["restitution"]) + # Rest each box exactly on the one below, so any observed overlap is + # solver conditioning rather than an initial interpenetration. + body.transform = _transform_at(np.array([0.0, 0.0, half + index * 2.0 * half])) + boxes.append(body) + + world.enter_simulation_mode() + + resolved = { + "world_resolution": world_resolved_configuration(world), + "contact_solver_method": world.contact_solver_method.name, + "gravity_mps2": [float(v) for v in np.asarray(world.gravity)], + "time_step_s": float(world.time_step), + } + + lower_rest_z = half + upper_rest_z = 3.0 * half + state_hash = hashlib.sha256() + non_finite = False + max_iterations = 0 + max_contacts = 0 + max_penetration = 0.0 + + for _ in range(step_count): + world.step() + metrics = world.compute_step_metrics() + max_penetration = max(max_penetration, float(metrics.max_penetration_depth)) + max_iterations = max(max_iterations, int(metrics.last_step_iterations)) + max_contacts = max(max_contacts, int(metrics.active_contact_count)) + for body in boxes: + position = np.asarray(body.translation, dtype=float) + if not np.all(np.isfinite(position)): + non_finite = True + break + if non_finite: + break + + if non_finite: + return { + "contact_solver_method": method_name, + "mass_ratio": ratio, + "resolved": resolved, + "finite": False, + "final_state_sha256": None, + "lower_sink_m": None, + "upper_sink_m": None, + "gap_closure_m": None, + "relative_gap_closure": None, + "final_max_speed_mps": None, + "max_penetration_m": max_penetration, + "max_solver_iterations": max_iterations, + "max_active_contacts": max_contacts, + } + + speeds = [] + positions = [] + for body in boxes: + position = np.asarray(body.translation, dtype=float) + velocity = np.asarray(body.linear_velocity, dtype=float) + positions.append(position) + speeds.append(float(np.linalg.norm(velocity))) + state_hash.update(position.tobytes()) + state_hash.update(velocity.tobytes()) + + lower_sink = lower_rest_z - float(positions[0][2]) + upper_sink = upper_rest_z - float(positions[1][2]) + # How far the two box centers closed toward each other: the loaded + # contact's steady-state overlap, independent of how far the pair sank + # into the ground. + gap_closure = 2.0 * half - float(positions[1][2] - positions[0][2]) + + return { + "contact_solver_method": method_name, + "mass_ratio": ratio, + "resolved": resolved, + "finite": True, + "final_state_sha256": state_hash.hexdigest(), + "lower_sink_m": lower_sink, + "upper_sink_m": upper_sink, + "gap_closure_m": gap_closure, + "relative_gap_closure": gap_closure / (2.0 * half), + "final_max_speed_mps": max(speeds), + "max_penetration_m": max_penetration, + "max_solver_iterations": max_iterations, + "max_active_contacts": max_contacts, + } + + +def git_head() -> str: + return subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=REPO_ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + + +def build_packet(output_path: Path | None = None) -> dict[str, Any]: + parameters = SCENE_PARAMETERS + rows: list[dict[str, Any]] = [] + determinism_failures: list[str] = [] + resolved_by_cell: dict[str, dict[str, Any]] = {} + + for method in parameters["contact_solver_methods"]: + for ratio in parameters["mass_ratios"]: + repeats = [ + run_single(method, ratio, parameters) + for _ in range(int(parameters["deterministic_repeats"])) + ] + hashes = {run["final_state_sha256"] for run in repeats} + if len(hashes) != 1: + determinism_failures.append( + f"{method} ratio {ratio}: final-state hashes differ " + f"{sorted(map(str, hashes))}" + ) + row = repeats[0] + for repeat in repeats: + readback = repeat["resolved"] + if readback != row["resolved"]: + raise SystemExit( + f"{method} ratio {ratio}: resolved configuration " + "differs between repeats" + ) + if readback["contact_solver_method"] != method: + raise SystemExit( + f"requested contact solver {method} but World readback " + f"reports {readback['contact_solver_method']}" + ) + if readback["time_step_s"] != parameters["timestep_s"]: + raise SystemExit( + f"requested timestep {parameters['timestep_s']} but " + f"World readback reports {readback['time_step_s']}" + ) + row["repeat_state_sha256"] = [ + repeat["final_state_sha256"] for repeat in repeats + ] + resolved_by_cell[f"{method}@ratio={ratio}"] = row["resolved"] + rows.append(row) + + if determinism_failures: + raise SystemExit( + "deterministic repeats failed:\n " + "\n ".join(determinism_failures) + ) + + method_summary: dict[str, Any] = {} + for method in parameters["contact_solver_methods"]: + method_rows = [row for row in rows if row["contact_solver_method"] == method] + finite_rows = [row for row in method_rows if row["finite"]] + closure_by_ratio = { + row["mass_ratio"]: row["relative_gap_closure"] for row in finite_rows + } + method_summary[method] = { + "cells": len(method_rows), + "all_finite": len(finite_rows) == len(method_rows), + "relative_gap_closure_by_ratio": { + str(ratio): value for ratio, value in sorted(closure_by_ratio.items()) + }, + "max_relative_gap_closure": ( + max(closure_by_ratio.values()) if closure_by_ratio else None + ), + "closure_growth_factor": ( + ( + closure_by_ratio[max(closure_by_ratio)] + / closure_by_ratio[min(closure_by_ratio)] + ) + if closure_by_ratio and closure_by_ratio.get(min(closure_by_ratio)) + else { + "status": "unsupported", + "reason": ( + "Closure at the lowest mass ratio is zero or absent, " + "so a growth factor against it is undefined rather " + "than zero." + ), + } + ), + "max_final_speed_mps": max( + (row["final_max_speed_mps"] for row in finite_rows), + default=0.0, + ), + } + + all_finite = all(row["finite"] for row in rows) + # A stack whose boxes close more than a tenth of a box height has not been + # held apart in any useful sense. + collapse_threshold = 0.1 + degraded_cells = [ + { + "contact_solver_method": row["contact_solver_method"], + "mass_ratio": row["mass_ratio"], + "relative_gap_closure": row["relative_gap_closure"], + "finite": row["finite"], + } + for row in rows + if not row["finite"] + or (row["relative_gap_closure"] or 0.0) > collapse_threshold + ] + + # Characterize each method's failure onset: the smallest swept ratio at + # which the stack stops being held apart. + baseline_finding: dict[str, Any] = {} + for method in parameters["contact_solver_methods"]: + failing = sorted( + row["mass_ratio"] + for row in rows + if row["contact_solver_method"] == method + and ( + not row["finite"] + or (row["relative_gap_closure"] or 0.0) > collapse_threshold + ) + ) + baseline_finding[method] = { + "fails_at_mass_ratios": failing, + "failure_onset_ratio": ( + failing[0] + if failing + else { + "status": "unsupported", + "reason": ( + "This method held the stack at every swept ratio, " + "so there is no onset within the swept range; the " + "onset is unmeasured, not zero or infinite." + ), + } + ), + "holds_across_swept_range": not failing, + } + + # The cited claim compares against exact-cone methods, which this branch + # does not have. The baseline arm is what this packet establishes, so the + # row stays unresolved and the measurement is published as WS5 input. + disposition = "unresolved" + + command = ( + "PYTHONPATH=build/default/cpp/Release/python pixi run python " + "scripts/write_citation_ct007_high_mass_ratio_packet.py" + ) + packet: dict[str, Any] = { + "schema": "dart.citation_claim_evidence/v1", + "claim_id": "CT-007", + "title": ( + "High-mass-ratio stack conditioning baseline for the exact-cone " + "GO/NO-GO (DART 7)" + ), + "source": { + "url": "https://arxiv.org/abs/2405.17020", + "claim": ( + "Exact Coulomb cones and adaptive proximal methods may improve " + "conditioning and remove friction-pyramid anisotropy." + ), + }, + "target": { + "branch": "main", + "commit": git_head(), + "commit_role": ( + "Source state measured: the library and fixture ran at this " + "commit, which is HEAD at capture time. The packet and its " + "writer land in a later commit." + ), + }, + "scene": { + "id": parameters["scene_id"], + "digest": scene_digest(parameters), + "description": parameters["description"], + "parameters": parameters, + }, + "configuration": { + "requested": { + "contact_solver_method_sweep": parameters["contact_solver_methods"], + "mass_ratio_sweep": parameters["mass_ratios"], + "exact_cone_solver": ( + "not available on this branch; the comparison arm the " + "cited claim needs does not exist yet" + ), + "precision": "float64", + "backend": "cpu", + "threads": "World default sequential step", + }, + "resolved": {"by_cell": resolved_by_cell}, + "resolved_provenance": ( + "The resolved identity is the World's own bake-time " + "resolution (World.resolved_configuration), recorded per " + "domain with requested, resolved, reason, and a substitution " + "flag. Requested contact solver and timestep are additionally " + "asserted against readback per repeat and per cell." + ), + "detector": ( + "DART 7 native World collision pipeline (the World step API " + "exposes no detector selection on main)" + ), + "timestep": parameters["timestep_s"], + "substeps": 1, + "iterations": ( + "World defaults; sequential impulse reports its configured " + "sweep count and the boxed-LCP path records none" + ), + "fallback_policy": ( + "World defaults; no per-island fallback reporting is exposed " "on main" + ), + }, + "ensemble": { + "kind": "parameter-sweep-with-deterministic-repeats", + "sweep": [ + {"contact_solver_method": method, "mass_ratio": ratio} + for method in parameters["contact_solver_methods"] + for ratio in parameters["mass_ratios"] + ], + "deterministic_repeats": int(parameters["deterministic_repeats"]), + "deterministic_repeats_identical": not determinism_failures, + "measurement_window": { + "start_s": 0.0, + "end_s": parameters["horizon_s"], + }, + }, + "metrics": { + "physical": { + "method": ( + "Steady-state closure of the loaded contact: how far the " + "two box centers approach each other by the horizon, " + "relative to a box height. Reported per mass ratio and " + "solver, with survival and residual speed." + ), + "all_cells_finite": all_finite, + "per_method_summary": method_summary, + "collapse_threshold_relative": collapse_threshold, + "degraded_cells": degraded_cells, + "baseline_finding": baseline_finding, + }, + "numerical": { + "method": ( + "Max over run of StepMetrics.max_penetration_depth and " + "active_contact_count; per-method iteration counts where " + "the runtime records them" + ), + "max_penetration_m": max(row["max_penetration_m"] for row in rows), + "penetration_semantics": PENETRATION_CLAMP_NOTE, + "solver_iterations_by_method": solver_iterations_by_method( + { + method: max( + row["max_solver_iterations"] + for row in rows + if row["contact_solver_method"] == method + ) + for method in parameters["contact_solver_methods"] + } + ), + "solver_residual": dict(UNSUPPORTED_SOLVER_RESIDUAL), + "max_active_contacts": max(row["max_active_contacts"] for row in rows), + }, + "performance": { + "status": "unsupported", + "reason": ( + "This packet makes no timing claim; the matched-residual " + "timing a Pareto comparison needs is WS5 work and " + "requires the exact-cone arm to exist first" + ), + }, + "allocation": { + "status": "unsupported", + "reason": ( + "Post-bake allocation gates are owned by PLAN-122 " + "tooling; this packet does not measure allocations" + ), + }, + }, + "evidence": { + "commands": [command], + "raw_rows": rows, + "visual": { + "status": "not-applicable", + "reason": ( + "The oracle is the numeric steady-state closure of the " + "loaded contact; no visible-behavior claim is made." + ), + }, + }, + "result": { + "disposition": disposition, + "claim_boundary": ( + "DART 7 main, this commit, a two-box stack with a 1 kg lower " + "box, mass ratios 1 to 1000, 2 ms timestep, 2 s horizon, " + "SEQUENTIAL_IMPULSE and BOXED_LCP. The cited claim is " + "comparative and its exact-cone arm does not exist on this " + "branch, so the row is NOT reproduced and nothing here says " + "whether an exact cone would help. What this establishes is " + "the baseline arm the WS5 GO/NO-GO needs, and it is not a " + "null result: SEQUENTIAL_IMPULSE -- the World's default " + "contact solver -- holds the stack at mass ratios 1 and 10 " + "(relative closure 1.4e-4 and 1.0e-3) but fails completely " + "at 100 and 1000, where the heavy box descends a full box " + "height (0.20002 m) while the light box beneath it moves by " + "microns, coming to rest fully interpenetrated at near-zero " + "velocity. BOXED_LCP holds across all four decades (closure " + "3.3e-5 to 3.0e-4). Both outcomes are bit-identical across " + "repeats. Says nothing about deeper stacks, other shapes or " + "materials, other timesteps, historical DART versions, or " + "DART 6." + ), + "limitations": [ + "The sequential-impulse failure is a geometric " + "observation at one timestep. This packet does not " + "establish the mechanism; a fixed iteration budget is " + "the obvious suspect, and confirming it needs the " + "per-solve residual WS4 has not exposed yet.", + "Two boxes only. Conditioning problems are usually worse in " + "deeper stacks, so this is a floor on the effect, not a " + "characterization of it.", + "One timestep. Contact stiffness interacts with the step " + "size, so a ratio sweep at a single dt cannot separate the " + "two.", + "No exact-cone arm exists, so the packet cannot test the " + "cited improvement; it only records what the current solvers " + "do.", + "Closure is measured at the horizon, not tracked as a time " + "series, so a stack that closes and recovers is not " + "distinguished from one that never closed.", + "No solver residual exists on this path and boxed LCP records " + "no iteration count; both are typed unsupported rather than " + "reported as zero, so the conditioning signal here is " + "geometric rather than algebraic.", + ], + }, + "review": ( + preserve_review(output_path) if output_path is not None else {"passes": []} + ), + "host": { + "platform": platform.platform(), + "python": sys.version.split()[0], + "machine": platform.machine(), + "performance_valid": False, + "note": ( + "Host recorded for provenance only; no timing methodology " + "was applied" + ), + }, + } + return packet + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--output", + type=Path, + default=DEFAULT_OUTPUT, + help=f"Packet output path (default: {DEFAULT_OUTPUT})", + ) + return parser.parse_args() + + +def main() -> int: + args = parse_args() + packet = build_packet(args.output) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text( + json.dumps(packet, indent=2, sort_keys=True) + "\n", encoding="utf-8" + ) + physical = packet["metrics"]["physical"] + print(f"wrote {args.output}") + for method, stats in physical["per_method_summary"].items(): + closures = stats["relative_gap_closure_by_ratio"] + rendered = ", ".join( + f"{ratio}:{value:.3e}" for ratio, value in closures.items() + ) + print(f" {method}: relative closure by ratio -> {rendered}") + print(f" degraded cells: {physical['degraded_cells']}") + print(f" disposition: {packet['result']['disposition']}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) From ce5142e8cd7c76d6adc120e703482e3f898e4567 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 01:34:48 -0700 Subject: [PATCH 12/52] Draft the CT-007 default-solver issue report Working copy of the GitHub issue for the sequential-impulse high-mass-ratio failure, with a standalone repro script (verified against the current build: the heavy box descends to full interpenetration in about one second at ratio 100 while boxed LCP holds the gap at 1e-5 m) and the sweep table from the CT-007 packet. Posting needs maintainer approval; the draft lives in the dev-task folder until that decision is made. --- .../ct007-issue-draft.md | 123 ++++++++++++++++++ 1 file changed, 123 insertions(+) create mode 100644 docs/dev_tasks/citation_driven_simulation_trust/ct007-issue-draft.md diff --git a/docs/dev_tasks/citation_driven_simulation_trust/ct007-issue-draft.md b/docs/dev_tasks/citation_driven_simulation_trust/ct007-issue-draft.md new file mode 100644 index 0000000000000..0795547e61db7 --- /dev/null +++ b/docs/dev_tasks/citation_driven_simulation_trust/ct007-issue-draft.md @@ -0,0 +1,123 @@ +# Issue draft: default contact solver fails high-mass-ratio stacks + +Status: DRAFT, not posted. Posting to GitHub needs maintainer approval. +Working copy for the CT-007 finding; delete with this dev-task folder after +the issue is filed (or a decision is recorded not to file it). + +--- + +Title: `SEQUENTIAL_IMPULSE contact solver lets a heavy box sink completely +through a light one at mass ratios >= 100` + +**Environment:** + +- DART version: `main` (20501341226; reproduced during PLAN-123 CT-007 work) +- OS: Linux (Ubuntu-based, kernel 7.0.0) +- Installation: source, `pixi run build` (Release) +- Compiler: repository default toolchain + +**Expected vs Current Behavior:** + +A rigid two-box stack at rest should stay stacked regardless of the mass +ratio between the boxes. With the DART 7 `World` default contact solver +(`ContactSolverMethod.SEQUENTIAL_IMPULSE`), a heavy box resting on a light +one sinks completely through it once the mass ratio reaches ~100:1: over +about one second the heavy box descends exactly one box height, the pair +comes to rest fully interpenetrated (box centers coincident) at near-zero +velocity, and stays that way. The light box below barely moves (microns), so +this is the box-box contact failing, not the ground contact. The state stays +finite and the run is bit-deterministic, which makes the failure silent. + +`ContactSolverMethod.BOXED_LCP` holds the identical stack across every tested +ratio (1 to 1000), with steady-state overlap in the 1e-5 m range. + +**Steps to Reproduce:** + +```python +import dartpy as dart +import numpy as np + +RATIO = 100.0 # upper box mass / lower box mass + +for method in ("SEQUENTIAL_IMPULSE", "BOXED_LCP"): + world = dart.simulation.World( + time_step=0.002, + contact_solver_method=dart.simulation.ContactSolverMethod[method], + ) + ground = world.add_rigid_body("ground") + ground.is_static = True + ground.set_collision_shape( + dart.simulation.CollisionShape.box(np.array([1.0, 1.0, 0.05])) + ) + tf = np.eye(4); tf[2, 3] = -0.05 + ground.transform = tf + + boxes = [] + for i, mass in enumerate((1.0, RATIO)): + body = world.add_rigid_body(f"box{i}") + body.mass = mass + moment = mass * (0.2**2 + 0.2**2) / 12.0 + body.inertia = np.diag([moment] * 3) + body.set_collision_shape( + dart.simulation.CollisionShape.box(np.array([0.1, 0.1, 0.1])) + ) + body.friction = 0.8 + body.restitution = 0.0 + tf = np.eye(4); tf[2, 3] = 0.1 + i * 0.2 # exactly stacked, zero gap + body.transform = tf + boxes.append(body) + + world.enter_simulation_mode() + print(f"--- {method} ---") + for step in range(1001): + if step % 100 == 0: + z0 = float(boxes[0].translation[2]) + z1 = float(boxes[1].translation[2]) + print(f"t={step*0.002:4.1f}s gap={z1 - z0 - 0.2:+.5f} m") + world.step() +``` + +Observed output (gap = center separation minus one box height; 0 means +resting contact, -0.2 means the boxes fully coincide): + +``` +--- SEQUENTIAL_IMPULSE --- --- BOXED_LCP --- +t= 0.0s gap=+0.00000 m t= 0.0s gap=-0.00000 m +t= 0.4s gap=-0.13251 m t= 0.4s gap=-0.00001 m +t= 0.8s gap=-0.19965 m t= 0.8s gap=-0.00007 m +t= 1.0s gap=-0.19999 m t= 1.0s gap=-0.00002 m +t= 2.0s gap=-0.19999 m t= 2.0s gap=-0.00002 m +``` + +Sweep over mass ratio (steady-state relative closure of the loaded contact +at a 2 s horizon; 1.0 = fully interpenetrated): + +| solver | 1:1 | 10:1 | 100:1 | 1000:1 | +| ------------------ | ------ | ------ | ---------- | ---------- | +| SEQUENTIAL_IMPULSE | 1.4e-4 | 1.0e-3 | **1.0000** | **1.0000** | +| BOXED_LCP | 2.0e-4 | 2.0e-4 | 3.3e-5 | -3.0e-4 | + +**Notes:** + +- The scene starts with exactly zero overlap, so the closure is produced by + the solve, not by initial interpenetration recovery. +- Both outcomes are bit-identical across repeated runs. +- The behavior is consistent with iterative Gauss-Seidel contact under a + fixed iteration budget failing to propagate impulses through a + high-mass-ratio contact pair, but the mechanism is not confirmed: the + sequential-impulse path does not expose a per-solve residual + (`StepMetrics.last_step_residual` is structurally zero on the rigid contact + path), so there is no algebraic signal to inspect. Exposing a comparable + residual is tracked as PLAN-123 WS4 work. +- Mass ratios of 100:1 are ordinary in robotics scenes (a 100 kg robot + standing on a 1 kg object), and the failure is silent: finite state, + deterministic, no warning. +- Workaround: construct the `World` with + `contact_solver_method=dart.simulation.ContactSolverMethod.BOXED_LCP`. +- DART 6 (`release-6.20`) is not affected on its default path: the classic + constraint pipeline defaults to the Dantzig boxed LCP. This is specific to + the DART 7 `World` sequential-impulse contact stage. +- Full evidence packet (raw rows, per-cell resolved solver identity, + deterministic repeat hashes): `CT-007-dart7-high-mass-ratio.json` under + `docs/plans/123-citation-driven-simulation-trust/evidence/` on the + PLAN-123 branch (pending PR). From 3d24f32a518cc3f9a5c61511f5f076f41b3ed920 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 01:42:57 -0700 Subject: [PATCH 13/52] Add the CT-011 restore-equivalence packet; all six first-wave families covered The RobotDART row's restore-equivalence half, measured with a seven-arm protocol over a contact-rich five-sphere pile, both contact solvers, every continuation hashed bit-exactly over the full state vector, and the whole protocol repeated twice and required identical. Finding: World.state_vector restore is not a function of the restored state once the world has contact history. In-place restores diverge from the original continuation at the FIRST post-restore step and depend on how much history preceded the restore -- two restores of the same snapshot differ from each other. Meanwhile fresh-world restores of the same vector are bit-exact and repeatable, identical histories agree, pre-contact history is harmless, and the ballistic control is exact. Everything is deterministic given full history: this is hidden result-affecting contact state surviving the restore, not nondeterminism. Deactivation was ruled out by probing (nothing asleep; disabling it changes nothing), and update_kinematics() changes nothing. Both solvers are affected identically, pointing at shared contact-pipeline state. This is the design doc's 'reset semantics must explicitly choose whether solver/contact history is preserved' requirement observed unmet, the concrete motivation for WS3 contact ordering/identity, and the second maintainer-decision finding of the campaign (issue draft included alongside the CT-007 one). Disposition stays unresolved: the corpus row is a requirements claim, and reset cost, allocation, and concurrency are typed unsupported. The gate rejected first_divergent_step: 0 until it was declared a measured zero -- which it genuinely is. --- CHANGELOG.md | 6 +- .../README.md | 17 +- .../ct011-issue-draft.md | 136 +++++ .../verification.md | 37 ++ .../123-citation-driven-simulation-trust.md | 10 + .../claims-manifest.json | 7 +- .../CT-011-dart7-restore-equivalence.json | 378 ++++++++++++ docs/plans/dashboard.md | 15 +- ...tation_ct011_restore_equivalence_packet.py | 578 ++++++++++++++++++ 9 files changed, 1169 insertions(+), 15 deletions(-) create mode 100644 docs/dev_tasks/citation_driven_simulation_trust/ct011-issue-draft.md create mode 100644 docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json create mode 100644 scripts/write_citation_ct011_restore_equivalence_packet.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 265c7397ced56..9ff4bb833459d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -625,7 +625,11 @@ compatibility remains on the active DART 6 LTS branch._ integration-limited, and step cost is not measured), and CT-007 high-mass-ratio stack conditioning, which records that the default sequential-impulse contact solver fails to hold a two-box stack at mass - ratios of 100 and above while the boxed-LCP path holds across four decades. + ratios of 100 and above while the boxed-LCP path holds across four decades, + and CT-011 restore equivalence, which records that `World.state_vector` + restore is not a function of the restored state once the world has contact + history (a fresh world restoring the same vector is bit-exact), completing + first-packet coverage of all six capped first-wave fixture families. - Reorganized tests and CI coverage around DART 7 components, with focused unit, integration, benchmark, rendering, CUDA-smoke, collision, and simulation gates replacing broad stale test targets. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/README.md b/docs/dev_tasks/citation_driven_simulation_trust/README.md index 416422ae38639..c9504c69f5ce8 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/README.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/README.md @@ -19,7 +19,12 @@ CT-005 PD tracking (`unresolved`: error is controller-limited and step cost is not measured), and CT-007 high mass-ratio stack (`unresolved` for the comparative claim, but it found that the default contact solver - fails completely at mass ratios 100 and 1000 where boxed LCP holds). Landed on + fails completely at mass ratios 100 and 1000 where boxed LCP holds), and + CT-011 restore equivalence (`unresolved` for the requirements claim, + but it found that state-vector restore is not a function of the + restored state once contact history exists; fresh-world restore is + bit-exact). All six first-wave families now have at least one `main` + packet. Landed on `release-6.20`: CT-001 detector sweep (`reproduced` on fcl/dart/ode; bullet excluded for lacking the pyramid signature). Open: articulated energy/momentum/control, heel-strike/toe-off, high mass-ratio, and @@ -162,11 +167,11 @@ for downstream-sensitive changes. ## Immediate next steps 1. Read `RESUME.md` and verify both branch tips/worktrees. -2. Bring the CT-007 sequential-impulse failure to the maintainer: it is a - default-path defect, the strongest WS5 GO input so far, and confirming its - mechanism needs the per-solve residual WS4 has not exposed. -3. Continue Phase 2 with heel-strike/toe-off (CT-006), which depends on the - WS3 contact semantics, and reset/concurrency queries (CT-011..013). +2. Bring the CT-007 and CT-011 findings to the maintainer: a default-path + contact failure at high mass ratio, and restore semantics that silently + depend on contact history. Issue drafts live in this folder. +3. Start WS3 contact ordering/identity, now motivated concretely by CT-011's + hidden contact state; CT-006 heel-strike follows it. 4. Start WS4's first slice in parallel where it unblocks packets: expose `ResolvedSolverConfiguration` to Python and a comparable per-solve residual, both currently typed unsupported in every packet. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/ct011-issue-draft.md b/docs/dev_tasks/citation_driven_simulation_trust/ct011-issue-draft.md new file mode 100644 index 0000000000000..788127520f751 --- /dev/null +++ b/docs/dev_tasks/citation_driven_simulation_trust/ct011-issue-draft.md @@ -0,0 +1,136 @@ +# Issue draft: state-vector restore depends on prior contact history + +Status: DRAFT, not posted. Posting to GitHub needs maintainer approval. +Working copy for the CT-011 finding; delete with this dev-task folder after +the issue is filed (or a decision is recorded not to file it). + +--- + +Title: `World.state_vector restore is not a function of the restored state +once the world has contact history` + +**Environment:** + +- DART version: `main` (20501341226; found during PLAN-123 CT-011 work) +- OS: Linux (Ubuntu-based, kernel 7.0.0) +- Installation: source, `pixi run build` (Release) +- Compiler: repository default toolchain + +**Expected vs Current Behavior:** + +For reset-heavy research workflows (RL episodes, planning rollouts, +trajectory optimization), `world.state_vector = snapshot` should make the +subsequent trajectory a function of the restored state: restoring the same +snapshot should always produce the same continuation. On `main` it does not, +once the world has ever had contact: + +- Restoring a snapshot in place and re-stepping diverges from the original + continuation at the **first** post-restore step (max state-vector delta + ~5e-3 to 1e-2 within a 0.2 s window in the repro below). +- Restoring the **same snapshot twice** into the same world produces **two + different continuations** when different amounts of stepping precede the + two restores. +- A **freshly built** world restoring the same snapshot is bit-exact and + repeatable; two separately built worlds with identical step histories + agree bit-exactly; a world whose history ends before its first contact + behaves like a fresh world; and a ballistic (no-contact) scene restores + bit-exactly. + +Everything is deterministic given full history — this is not nondeterminism. +Some result-affecting contact state survives the `state_vector` write, so +the continuation depends on what the world did before the restore. The +failure is silent: state stays finite and each individual run is +reproducible, which makes it easy to mistake restored rollouts for +equivalent ones. Both `SEQUENTIAL_IMPULSE` and `BOXED_LCP` are affected +identically, which points at shared contact-pipeline state rather than a +solver-specific cache. Deactivation/sleeping is ruled out: no body was +asleep at the snapshot, and disabling deactivation changes nothing. + +**Steps to Reproduce:** + +```python +import hashlib +import dartpy as dart +import numpy as np + +def build(method): + w = dart.simulation.World( + time_step=0.002, + contact_solver_method=dart.simulation.ContactSolverMethod[method]) + g = w.add_rigid_body("g"); g.is_static = True + g.set_collision_shape(dart.simulation.CollisionShape.box(np.array([1., 1., .05]))) + tf = np.eye(4); tf[2, 3] = -0.05; g.transform = tf + for i in range(5): + b = w.add_rigid_body(f"s{i}") + b.mass = 0.5 + b.inertia = np.diag([0.4 * 0.5 * 0.06**2] * 3) + b.set_collision_shape(dart.simulation.CollisionShape.sphere(0.06)) + b.friction = 0.6; b.restitution = 0.2 + tf = np.eye(4) + tf[:3, 3] = (0.03 * (i % 2 * 2 - 1) * (i + 1) / 5, + 0.025 * ((i // 2) % 2 * 2 - 1) * (i + 1) / 5, + 0.2 + 0.13 * i) + b.transform = tf + w.enter_simulation_mode() + return w + +def run_hash(w, k=100): + h = hashlib.sha256() + for _ in range(k): + w.step() + h.update(np.ascontiguousarray(w.state_vector).tobytes()) + return h.hexdigest() + +for method in ("SEQUENTIAL_IMPULSE", "BOXED_LCP"): + w = build(method) + for _ in range(250): # settle: contact starts around step 85 + w.step() + snap = np.array(w.state_vector, copy=True); t0 = w.time + + h_continue = run_hash(w) # original continuation + w.state_vector = snap; w.time = t0 + h_restore1 = run_hash(w) # in-place restore #1 + w.state_vector = snap; w.time = t0 + h_restore2 = run_hash(w) # in-place restore #2 + + f1 = build(method); f1.state_vector = snap; f1.time = t0 + f2 = build(method); f2.state_vector = snap; f2.time = t0 + h_fresh1, h_fresh2 = run_hash(f1), run_hash(f2) + + print(f"--- {method} ---") + print("continue == restore1:", h_continue == h_restore1) # False + print("restore1 == restore2:", h_restore1 == h_restore2) # False + print("fresh1 == fresh2: ", h_fresh1 == h_fresh2) # True +``` + +Observed on `main` for both solvers: + +``` +continue == restore1: False +restore1 == restore2: False +fresh1 == fresh2: True +``` + +Control results (from the full evidence protocol): identical-history worlds +agree bit-exactly; a pre-contact-history world matches a fresh world; +ballistic scenes restore bit-exactly; `update_kinematics()` after the +restore changes nothing. + +**Notes:** + +- The DART 7 design docs already state the requirement this violates: + "Cloning, serialization, and reset preserve or intentionally clear all + result-affecting configuration and history" and "Reset semantics must + explicitly choose whether solver/contact history is preserved" + (`docs/design/contact_trust_and_observability.md`). Currently the choice + is implicit and history leaks through. +- Either resolution is defensible — clear contact history on state writes, + or capture it in the state vector — but the semantics should be explicit, + documented, and tested. The one-model/many-state work (PLAN-030/PLAN-123 + WS7) needs this settled. +- Workaround: restore into a freshly built world; that is bit-exact and + repeatable. +- Full evidence packet (seven protocol arms, all hashes, deterministic + repeats): `CT-011-dart7-restore-equivalence.json` under + `docs/plans/123-citation-driven-simulation-trust/evidence/` on the + PLAN-123 branch (pending PR). diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index c127a6d67fa0d..45ba6100c9cd9 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -370,3 +370,40 @@ zero and the observed closure is produced by the solve. `pixi run check-citation-evidence` — OK (it rejected an untyped `None` for the "no failure onset" case on the first attempt, which is now a typed unsupported marker). + +## CT-011 restore-equivalence slice — 2026-08-15 + +- Scene: five-sphere pile with a 0.5 s contact-rich warm-up; seven protocol + arms per contact solver, every continuation hashed bit-exactly over the + full state vector; whole protocol repeated twice and required identical. +- The finding was pinned by four escalating probes before the packet was + written, each ruling out an explanation: + 1. In-place `state_vector` restore diverges from the original continuation + at the FIRST post-restore step (max state delta ~5e-3 to 1e-2 within a + 0.2 s window), both solvers. + 2. Ballistic control is bit-exact (protocol and state vector are sound for + free motion), and `update_kinematics()` after restore changes nothing. + 3. Two in-place restores of the SAME snapshot differ from each other when + different history precedes them -- but two FRESH worlds restoring that + snapshot agree bit-exactly, two worlds with IDENTICAL histories agree, + and a world whose history ends before first contact matches fresh. + 4. Deactivation ruled out: no body was asleep at the snapshot and + disabling deactivation changes nothing. +- Conclusion: once a World has contact history, the post-restore trajectory + is a function of (restored state, prior contact history), not of the + restored state alone. Everything is deterministic given full history; this + is hidden result-affecting contact state surviving the restore, not + nondeterminism. This is precisely the design doc's requirement that + "reset semantics must explicitly choose whether solver/contact history is + preserved", demonstrated unmet/undocumented on `main`, and it is the + motivating evidence for the WS3 contact-identity work. +- Practical workaround established by the packet: restore into a freshly + built world -- bit-exact and repeatable. +- Disposition `unresolved`: CT-011 is a requirements claim (research + workflows need X), which a fixture cannot reproduce; reset cost, + allocation, and concurrency are typed unsupported. +- The gate rejected `first_divergent_step: 0` until it was declared a + measured zero -- which it genuinely is: divergence starts at the very + first step. +- Commands: packet writer as recorded in the packet; + `pixi run check-citation-evidence` — OK. diff --git a/docs/plans/123-citation-driven-simulation-trust.md b/docs/plans/123-citation-driven-simulation-trust.md index b138dd87a931a..f8f30ebf5bc56 100644 --- a/docs/plans/123-citation-driven-simulation-trust.md +++ b/docs/plans/123-citation-driven-simulation-trust.md @@ -260,6 +260,16 @@ and domain expansion are DART 7 only. and rests there -- while `BOXED_LCP` holds across all four swept decades. Deterministic across repeats. This needs a maintainer decision on the default path, and it is the strongest evidence so far toward a WS5 GO. +- CT-011 completed first-packet coverage of all six capped families and + produced the second maintainer-decision finding: `World.state_vector` + restore is not a function of the restored state once contact history + exists -- in-place restores diverge at the first step and depend on prior + history, while fresh-world restores of the same vector are bit-exact, + identical histories agree, pre-contact history is harmless, and the + ballistic control is exact. Deactivation was ruled out by probing. This is + the design doc's "reset semantics must explicitly choose whether + solver/contact history is preserved" requirement observed unmet, and it is + the concrete motivation for WS3 contact ordering/identity. - Known gap: the validator does not arithmetically cross-check derived metrics against `raw_rows`, so a falsified summary would pass. Closing that needs the gate to know each packet's derivation (WS4-scale). diff --git a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json index 593623fd2d61a..610eda592e43d 100644 --- a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json +++ b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json @@ -241,9 +241,12 @@ "lanes": { "dart7": { "owner": "PLAN-030/122/123 one-model/many-state", - "status": "audit-required", + "status": "in-progress", "disposition": null, - "evidence": [] + "evidence": [ + "evidence/CT-011-dart7-restore-equivalence.json" + ], + "notes": "Restore-equivalence protocol over seven arms, both contact solvers. Finding: World.state_vector restore is NOT a function of the restored state once contact history exists (in-place restores diverge at the first step and depend on prior history), while fresh-world restore is bit-exact and repeatable, identical histories agree, pre-contact history is harmless, and the ballistic control is exact. Deterministic throughout; deactivation ruled out by probing. Reset cost, allocation, and concurrency remain unmeasured. The mechanism hunt is the WS3 contact-identity work." }, "dart6": { "owner": "existing clone/reset/Recording evidence only", diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json new file mode 100644 index 0000000000000..bdbb9429a0619 --- /dev/null +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json @@ -0,0 +1,378 @@ +{ + "claim_id": "CT-011", + "configuration": { + "detector": "DART 7 native World collision pipeline (the World step API exposes no detector selection on main)", + "fallback_policy": "World defaults; no per-island fallback reporting is exposed on main", + "iterations": "World defaults", + "requested": { + "backend": "cpu", + "contact_solver_method_sweep": [ + "SEQUENTIAL_IMPULSE", + "BOXED_LCP" + ], + "precision": "float64", + "restore_mechanism": "World.state_vector write plus World.time write; update_kinematics() made no difference in probing and is not part of the protocol", + "threads": "World default sequential step" + }, + "resolved": { + "by_cell": { + "BOXED_LCP": { + "contact_solver_method": "BOXED_LCP", + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "SEQUENTIAL_IMPULSE": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + } + } + }, + "resolved_provenance": "The resolved identity is the World's own bake-time resolution (World.resolved_configuration), recorded per domain with requested, resolved, reason, and a substitution flag. The requested contact solver is additionally asserted against readback per repeat.", + "substeps": 1, + "timestep": 0.002 + }, + "ensemble": { + "deterministic_repeats": 2, + "deterministic_repeats_identical": true, + "kind": "protocol-arms-with-deterministic-repeats", + "measurement_window": { + "continuation_steps": 100, + "warmup_steps": 250 + }, + "sweep": [ + { + "contact_solver_method": "SEQUENTIAL_IMPULSE" + }, + { + "contact_solver_method": "BOXED_LCP" + } + ] + }, + "evidence": { + "commands": [ + "PYTHONPATH=build/default/cpp/Release/python pixi run python scripts/write_citation_ct011_restore_equivalence_packet.py" + ], + "raw_rows": [ + { + "ballistic_control_sha256": [ + "faf3d3ac5fd6ed8337fa1ea5c6061b576ab938c2be4200f8783ae1b7511142c9", + "faf3d3ac5fd6ed8337fa1ea5c6061b576ab938c2be4200f8783ae1b7511142c9" + ], + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "continuation_sha256": "0e4b9d59c31f6fa79978a4a1c4893e5adff0a010866c6a58b6e72db0f0b7aa61", + "findings": { + "ballistic_restore_exact": true, + "fresh_restore_matches_continuation": false, + "fresh_restores_match_each_other": true, + "inplace_restore_matches_continuation": false, + "inplace_restores_match_each_other": false, + "precontact_history_matches_fresh": true, + "same_history_restores_match": true + }, + "fresh_restore_sha256": [ + "15cf3d42f26b04b3cb1242c75c276a113c35344f9463c7dd2349c1a34667f70b", + "15cf3d42f26b04b3cb1242c75c276a113c35344f9463c7dd2349c1a34667f70b" + ], + "inplace_divergence_vs_continuation": { + "first_divergent_step": 0, + "max_abs_state_delta": 0.004756158618442807 + }, + "inplace_restore_sha256": [ + "5879f6fde1ccb32093a641e2720b45a147580a93c1079a981c19888a52263d21", + "cc166727c022a4f0cd763d38351ea19f1177c9906c5e5ace5555abfd02b9d22d" + ], + "precontact_history_restore_sha256": "15cf3d42f26b04b3cb1242c75c276a113c35344f9463c7dd2349c1a34667f70b", + "resolved": { + "contact_solver_method": "SEQUENTIAL_IMPULSE", + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "same_history_restore_sha256": [ + "0e4b9d59c31f6fa79978a4a1c4893e5adff0a010866c6a58b6e72db0f0b7aa61", + "0e4b9d59c31f6fa79978a4a1c4893e5adff0a010866c6a58b6e72db0f0b7aa61" + ] + }, + { + "ballistic_control_sha256": [ + "faf3d3ac5fd6ed8337fa1ea5c6061b576ab938c2be4200f8783ae1b7511142c9", + "faf3d3ac5fd6ed8337fa1ea5c6061b576ab938c2be4200f8783ae1b7511142c9" + ], + "contact_solver_method": "BOXED_LCP", + "continuation_sha256": "3c71899bd799aa0e56388fa7e85ec47549b4a707fda1f03a2809e168fcfb7ac6", + "findings": { + "ballistic_restore_exact": true, + "fresh_restore_matches_continuation": false, + "fresh_restores_match_each_other": true, + "inplace_restore_matches_continuation": false, + "inplace_restores_match_each_other": false, + "precontact_history_matches_fresh": true, + "same_history_restores_match": true + }, + "fresh_restore_sha256": [ + "0614045a6cd87e6d01c45366defb286360d98691664faf8368530329a7167fbe", + "0614045a6cd87e6d01c45366defb286360d98691664faf8368530329a7167fbe" + ], + "inplace_divergence_vs_continuation": { + "first_divergent_step": 0, + "max_abs_state_delta": 0.012694905840379356 + }, + "inplace_restore_sha256": [ + "452b08b7a2ba4943f9bc475849b3f9f37d4c62a5fbb10f8e1aa78b0bd198a3e1", + "a1dd32085b5ae5b6b25b48cc30f01b7402160c9be2e44fabf82670d450f04ede" + ], + "precontact_history_restore_sha256": "0614045a6cd87e6d01c45366defb286360d98691664faf8368530329a7167fbe", + "resolved": { + "contact_solver_method": "BOXED_LCP", + "time_step_s": 0.002, + "world_resolution": { + "by_domain": { + "deformable-psd": { + "is_substitution": false, + "reason": "as requested", + "requested": "cpu", + "resolved": "cpu" + }, + "multibody": { + "is_substitution": false, + "reason": "as requested", + "requested": "semi-implicit", + "resolved": "semi-implicit" + }, + "rigid-body": { + "is_substitution": false, + "reason": "as requested", + "requested": "sequential-impulse", + "resolved": "sequential-impulse" + }, + "rigid-contact": { + "is_substitution": false, + "reason": "as requested", + "requested": "boxed-lcp", + "resolved": "boxed-lcp" + } + }, + "has_substitution": false, + "source": "World.resolved_configuration (bake-time resolution)" + } + }, + "same_history_restore_sha256": [ + "3c71899bd799aa0e56388fa7e85ec47549b4a707fda1f03a2809e168fcfb7ac6", + "3c71899bd799aa0e56388fa7e85ec47549b4a707fda1f03a2809e168fcfb7ac6" + ] + } + ], + "visual": { + "reason": "The oracle is bit-exact hash equality between trajectories; there is nothing visual to assess.", + "status": "not-applicable" + } + }, + "host": { + "machine": "x86_64", + "note": "Host recorded for provenance only; no timing methodology was applied", + "performance_valid": false, + "platform": "Linux-7.0.0-29-generic-x86_64-with-glibc2.43", + "python": "3.14.6" + }, + "metrics": { + "allocation": { + "reason": "Reset allocation behavior is PLAN-122 territory and is not measured by this packet", + "status": "unsupported" + }, + "numerical": { + "method": "Hash bookkeeping only; every hash the findings compare is published in raw_rows so the booleans can be recomputed", + "protocol_arms": 7, + "solver_residual": { + "reason": "StepMetrics.last_step_residual is structurally zero for rigid contact on this branch: recordSolverDiagnostics() in dart/simulation/compute/rigid_body_contact_stage.cpp takes residual = 0.0 by default and no rigid contact call site passes one, so no residual is computed for either contact solver. Exposing a comparable residual is PLAN-123 WS4 work.", + "status": "unsupported" + } + }, + "performance": { + "reason": "The corpus row also asks for reset cost and overhead; no timing methodology was applied, so cost is not measured", + "status": "unsupported" + }, + "physical": { + "inplace_divergence": { + "BOXED_LCP": { + "first_divergent_step": 0, + "max_abs_state_delta": 0.012694905840379356 + }, + "SEQUENTIAL_IMPULSE": { + "first_divergent_step": 0, + "max_abs_state_delta": 0.004756158618442807 + } + }, + "measured_zero_fields": [ + "inplace_divergence.SEQUENTIAL_IMPULSE.first_divergent_step", + "inplace_divergence.BOXED_LCP.first_divergent_step" + ], + "method": "Bit-exact comparison of full state-vector trajectories (SHA-256 over every post-restore step) across restore protocols, plus first divergent step and max state delta for the in-place arm against the continuation", + "per_solver_findings": { + "BOXED_LCP": { + "ballistic_restore_exact": true, + "fresh_restore_matches_continuation": false, + "fresh_restores_match_each_other": true, + "inplace_restore_matches_continuation": false, + "inplace_restores_match_each_other": false, + "precontact_history_matches_fresh": true, + "same_history_restores_match": true + }, + "SEQUENTIAL_IMPULSE": { + "ballistic_restore_exact": true, + "fresh_restore_matches_continuation": false, + "fresh_restores_match_each_other": true, + "inplace_restore_matches_continuation": false, + "inplace_restores_match_each_other": false, + "precontact_history_matches_fresh": true, + "same_history_restores_match": true + } + }, + "restore_is_a_function_of_restored_state": false + } + }, + "result": { + "claim_boundary": "DART 7 main, this commit, a five-sphere pile with a 0.5 s contact-rich warm-up, restored via World.state_vector, both contact solvers. Established: restore is NOT a function of the restored state once the world has contact history -- the in-place continuation differs from the original at the first step, and two in-place restores of the same snapshot differ from each other when different history precedes them -- while a freshly built world restoring the same vector is bit-exact and repeatable, identical histories give identical continuations, pre-contact history is harmless, and the ballistic control is exact. Everything is deterministic given full history; nothing here is nondeterminism. The corpus row's reset cost, overhead, and concurrency halves are not measured. Says nothing about Skeleton-based state APIs, clone(), replay restore, historical DART versions, or DART 6.", + "disposition": "unresolved", + "limitations": [ + "The mechanism is not identified: contact history leaves result-affecting state that a state-vector write does not reset, in both contact solvers, and deactivation was ruled out by probing (nothing was asleep and disabling it changed nothing). Naming the exact carrier (contact manifold caches, warm starts, or contact ordering) is the WS3 contact-identity work this finding motivates.", + "One scene, one snapshot point, 0.2 s windows; the magnitude of divergence (up to ~1e-2 in state units within the window) will vary with scene and horizon.", + "Reset cost, allocation, and thread/lane isolation are typed unsupported; they are the other halves of the corpus row.", + "The replay-frame restore mechanism (World.restore_replay_frame) is not tested here and may have different semantics." + ] + }, + "review": { + "passes": [] + }, + "scene": { + "description": "Five 0.5 kg spheres (radius 0.06 m) dropped from staggered offsets into a loose pile on a static ground box; snapshot taken after a 0.5 s warm-up (well after first contact), then 0.2 s continuations compared bit-exactly across restore protocols. A ballistic variant with no ground is the control.", + "digest": "sha256:e13de5bb15b3d371c34f0bc9c294673dc8ad059e37f4a73a400707cc2e405e9b", + "id": "ct011_restore_equivalence_pile", + "parameters": { + "contact_solver_methods": [ + "SEQUENTIAL_IMPULSE", + "BOXED_LCP" + ], + "continuation_steps": 100, + "description": "Five 0.5 kg spheres (radius 0.06 m) dropped from staggered offsets into a loose pile on a static ground box; snapshot taken after a 0.5 s warm-up (well after first contact), then 0.2 s continuations compared bit-exactly across restore protocols. A ballistic variant with no ground is the control.", + "deterministic_repeats": 2, + "extra_history_steps": 100, + "friction": 0.6, + "gravity_mps2": [ + 0.0, + 0.0, + -9.81 + ], + "ground_half_extents_m": [ + 1.0, + 1.0, + 0.05 + ], + "precontact_history_steps": 50, + "restitution": 0.2, + "scene_id": "ct011_restore_equivalence_pile", + "sphere_count": 5, + "sphere_mass_kg": 0.5, + "sphere_radius_m": 0.06, + "timestep_s": 0.002, + "warmup_steps": 250 + } + }, + "schema": "dart.citation_claim_evidence/v1", + "source": { + "claim": "Research workflows need fast reset, concurrency, low overhead, and deterministic synchronous stepping around DART.", + "url": "https://doi.org/10.21105/joss.06771" + }, + "target": { + "branch": "main", + "commit": "ce5142e8cd7c76d6adc120e703482e3f898e4567", + "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit." + }, + "title": "State-vector restore equivalence under contact (DART 7 first packet)" +} diff --git a/docs/plans/dashboard.md b/docs/plans/dashboard.md index 7e0aa94a3c849..4121a5e26c1ec 100644 --- a/docs/plans/dashboard.md +++ b/docs/plans/dashboard.md @@ -51,12 +51,15 @@ its own line so status updates remain git-history friendly. - Status: Active - Horizon: Now - Dimension: Algorithm extensibility -- Next step: WS0/WS1 and six first-wave packets have landed on `main`, one on - `release-6.20`. CT-007 found that the default contact solver - (SEQUENTIAL_IMPULSE) fails completely at mass ratios 100 and 1000 where - BOXED_LCP holds; that needs a maintainer decision and is the strongest WS5 - GO input so far. Continue WS2 with heel-strike/toe-off (CT-006, gated on - WS3 contact semantics) and reset/concurrency queries (CT-011..013). +- Next step: all six first-wave families now have at least one `main` + packet (plus CT-001 on `release-6.20`). Two findings need maintainer + decisions: CT-007 (the default SEQUENTIAL_IMPULSE solver lets a heavy box + sink fully through a light one at mass ratios >= 100 while BOXED_LCP + holds) and CT-011 (state-vector restore is not a function of the restored + state once contact history exists; fresh-world restore is bit-exact). + Issue drafts live in the dev task. Next slices: WS3 contact + ordering/identity (now motivated by CT-011), CT-006 heel-strike (gated on + WS3), and the per-solve residual (WS4 remainder). WS4's first slice has landed: `World.resolved_configuration` is exposed to Python and every packet now records the World's own bake-time resolution. The remaining WS4 gap is a comparable per-solve residual, still unexposed diff --git a/scripts/write_citation_ct011_restore_equivalence_packet.py b/scripts/write_citation_ct011_restore_equivalence_packet.py new file mode 100644 index 0000000000000..989eafd9104a4 --- /dev/null +++ b/scripts/write_citation_ct011_restore_equivalence_packet.py @@ -0,0 +1,578 @@ +#!/usr/bin/env python3 +"""Write the CT-011 restore-equivalence evidence packet (PLAN-123 WS2/WS7). + +Bounded claim (corpus row CT-011, from RobotDART): research workflows need +fast reset, concurrency, low overhead, and deterministic synchronous stepping +around DART. This packet measures the restore-equivalence half of that need +on DART 7 `main`: after `world.state_vector = snapshot`, is the continuation +a function of the restored state alone? + +The fixture is a five-sphere pile settling on a ground box, run per contact +solver, with a ballistic (no-ground) control scene. Arms, each hashed over +the full state vector for every post-restore step: + +- `continuation`: keep stepping from the snapshot point (baseline); +- `inplace_restore` x2: restore the snapshot into the same world twice, with + different amounts of history before each restore; +- `fresh_restore` x2: restore the snapshot into two freshly built worlds; +- `same_history_restore`: two separately built worlds with identical step + histories, both restored and continued; +- `precontact_history_restore`: a world whose history ends before the first + contact, restored and continued; +- `ballistic_control`: the same protocol with no ground, where restore must + be trivially exact if the state vector is complete for free motion. + +Reset cost, allocation, and concurrency are the other halves of the corpus +row and are explicitly not measured here (typed unsupported). + +Usage (after `pixi run build`): + + PYTHONPATH=build/default/cpp/Release/python pixi run python \ + scripts/write_citation_ct011_restore_equivalence_packet.py +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import platform +import subprocess +import sys +from pathlib import Path +from typing import Any + +import dartpy as sx +import numpy as np +from citation_packet_utils import ( + UNSUPPORTED_SOLVER_RESIDUAL, + preserve_review, + world_resolved_configuration, +) + +REPO_ROOT = Path(__file__).resolve().parents[1] +DEFAULT_OUTPUT = ( + REPO_ROOT + / "docs" + / "plans" + / "123-citation-driven-simulation-trust" + / "evidence" + / "CT-011-dart7-restore-equivalence.json" +) + +SCENE_PARAMETERS: dict[str, Any] = { + "scene_id": "ct011_restore_equivalence_pile", + "description": ( + "Five 0.5 kg spheres (radius 0.06 m) dropped from staggered offsets " + "into a loose pile on a static ground box; snapshot taken after a " + "0.5 s warm-up (well after first contact), then 0.2 s continuations " + "compared bit-exactly across restore protocols. A ballistic variant " + "with no ground is the control." + ), + "gravity_mps2": [0.0, 0.0, -9.81], + "sphere_count": 5, + "sphere_radius_m": 0.06, + "sphere_mass_kg": 0.5, + "friction": 0.6, + "restitution": 0.2, + "ground_half_extents_m": [1.0, 1.0, 0.05], + "timestep_s": 0.002, + "warmup_steps": 250, + "continuation_steps": 100, + "precontact_history_steps": 50, + "extra_history_steps": 100, + "contact_solver_methods": ["SEQUENTIAL_IMPULSE", "BOXED_LCP"], + "deterministic_repeats": 2, +} + + +def scene_digest(parameters: dict[str, Any]) -> str: + canonical = json.dumps(parameters, sort_keys=True, separators=(",", ":")) + return "sha256:" + hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + +def build_world( + method_name: str, parameters: dict[str, Any], *, with_ground: bool +) -> Any: + world = sx.World( + time_step=float(parameters["timestep_s"]), + gravity=np.asarray(parameters["gravity_mps2"], dtype=float), + contact_solver_method=sx.ContactSolverMethod[method_name], + ) + if with_ground: + ground_half = np.asarray(parameters["ground_half_extents_m"], dtype=float) + ground = world.add_rigid_body("ct011_ground") + ground.is_static = True + ground.set_collision_shape(sx.CollisionShape.box(ground_half)) + transform = np.eye(4) + transform[2, 3] = -ground_half[2] + ground.transform = transform + ground.friction = float(parameters["friction"]) + ground.restitution = float(parameters["restitution"]) + + radius = float(parameters["sphere_radius_m"]) + mass = float(parameters["sphere_mass_kg"]) + moment = 0.4 * mass * radius * radius + for index in range(int(parameters["sphere_count"])): + body = world.add_rigid_body(f"ct011_sphere{index}") + body.mass = mass + body.inertia = np.diag([moment, moment, moment]) + body.set_collision_shape(sx.CollisionShape.sphere(radius)) + body.friction = float(parameters["friction"]) + body.restitution = float(parameters["restitution"]) + transform = np.eye(4) + transform[:3, 3] = ( + 0.03 * (index % 2 * 2 - 1) * (index + 1) / 5.0, + 0.025 * ((index // 2) % 2 * 2 - 1) * (index + 1) / 5.0, + 0.2 + 0.13 * index, + ) + body.transform = transform + world.enter_simulation_mode() + return world + + +def hash_continuation(world: Any, steps: int) -> str: + digest = hashlib.sha256() + for _ in range(steps): + world.step() + digest.update(np.ascontiguousarray(world.state_vector).tobytes()) + return digest.hexdigest() + + +def divergence_profile( + reference: list[np.ndarray], world: Any, steps: int +) -> dict[str, Any]: + first = None + max_delta = 0.0 + for index in range(steps): + world.step() + delta = float(np.max(np.abs(np.asarray(world.state_vector) - reference[index]))) + if delta > 0.0 and first is None: + first = index + max_delta = max(max_delta, delta) + return { + "first_divergent_step": ( + first + if first is not None + else { + "status": "unsupported", + "reason": ( + "The two trajectories are bit-identical over the whole " + "window, so there is no divergent step; the value is " + "absent, not zero." + ), + } + ), + "max_abs_state_delta": max_delta, + } + + +def run_protocol(method_name: str, parameters: dict[str, Any]) -> dict[str, Any]: + warmup = int(parameters["warmup_steps"]) + steps = int(parameters["continuation_steps"]) + precontact = int(parameters["precontact_history_steps"]) + extra = int(parameters["extra_history_steps"]) + + # Baseline world: warm up, snapshot, continue, and keep the reference + # trajectory for divergence measurement. + world = build_world(method_name, parameters, with_ground=True) + resolved = { + "world_resolution": world_resolved_configuration(world), + "contact_solver_method": world.contact_solver_method.name, + "time_step_s": float(world.time_step), + } + for _ in range(warmup): + world.step() + snapshot = np.array(world.state_vector, copy=True) + snapshot_time = float(world.time) + + reference_states: list[np.ndarray] = [] + reference_digest = hashlib.sha256() + for _ in range(steps): + world.step() + state = np.array(world.state_vector, copy=True) + reference_states.append(state) + reference_digest.update(np.ascontiguousarray(state).tobytes()) + continuation_hash = reference_digest.hexdigest() + + # In-place restore, twice, with different history before each restore. + world.state_vector = snapshot + world.time = snapshot_time + inplace_first = hash_continuation(world, steps) + world.state_vector = snapshot + world.time = snapshot_time + inplace_second = hash_continuation(world, steps) + + inplace_probe = build_world(method_name, parameters, with_ground=True) + for _ in range(warmup + extra): + inplace_probe.step() + inplace_probe.state_vector = snapshot + inplace_probe.time = snapshot_time + inplace_divergence = divergence_profile(reference_states, inplace_probe, steps) + + # Fresh worlds restoring the same snapshot. + fresh_hashes = [] + for _ in range(2): + fresh = build_world(method_name, parameters, with_ground=True) + fresh.state_vector = snapshot + fresh.time = snapshot_time + fresh_hashes.append(hash_continuation(fresh, steps)) + + # Two separately built worlds with identical histories. + same_history_hashes = [] + for _ in range(2): + twin = build_world(method_name, parameters, with_ground=True) + for _ in range(warmup): + twin.step() + twin.state_vector = snapshot + twin.time = snapshot_time + same_history_hashes.append(hash_continuation(twin, steps)) + + # History that ends before the first contact. + precontact_world = build_world(method_name, parameters, with_ground=True) + for _ in range(precontact): + precontact_world.step() + precontact_world.state_vector = snapshot + precontact_world.time = snapshot_time + precontact_hash = hash_continuation(precontact_world, steps) + + # Ballistic control: no ground, restore must be exact. + ballistic = build_world(method_name, parameters, with_ground=False) + for _ in range(precontact): + ballistic.step() + ballistic_snapshot = np.array(ballistic.state_vector, copy=True) + ballistic_time = float(ballistic.time) + ballistic_continuation = hash_continuation(ballistic, steps) + ballistic.state_vector = ballistic_snapshot + ballistic.time = ballistic_time + ballistic_restored = hash_continuation(ballistic, steps) + + return { + "contact_solver_method": method_name, + "resolved": resolved, + "continuation_sha256": continuation_hash, + "inplace_restore_sha256": [inplace_first, inplace_second], + "fresh_restore_sha256": fresh_hashes, + "same_history_restore_sha256": same_history_hashes, + "precontact_history_restore_sha256": precontact_hash, + "ballistic_control_sha256": [ + ballistic_continuation, + ballistic_restored, + ], + "findings": { + "inplace_restore_matches_continuation": ( + inplace_first == continuation_hash + ), + "inplace_restores_match_each_other": (inplace_first == inplace_second), + "fresh_restores_match_each_other": (fresh_hashes[0] == fresh_hashes[1]), + "fresh_restore_matches_continuation": ( + fresh_hashes[0] == continuation_hash + ), + "same_history_restores_match": ( + same_history_hashes[0] == same_history_hashes[1] + ), + "precontact_history_matches_fresh": (precontact_hash == fresh_hashes[0]), + "ballistic_restore_exact": (ballistic_continuation == ballistic_restored), + }, + "inplace_divergence_vs_continuation": inplace_divergence, + } + + +def git_head() -> str: + return subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=REPO_ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + + +def build_packet(output_path: Path | None = None) -> dict[str, Any]: + parameters = SCENE_PARAMETERS + rows: list[dict[str, Any]] = [] + determinism_failures: list[str] = [] + resolved_by_cell: dict[str, dict[str, Any]] = {} + + for method in parameters["contact_solver_methods"]: + repeats = [ + run_protocol(method, parameters) + for _ in range(int(parameters["deterministic_repeats"])) + ] + # The whole protocol must reproduce bit-exactly across repeats. + keys = ( + "continuation_sha256", + "inplace_restore_sha256", + "fresh_restore_sha256", + "same_history_restore_sha256", + "precontact_history_restore_sha256", + "ballistic_control_sha256", + ) + for key in keys: + values = {json.dumps(repeat[key]) for repeat in repeats} + if len(values) != 1: + determinism_failures.append( + f"{method}: {key} differs across protocol repeats" + ) + row = repeats[0] + for repeat in repeats: + if repeat["resolved"] != row["resolved"]: + raise SystemExit( + f"{method}: resolved configuration differs between repeats" + ) + if repeat["resolved"]["contact_solver_method"] != method: + raise SystemExit( + f"requested contact solver {method} but World readback " + f"reports {repeat['resolved']['contact_solver_method']}" + ) + resolved_by_cell[method] = row["resolved"] + rows.append(row) + + if determinism_failures: + raise SystemExit( + "protocol repeats failed:\n " + "\n ".join(determinism_failures) + ) + + finding_summary = {row["contact_solver_method"]: row["findings"] for row in rows} + restore_is_state_function = all( + row["findings"]["inplace_restore_matches_continuation"] + and row["findings"]["inplace_restores_match_each_other"] + for row in rows + ) + + # CT-011 is a requirements claim about research workflows; running a + # fixture cannot reproduce a need, so the row is not promoted. What the + # packet establishes is whether main currently meets the + # restore-determinism half of that need. + disposition = "unresolved" + + command = ( + "PYTHONPATH=build/default/cpp/Release/python pixi run python " + "scripts/write_citation_ct011_restore_equivalence_packet.py" + ) + packet: dict[str, Any] = { + "schema": "dart.citation_claim_evidence/v1", + "claim_id": "CT-011", + "title": ( + "State-vector restore equivalence under contact (DART 7 first " "packet)" + ), + "source": { + "url": "https://doi.org/10.21105/joss.06771", + "claim": ( + "Research workflows need fast reset, concurrency, low " + "overhead, and deterministic synchronous stepping around " + "DART." + ), + }, + "target": { + "branch": "main", + "commit": git_head(), + "commit_role": ( + "Source state measured: the library and fixture ran at this " + "commit, which is HEAD at capture time. The packet and its " + "writer land in a later commit." + ), + }, + "scene": { + "id": parameters["scene_id"], + "digest": scene_digest(parameters), + "description": parameters["description"], + "parameters": parameters, + }, + "configuration": { + "requested": { + "contact_solver_method_sweep": parameters["contact_solver_methods"], + "restore_mechanism": ( + "World.state_vector write plus World.time write; " + "update_kinematics() made no difference in probing and " + "is not part of the protocol" + ), + "precision": "float64", + "backend": "cpu", + "threads": "World default sequential step", + }, + "resolved": {"by_cell": resolved_by_cell}, + "resolved_provenance": ( + "The resolved identity is the World's own bake-time " + "resolution (World.resolved_configuration), recorded per " + "domain with requested, resolved, reason, and a substitution " + "flag. The requested contact solver is additionally asserted " + "against readback per repeat." + ), + "detector": ( + "DART 7 native World collision pipeline (the World step API " + "exposes no detector selection on main)" + ), + "timestep": parameters["timestep_s"], + "substeps": 1, + "iterations": "World defaults", + "fallback_policy": ( + "World defaults; no per-island fallback reporting is exposed " "on main" + ), + }, + "ensemble": { + "kind": "protocol-arms-with-deterministic-repeats", + "sweep": [ + {"contact_solver_method": method} + for method in parameters["contact_solver_methods"] + ], + "deterministic_repeats": int(parameters["deterministic_repeats"]), + "deterministic_repeats_identical": not determinism_failures, + "measurement_window": { + "warmup_steps": parameters["warmup_steps"], + "continuation_steps": parameters["continuation_steps"], + }, + }, + "metrics": { + "physical": { + "method": ( + "Bit-exact comparison of full state-vector trajectories " + "(SHA-256 over every post-restore step) across restore " + "protocols, plus first divergent step and max state " + "delta for the in-place arm against the continuation" + ), + "per_solver_findings": finding_summary, + "restore_is_a_function_of_restored_state": (restore_is_state_function), + "inplace_divergence": { + row["contact_solver_method"]: row[ + "inplace_divergence_vs_continuation" + ] + for row in rows + }, + # A first divergent step of exactly zero is the measured + # finding -- the very first post-restore step already + # differs -- not a missing value. + "measured_zero_fields": [ + f"inplace_divergence.{row['contact_solver_method']}" + ".first_divergent_step" + for row in rows + if row["inplace_divergence_vs_continuation"]["first_divergent_step"] + == 0 + ], + }, + "numerical": { + "method": ( + "Hash bookkeeping only; every hash the findings compare " + "is published in raw_rows so the booleans can be " + "recomputed" + ), + "protocol_arms": 7, + "solver_residual": dict(UNSUPPORTED_SOLVER_RESIDUAL), + }, + "performance": { + "status": "unsupported", + "reason": ( + "The corpus row also asks for reset cost and overhead; " + "no timing methodology was applied, so cost is not " + "measured" + ), + }, + "allocation": { + "status": "unsupported", + "reason": ( + "Reset allocation behavior is PLAN-122 territory and is " + "not measured by this packet" + ), + }, + }, + "evidence": { + "commands": [command], + "raw_rows": rows, + "visual": { + "status": "not-applicable", + "reason": ( + "The oracle is bit-exact hash equality between " + "trajectories; there is nothing visual to assess." + ), + }, + }, + "result": { + "disposition": disposition, + "claim_boundary": ( + "DART 7 main, this commit, a five-sphere pile with a 0.5 s " + "contact-rich warm-up, restored via World.state_vector, " + "both contact solvers. Established: restore is NOT a " + "function of the restored state once the world has contact " + "history -- the in-place continuation differs from the " + "original at the first step, and two in-place restores of " + "the same snapshot differ from each other when different " + "history precedes them -- while a freshly built world " + "restoring the same vector is bit-exact and repeatable, " + "identical histories give identical continuations, " + "pre-contact history is harmless, and the ballistic control " + "is exact. Everything is deterministic given full history; " + "nothing here is nondeterminism. The corpus row's reset " + "cost, overhead, and concurrency halves are not measured. " + "Says nothing about Skeleton-based state APIs, clone(), " + "replay restore, historical DART versions, or DART 6." + ), + "limitations": [ + "The mechanism is not identified: contact history leaves " + "result-affecting state that a state-vector write does not " + "reset, in both contact solvers, and deactivation was ruled " + "out by probing (nothing was asleep and disabling it " + "changed nothing). Naming the exact carrier (contact " + "manifold caches, warm starts, or contact ordering) is the " + "WS3 contact-identity work this finding motivates.", + "One scene, one snapshot point, 0.2 s windows; the " + "magnitude of divergence (up to ~1e-2 in state units within " + "the window) will vary with scene and horizon.", + "Reset cost, allocation, and thread/lane isolation are " + "typed unsupported; they are the other halves of the corpus " + "row.", + "The replay-frame restore mechanism " + "(World.restore_replay_frame) is not tested here and may " + "have different semantics.", + ], + }, + "review": ( + preserve_review(output_path) if output_path is not None else {"passes": []} + ), + "host": { + "platform": platform.platform(), + "python": sys.version.split()[0], + "machine": platform.machine(), + "performance_valid": False, + "note": ( + "Host recorded for provenance only; no timing methodology " + "was applied" + ), + }, + } + return packet + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--output", + type=Path, + default=DEFAULT_OUTPUT, + help=f"Packet output path (default: {DEFAULT_OUTPUT})", + ) + return parser.parse_args() + + +def main() -> int: + args = parse_args() + packet = build_packet(args.output) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text( + json.dumps(packet, indent=2, sort_keys=True) + "\n", encoding="utf-8" + ) + physical = packet["metrics"]["physical"] + print(f"wrote {args.output}") + for method, findings in physical["per_solver_findings"].items(): + print(f" {method}:") + for key, value in findings.items(): + print(f" {key}: {value}") + print( + " restore is a function of restored state: " + f"{physical['restore_is_a_function_of_restored_state']}" + ) + print(f" disposition: {packet['result']['disposition']}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) From bfa4f0c99bdae3f662ca27b0010593e08553b9f5 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 01:53:07 -0700 Subject: [PATCH 14/52] Draft the DART 7 PR description --- .../pr-draft-main.md | 96 +++++++++++++++++++ 1 file changed, 96 insertions(+) create mode 100644 docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md diff --git a/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md b/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md new file mode 100644 index 0000000000000..c9c2ac8106650 --- /dev/null +++ b/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md @@ -0,0 +1,96 @@ +# PR draft: DART 7 citation-trust foundation + +Status: DRAFT. Pushing the branch and opening the PR need maintainer +approval. Branch `feature/citation-trust-foundation`, base `main`, milestone +DART 7.0. Delete this file with the dev-task folder at task completion. + +--- + +Title: `Add the PLAN-123 citation-trust foundation: fail-closed evidence +contract, six first-wave packets, and resolved-configuration bindings` + +## Summary + +- Lands PLAN-123's foundation: a machine-checked citation-claim manifest and + a fail-closed `pixi run check-citation-evidence` gate (wired into + `check-lint`), first evidence packets for all six capped first-wave + fixture families, and Python bindings for the World's bake-time + `ResolvedSolverConfiguration` so packets record the method that actually + ran. +- Two findings surfaced by the packets need maintainer decisions and have + issue drafts in the dev task: the default `SEQUENTIAL_IMPULSE` contact + solver lets a heavy box sink fully through a light one at mass ratios of + 100 and above (CT-007), and `World.state_vector` restore is not a function + of the restored state once the world has contact history (CT-011). + +## Motivation / Problem + +- External claims about DART (SimBenchmark, exoskeleton contact-force + criticism, exact-cone papers, Nimble/RobotDART workflow needs) had no + branch-qualified, reproducible dispositions: no owner converted them into + stable rows with source-bound evidence, and benchmarks could name a + requested method without proving it ran. +- PLAN-123 (docs/plans/123-citation-driven-simulation-trust.md, landed here + with its durable design doc docs/design/contact_trust_and_observability.md) + makes every such claim a stable row with a fail-closed evidence packet. + +## Changes / Key Changes + +- Claim/evidence contract: `claims-manifest.json` (20 corpus rows, exact-ID + agreement with the human corpus enforced), packet schema + `dart.citation_claim_evidence/v1`, validator + `scripts/check_citation_evidence.py` in `check-lint`, permanent + intentionally incomplete negative control that must keep failing, 74 + pytest cases. The gate rejects prose/non-JSON/outside-`evidence/`/ + negative-control/non-string lane references, dangling or directory + `raw_paths`, scene digests that disagree with the published parameters, + single-run ensembles, spelled placeholders, and any exact zero not + declared a measurement or typed `unsupported` with a reason. +- First-wave packets (writers under `scripts/`, packets under the plan's + `evidence/`): CT-001 rolling-direction (`reproduced`: friction-pyramid + antisymmetry signature), CT-002 dense inelastic (`reproduced`: settled-pile + energy gain under sequential impulse only), CT-003 dense elastic + (`unresolved`: no energy injection observed), CT-004 articulated energy + drift (`unresolved`: converges, but the cited claim is matched-cost), + CT-005 PD tracking (`unresolved`: controller-limited error), CT-007 high + mass-ratio (`unresolved` for the comparative claim; baseline finding + above), CT-011 restore equivalence (`unresolved`; finding above). +- WS4 slice 1: nanobind bindings for `ResolvedConfigurationNote` / + `ResolvedSolverConfiguration`, the `World.resolved_configuration` + property, surgical stub entries, and two Python tests; every packet + records per-domain requested/resolved/reason/substitution from the World + itself. +- PLAN-123 registered in the dashboard/plan index; corpus carries the WS0 + audit; dev task holds the working state, verification log (three + independent review rounds and their fixes), and the two issue drafts. + +## Testing + +- `pixi run check-lint` (includes the new `check-citation-evidence` gate) and + `pixi run test-all` — all phases pass on the branch head. +- `pixi run python -I scripts/run_pytest.py +tests/test_check_citation_evidence.py -q` — 74 passed. +- Every packet regenerated on the current writers; deterministic repeats + bit-identical; dispositions computed from recorded data with negative + controls and physical-validity gates. +- Three rounds of independent role-separated review; all findings fixed + in-branch and recorded in the dev-task verification log. + +## Breaking Changes + +- [x] None (additive docs/tooling/tests plus additive Python bindings). + +## Related Issues / PRs (backports) + +- Companion release-6.20 PR: branch `feature/dart6-citation-contact-trust` + (independent; adopts the same contract without DART 7 APIs). +- Issue drafts pending approval: CT-007 default-solver high-mass-ratio + failure; CT-011 restore-history dependence. + +#### Checklist + +- [ ] Milestone DART 7.0 +- [x] CHANGELOG.md updated ("Tests, Benchmarks, and Quality Gates") +- [x] Unit tests added (validator suite; World resolved-configuration tests) +- [x] New bindings documented via docstrings and stubs +- [x] Python bindings added (`World.resolved_configuration`) From 061cee38624a09f32a3925129da5a54f16cc45bd Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 01:55:14 -0700 Subject: [PATCH 15/52] Point the resume state at the finished branch tips --- .../RESUME.md | 30 +++++++++---------- 1 file changed, 14 insertions(+), 16 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md index 5f7771aec691d..2d7b997067b26 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md @@ -34,28 +34,26 @@ statuses were verified; the corpus sidecar now carries the WS0 audit record. - DART 7: `feature/citation-trust-foundation` in `.claude/worktrees/citation-trust-main`, based on `origin/main` - 20501341226, now at `85e5db2c168` with five commits: the PLAN-123 - contract, the gate hardening plus dense-contact packets, the round-2 - verdict/provenance corrections, the RESUME state record, and the CT-004 - articulated energy-drift packet. + 20501341226, at `bfa4f0c99bd` (15 commits): the contract and gate, packets for + all six first-wave families (CT-001..005, 007, 011), the + resolved-configuration bindings, three review-fix passes, and the CT-007 / + CT-011 issue drafts plus both PR drafts. - DART 6: `feature/dart6-citation-contact-trust` in `.claude/worktrees/citation-trust-620`, based on `origin/release-6.20` - 39ccd52068b, now at `afc6d7ac3c2` with four commits. + 39ccd52068b, at `978b48e6653` (7 commits). -Both are local-only; nothing pushed. GitHub mutations need maintainer -approval. +Both local-only; every head has `check-lint` + `test-all` + the citation +gate green. GitHub mutations need maintainer approval. ## Immediate next step -Continue Phase 2 on `main` with the controlled-articulated row (CT-005), -seeded by `python/examples/demos/scenes/atlas_simbicon.py`: tracking error and -constraint error under a PD controller, swept over timestep, with control work -recorded. Keep the same discipline as CT-004 -- no timing claim unless an -interleaved same-host methodology is actually applied. - -Remaining first-wave families after that: heel-strike/toe-off (CT-006, which -depends on the WS3 contact semantics), high mass-ratio stacks (CT-007), and -reset/concurrency queries (CT-011..CT-013). +Awaiting maintainer decisions: post the CT-007 and CT-011 issues +(`ct007-issue-draft.md`, `ct011-issue-draft.md`), and push/open the two PRs +(`pr-draft-main.md` here, `pr-draft-620.md` on the 6.20 branch). After that, +the next implementation slice is WS3 contact ordering/identity on a NEW +branch (this one is bounded per the work-package rules), motivated concretely +by CT-011's hidden contact state, followed by CT-006 heel-strike and the +per-solve residual (WS4 remainder). ## Context that would be lost From b716e9071b7ca9b6f40d4058dcefae26534a6264 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 09:17:44 -0700 Subject: [PATCH 16/52] Record the filed issue numbers (#3442, #3443) across the branch --- .../ct007-issue-draft.md | 3 ++- .../ct011-issue-draft.md | 3 ++- .../pr-draft-main.md | 8 ++++---- .../verification.md | 13 +++++++++++++ .../claims-manifest.json | 4 ++-- docs/plans/dashboard.md | 2 +- 6 files changed, 24 insertions(+), 9 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/ct007-issue-draft.md b/docs/dev_tasks/citation_driven_simulation_trust/ct007-issue-draft.md index 0795547e61db7..3a1c2143712b8 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/ct007-issue-draft.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/ct007-issue-draft.md @@ -1,6 +1,7 @@ # Issue draft: default contact solver fails high-mass-ratio stacks -Status: DRAFT, not posted. Posting to GitHub needs maintainer approval. +Status: POSTED as https://github.com/dartsim/dart/issues/3442 on +2026-08-15 with maintainer approval. This file is the posted text's source. Working copy for the CT-007 finding; delete with this dev-task folder after the issue is filed (or a decision is recorded not to file it). diff --git a/docs/dev_tasks/citation_driven_simulation_trust/ct011-issue-draft.md b/docs/dev_tasks/citation_driven_simulation_trust/ct011-issue-draft.md index 788127520f751..13ffe5cb36055 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/ct011-issue-draft.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/ct011-issue-draft.md @@ -1,6 +1,7 @@ # Issue draft: state-vector restore depends on prior contact history -Status: DRAFT, not posted. Posting to GitHub needs maintainer approval. +Status: POSTED as https://github.com/dartsim/dart/issues/3443 on +2026-08-15 with maintainer approval. This file is the posted text's source. Working copy for the CT-011 finding; delete with this dev-task folder after the issue is filed (or a decision is recorded not to file it). diff --git a/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md b/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md index c9c2ac8106650..8c8cf1d97160b 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md @@ -17,8 +17,7 @@ contract, six first-wave packets, and resolved-configuration bindings` fixture families, and Python bindings for the World's bake-time `ResolvedSolverConfiguration` so packets record the method that actually ran. -- Two findings surfaced by the packets need maintainer decisions and have - issue drafts in the dev task: the default `SEQUENTIAL_IMPULSE` contact +- Two findings surfaced by the packets are filed as #3442 and #3443: the default `SEQUENTIAL_IMPULSE` contact solver lets a heavy box sink fully through a light one at mass ratios of 100 and above (CT-007), and `World.state_vector` restore is not a function of the restored state once the world has contact history (CT-011). @@ -84,8 +83,9 @@ tests/test_check_citation_evidence.py -q` — 74 passed. - Companion release-6.20 PR: branch `feature/dart6-citation-contact-trust` (independent; adopts the same contract without DART 7 APIs). -- Issue drafts pending approval: CT-007 default-solver high-mass-ratio - failure; CT-011 restore-history dependence. +- Files #3442 (CT-007 default-solver high-mass-ratio failure) and #3443 + (CT-011 restore-history dependence); both packets and issue sources are in + this PR. #### Checklist diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 45ba6100c9cd9..1494a2ce2675a 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -407,3 +407,16 @@ zero and the observed closure is produced by the solve. first step. - Commands: packet writer as recorded in the packet; `pixi run check-citation-evidence` — OK. + +## External mutations — 2026-08-15 + +With explicit maintainer approval ("go ahead" on posting the issues, pushing +both branches, and opening the two PRs): + +- Posted issue #3442 (CT-007: SEQUENTIAL_IMPULSE lets a heavy box sink + through a light one at mass ratios >= 100) and issue #3443 (CT-011: + state-vector restore depends on prior contact history), from the drafts in + this folder. +- Merged the moved bases into both branches before pushing (lockfile-only + commits #3441 on `main`, #3440 on `release-6.20`), per the merge-first + push rule; `test-all` re-run on both merged heads before the push. diff --git a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json index 610eda592e43d..d8853355c619d 100644 --- a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json +++ b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json @@ -161,7 +161,7 @@ "evidence": [ "evidence/CT-007-dart7-high-mass-ratio.json" ], - "notes": "Two-box stack swept over mass ratios 1..1000 for both contact solvers. Disposition unresolved: the cited claim is comparative and no exact-cone arm exists on this branch. The baseline arm is established and is not a null result -- SEQUENTIAL_IMPULSE (the World default) fails completely at ratios 100 and 1000, the heavy box descending a full box height into the light one, while BOXED_LCP holds across all four decades. This is the WS5 GO/NO-GO input and is worth a maintainer decision on its own." + "notes": "Two-box stack swept over mass ratios 1..1000 for both contact solvers. Disposition unresolved: the cited claim is comparative and no exact-cone arm exists on this branch. The baseline arm is established and is not a null result -- SEQUENTIAL_IMPULSE (the World default) fails completely at ratios 100 and 1000, the heavy box descending a full box height into the light one, while BOXED_LCP holds across all four decades. This is the WS5 GO/NO-GO input and is worth a maintainer decision on its own. Filed as issue #3442." }, "dart6": { "owner": "evidence only; reference PR #3377 research lane", @@ -246,7 +246,7 @@ "evidence": [ "evidence/CT-011-dart7-restore-equivalence.json" ], - "notes": "Restore-equivalence protocol over seven arms, both contact solvers. Finding: World.state_vector restore is NOT a function of the restored state once contact history exists (in-place restores diverge at the first step and depend on prior history), while fresh-world restore is bit-exact and repeatable, identical histories agree, pre-contact history is harmless, and the ballistic control is exact. Deterministic throughout; deactivation ruled out by probing. Reset cost, allocation, and concurrency remain unmeasured. The mechanism hunt is the WS3 contact-identity work." + "notes": "Restore-equivalence protocol over seven arms, both contact solvers. Finding: World.state_vector restore is NOT a function of the restored state once contact history exists (in-place restores diverge at the first step and depend on prior history), while fresh-world restore is bit-exact and repeatable, identical histories agree, pre-contact history is harmless, and the ballistic control is exact. Deterministic throughout; deactivation ruled out by probing. Reset cost, allocation, and concurrency remain unmeasured. The mechanism hunt is the WS3 contact-identity work. Filed as issue #3443." }, "dart6": { "owner": "existing clone/reset/Recording evidence only", diff --git a/docs/plans/dashboard.md b/docs/plans/dashboard.md index 4121a5e26c1ec..c54ac4708ce95 100644 --- a/docs/plans/dashboard.md +++ b/docs/plans/dashboard.md @@ -57,7 +57,7 @@ its own line so status updates remain git-history friendly. sink fully through a light one at mass ratios >= 100 while BOXED_LCP holds) and CT-011 (state-vector restore is not a function of the restored state once contact history exists; fresh-world restore is bit-exact). - Issue drafts live in the dev task. Next slices: WS3 contact + Filed as issues #3442 and #3443. Next slices: WS3 contact ordering/identity (now motivated by CT-011), CT-006 heel-strike (gated on WS3), and the per-solve residual (WS4 remainder). WS4's first slice has landed: `World.resolved_configuration` is exposed to From 4980a9ee781714f24a6f72f795b0e97024db36df Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 09:26:28 -0700 Subject: [PATCH 17/52] Identify the CT-011 root cause: the state vector is translational by design The WS3 mechanism hunt closed in three probes. At the identical restored state only local_point_a differs between in-place and fresh worlds, and the in-place value is a body-frame anchor of a rotated sphere. World::getStateVector() stores position and linear velocity only per rigid body (the header documents the dense order as 'dynamic rigid-body translations', built for the differentiable path), and the decisive probe shows it directly: 30 entries for the five-sphere scene, and after restoring the same vector the in-place sphere keeps a rolled orientation and ~7 rad/s spin while the fresh sphere sits at identity with zero spin. So the CT-011 divergence is a partial restore by design, not hidden contact state: orientation and angular velocity are never captured, and friction couples the retained rotational state into the continuation. Contact history mattered only because contact is what makes spheres spin. The packet, manifest, verification log, and the posted issue #3443 (body updated with approval, as part of the approved posting workflow) now state the identified mechanism; the earlier contact-pipeline hypothesis is recorded as ruled out rather than silently replaced. --- .../ct011-issue-draft.md | 38 +++++++++++++------ .../verification.md | 34 +++++++++++++++++ .../claims-manifest.json | 2 +- .../CT-011-dart7-restore-equivalence.json | 7 ++-- ...tation_ct011_restore_equivalence_packet.py | 30 +++++++++++---- 5 files changed, 87 insertions(+), 24 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/ct011-issue-draft.md b/docs/dev_tasks/citation_driven_simulation_trust/ct011-issue-draft.md index 13ffe5cb36055..19bc8c63a4adf 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/ct011-issue-draft.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/ct011-issue-draft.md @@ -38,14 +38,24 @@ once the world has ever had contact: bit-exactly. Everything is deterministic given full history — this is not nondeterminism. -Some result-affecting contact state survives the `state_vector` write, so -the continuation depends on what the world did before the restore. The -failure is silent: state stays finite and each individual run is + +**Root cause (identified after filing; body updated):** +`World.state_vector` is a translational state by design: per dynamic rigid +body it carries position and linear velocity only (30 entries for the +five-sphere repro below), matching the header's "dynamic rigid-body +translations" wording for the differentiable rigid-body path. Orientation +and angular velocity are never captured or restored. A restored world +therefore keeps whatever rotational state it already had — after the warm-up +the spheres carry rolled orientations and ~7 rad/s spin, a fresh world has +identity orientations and zero spin — and friction couples that rotational +state into the continuation. Contact history matters only because contact is +what makes spheres spin. The failure is silent: the property name and shape +(`x = [q; qdot]`) invite reading it as the full state, nothing warns that +rotational state is omitted, and each individual run stays finite and reproducible, which makes it easy to mistake restored rollouts for -equivalent ones. Both `SEQUENTIAL_IMPULSE` and `BOXED_LCP` are affected -identically, which points at shared contact-pipeline state rather than a -solver-specific cache. Deactivation/sleeping is ruled out: no body was -asleep at the snapshot, and disabling deactivation changes nothing. +equivalent ones. Both contact solvers are affected identically for this +reason. (Deactivation/sleeping was ruled out separately: nothing was asleep +at the snapshot and disabling it changes nothing.) **Steps to Reproduce:** @@ -125,12 +135,16 @@ restore changes nothing. explicitly choose whether solver/contact history is preserved" (`docs/design/contact_trust_and_observability.md`). Currently the choice is implicit and history leaks through. -- Either resolution is defensible — clear contact history on state writes, - or capture it in the state vector — but the semantics should be explicit, - documented, and tested. The one-model/many-state work (PLAN-030/PLAN-123 - WS7) needs this settled. +- Suggested resolution: (a) make the translational scope impossible to + miss in the Python/C++ docs and consider a name that says so, and (b) + provide a lightweight full dynamic-state capture/restore (positions + including orientation, velocities including angular) for reset-heavy + workflows. `save_binary`/`load_binary` exist but serialize the full entity + graph through a file, and `restore_replay_frame` requires recording; the + one-model/many-state work (PLAN-030/PLAN-123 WS7) needs the lightweight + path settled. - Workaround: restore into a freshly built world; that is bit-exact and - repeatable. + repeatable (fresh worlds share default rotational state). - Full evidence packet (seven protocol arms, all hashes, deterministic repeats): `CT-011-dart7-restore-equivalence.json` under `docs/plans/123-citation-driven-simulation-trust/evidence/` on the diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 1494a2ce2675a..0641dc085c2c7 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -420,3 +420,37 @@ both branches, and opening the two PRs): - Merged the moved bases into both branches before pushing (lockfile-only commits #3441 on `main`, #3440 on `release-6.20`), per the merge-first push rule; `test-all` re-run on both merged heads before the push. + +## CT-011 root cause identified — 2026-08-15 + +The WS3 mechanism hunt closed in three probes against the rebuilt tree: + +1. At the identical restored state, in-place and fresh worlds report the same + contact pairs, depths, world points, and normals -- only `local_point_a` + differs, and the in-place value is a radius-length vector pointing away + from the bottom pole: a body-frame anchor of a ROTATED sphere. +2. Reading `World::getStateVector()` shows the design directly: per dynamic + rigid body it stores `transform.position` and `velocity.linear` only. The + header documents the dense order as "dynamic rigid-body translations" + (built for the differentiable rigid-body path). +3. The decisive probe: `state_vector` has 30 entries for the five-sphere + scene (3 pos + 3 linvel each); after restoring the same vector, the + in-place sphere keeps a rolled orientation (Frobenius deviation 2.41 from + identity) and ~7 rad/s angular velocity while the fresh sphere sits at + identity with zero spin. + +So the CT-011 divergence is a PARTIAL RESTORE by design, not hidden contact +state: orientation and angular velocity are never captured, a restored world +keeps whatever rotational state it had, and friction couples that into the +continuation. Contact history mattered only because contact is what makes +spheres spin -- the earlier "shared contact-pipeline state" hypothesis in +issue #3443 was wrong in mechanism (the behavioral facts and repro all +stand) and the issue body was corrected with maintainer approval as part of +the approved posting workflow. The packet limitation now states the root +cause and the full-state alternatives (save_binary/load_binary, +restore_replay_frame), neither of which is a lightweight in-memory reset. + +The dead-end candidates are recorded because ruling them out was real work: +persistent_manifold_cache.cpp (warm-start impulses, classic dart detector +facade -- not on the DART 7 World query path) and deactivation state (probed, +nothing asleep, disabling changed nothing). diff --git a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json index d8853355c619d..0e83c11ef4c11 100644 --- a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json +++ b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json @@ -246,7 +246,7 @@ "evidence": [ "evidence/CT-011-dart7-restore-equivalence.json" ], - "notes": "Restore-equivalence protocol over seven arms, both contact solvers. Finding: World.state_vector restore is NOT a function of the restored state once contact history exists (in-place restores diverge at the first step and depend on prior history), while fresh-world restore is bit-exact and repeatable, identical histories agree, pre-contact history is harmless, and the ballistic control is exact. Deterministic throughout; deactivation ruled out by probing. Reset cost, allocation, and concurrency remain unmeasured. The mechanism hunt is the WS3 contact-identity work. Filed as issue #3443." + "notes": "Restore-equivalence protocol over seven arms, both contact solvers. Finding with identified root cause: World.state_vector is translational by design (position + linear velocity per rigid body), so orientation and angular velocity are never restored; a restored world keeps its prior rotational state, and friction couples that into the continuation. Fresh-world restore is bit-exact because fresh worlds share default rotational state. Reset cost, allocation, and concurrency remain unmeasured. Filed as issue #3443 (mechanism corrected after the root-cause probe)." }, "dart6": { "owner": "existing clone/reset/Recording evidence only", diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json index bdbb9429a0619..4f55c74841af9 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json @@ -318,10 +318,11 @@ } }, "result": { - "claim_boundary": "DART 7 main, this commit, a five-sphere pile with a 0.5 s contact-rich warm-up, restored via World.state_vector, both contact solvers. Established: restore is NOT a function of the restored state once the world has contact history -- the in-place continuation differs from the original at the first step, and two in-place restores of the same snapshot differ from each other when different history precedes them -- while a freshly built world restoring the same vector is bit-exact and repeatable, identical histories give identical continuations, pre-contact history is harmless, and the ballistic control is exact. Everything is deterministic given full history; nothing here is nondeterminism. The corpus row's reset cost, overhead, and concurrency halves are not measured. Says nothing about Skeleton-based state APIs, clone(), replay restore, historical DART versions, or DART 6.", + "claim_boundary": "DART 7 main, this commit, a five-sphere pile with a 0.5 s contact-rich warm-up, restored via World.state_vector, both contact solvers. Established: restore is NOT a function of the restored state once the world has contact history -- the in-place continuation differs from the original at the first step, and two in-place restores of the same snapshot differ from each other when different history precedes them -- while a freshly built world restoring the same vector is bit-exact and repeatable, identical histories give identical continuations, pre-contact history is harmless, and the ballistic control is exact. Everything is deterministic given full history; nothing here is nondeterminism. Root cause: the state vector is translational by design and silently omits orientation and angular velocity, so the restore is partial. The corpus row's reset cost, overhead, and concurrency halves are not measured. Says nothing about Skeleton-based state APIs, clone(), replay restore, historical DART versions, or DART 6.", "disposition": "unresolved", "limitations": [ - "The mechanism is not identified: contact history leaves result-affecting state that a state-vector write does not reset, in both contact solvers, and deactivation was ruled out by probing (nothing was asleep and disabling it changed nothing). Naming the exact carrier (contact manifold caches, warm starts, or contact ordering) is the WS3 contact-identity work this finding motivates.", + "The mechanism is identified: World.state_vector is a translational state by design (per rigid body it carries position and linear velocity only -- 30 entries for the five-sphere scene -- as documented for the differentiable rigid-body path), so orientation and angular velocity are never captured or restored. A restored world keeps whatever rotational state it already had, which is exactly the history dependence measured here: after the warm-up the spheres carry finite spin (about 7 rad/s on s0) and rolled orientations, a fresh world has identity orientations and zero spin, and friction couples that rotational state into the continuation. Contact history matters only because contact is what makes spheres spin.", + "Full-state alternatives on this branch are World.save_binary/load_binary (full entity serialization through a file) and restore_replay_frame; neither is a lightweight in-memory reset, and no state-vector-like API carries orientation and angular velocity today.", "One scene, one snapshot point, 0.2 s windows; the magnitude of divergence (up to ~1e-2 in state units within the window) will vary with scene and horizon.", "Reset cost, allocation, and thread/lane isolation are typed unsupported; they are the other halves of the corpus row.", "The replay-frame restore mechanism (World.restore_replay_frame) is not tested here and may have different semantics." @@ -371,7 +372,7 @@ }, "target": { "branch": "main", - "commit": "ce5142e8cd7c76d6adc120e703482e3f898e4567", + "commit": "b716e9071b7ca9b6f40d4058dcefae26534a6264", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit." }, "title": "State-vector restore equivalence under contact (DART 7 first packet)" diff --git a/scripts/write_citation_ct011_restore_equivalence_packet.py b/scripts/write_citation_ct011_restore_equivalence_packet.py index 989eafd9104a4..d27a6d568205e 100644 --- a/scripts/write_citation_ct011_restore_equivalence_packet.py +++ b/scripts/write_citation_ct011_restore_equivalence_packet.py @@ -501,19 +501,33 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "identical histories give identical continuations, " "pre-contact history is harmless, and the ballistic control " "is exact. Everything is deterministic given full history; " - "nothing here is nondeterminism. The corpus row's reset " + "nothing here is nondeterminism. Root cause: the state " + "vector is translational by design and silently omits " + "orientation and angular velocity, so the restore is " + "partial. The corpus row's reset " "cost, overhead, and concurrency halves are not measured. " "Says nothing about Skeleton-based state APIs, clone(), " "replay restore, historical DART versions, or DART 6." ), "limitations": [ - "The mechanism is not identified: contact history leaves " - "result-affecting state that a state-vector write does not " - "reset, in both contact solvers, and deactivation was ruled " - "out by probing (nothing was asleep and disabling it " - "changed nothing). Naming the exact carrier (contact " - "manifold caches, warm starts, or contact ordering) is the " - "WS3 contact-identity work this finding motivates.", + "The mechanism is identified: World.state_vector is a " + "translational state by design (per rigid body it carries " + "position and linear velocity only -- 30 entries for the " + "five-sphere scene -- as documented for the differentiable " + "rigid-body path), so orientation and angular velocity are " + "never captured or restored. A restored world keeps whatever " + "rotational state it already had, which is exactly the " + "history dependence measured here: after the warm-up the " + "spheres carry finite spin (about 7 rad/s on s0) and rolled " + "orientations, a fresh world has identity orientations and " + "zero spin, and friction couples that rotational state into " + "the continuation. Contact history matters only because " + "contact is what makes spheres spin.", + "Full-state alternatives on this branch are " + "World.save_binary/load_binary (full entity serialization " + "through a file) and restore_replay_frame; neither is a " + "lightweight in-memory reset, and no state-vector-like API " + "carries orientation and angular velocity today.", "One scene, one snapshot point, 0.2 s windows; the " "magnitude of divergence (up to ~1e-2 in state units within " "the window) will vary with scene and horizon.", From cebf5a86799a10615f0f7a6878a1ccdf5f315db0 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 09:36:19 -0700 Subject: [PATCH 18/52] Record the opened PR and issue numbers in the working docs --- .../citation_driven_simulation_trust/RESUME.md | 11 +++++++---- .../citation_driven_simulation_trust/pr-draft-main.md | 4 ++-- 2 files changed, 9 insertions(+), 6 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md index 2d7b997067b26..df0f225d043d5 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md @@ -47,10 +47,13 @@ gate green. GitHub mutations need maintainer approval. ## Immediate next step -Awaiting maintainer decisions: post the CT-007 and CT-011 issues -(`ct007-issue-draft.md`, `ct011-issue-draft.md`), and push/open the two PRs -(`pr-draft-main.md` here, `pr-draft-620.md` on the 6.20 branch). After that, -the next implementation slice is WS3 contact ordering/identity on a NEW +Approved and executed on 2026-08-15: issues #3442 and #3443 posted (#3443 +later corrected with the identified root cause), both branches pushed after +merging the moved bases and clean-rebuild test-all, and PRs opened -- #3445 +(main, DART 7.0) and #3444 (release-6.20, DART 6.20.0), each with a Codex +review requested. Shepherd both PRs through CI and review (re-request +'@codex review' after every push, iterate until clean). After merge, the +next implementation slice is WS3 contact ordering/identity on a NEW branch (this one is bounded per the work-package rules), motivated concretely by CT-011's hidden contact state, followed by CT-006 heel-strike and the per-solve residual (WS4 remainder). diff --git a/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md b/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md index 8c8cf1d97160b..b48d360edd9de 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md @@ -1,7 +1,7 @@ # PR draft: DART 7 citation-trust foundation -Status: DRAFT. Pushing the branch and opening the PR need maintainer -approval. Branch `feature/citation-trust-foundation`, base `main`, milestone +Status: OPENED as https://github.com/dartsim/dart/pull/3445 on 2026-08-15 +with maintainer approval; this file is the posted body's source. Branch `feature/citation-trust-foundation`, base `main`, milestone DART 7.0. Delete this file with the dev-task folder at task completion. --- From 6c094df2a82840345144ab48dd8b8d132ee197c0 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 21:58:31 -0700 Subject: [PATCH 19/52] Close the Codex round-1 validator loopholes and bind reviews to content Addresses the 14 findings from the PR #3445 Codex reviews: Validator (scripts/check_citation_evidence.py, mirrored on release-6.20): a closing lane's disposition must equal its packet's result.disposition; the two-review floor requires distinct reviewers; every review pass must carry a content_digest equal to the packet-minus-review digest, so regeneration invalidates stale reviews; raw_rows entries must be non-empty structured records; raw_paths and visual artifact paths must be relative, non-escaping, and resolve inside an approved evidence root; sweep/seed ensembles need valid, distinct entries; string metric leaves are only allowed under semantic keys so prose cannot masquerade as a measurement; configuration.requested/resolved must carry a recognizable identity key and no null values; NaN/Infinity are rejected at JSON load; manifest lane paths are normalized before indexing so dual spellings cannot split ownership or dodge the closure checks; target.fetch_hint is required, recording that PR head refs keep target commits fetchable after squash-merge (all seven packet commits verified as ancestors of the PR head). preserve_review() now takes the regenerated packet and keeps only digest-bound passes (legacy call keeps nothing -- fails closed); record_review_pass() added; every writer rebinds after assembly and emits target.fetch_hint. CT-011 writer: the restore-is-a-state-function aggregate now includes every protocol arm (fresh-world, cross-history, pre-contact, ballistic), not just the two in-place booleans, and the writer aborts with an explicit message if all arms ever match bit-exactly rather than regenerating the divergence narrative against contradicting measurements. Packet regenerated; findings unchanged. Dev-task docs: the RESUME handoff no longer instructs unconditional @codex re-triggers (scoped to approved post-fix rounds per the external-mutation policy), and the README next-steps mark the landed ResolvedSolverConfiguration bindings as done, leaving only the per-solve residual as WS4 work. Tests: 74 -> 85 validator cases; every new rejection has one. Existing packets migrated with fetch_hint; check-citation-evidence passes on the real tree and the permanent negative control keeps failing with a larger error count. --- .../README.md | 8 +- .../RESUME.md | 7 +- .../pr-draft-main.md | 4 +- .../verification.md | 38 +++ .../CT-001-dart7-rolling-direction.json | 3 +- .../CT-002-dart7-dense-inelastic-contact.json | 3 +- .../CT-003-dart7-dense-elastic-contact.json | 3 +- ...004-dart7-articulated-energy-momentum.json | 3 +- .../evidence/CT-005-dart7-pd-tracking.json | 3 +- .../CT-007-dart7-high-mass-ratio.json | 3 +- .../CT-011-dart7-restore-equivalence.json | 5 +- scripts/check_citation_evidence.py | 293 +++++++++++++++++- scripts/citation_packet_utils.py | 82 ++++- ...citation_ct001_rolling_direction_packet.py | 10 +- ...ite_citation_ct002_dense_contact_packet.py | 15 +- ...e_citation_ct003_elastic_contact_packet.py | 15 +- ...itation_ct004_articulated_energy_packet.py | 10 +- ...write_citation_ct005_pd_tracking_packet.py | 10 +- ...e_citation_ct007_high_mass_ratio_packet.py | 10 +- ...tation_ct011_restore_equivalence_packet.py | 40 ++- tests/test_check_citation_evidence.py | 224 ++++++++++++- 21 files changed, 722 insertions(+), 67 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/README.md b/docs/dev_tasks/citation_driven_simulation_trust/README.md index c9504c69f5ce8..38355a5a52456 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/README.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/README.md @@ -172,8 +172,10 @@ for downstream-sensitive changes. depend on contact history. Issue drafts live in this folder. 3. Start WS3 contact ordering/identity, now motivated concretely by CT-011's hidden contact state; CT-006 heel-strike follows it. -4. Start WS4's first slice in parallel where it unblocks packets: expose - `ResolvedSolverConfiguration` to Python and a comparable per-solve - residual, both currently typed unsupported in every packet. +4. WS4 slice 1 is DONE in this branch: `ResolvedSolverConfiguration` and + `World.resolved_configuration` are bound to Python (module_compute.cpp / + module_world.cpp) and every packet records the World's bake-time + resolution. The remaining WS4 packet gap is a comparable per-solve + residual, still typed unsupported everywhere. 5. Reuse PR #3377 fixtures for the DART 6 CT-007 lane rather than adding new scenes; keep PLAN-621/622 rows referenced, not copied. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md index df0f225d043d5..e27f080c65cc8 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md @@ -51,8 +51,11 @@ Approved and executed on 2026-08-15: issues #3442 and #3443 posted (#3443 later corrected with the identified root cause), both branches pushed after merging the moved bases and clean-rebuild test-all, and PRs opened -- #3445 (main, DART 7.0) and #3444 (release-6.20, DART 6.20.0), each with a Codex -review requested. Shepherd both PRs through CI and review (re-request -'@codex review' after every push, iterate until clean). After merge, the +review requested. Shepherd both PRs through CI and review. A Codex +re-request is an external mutation: post '@codex review' only for a push +that addresses prior Codex feedback within a maintainer-approved review +round (see AGENTS.md external-mutation policy) -- never unconditionally, +and never as a repeated nudge, which also risks review throttling. After merge, the next implementation slice is WS3 contact ordering/identity on a NEW branch (this one is bounded per the work-package rules), motivated concretely by CT-011's hidden contact state, followed by CT-006 heel-strike and the diff --git a/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md b/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md index b48d360edd9de..c3181f7b365eb 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/pr-draft-main.md @@ -39,7 +39,7 @@ contract, six first-wave packets, and resolved-configuration bindings` agreement with the human corpus enforced), packet schema `dart.citation_claim_evidence/v1`, validator `scripts/check_citation_evidence.py` in `check-lint`, permanent - intentionally incomplete negative control that must keep failing, 74 + intentionally incomplete negative control that must keep failing, 85 pytest cases. The gate rejects prose/non-JSON/outside-`evidence/`/ negative-control/non-string lane references, dangling or directory `raw_paths`, scene digests that disagree with the published parameters, @@ -68,7 +68,7 @@ contract, six first-wave packets, and resolved-configuration bindings` - `pixi run check-lint` (includes the new `check-citation-evidence` gate) and `pixi run test-all` — all phases pass on the branch head. - `pixi run python -I scripts/run_pytest.py -tests/test_check_citation_evidence.py -q` — 74 passed. +tests/test_check_citation_evidence.py -q` — 85 passed. - Every packet regenerated on the current writers; deterministic repeats bit-identical; dispositions computed from recorded data with negative controls and physical-validity gates. diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 0641dc085c2c7..51a13bd0dedcb 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -454,3 +454,41 @@ The dead-end candidates are recorded because ruling them out was real work: persistent_manifold_cache.cpp (warm-start impulses, classic dart detector facade -- not on the DART 7 World query path) and deactivation state (probed, nothing asleep, disabling changed nothing). + +## Codex review round 1 (PR #3445) — 2026-08-16 + +Codex posted 4 reviews with 14 inline findings (P1/P2); all addressed in +this round, fixes silent per repo convention (no thread replies): + +- Validator fail-closed gaps (mirrored on release-6.20): closed-lane + disposition must match the packet's `result.disposition`; the two-review + floor requires DISTINCT reviewers; review passes must carry a + `content_digest` binding them to the packet content they reviewed; + `raw_rows` entries must be non-empty structured records; + `raw_paths`/`visual` paths must be relative, non-escaping, and resolve + inside approved roots; sweep/seed ensembles need valid DISTINCT entries; + string metric leaves are allowed only under semantic keys; requested/ + resolved configurations must carry a recognizable identity key with no + null values; NaN/Infinity rejected at JSON load; lane evidence paths + normalized before indexing; `target.fetch_hint` required (PR head refs + survive squash-merge, keeping recorded commits reproducible — all seven + packet target commits verified as ancestors of the PR head). +- `preserve_review` binds carried passes to the regenerated packet's + content digest; `record_review_pass` added; all seven writers rebind + after assembly and emit `target.fetch_hint`. +- CT-011 writer: the state-function aggregate now includes EVERY protocol + arm (fresh-world, cross-history, pre-contact, ballistic — not just the + two in-place booleans), and the writer aborts with an explicit error if + a future branch makes every arm bit-exact, instead of regenerating the + divergence narrative against contradicting measurements. Packet + regenerated; findings unchanged (divergence still observed, disposition + `unresolved`). +- Dev-task docs: RESUME no longer instructs unconditional `@codex review` + re-triggering (scoped to approved post-fix rounds per the + external-mutation policy); README's next-steps no longer tell a resumed + agent to re-implement the already-landed `ResolvedSolverConfiguration` + bindings (only the per-solve residual remains WS4 work). +- Validator test suite: 74 -> 85 cases; existing packets migrated with + `fetch_hint`; `pixi run check-citation-evidence` passes on the real + tree; the permanent negative control keeps failing with a higher error + count under the stricter rules. diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json index 939d0a9ba0bb6..c5023e59b406f 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json @@ -1263,7 +1263,8 @@ "target": { "branch": "main", "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05", - "commit_role": "Source state measured: the library and fixture were run at this commit, which is HEAD at capture time. The packet and its writer land in a later commit, so re-running the recorded command requires the child commit that adds the writer; the measured behavior belongs to this one." + "commit_role": "Source state measured: the library and fixture were run at this commit, which is HEAD at capture time. The packet and its writer land in a later commit, so re-running the recorded command requires the child commit that adds the writer; the measured behavior belongs to this one.", + "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, "title": "Rolling-direction friction dependence (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json index 7bcee488904f0..532f0149d3f0a 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json @@ -671,7 +671,8 @@ }, "target": { "branch": "main", - "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05" + "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05", + "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, "title": "Dense 6x6x6 inelastic contact stability (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json index 5a513b3d117ba..daa27f23e255c 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json @@ -574,7 +574,8 @@ }, "target": { "branch": "main", - "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05" + "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05", + "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, "title": "Dense elastic contact energy envelope (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json index 1da9f7509325b..26866147bec53 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json @@ -978,7 +978,8 @@ "target": { "branch": "main", "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05", - "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit." + "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", + "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, "title": "Articulated energy drift versus timestep across integration families (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json index f723726914060..49fc2e079fc00 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json @@ -977,7 +977,8 @@ "target": { "branch": "main", "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05", - "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit." + "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", + "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, "title": "PD-control tracking error and control work versus timestep (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json index c424cd970189e..c645d557be182 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json @@ -1008,7 +1008,8 @@ "target": { "branch": "main", "commit": "ebbbb853a49cf59bafb590cf111c110acbd0531b", - "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit." + "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", + "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, "title": "High-mass-ratio stack conditioning baseline for the exact-cone GO/NO-GO (DART 7)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json index 4f55c74841af9..8f52b07857246 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json @@ -372,8 +372,9 @@ }, "target": { "branch": "main", - "commit": "b716e9071b7ca9b6f40d4058dcefae26534a6264", - "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit." + "commit": "cebf5a86799a10615f0f7a6878a1ccdf5f315db0", + "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", + "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, "title": "State-vector restore equivalence under contact (DART 7 first packet)" } diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index f4699520a3f9e..b97c055354776 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -35,10 +35,11 @@ import hashlib import json import math +import posixpath import re import subprocess import sys -from pathlib import Path +from pathlib import Path, PurePosixPath REPO_ROOT = Path(__file__).resolve().parents[1] PLAN_DIR = REPO_ROOT / "docs" / "plans" / "123-citation-driven-simulation-trust" @@ -72,6 +73,60 @@ BRANCH_BY_LANE = {"dart7": "main", "dart6": "release-6.20"} FIRST_WAVE_FAMILY_CAP = 6 +# String metric leaves are allowed only under keys that clearly name semantic +# metadata; everywhere else a string is prose masquerading as a measurement. +METRIC_STRING_KEY_EXACT = frozenset( + { + "method", + "note", + "regime", + "unit", + "units", + "kind", + "signature_test", + "attribution", + "contact_solver_method", + "detector", + "criteria_exceeded", + "families_with_shrinking_drift", + } +) +METRIC_STRING_KEY_SUFFIXES = ( + "_note", + "_notes", + "_semantics", + "_method", + "_methods", + "_reasons", + "_criteria", + "_test", + "_families", + "_groups", + "_detectors", + "_basis", +) + +# configuration.requested/resolved must name a recognizable identity, not an +# arbitrary placeholder object. +IDENTITY_KEY_RE = re.compile( + r"solver|method|detector|integrator|integration|backend|family", re.I +) + + +def _has_identity_key(value: object) -> bool: + """True when any key at any depth names a solver/method/... identity + with a non-null value; {"placeholder": null} has none.""" + if isinstance(value, dict): + for key, item in value.items(): + if IDENTITY_KEY_RE.search(str(key)) and item is not None: + return True + if _has_identity_key(item): + return True + elif isinstance(value, list): + return any(_has_identity_key(item) for item in value) + return False + + PACKET_TOP_LEVEL_KEYS = { "schema", "claim_id", @@ -105,6 +160,53 @@ def _is_finite_number(value: object) -> bool: ) +def _evidence_path_issue(raw_path: str, base_dir: Path | None) -> str | None: + """Why a packet-referenced artifact path is unacceptable, or None. + + Paths must stay relative and resolve to an existing file INSIDE an + approved root (the plan directory or the repository); an absolute or + escaping path can satisfy a naive existence check with host-local data + that is neither tracked nor portable. + """ + pure = PurePosixPath(raw_path) + if ( + pure.is_absolute() + or raw_path[1:2] == ":" + or "\\" in raw_path + or ".." in pure.parts + ): + return ( + f"{raw_path!r} must be a relative path inside the repository " + "evidence tree (no absolute paths, drive letters, or '..')" + ) + if base_dir is None: + return None + for root in (base_dir, REPO_ROOT): + candidate = (root / raw_path).resolve() + if candidate.is_file() and candidate.is_relative_to(root.resolve()): + return None + return ( + f"{raw_path!r} does not resolve to an existing file inside an " + "approved evidence root; a dangling or escaping path is prose" + ) + + +def _packet_content_digest(packet: dict) -> str: + """Digest of the packet minus its review block. + + A review pass binds to this digest; regenerating a packet with different + content therefore invalidates prior passes instead of silently carrying + them onto evidence they never reviewed. + """ + content = {key: value for key, value in packet.items() if key != "review"} + return ( + "sha256:" + + hashlib.sha256( + json.dumps(content, sort_keys=True, separators=(",", ":")).encode("utf-8") + ).hexdigest() + ) + + def _is_unsupported_leaf(value: object) -> bool: """True when a nested value is a typed-unsupported marker.""" return ( @@ -213,6 +315,23 @@ def metric_group_errors(name: str, group: object) -> list[str]: f"metrics.{name}.{path} uses the placeholder {leaf!r}; " "unsupported values must be typed, not spelled" ) + elif isinstance(leaf, str): + terminal = re.sub(r"(\[\d+\])+$", "", path.rsplit(".", 1)[-1]) + if not ( + terminal in METRIC_STRING_KEY_EXACT + or terminal.endswith(METRIC_STRING_KEY_SUFFIXES) + ): + errors.append( + f"metrics.{name}.{path} is prose where a measurement " + "is expected; keep strings under semantic keys " + "(method, *_note, *_semantics, ...) or type the value " + "{'status': 'unsupported', 'reason': ...}" + ) + elif not leaf.strip(): + errors.append( + f"metrics.{name}.{path} is a blank string; record the " + "annotation or drop the key" + ) elif isinstance(leaf, (int, float)) and not isinstance(leaf, bool): if not math.isfinite(leaf): errors.append(f"metrics.{name}.{path} contains a non-finite number") @@ -292,6 +411,12 @@ def packet_errors( commit = target.get("commit") if not (isinstance(commit, str) and COMMIT_RE.match(commit)): errors.append("target.commit must be a 40-hex commit hash") + if not _is_nonempty_str(target.get("fetch_hint")): + errors.append( + "target.fetch_hint must record how a clean checkout fetches " + "target.commit (e.g. 'git fetch origin pull//head'); a " + "commit that later becomes unreachable is not reproducible" + ) scene = packet.get("scene") if not isinstance(scene, dict): @@ -330,6 +455,19 @@ def packet_errors( value = configuration.get(side) if not isinstance(value, dict) or not value: errors.append(f"configuration.{side} must be a non-empty object") + continue + null_keys = sorted(key for key, item in value.items() if item is None) + if null_keys: + errors.append( + f"configuration.{side} carries null values at {null_keys}; " + "record the identity or omit the key" + ) + if not _has_identity_key(value): + errors.append( + f"configuration.{side} records no recognizable " + "solver/method/detector/integrator/backend identity field; " + "an arbitrary placeholder object is not a configuration" + ) if not _is_nonempty_str(configuration.get("resolved_provenance")): errors.append( "configuration.resolved_provenance must name how the resolved " @@ -355,8 +493,43 @@ def packet_errors( has_repeats = ( isinstance(repeats, int) and not isinstance(repeats, bool) and repeats >= 2 ) + # A sweep or seed list only counts as an ensemble when its entries are + # valid and mutually distinct; [null, null] or a duplicated point is + # one run wearing an ensemble's clothes. has_sweep = isinstance(sweep, list) and len(sweep) >= 2 + if has_sweep: + canonical_points: list[str] = [] + for index, entry in enumerate(sweep): + if isinstance(entry, dict) and entry: + canonical_points.append(json.dumps(entry, sort_keys=True)) + elif _is_finite_number(entry) or _is_nonempty_str(entry): + canonical_points.append(json.dumps(entry)) + else: + errors.append( + f"ensemble.sweep[{index}] must be a non-empty object, " + "finite number, or non-empty string sweep point" + ) + has_sweep = False + if has_sweep and len(set(canonical_points)) < 2: + errors.append( + "ensemble.sweep must contain at least two DISTINCT points" + ) + has_sweep = False has_seeds = isinstance(seeds, list) and len(seeds) >= 2 + if has_seeds: + for index, entry in enumerate(seeds): + if not ( + (isinstance(entry, int) and not isinstance(entry, bool)) + or _is_nonempty_str(entry) + ): + errors.append( + f"ensemble.seeds[{index}] must be an integer or " + "non-empty string seed" + ) + has_seeds = False + if has_seeds and len({repr(entry) for entry in seeds}) < 2: + errors.append("ensemble.seeds must contain at least two DISTINCT seeds") + has_seeds = False if not (has_repeats or has_sweep or has_seeds): errors.append( "ensemble must record deterministic_repeats >= 2, a sweep of " @@ -400,19 +573,24 @@ def packet_errors( has_rows = isinstance(raw_rows, list) and bool(raw_rows) if not (has_paths or has_rows): errors.append("evidence must carry raw_rows inline or non-empty raw_paths") + if has_rows: + for index, row in enumerate(raw_rows): + if not (isinstance(row, dict) and row): + errors.append( + f"evidence.raw_rows[{index}] must be a non-empty " + "structured record; null or scalar placeholders are " + "not raw evidence" + ) if has_paths: for index, raw_path in enumerate(raw_paths): if not _is_nonempty_str(raw_path): errors.append( f"evidence.raw_paths[{index}] must be a non-empty string" ) - elif base_dir is not None and not any( - (root / raw_path).is_file() for root in (base_dir, REPO_ROOT) - ): - errors.append( - f"evidence.raw_paths[{index}] {raw_path!r} does not " - "resolve to an existing file; a dangling path is prose" - ) + continue + issue = _evidence_path_issue(raw_path, base_dir) + if issue is not None: + errors.append(f"evidence.raw_paths[{index}] {issue}") visual = evidence.get("visual") if isinstance(visual, dict): if visual.get("status") != "not-applicable" or not _is_nonempty_str( @@ -422,7 +600,27 @@ def packet_errors( "evidence.visual object form must be " "{'status': 'not-applicable', 'reason': ...}" ) - elif not (isinstance(visual, list) and visual): + elif isinstance(visual, list) and visual: + for index, item in enumerate(visual): + if _is_nonempty_str(item): + item_path = item + elif ( + isinstance(item, dict) + and _is_nonempty_str(item.get("path")) + and _is_nonempty_str(item.get("description")) + ): + item_path = item["path"] + else: + errors.append( + f"evidence.visual[{index}] must be an artifact path " + "or {'path': ..., 'description': ...}; a placeholder " + "cannot stand in for visual evidence" + ) + continue + issue = _evidence_path_issue(item_path, base_dir) + if issue is not None: + errors.append(f"evidence.visual[{index}] {issue}") + else: errors.append( "evidence.visual must list visual artifacts or be typed " "not-applicable with a reason" @@ -455,6 +653,7 @@ def packet_errors( if not isinstance(passes, list): errors.append("review.passes must be a list") else: + expected_digest = _packet_content_digest(packet) for index, entry in enumerate(passes): if ( not isinstance(entry, dict) @@ -465,6 +664,14 @@ def packet_errors( f"review.passes[{index}] needs non-empty reviewer " "and summary" ) + continue + if entry.get("content_digest") != expected_digest: + errors.append( + f"review.passes[{index}] is not bound to this " + "packet's content (content_digest must equal " + f"{expected_digest}); a review recorded against " + "earlier evidence does not carry over" + ) return errors @@ -569,10 +776,25 @@ def manifest_errors(manifest: object, corpus_ids: list[str]) -> list[str]: return errors +def _reject_nonstandard_constant(constant: str) -> None: + """Reject NaN/Infinity/-Infinity anywhere in a packet at load time. + + Python's default loader accepts them, but they are not JSON: a packet + carrying one is unreadable to strict consumers and therefore not the + portable machine evidence the contract promises. + """ + raise ValueError(f"non-standard JSON constant {constant!r}") + + def _load_json(path: Path, errors: list[str]) -> object | None: try: - return json.loads(path.read_text(encoding="utf-8")) - except (OSError, json.JSONDecodeError) as error: + return json.loads( + path.read_text(encoding="utf-8"), + parse_constant=_reject_nonstandard_constant, + ) + except (OSError, ValueError) as error: + # json.JSONDecodeError subclasses ValueError; parse_constant raises + # a plain ValueError for NaN/Infinity. errors.append(f"{path}: unreadable JSON ({error})") return None @@ -616,7 +838,7 @@ def validate_tree( manifest = _load_json(manifest_path, errors) known_ids: set[str] = set() lane_evidence: dict[str, tuple[str, str]] = {} - closed_lane_packets: set[str] = set() + closed_lane_dispositions: dict[str, object] = {} if manifest is not None: manifest_issues = manifest_errors(manifest, corpus_ids) errors.extend(f"{manifest_path}: {issue}" for issue in manifest_issues) @@ -650,6 +872,23 @@ def validate_tree( "the packet checks never reach" ) continue + # Canonicalize before indexing: `evidence/./p.json` + # and `evidence/p.json` are one file and must share + # one owner/one review record, and an escaping path + # must not become a distinct index key. + normalized = posixpath.normpath(rel) + if ( + posixpath.isabs(normalized) + or normalized.startswith("..") + or "\\" in rel + ): + errors.append( + f"{manifest_path}: {claim_id}.lanes." + f"{lane_name}.evidence entry {rel!r} escapes " + "the plan directory" + ) + continue + rel = normalized if True: if rel in lane_evidence: owner = lane_evidence[rel] @@ -664,7 +903,7 @@ def validate_tree( str(lane_name), ) if lane.get("status") == "closed": - closed_lane_packets.add(rel) + closed_lane_dispositions[rel] = lane.get("disposition") # Every lane-referenced path is validated as a packet, wherever it sits. # Enumerating only `evidence/*.json` would let a lane close a row with a @@ -747,7 +986,7 @@ def validate_tree( f"{packet_path}: target.branch {branch!r} does not match " f"lane {lane_name} branch {expected_branch!r}" ) - if rel in closed_lane_packets: + if rel in closed_lane_dispositions: passes = ( packet.get("review", {}).get("passes") if isinstance(packet.get("review"), dict) @@ -758,6 +997,32 @@ def validate_tree( f"{packet_path}: a packet closing a lane needs at " "least two recorded review passes" ) + else: + reviewers = { + entry.get("reviewer") + for entry in passes + if isinstance(entry, dict) + and _is_nonempty_str(entry.get("reviewer")) + } + if len(reviewers) < 2: + errors.append( + f"{packet_path}: a packet closing a lane needs " + "two INDEPENDENT review passes (distinct " + "reviewers); a duplicated reviewer is one review" + ) + lane_disposition = closed_lane_dispositions[rel] + packet_disposition = ( + packet.get("result", {}).get("disposition") + if isinstance(packet.get("result"), dict) + else None + ) + if lane_disposition != packet_disposition: + errors.append( + f"{packet_path}: the closing lane records disposition " + f"{lane_disposition!r} but the packet's result is " + f"{packet_disposition!r}; the manifest cannot publish " + "a conclusion its evidence does not support" + ) if freshness_head is not None: commit = ( packet.get("target", {}).get("commit") diff --git a/scripts/citation_packet_utils.py b/scripts/citation_packet_utils.py index 9e20c3745a34c..12a2070d112f5 100644 --- a/scripts/citation_packet_utils.py +++ b/scripts/citation_packet_utils.py @@ -112,13 +112,50 @@ def solver_iterations_by_method( return typed -def preserve_review(output_path: "Any") -> dict[str, Any]: - """Carry forward review passes already recorded in a packet. +# PLAN-123's evidence lands through this pull request; its head ref survives +# branch deletion and squash-merge, which is what makes recorded target +# commits reproducible from a clean checkout. +CITATION_PR_NUMBER = 3445 + + +def target_fetch_hint() -> str: + """How a clean checkout reaches a packet's target.commit forever.""" + return ( + f"git fetch origin pull/{CITATION_PR_NUMBER}/head && " + "git checkout " + ) + + +def packet_content_digest(packet: dict[str, Any]) -> str: + """Digest of a packet minus its review block (the validator's algorithm). + + A review pass binds to this digest, so regenerating a packet with + different content invalidates prior passes instead of carrying them onto + evidence they never reviewed. + """ + import hashlib as _hashlib + import json as _json + + content = {key: value for key, value in packet.items() if key != "review"} + return ( + "sha256:" + + _hashlib.sha256( + _json.dumps(content, sort_keys=True, separators=(",", ":")).encode("utf-8") + ).hexdigest() + ) + + +def preserve_review( + output_path: "Any", new_packet: dict[str, Any] | None = None +) -> dict[str, Any]: + """Carry forward review passes that still review THIS packet. Packets are generated, but review passes are recorded by people and other - agents after the fact. Without this, every regeneration silently erases - them, and the two-review floor a lane needs to close could only be met by - hand-editing a file the next run clobbers. + agents after the fact. A pass is kept only when its `content_digest` + matches the regenerated packet's content: reviews of an earlier packet + (different target commit, scene, metrics, or conclusion) are dropped so + the two-review floor cannot be satisfied by evidence nobody re-reviewed. + Passing `new_packet=None` (legacy call) keeps nothing, which fails closed. """ try: import json as _json @@ -128,9 +165,38 @@ def preserve_review(output_path: "Any") -> dict[str, Any]: except OSError, ValueError: return {"passes": []} review = existing.get("review") - if isinstance(review, dict) and isinstance(review.get("passes"), list): - return {"passes": review["passes"]} - return {"passes": []} + if not (isinstance(review, dict) and isinstance(review.get("passes"), list)): + return {"passes": []} + if new_packet is None: + return {"passes": []} + expected = packet_content_digest(new_packet) + kept = [ + entry + for entry in review["passes"] + if isinstance(entry, dict) and entry.get("content_digest") == expected + ] + return {"passes": kept} + + +def record_review_pass(packet_path: "Any", reviewer: str, summary: str) -> None: + """Append a review pass bound to the packet's current content digest.""" + import json as _json + from pathlib import Path as _Path + + path = _Path(packet_path) + packet = _json.loads(path.read_text(encoding="utf-8")) + review = packet.setdefault("review", {}) + passes = review.setdefault("passes", []) + passes.append( + { + "reviewer": reviewer, + "summary": summary, + "content_digest": packet_content_digest(packet), + } + ) + path.write_text( + _json.dumps(packet, indent=2, sort_keys=True) + "\n", encoding="utf-8" + ) def world_resolved_configuration(world: Any) -> dict[str, Any]: diff --git a/scripts/write_citation_ct001_rolling_direction_packet.py b/scripts/write_citation_ct001_rolling_direction_packet.py index c6810d3178b16..9b22eb183a4f8 100644 --- a/scripts/write_citation_ct001_rolling_direction_packet.py +++ b/scripts/write_citation_ct001_rolling_direction_packet.py @@ -51,6 +51,7 @@ UNSUPPORTED_SOLVER_RESIDUAL, preserve_review, solver_iterations_by_method, + target_fetch_hint, world_resolved_configuration, ) @@ -441,6 +442,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "target": { "branch": "main", "commit": git_head(), + "fetch_hint": target_fetch_hint(), "commit_role": ( "Source state measured: the library and fixture were run at " "this commit, which is HEAD at capture time. The packet and " @@ -625,9 +627,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "as one.", ], }, - "review": ( - preserve_review(output_path) if output_path is not None else {"passes": []} - ), + "review": {"passes": []}, "host": { "platform": platform.platform(), "python": sys.version.split()[0], @@ -639,6 +639,10 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: ), }, } + if output_path is not None: + # Rebind after assembly: only passes whose content_digest matches the + # regenerated packet survive (see preserve_review). + packet["review"] = preserve_review(output_path, packet) return packet diff --git a/scripts/write_citation_ct002_dense_contact_packet.py b/scripts/write_citation_ct002_dense_contact_packet.py index 07c15a8534566..8a1e3911dace0 100644 --- a/scripts/write_citation_ct002_dense_contact_packet.py +++ b/scripts/write_citation_ct002_dense_contact_packet.py @@ -47,6 +47,7 @@ UNSUPPORTED_SOLVER_RESIDUAL, preserve_review, solver_iterations_by_method, + target_fetch_hint, world_resolved_configuration, ) @@ -411,7 +412,11 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "for some timestep/solver settings." ), }, - "target": {"branch": "main", "commit": git_head()}, + "target": { + "branch": "main", + "commit": git_head(), + "fetch_hint": target_fetch_hint(), + }, "scene": { "id": parameters["scene_id"], "digest": scene_digest(parameters), @@ -614,9 +619,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "to a follow-up once per-island diagnostics exist.", ], }, - "review": ( - preserve_review(output_path) if output_path is not None else {"passes": []} - ), + "review": {"passes": []}, "host": { "platform": platform.platform(), "python": sys.version.split()[0], @@ -628,6 +631,10 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: ), }, } + if output_path is not None: + # Rebind after assembly: only passes whose content_digest matches the + # regenerated packet survive (see preserve_review). + packet["review"] = preserve_review(output_path, packet) return packet diff --git a/scripts/write_citation_ct003_elastic_contact_packet.py b/scripts/write_citation_ct003_elastic_contact_packet.py index 52e92041b1c25..1f1f824fc0d94 100644 --- a/scripts/write_citation_ct003_elastic_contact_packet.py +++ b/scripts/write_citation_ct003_elastic_contact_packet.py @@ -48,6 +48,7 @@ UNSUPPORTED_SOLVER_RESIDUAL, preserve_review, solver_iterations_by_method, + target_fetch_hint, world_resolved_configuration, ) @@ -325,7 +326,11 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "Elastic dense contact may inject energy or expose solver " "failure." ), }, - "target": {"branch": "main", "commit": git_head()}, + "target": { + "branch": "main", + "commit": git_head(), + "fetch_hint": target_fetch_hint(), + }, "scene": { "id": parameters["scene_id"], "digest": scene_digest(parameters), @@ -494,9 +499,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "Two timesteps and two solvers only.", ], }, - "review": ( - preserve_review(output_path) if output_path is not None else {"passes": []} - ), + "review": {"passes": []}, "host": { "platform": platform.platform(), "python": sys.version.split()[0], @@ -508,6 +511,10 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: ), }, } + if output_path is not None: + # Rebind after assembly: only passes whose content_digest matches the + # regenerated packet survive (see preserve_review). + packet["review"] = preserve_review(output_path, packet) return packet diff --git a/scripts/write_citation_ct004_articulated_energy_packet.py b/scripts/write_citation_ct004_articulated_energy_packet.py index 84b7c4e9d2d7d..f57007e6111b6 100644 --- a/scripts/write_citation_ct004_articulated_energy_packet.py +++ b/scripts/write_citation_ct004_articulated_energy_packet.py @@ -46,6 +46,7 @@ from citation_packet_utils import ( UNSUPPORTED_SOLVER_RESIDUAL, preserve_review, + target_fetch_hint, world_resolved_configuration, ) @@ -353,6 +354,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "target": { "branch": "main", "commit": git_head(), + "fetch_hint": target_fetch_hint(), "commit_role": ( "Source state measured: the library and fixture ran at this " "commit, which is HEAD at capture time. The packet and its " @@ -517,9 +519,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "trend summary, not a certified order of accuracy.", ], }, - "review": ( - preserve_review(output_path) if output_path is not None else {"passes": []} - ), + "review": {"passes": []}, "host": { "platform": platform.platform(), "python": sys.version.split()[0], @@ -531,6 +531,10 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: ), }, } + if output_path is not None: + # Rebind after assembly: only passes whose content_digest matches the + # regenerated packet survive (see preserve_review). + packet["review"] = preserve_review(output_path, packet) return packet diff --git a/scripts/write_citation_ct005_pd_tracking_packet.py b/scripts/write_citation_ct005_pd_tracking_packet.py index d71b4e873f266..b46abedc722e7 100644 --- a/scripts/write_citation_ct005_pd_tracking_packet.py +++ b/scripts/write_citation_ct005_pd_tracking_packet.py @@ -46,6 +46,7 @@ from citation_packet_utils import ( UNSUPPORTED_SOLVER_RESIDUAL, preserve_review, + target_fetch_hint, world_resolved_configuration, ) @@ -433,6 +434,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "target": { "branch": "main", "commit": git_head(), + "fetch_hint": target_fetch_hint(), "commit_role": ( "Source state measured: the library and fixture ran at this " "commit, which is HEAD at capture time. The packet and its " @@ -596,9 +598,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "unequal, which is why the flag is published per family.", ], }, - "review": ( - preserve_review(output_path) if output_path is not None else {"passes": []} - ), + "review": {"passes": []}, "host": { "platform": platform.platform(), "python": sys.version.split()[0], @@ -610,6 +610,10 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: ), }, } + if output_path is not None: + # Rebind after assembly: only passes whose content_digest matches the + # regenerated packet survive (see preserve_review). + packet["review"] = preserve_review(output_path, packet) return packet diff --git a/scripts/write_citation_ct007_high_mass_ratio_packet.py b/scripts/write_citation_ct007_high_mass_ratio_packet.py index 0f7d98ab71db6..b4010307fbacc 100644 --- a/scripts/write_citation_ct007_high_mass_ratio_packet.py +++ b/scripts/write_citation_ct007_high_mass_ratio_packet.py @@ -48,6 +48,7 @@ UNSUPPORTED_SOLVER_RESIDUAL, preserve_review, solver_iterations_by_method, + target_fetch_hint, world_resolved_configuration, ) @@ -386,6 +387,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "target": { "branch": "main", "commit": git_head(), + "fetch_hint": target_fetch_hint(), "commit_role": ( "Source state measured: the library and fixture ran at this " "commit, which is HEAD at capture time. The packet and its " @@ -554,9 +556,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "geometric rather than algebraic.", ], }, - "review": ( - preserve_review(output_path) if output_path is not None else {"passes": []} - ), + "review": {"passes": []}, "host": { "platform": platform.platform(), "python": sys.version.split()[0], @@ -568,6 +568,10 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: ), }, } + if output_path is not None: + # Rebind after assembly: only passes whose content_digest matches the + # regenerated packet survive (see preserve_review). + packet["review"] = preserve_review(output_path, packet) return packet diff --git a/scripts/write_citation_ct011_restore_equivalence_packet.py b/scripts/write_citation_ct011_restore_equivalence_packet.py index d27a6d568205e..af993beca7131 100644 --- a/scripts/write_citation_ct011_restore_equivalence_packet.py +++ b/scripts/write_citation_ct011_restore_equivalence_packet.py @@ -47,6 +47,7 @@ from citation_packet_utils import ( UNSUPPORTED_SOLVER_RESIDUAL, preserve_review, + target_fetch_hint, world_resolved_configuration, ) @@ -334,11 +335,35 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: ) finding_summary = {row["contact_solver_method"]: row["findings"] for row in rows} + # "Function of the restored state alone" must hold across EVERY protocol + # arm: two worlds given the same restored vector must continue + # identically regardless of what preceded the restore. Considering only + # the in-place arms would let a fresh-world or cross-history divergence + # (a direct counterexample) flip this aggregate to true. + state_function_arm_keys = ( + "inplace_restore_matches_continuation", + "inplace_restores_match_each_other", + "fresh_restores_match_each_other", + "fresh_restore_matches_continuation", + "same_history_restores_match", + "precontact_history_matches_fresh", + "ballistic_restore_exact", + ) restore_is_state_function = all( - row["findings"]["inplace_restore_matches_continuation"] - and row["findings"]["inplace_restores_match_each_other"] - for row in rows + all(row["findings"][key] for key in state_function_arm_keys) for row in rows ) + if restore_is_state_function: + # This writer's claim boundary and limitations describe the measured + # divergence and its identified root cause. If a future branch makes + # every arm bit-exact, regenerating that narrative would publish a + # false conclusion; stop and force a conscious rewrite instead. + raise SystemExit( + "CT-011: every restore arm matched bit-exactly, contradicting " + "this packet's divergence narrative. Restore behavior on this " + "branch has changed; rewrite the claim boundary, limitations, " + "and disposition from the new findings instead of regenerating " + "the old conclusion." + ) # CT-011 is a requirements claim about research workflows; running a # fixture cannot reproduce a need, so the row is not promoted. What the @@ -367,6 +392,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "target": { "branch": "main", "commit": git_head(), + "fetch_hint": target_fetch_hint(), "commit_role": ( "Source state measured: the library and fixture ran at this " "commit, which is HEAD at capture time. The packet and its " @@ -539,9 +565,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "have different semantics.", ], }, - "review": ( - preserve_review(output_path) if output_path is not None else {"passes": []} - ), + "review": {"passes": []}, "host": { "platform": platform.platform(), "python": sys.version.split()[0], @@ -553,6 +577,10 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: ), }, } + if output_path is not None: + # Rebind after assembly: only passes whose content_digest matches the + # regenerated packet survive (see preserve_review). + packet["review"] = preserve_review(output_path, packet) return packet diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index d281a1007e9b0..ec450a81ec70f 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -36,7 +36,11 @@ def complete_packet() -> dict: "claim_id": "CT-001", "title": "Test packet", "source": {"url": "https://example.org/claim", "claim": "A claim."}, - "target": {"branch": "main", "commit": "0" * 40}, + "target": { + "branch": "main", + "commit": "0" * 40, + "fetch_hint": "git fetch origin pull/3445/head", + }, "scene": { "id": "test_scene", "digest": "sha256:" + "a" * 64, @@ -542,26 +546,93 @@ def test_validate_tree_rejects_shared_packet_owner(tmp_path): assert any("one packet has one owner" in error for error in errors) +def _bound_pass(packet: dict, reviewer: str) -> dict: + """A review pass bound to the packet's current non-review content.""" + return { + "reviewer": reviewer, + "summary": "clean", + "content_digest": MODULE._packet_content_digest(packet), + } + + def test_validate_tree_closed_lane_needs_two_review_passes(tmp_path): packet = copy.deepcopy(complete_packet()) - packet["review"]["passes"] = [{"reviewer": "first", "summary": "clean"}] + packet["review"]["passes"] = [_bound_pass(packet, "first")] manifest = _minimal_manifest(["CT-001"]) lane = manifest["claims"][0]["lanes"]["dart7"] lane["status"] = "closed" - lane["disposition"] = "reproduced" + lane["disposition"] = "unresolved" lane["evidence"] = ["evidence/packet.json"] plan_dir = _write_tree( tmp_path, packet=packet, negative={"schema": "x"}, manifest=manifest ) errors = MODULE.validate_tree(plan_dir) assert any("at least two recorded review passes" in error for error in errors) - packet["review"]["passes"].append({"reviewer": "second", "summary": "clean"}) + packet["review"]["passes"].append(_bound_pass(packet, "second")) (plan_dir / "evidence" / "packet.json").write_text( json.dumps(packet), encoding="utf-8" ) assert MODULE.validate_tree(plan_dir) == [] +def test_validate_tree_closed_lane_needs_distinct_reviewers(tmp_path): + packet = copy.deepcopy(complete_packet()) + packet["review"]["passes"] = [ + _bound_pass(packet, "same"), + _bound_pass(packet, "same"), + ] + manifest = _minimal_manifest(["CT-001"]) + lane = manifest["claims"][0]["lanes"]["dart7"] + lane["status"] = "closed" + lane["disposition"] = "unresolved" + lane["evidence"] = ["evidence/packet.json"] + plan_dir = _write_tree( + tmp_path, packet=packet, negative={"schema": "x"}, manifest=manifest + ) + errors = MODULE.validate_tree(plan_dir) + assert any("INDEPENDENT" in error for error in errors) + + +def test_validate_tree_closed_lane_disposition_must_match_packet(tmp_path): + packet = copy.deepcopy(complete_packet()) # result.disposition: unresolved + packet["review"]["passes"] = [ + _bound_pass(packet, "first"), + _bound_pass(packet, "second"), + ] + manifest = _minimal_manifest(["CT-001"]) + lane = manifest["claims"][0]["lanes"]["dart7"] + lane["status"] = "closed" + lane["disposition"] = "reproduced" + lane["evidence"] = ["evidence/packet.json"] + plan_dir = _write_tree( + tmp_path, packet=packet, negative={"schema": "x"}, manifest=manifest + ) + errors = MODULE.validate_tree(plan_dir) + assert any( + "cannot publish a conclusion its evidence does not support" in error + for error in errors + ) + + +def test_review_pass_must_bind_to_packet_content(): + packet = complete_packet() + packet["review"]["passes"] = [ + { + "reviewer": "first", + "summary": "clean", + "content_digest": "sha256:" + "0" * 64, + } + ] + errors = MODULE.packet_errors(packet) + assert any("not bound to this packet's content" in error for error in errors) + packet["review"]["passes"] = [_bound_pass(packet, "first")] + assert MODULE.packet_errors(packet) == [] + # Changing any non-review content invalidates the binding. + packet["result"]["claim_boundary"] = "Changed after review." + errors = MODULE.packet_errors(packet) + assert any("not bound to this packet's content" in error for error in errors) + + def test_validate_tree_freshness_flags_stale_commit(tmp_path): plan_dir = _write_tree(tmp_path, packet=complete_packet(), negative={"schema": "x"}) errors = MODULE.validate_tree(plan_dir, freshness_head="1" * 40) @@ -574,6 +645,151 @@ def test_repository_tree_validates(): assert errors == [] +def test_fetch_hint_is_required(): + packet = complete_packet() + del packet["target"]["fetch_hint"] + errors = MODULE.packet_errors(packet) + assert any("fetch_hint" in error for error in errors) + + +def test_raw_rows_placeholders_fail(): + for bad in ([None], ["prose"], [{}], [{"k": 1}, None]): + packet = complete_packet() + packet["evidence"]["raw_rows"] = bad + errors = MODULE.packet_errors(packet) + assert any( + "raw_rows" in error and "structured record" in error for error in errors + ), bad + + +def test_raw_paths_must_stay_inside_evidence_roots(tmp_path): + for bad in ("/etc/passwd", "../escape.json", "C:\\evil.json", "a/../../b"): + packet = complete_packet() + del packet["evidence"]["raw_rows"] + packet["evidence"]["raw_paths"] = [bad] + errors = MODULE.packet_errors(packet) + assert any( + "relative path inside the repository" in error for error in errors + ), bad + # A relative path escaping via symlink-free resolution is caught with a + # base_dir even when the file exists outside the roots. + outside = tmp_path / "outside.json" + outside.write_text("{}", encoding="utf-8") + plan = tmp_path / "plan" + plan.mkdir() + packet = complete_packet() + del packet["evidence"]["raw_rows"] + packet["evidence"]["raw_paths"] = ["missing/nowhere.json"] + errors = MODULE.packet_errors(packet, base_dir=plan) + assert any("does not resolve to an existing file" in error for error in errors) + + +def test_ensemble_sweep_and_seed_entries_must_be_valid_and_distinct(): + base = complete_packet() + del base["ensemble"]["deterministic_repeats"] + for bad, needle in ( + ([None, None], "sweep[0]"), + ([{"a": 1}, {"a": 1}], "DISTINCT points"), + ([{}, {"a": 1}], "sweep[0]"), + ): + packet = copy.deepcopy(base) + packet["ensemble"]["sweep"] = bad + errors = MODULE.packet_errors(packet) + assert any(needle in error for error in errors), (bad, errors) + for bad, needle in ( + ([None, None], "seeds[0]"), + ([7, 7], "DISTINCT seeds"), + ([True, False], "seeds[0]"), + ): + packet = copy.deepcopy(base) + packet["ensemble"]["seeds"] = bad + errors = MODULE.packet_errors(packet) + assert any(needle in error for error in errors), (bad, errors) + good = copy.deepcopy(base) + good["ensemble"]["sweep"] = [{"angle_deg": 0.0}, {"angle_deg": 15.0}] + assert MODULE.packet_errors(good) == [] + + +def test_visual_entries_are_validated(): + for bad in ([None], [{"path": "x.png"}], [123]): + packet = complete_packet() + packet["evidence"]["visual"] = bad + errors = MODULE.packet_errors(packet) + assert any("evidence.visual[0]" in error for error in errors), bad + packet = complete_packet() + packet["evidence"]["visual"] = ["/abs/frame.png"] + errors = MODULE.packet_errors(packet) + assert any("relative path inside the repository" in error for error in errors) + + +def test_prose_cannot_masquerade_as_measurement(): + packet = complete_packet() + packet["metrics"]["numerical"]["max_penetration_m"] = "not measured yet" + errors = MODULE.packet_errors(packet) + assert any("prose where a measurement is expected" in error for error in errors) + # Semantic annotation keys stay allowed. + packet = complete_packet() + packet["metrics"]["numerical"]["penetration_semantics"] = "clamped at zero" + packet["metrics"]["numerical"]["clamp_note"] = "runtime clamps depth" + assert MODULE.packet_errors(packet) == [] + + +def test_configuration_placeholder_objects_fail(): + packet = complete_packet() + packet["configuration"]["resolved"] = {"placeholder": None} + errors = MODULE.packet_errors(packet) + assert any("null values" in error for error in errors) + assert any("no recognizable" in error for error in errors) + + +def test_nonstandard_json_constants_fail_at_load(tmp_path): + packet = complete_packet() + packet["host"] = {"skew": float("nan")} + plan_dir = _write_tree(tmp_path, packet=None, negative={"schema": "x"}) + (plan_dir / "evidence" / "packet.json").write_text( + json.dumps(packet, allow_nan=True), encoding="utf-8" + ) + manifest = _minimal_manifest(["CT-001"]) + manifest["claims"][0]["lanes"]["dart7"]["status"] = "in-progress" + manifest["claims"][0]["lanes"]["dart7"]["evidence"] = ["evidence/packet.json"] + (plan_dir / "claims-manifest.json").write_text( + json.dumps(manifest), encoding="utf-8" + ) + errors = MODULE.validate_tree(plan_dir) + assert any("unreadable JSON" in error for error in errors) + + +def test_lane_evidence_paths_are_canonicalized(tmp_path): + packet = copy.deepcopy(complete_packet()) + manifest = _minimal_manifest(["CT-001", "CT-002"]) + first = manifest["claims"][0]["lanes"]["dart7"] + second = manifest["claims"][1]["lanes"]["dart7"] + first["status"] = "in-progress" + first["evidence"] = ["evidence/packet.json"] + second["status"] = "in-progress" + second["evidence"] = ["evidence/./packet.json"] + plan_dir = _write_tree( + tmp_path, packet=packet, negative={"schema": "x"}, manifest=manifest + ) + (plan_dir / "citation-claim-corpus.md").write_text( + "| CT-001 | claim |\n| CT-002 | claim |\n", encoding="utf-8" + ) + errors = MODULE.validate_tree(plan_dir) + assert any("one packet has one owner" in error for error in errors) + + +def test_lane_evidence_paths_cannot_escape(tmp_path): + manifest = _minimal_manifest(["CT-001"]) + lane = manifest["claims"][0]["lanes"]["dart7"] + lane["status"] = "in-progress" + lane["evidence"] = ["../outside.json"] + plan_dir = _write_tree( + tmp_path, packet=None, negative={"schema": "x"}, manifest=manifest + ) + errors = MODULE.validate_tree(plan_dir) + assert any("escapes the plan directory" in error for error in errors) + + def test_non_string_lane_evidence_entry_fails(tmp_path): """A lane must not be closable by an entry the packet checks cannot read.""" for bad in ({"path": "evidence/x.json"}, None, 42, True, ["evidence/x.json"]): From f4e9e5c11d5d3b443693cde9142bf317e2115cd3 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 22:18:05 -0700 Subject: [PATCH 20/52] Close the Codex round-2 validator bypasses Round-2 findings were adversarial re-tests of the round-1 fixes; all addressed on both branches: - scene.parameters is now REQUIRED: a well-formed sha256 with no published content binds nothing, so every digest is recomputed against the packet's own parameters. - The metric-leaf type chain is exhaustive: booleans are accepted explicitly as measured findings (stability flags, signature verdicts) and any other leaf type is rejected, so nothing falls through unvalidated. - Reviewer identities are normalized (strip + casefold) before the distinct-reviewer closure count; whitespace/case variants of one reviewer no longer count as two. - Visual evidence entries must carry a recognized media suffix; an ordinary text file that merely exists can no longer satisfy the visual requirement. - The CT-011 writer now asserts the EXACT measured arm pattern its claim boundary describes (in-place arms diverge; fresh-world, same-history, pre-contact, ballistic match) and aborts naming the deviating arm on any change, not only when every arm matches. - RESUME no longer lists the landed ResolvedSolverConfiguration binding among unsupported quantities (do-not-reimplement note added). Tests: 85 -> 89 validator cases. Packet regeneration at this commit follows in the next commit so every packet's recorded command runs a writer that exists (and is current) at its recorded target commit -- the round-2 CT-007 finding generalized. --- .../RESUME.md | 6 ++- scripts/check_citation_evidence.py | 46 ++++++++++++++++- ...tation_ct011_restore_equivalence_packet.py | 45 ++++++++++++----- tests/test_check_citation_evidence.py | 49 ++++++++++++++++++- 4 files changed, 129 insertions(+), 17 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md index e27f080c65cc8..795c3d7abd837 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/RESUME.md @@ -72,8 +72,10 @@ per-solve residual (WS4 remainder). - Quantities that do not exist on `main` today, and must stay typed unsupported until WS4 lands them: per-solve solver residual (rigid contact never passes one to `recordSolverDiagnostics`), boxed-LCP iteration counts - (its branch returns before the diagnostics call), per-island fallback - reporting, and `ResolvedSolverConfiguration` in Python. + (its branch returns before the diagnostics call), and per-island fallback + reporting. `ResolvedSolverConfiguration` IS bound to Python in this branch + (`World.resolved_configuration`; module_compute.cpp / module_world.cpp) + and every packet records it — do not re-implement it. - Packet dispositions must survive an adversarial read. Review killed two verdicts: one rested on a hardcoded threshold (a settle speed exactly linear in dt is integrator residual, not instability), and one measured diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index b97c055354776..dfaede25ad737 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -106,6 +106,20 @@ "_basis", ) +# Visual evidence must be actual media; an ordinary text file satisfying a +# path-existence check is not a capture. +VISUAL_MEDIA_SUFFIXES = ( + ".png", + ".jpg", + ".jpeg", + ".gif", + ".webp", + ".svg", + ".apng", + ".mp4", + ".webm", +) + # configuration.requested/resolved must name a recognizable identity, not an # arbitrary placeholder object. IDENTITY_KEY_RE = re.compile( @@ -337,6 +351,19 @@ def metric_group_errors(name: str, group: object) -> list[str]: errors.append(f"metrics.{name}.{path} contains a non-finite number") elif leaf == 0: observed_zero_fields.append(path) + elif isinstance(leaf, bool): + # Boolean findings (stability flags, signature verdicts) are + # legitimate measured outcomes; accepting them EXPLICITLY here + # keeps the type chain exhaustive so nothing falls through + # unvalidated. + pass + else: + errors.append( + f"metrics.{name}.{path} has unrecognized leaf type " + f"{type(leaf).__name__}; a measurement is a finite " + "number, a boolean finding, a whitelisted semantic " + "string, or a typed-unsupported marker" + ) if value_keys and leaf_count == 0: errors.append( @@ -427,7 +454,14 @@ def packet_errors( digest = scene.get("digest") if not (isinstance(digest, str) and SCENE_DIGEST_RE.match(digest)): errors.append("scene.digest must match sha256:<64 hex>") - elif isinstance(scene.get("parameters"), dict): + parameters = scene.get("parameters") + if not (isinstance(parameters, dict) and parameters): + errors.append( + "scene.parameters must publish the non-empty parameter " + "object the digest was computed over; a well-formed digest " + "with no content binds nothing" + ) + elif isinstance(digest, str) and SCENE_DIGEST_RE.match(digest): # When the packet publishes the parameters the digest was taken # over, recompute it. A digest that cannot be reproduced from the # packet's own scene description binds nothing, and a hand-edited @@ -617,6 +651,14 @@ def packet_errors( "cannot stand in for visual evidence" ) continue + if PurePosixPath(item_path).suffix.lower() not in VISUAL_MEDIA_SUFFIXES: + errors.append( + f"evidence.visual[{index}] {item_path!r} is not a " + "recognized visual media artifact " + f"({', '.join(VISUAL_MEDIA_SUFFIXES)}); an ordinary " + "file cannot stand in for visual evidence" + ) + continue issue = _evidence_path_issue(item_path, base_dir) if issue is not None: errors.append(f"evidence.visual[{index}] {issue}") @@ -999,7 +1041,7 @@ def validate_tree( ) else: reviewers = { - entry.get("reviewer") + entry["reviewer"].strip().casefold() for entry in passes if isinstance(entry, dict) and _is_nonempty_str(entry.get("reviewer")) diff --git a/scripts/write_citation_ct011_restore_equivalence_packet.py b/scripts/write_citation_ct011_restore_equivalence_packet.py index af993beca7131..55304677cd0db 100644 --- a/scripts/write_citation_ct011_restore_equivalence_packet.py +++ b/scripts/write_citation_ct011_restore_equivalence_packet.py @@ -352,18 +352,39 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: restore_is_state_function = all( all(row["findings"][key] for key in state_function_arm_keys) for row in rows ) - if restore_is_state_function: - # This writer's claim boundary and limitations describe the measured - # divergence and its identified root cause. If a future branch makes - # every arm bit-exact, regenerating that narrative would publish a - # false conclusion; stop and force a conscious rewrite instead. - raise SystemExit( - "CT-011: every restore arm matched bit-exactly, contradicting " - "this packet's divergence narrative. Restore behavior on this " - "branch has changed; rewrite the claim boundary, limitations, " - "and disposition from the new findings instead of regenerating " - "the old conclusion." - ) + # This writer's claim boundary and limitations assert one SPECIFIC + # measured pattern: the two in-place arms diverge while the fresh-world, + # same-history, pre-contact, and ballistic arms match (the omitted + # rotational state explains exactly that split). ANY deviation -- not + # just all-arms-exact -- would make the regenerated narrative false, so + # the writer aborts on the first arm whose outcome changes and forces a + # conscious rewrite. + expected_arm_pattern = { + "inplace_restore_matches_continuation": False, + "inplace_restores_match_each_other": False, + "fresh_restores_match_each_other": True, + "fresh_restore_matches_continuation": False, + "same_history_restores_match": True, + "precontact_history_matches_fresh": True, + "ballistic_restore_exact": True, + } + for row in rows: + deviations = { + key: row["findings"][key] + for key, expected in expected_arm_pattern.items() + if row["findings"][key] != expected + } + if deviations: + raise SystemExit( + f"CT-011 {row['contact_solver_method']}: measured arm " + f"outcomes {deviations} deviate from the divergence pattern " + "this packet's claim boundary asserts (in-place restores " + "diverge; fresh-world, same-history, pre-contact, and " + "ballistic arms match). Restore behavior has changed; " + "rewrite the claim boundary, limitations, and disposition " + "from the new findings instead of regenerating the old " + "conclusion." + ) # CT-011 is a requirements claim about research workflows; running a # fixture cannot reproduce a need, so the row is not promoted. What the diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index ec450a81ec70f..0486c3400c09c 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -5,6 +5,7 @@ """ import copy +import hashlib import importlib.util import json import sys @@ -43,7 +44,15 @@ def complete_packet() -> dict: }, "scene": { "id": "test_scene", - "digest": "sha256:" + "a" * 64, + "digest": ( + "sha256:" + + hashlib.sha256( + json.dumps( + {"gravity": -9.81}, sort_keys=True, separators=(",", ":") + ).encode("utf-8") + ).hexdigest() + ), + "parameters": {"gravity": -9.81}, "description": "A scene.", }, "configuration": { @@ -843,3 +852,41 @@ def test_nested_negative_control_is_enumerated(tmp_path): ) errors = MODULE.validate_tree(tree) assert any("fail-closed proof is vacuous" in error for error in errors) + + +def test_scene_parameters_are_required(): + packet = complete_packet() + del packet["scene"]["parameters"] + errors = MODULE.packet_errors(packet) + assert any("binds nothing" in error for error in errors) + + +def test_boolean_metric_leaves_are_valid_but_odd_types_fail(): + packet = complete_packet() + packet["metrics"]["physical"]["pyramid_signature"] = True + assert MODULE.packet_errors(packet) == [] + + +def test_reviewer_identities_are_normalized_before_counting(tmp_path): + packet = copy.deepcopy(complete_packet()) + packet["review"]["passes"] = [ + _bound_pass(packet, "reviewer-a"), + _bound_pass(packet, " Reviewer-A "), + ] + manifest = _minimal_manifest(["CT-001"]) + lane = manifest["claims"][0]["lanes"][LANE] + lane["status"] = "closed" + lane["disposition"] = "unresolved" + lane["evidence"] = ["evidence/packet.json"] + tree_dir = _write_tree( + tmp_path, packet=packet, negative={"schema": "x"}, manifest=manifest + ) + errors = MODULE.validate_tree(tree_dir) + assert any("INDEPENDENT" in error for error in errors) + + +def test_visual_entries_must_be_media_artifacts(): + packet = complete_packet() + packet["evidence"]["visual"] = ["CHANGELOG.md"] + errors = MODULE.packet_errors(packet) + assert any("not a recognized visual media artifact" in error for error in errors) From f198a290c7b06a153cf35a2d179ab196ccb67a81 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 22:30:28 -0700 Subject: [PATCH 21/52] Regenerate every packet at the round-2 commit All seven packets now record target.commit f4e9e5c11d5, whose tree contains every current writer, so each packet's recorded command plus target.fetch_hint reproduces it from a clean checkout -- the round-2 CT-007 finding (writer absent at its recorded target) generalized to the whole evidence set. Dispositions unchanged: CT-001/002 reproduced; CT-003/004/005/007/011 unresolved. Round-2 verification entries recorded. --- .../verification.md | 18 ++++++++++++++++++ .../CT-001-dart7-rolling-direction.json | 2 +- .../CT-002-dart7-dense-inelastic-contact.json | 2 +- .../CT-003-dart7-dense-elastic-contact.json | 2 +- ...-004-dart7-articulated-energy-momentum.json | 2 +- .../evidence/CT-005-dart7-pd-tracking.json | 2 +- .../evidence/CT-007-dart7-high-mass-ratio.json | 2 +- .../CT-011-dart7-restore-equivalence.json | 2 +- 8 files changed, 25 insertions(+), 7 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 51a13bd0dedcb..83acc83f807c1 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -492,3 +492,21 @@ this round, fixes silent per repo convention (no thread replies): `fetch_hint`; `pixi run check-citation-evidence` passes on the real tree; the permanent negative control keeps failing with a higher error count under the stricter rules. + +## Codex review round 2 (PR #3445) — 2026-08-16 + +Round 2 re-tested the round-1 fixes adversarially; 3 new findings, all +fixed (mirrored where shared): scene.parameters is now required so a +digest cannot bind to absent content; the metric-leaf type chain is +exhaustive (booleans accepted explicitly as measured findings, anything +else rejected); reviewer identities normalized before the +distinct-reviewer count; visual entries must be recognized media; the +CT-011 writer asserts its EXACT measured arm pattern and aborts naming +any deviating arm; RESUME's unsupported-quantities list no longer names +the landed ResolvedSolverConfiguration binding. The CT-007 finding +(writer absent at its recorded target commit) was generalized: ALL seven +packets regenerated at the round-2 commit `f4e9e5c11d5`, whose tree +contains every current writer, so each recorded command plus fetch_hint +reproduces its packet from a clean checkout. All dispositions unchanged +(CT-001/002 reproduced; CT-003/004/005/007/011 unresolved). Tests: 85 -> +89; tree validates; negative control still fails. diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json index c5023e59b406f..dcead5644bf35 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json @@ -1262,7 +1262,7 @@ }, "target": { "branch": "main", - "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05", + "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", "commit_role": "Source state measured: the library and fixture were run at this commit, which is HEAD at capture time. The packet and its writer land in a later commit, so re-running the recorded command requires the child commit that adds the writer; the measured behavior belongs to this one.", "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json index 532f0149d3f0a..f01e87696eb96 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json @@ -671,7 +671,7 @@ }, "target": { "branch": "main", - "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05", + "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, "title": "Dense 6x6x6 inelastic contact stability (DART 7 first packet)" diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json index daa27f23e255c..e76fa8fd7b7de 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json @@ -574,7 +574,7 @@ }, "target": { "branch": "main", - "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05", + "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, "title": "Dense elastic contact energy envelope (DART 7 first packet)" diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json index 26866147bec53..f3feb12714ec7 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json @@ -977,7 +977,7 @@ }, "target": { "branch": "main", - "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05", + "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json index 49fc2e079fc00..42384b9ab9a69 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json @@ -976,7 +976,7 @@ }, "target": { "branch": "main", - "commit": "ca78b0980e88617b98bdbcb84cfeb01931238d05", + "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json index c645d557be182..97ad5d2950bca 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json @@ -1007,7 +1007,7 @@ }, "target": { "branch": "main", - "commit": "ebbbb853a49cf59bafb590cf111c110acbd0531b", + "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json index 8f52b07857246..18e7099565f64 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json @@ -372,7 +372,7 @@ }, "target": { "branch": "main", - "commit": "cebf5a86799a10615f0f7a6878a1ccdf5f315db0", + "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, From d2a4c6b1fdf958d3e91116e507646b1056165ba9 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 22:47:00 -0700 Subject: [PATCH 22/52] Close the Codex round-3 validator bypasses Round 3 re-tested rounds 1-2 adversarially; five findings, all fixed on both branches: configuration identity keys must carry non-empty string values; review passes must record an explicit verdict 'pass' (a digest-bound entry describing a FAILED review cannot count toward closure; record_review_pass stamps the verdict); measurement_window is structurally validated (non-empty string, or non-empty object of finite numeric values with start_s <= end_s); visual artifacts are verified by media signature, so a prose file renamed to capture.png is rejected by magic bytes rather than trusted by suffix; the signature reader closes its handle. Tests: 89 -> 93 validator cases. --- .../verification.md | 9 ++ scripts/check_citation_evidence.py | 102 ++++++++++++++++-- scripts/citation_packet_utils.py | 1 + tests/test_check_citation_evidence.py | 52 +++++++++ 4 files changed, 156 insertions(+), 8 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 83acc83f807c1..54286b518237a 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -510,3 +510,12 @@ contains every current writer, so each recorded command plus fetch_hint reproduces its packet from a clean checkout. All dispositions unchanged (CT-001/002 reproduced; CT-003/004/005/007/011 unresolved). Tests: 85 -> 89; tree validates; negative control still fails. + +## Codex review round 3 (PR #3445) — 2026-08-16 + +Two findings, both fixed on both branches: identity keys must carry +non-empty string values (`{"solver_method": ""}` no longer counts as a +recorded identity), and review passes must record an explicit +`verdict: "pass"` — a digest-bound entry summarizing a FAILED review can +no longer count toward the two-review closure floor +(`record_review_pass` stamps the verdict). Tests: 89 -> 93. diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index dfaede25ad737..8f14196dedc41 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -132,7 +132,7 @@ def _has_identity_key(value: object) -> bool: with a non-null value; {"placeholder": null} has none.""" if isinstance(value, dict): for key, item in value.items(): - if IDENTITY_KEY_RE.search(str(key)) and item is not None: + if IDENTITY_KEY_RE.search(str(key)) and _is_nonempty_str(item): return True if _has_identity_key(item): return True @@ -193,16 +193,64 @@ def _evidence_path_issue(raw_path: str, base_dir: Path | None) -> str | None: f"{raw_path!r} must be a relative path inside the repository " "evidence tree (no absolute paths, drive letters, or '..')" ) + if base_dir is None: + return None + if _resolve_evidence_path(raw_path, base_dir) is None: + return ( + f"{raw_path!r} does not resolve to an existing file inside an " + "approved evidence root; a dangling or escaping path is prose" + ) + return None + + +def _resolve_evidence_path(raw_path: str, base_dir: "Path | None") -> "Path | None": + """The resolved in-root file a relative evidence path names, or None.""" if base_dir is None: return None for root in (base_dir, REPO_ROOT): candidate = (root / raw_path).resolve() if candidate.is_file() and candidate.is_relative_to(root.resolve()): - return None - return ( - f"{raw_path!r} does not resolve to an existing file inside an " - "approved evidence root; a dangling or escaping path is prose" + return candidate + return None + + +_VISUAL_SIGNATURES = { + ".png": (b"\x89PNG\r\n\x1a\n",), + ".apng": (b"\x89PNG\r\n\x1a\n",), + ".jpg": (b"\xff\xd8\xff",), + ".jpeg": (b"\xff\xd8\xff",), + ".gif": (b"GIF87a", b"GIF89a"), + ".webm": (b"\x1a\x45\xdf\xa3",), +} + + +def _visual_content_issue(path: "Path") -> "str | None": + """Why a resolved visual artifact's bytes do not match its claimed type. + + A prose file renamed to `capture.png` satisfies a suffix check; the + magic-signature check makes the artifact itself carry the evidence. + """ + suffix = path.suffix.lower() + try: + with path.open("rb") as stream: + header = stream.read(256) + except OSError: + return "could not be read for media-signature verification" + mismatch = ( + "does not carry the media signature its suffix claims; a renamed " + "non-media file is not visual evidence" ) + if suffix in _VISUAL_SIGNATURES: + if not any(header.startswith(sig) for sig in _VISUAL_SIGNATURES[suffix]): + return mismatch + return None + if suffix == ".webp": + return None if header[:4] == b"RIFF" and header[8:12] == b"WEBP" else mismatch + if suffix == ".mp4": + return None if header[4:8] == b"ftyp" else mismatch + if suffix == ".svg": + return None if b" str: @@ -572,10 +620,32 @@ def packet_errors( window = ensemble.get("measurement_window") if "measurement_window" not in ensemble: errors.append("ensemble.measurement_window is required") - elif not window or (isinstance(window, str) and not window.strip()): + elif isinstance(window, str): + if not window.strip(): + errors.append( + "ensemble.measurement_window must record an actual " + "window, not an empty value" + ) + elif isinstance(window, dict) and window: + bad_values = sorted( + key for key, item in window.items() if not _is_finite_number(item) + ) + if bad_values: + errors.append( + "ensemble.measurement_window values at " + f"{bad_values} must be finite numbers" + ) + start = window.get("start_s") + end = window.get("end_s") + if _is_finite_number(start) and _is_finite_number(end) and start > end: + errors.append( + "ensemble.measurement_window start_s must not exceed end_s" + ) + else: errors.append( - "ensemble.measurement_window must record an actual window, " - "not an empty value" + "ensemble.measurement_window must be a non-empty string or a " + "non-empty object of finite numeric bounds; a truthy " + "placeholder does not record when measurements were collected" ) metrics = packet.get("metrics") @@ -662,6 +732,15 @@ def packet_errors( issue = _evidence_path_issue(item_path, base_dir) if issue is not None: errors.append(f"evidence.visual[{index}] {issue}") + continue + resolved = _resolve_evidence_path(item_path, base_dir) + if resolved is not None: + content_issue = _visual_content_issue(resolved) + if content_issue is not None: + errors.append( + f"evidence.visual[{index}] {item_path!r} " + f"{content_issue}" + ) else: errors.append( "evidence.visual must list visual artifacts or be typed " @@ -707,6 +786,13 @@ def packet_errors( "and summary" ) continue + if entry.get("verdict") != "pass": + errors.append( + f"review.passes[{index}] must record verdict 'pass'; " + "an entry without an explicit passing verdict (or " + "one recording a failure) cannot count toward " + "closure" + ) if entry.get("content_digest") != expected_digest: errors.append( f"review.passes[{index}] is not bound to this " diff --git a/scripts/citation_packet_utils.py b/scripts/citation_packet_utils.py index 12a2070d112f5..6e3269984cffb 100644 --- a/scripts/citation_packet_utils.py +++ b/scripts/citation_packet_utils.py @@ -191,6 +191,7 @@ def record_review_pass(packet_path: "Any", reviewer: str, summary: str) -> None: { "reviewer": reviewer, "summary": summary, + "verdict": "pass", "content_digest": packet_content_digest(packet), } ) diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index 0486c3400c09c..4ecfe6c632aef 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -560,6 +560,7 @@ def _bound_pass(packet: dict, reviewer: str) -> dict: return { "reviewer": reviewer, "summary": "clean", + "verdict": "pass", "content_digest": MODULE._packet_content_digest(packet), } @@ -890,3 +891,54 @@ def test_visual_entries_must_be_media_artifacts(): packet["evidence"]["visual"] = ["CHANGELOG.md"] errors = MODULE.packet_errors(packet) assert any("not a recognized visual media artifact" in error for error in errors) + + +def test_review_entries_require_a_passing_verdict(): + packet = complete_packet() + entry = _bound_pass(packet, "first") + del entry["verdict"] + packet["review"]["passes"] = [entry] + errors = MODULE.packet_errors(packet) + assert any("verdict 'pass'" in error for error in errors) + entry["verdict"] = "fail" + errors = MODULE.packet_errors(packet) + assert any("verdict 'pass'" in error for error in errors) + + +def test_identity_values_must_be_non_empty(): + for bad in ("", [], False, {}): + packet = complete_packet() + packet["configuration"]["resolved"] = {"solver": bad} + errors = MODULE.packet_errors(packet) + assert any("no recognizable" in error for error in errors), bad + + +def test_measurement_window_placeholders_fail(): + for bad in (True, 1, [None], {"a": None}, {"start_s": 2.0, "end_s": 1.0}): + packet = complete_packet() + packet["ensemble"]["measurement_window"] = bad + errors = MODULE.packet_errors(packet) + assert any("measurement_window" in error for error in errors), bad + packet = complete_packet() + packet["ensemble"]["measurement_window"] = {"warmup_steps": 250} + assert MODULE.packet_errors(packet) == [] + packet["ensemble"]["measurement_window"] = "full 1 s horizon" + assert MODULE.packet_errors(packet) == [] + + +def test_visual_artifacts_are_verified_by_content(tmp_path): + fake = tmp_path / "design" + fake.mkdir() + (fake / "capture.png").write_text("not an image", encoding="utf-8") + packet = complete_packet() + packet["evidence"]["visual"] = ["capture.png"] + errors = MODULE.packet_errors(packet, base_dir=fake) + assert any("media signature" in error for error in errors) + png = b"\x89PNG\r\n\x1a\n" + bytes.fromhex( + "0000000d49484452000000010000000108060000001f15c489" + "0000000a49444154789c63000100000500010d0a2db4" + "0000000049454e44ae426082" + ) + (fake / "real.png").write_bytes(png) + packet["evidence"]["visual"] = ["real.png"] + assert MODULE.packet_errors(packet, base_dir=fake) == [] From 5abeeb81a701d353027e3027ae227287f8cac71f Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 23:07:01 -0700 Subject: [PATCH 23/52] Close the Codex round-4 validator bypasses Metadata-suffixed keys (method_note, backend_reason) no longer count as configuration identities; raw rows and measured metric groups must carry at least one numeric/boolean measurement or typed-unsupported marker (metadata-only records rejected); not-applicable lanes cannot publish a disposition or hold evidence packets; the Windows check-lint task now runs check-citation-evidence; the CT-001 claim boundary is formatted from the run's own summary (peak drift, tolerance, null angles, antisymmetry ratio, criteria, validity-gate state) instead of hardcoding one historical run's numbers. Tests: 93 -> 97 validator cases. CT-001 is regenerated at this commit in the follow-up commit so its recorded command reproduces it. --- .../verification.md | 19 ++++++ pixi.toml | 1 + scripts/check_citation_evidence.py | 65 +++++++++++++++++-- ...citation_ct001_rolling_direction_packet.py | 62 ++++++++++++++++-- tests/test_check_citation_evidence.py | 35 ++++++++++ 5 files changed, 168 insertions(+), 14 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 54286b518237a..f29e93517c89e 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -519,3 +519,22 @@ recorded identity), and review passes must record an explicit `verdict: "pass"` — a digest-bound entry summarizing a FAILED review can no longer count toward the two-review closure floor (`record_review_pass` stamps the verdict). Tests: 89 -> 93. + +## Codex review round 4 (PR #3445) — 2026-08-16 + +Four findings, all fixed (shared rules mirrored on release-6.20): +metadata-suffixed keys (method_note, backend_reason) no longer count as +configuration identities (one own-goal caught in the round: the first +exclusion list reused the metric-string suffixes, whose `_method` entry +would have knocked out `contact_solver_method` itself — the validator +suite caught it before commit); raw rows must carry at least one +numeric/boolean measurement (metadata-only rows rejected); a measured +metric group needs at least one numeric/boolean/typed-unsupported leaf +(method+note alone no longer passes); the CT-001 claim boundary is now +formatted from the run's own summary (peak drift, isotropy tolerance, +null angles, antisymmetry ratio, criteria, validity-gate state) so a +regeneration cannot pair fresh raw rows with stale quantitative prose. +Also from the 3444 side: not-applicable lanes cannot publish +dispositions or hold evidence, and the Windows check-lint variant now +includes the citation gate. Tests: 93 -> 97. CT-001 regenerated at the +round-4 commit in the follow-up commit. diff --git a/pixi.toml b/pixi.toml index a969eadf78c60..e4b31a74307fb 100644 --- a/pixi.toml +++ b/pixi.toml @@ -2793,6 +2793,7 @@ check-lint = { depends-on = [ "check-lcp-solver-roster", "check-ai-commands", "check-ai-infra", + "check-citation-evidence", ] } build = { cmd = """ diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 8f14196dedc41..d060cf87bdaf9 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -121,18 +121,38 @@ ) # configuration.requested/resolved must name a recognizable identity, not an -# arbitrary placeholder object. +# arbitrary placeholder object. Keys that merely mention an identity word in +# a metadata role (method_note, backend_reason, ...) do not count. IDENTITY_KEY_RE = re.compile( r"solver|method|detector|integrator|integration|backend|family", re.I ) +IDENTITY_METADATA_SUFFIXES = ( + "_note", + "_notes", + "_semantics", + "_reason", + "_reasons", + "_criteria", + "_test", + "_basis", + "_provenance", + "_policy", +) + + +def _is_identity_key(key: str) -> bool: + return bool(IDENTITY_KEY_RE.search(key)) and not key.endswith( + IDENTITY_METADATA_SUFFIXES + ) def _has_identity_key(value: object) -> bool: """True when any key at any depth names a solver/method/... identity - with a non-null value; {"placeholder": null} has none.""" + with a non-empty string value; {"placeholder": null} and + {"method_note": "..."} have none.""" if isinstance(value, dict): for key, item in value.items(): - if IDENTITY_KEY_RE.search(str(key)) and _is_nonempty_str(item): + if _is_identity_key(str(key)) and _is_nonempty_str(item): return True if _has_identity_key(item): return True @@ -354,11 +374,13 @@ def metric_group_errors(name: str, group: object) -> list[str]: observed_zero_fields: list[str] = [] leaf_count = 0 + measurement_leaves = 0 for key in value_keys: leaves = _metric_leaves(group[key], key) leaf_count += len(leaves) for path, leaf in leaves: if _is_unsupported_leaf(leaf): + measurement_leaves += 1 if not _is_nonempty_str(leaf.get("reason")): errors.append( f"metrics.{name}.{path} is typed unsupported but has " @@ -395,16 +417,17 @@ def metric_group_errors(name: str, group: object) -> list[str]: "annotation or drop the key" ) elif isinstance(leaf, (int, float)) and not isinstance(leaf, bool): + measurement_leaves += 1 if not math.isfinite(leaf): errors.append(f"metrics.{name}.{path} contains a non-finite number") elif leaf == 0: observed_zero_fields.append(path) elif isinstance(leaf, bool): # Boolean findings (stability flags, signature verdicts) are - # legitimate measured outcomes; accepting them EXPLICITLY here - # keeps the type chain exhaustive so nothing falls through - # unvalidated. - pass + # legitimate measured outcomes; accepting them EXPLICITLY + # here keeps the type chain exhaustive so nothing falls + # through unvalidated. + measurement_leaves += 1 else: errors.append( f"metrics.{name}.{path} has unrecognized leaf type " @@ -413,6 +436,12 @@ def metric_group_errors(name: str, group: object) -> list[str]: "string, or a typed-unsupported marker" ) + if value_keys and leaf_count > 0 and measurement_leaves == 0: + errors.append( + f"metrics.{name} carries only semantic annotations; a measured " + "group needs at least one numeric/boolean measurement or " + "typed-unsupported marker" + ) if value_keys and leaf_count == 0: errors.append( f"metrics.{name} has a method but only empty containers; that is " @@ -685,6 +714,16 @@ def packet_errors( "structured record; null or scalar placeholders are " "not raw evidence" ) + continue + if not any( + _is_finite_number(leaf) or isinstance(leaf, bool) + for _, leaf in _metric_leaves(row) + ): + errors.append( + f"evidence.raw_rows[{index}] carries no numeric or " + "boolean measurement; a metadata-only record is not " + "raw evidence" + ) if has_paths: for index, raw_path in enumerate(raw_paths): if not _is_nonempty_str(raw_path): @@ -873,6 +912,18 @@ def manifest_errors(manifest: object, corpus_ids: list[str]) -> list[str]: errors.append( f"{prefix} is not-applicable and must record a reason" ) + na_disposition = lane.get("disposition") + if na_disposition not in (None, "not-applicable"): + errors.append( + f"{prefix} is not-applicable and cannot publish " + f"disposition {na_disposition!r}; a lane that does " + "not apply concludes nothing" + ) + if evidence: + errors.append( + f"{prefix} is not-applicable and must not hold " + "evidence packets" + ) continue if not _is_nonempty_str(lane.get("owner")): errors.append(f"{prefix}.owner must be non-empty") diff --git a/scripts/write_citation_ct001_rolling_direction_packet.py b/scripts/write_citation_ct001_rolling_direction_packet.py index 9b22eb183a4f8..fc6d96d16d663 100644 --- a/scripts/write_citation_ct001_rolling_direction_packet.py +++ b/scripts/write_citation_ct001_rolling_direction_packet.py @@ -414,6 +414,58 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "reproduced" if anisotropic_methods and not validity_failures else "unresolved" ) + # The claim boundary quotes quantitative outcomes; format them from THIS + # run's summary so a regeneration with different rolling behavior cannot + # combine fresh raw rows with a stale conclusion. + peak_drift = max(stats["max_abs_lateral_drift_m"] for stats in summary.values()) + ratios = [ + stats["antisymmetry_residual_over_peak_drift"] + for stats in summary.values() + if isinstance(stats["antisymmetry_residual_over_peak_drift"], (int, float)) + ] + max_ratio = max(ratios) if ratios else None + swept_angles = sorted({row["angle_deg"] for row in rows}) + null_angles = [ + angle + for angle in swept_angles + if all( + row["lateral_drift_m"] == 0.0 for row in rows if row["angle_deg"] == angle + ) + ] + criteria_union = sorted( + { + criterion + for finding in anisotropy_findings.values() + for criterion in finding["criteria_exceeded"] + } + ) + solver_names = " and ".join(sorted(summary)) + outcome_sentence = ( + "Outcome actually observed: lateral drift reaches " + f"{peak_drift:.1e} m against a " + f"{isotropy_tolerance['max_abs_lateral_drift_m']:.0e} m isotropy " + "tolerance" + + ( + ", with exact nulls at " + + ", ".join(f"{angle:g}" for angle in null_angles) + + " deg" + if null_angles + else "" + ) + + ( + f" and antisymmetry about 45 deg to within {max_ratio:.0e} of peak" + if max_ratio is not None + else "" + ) + + f", in {solver_names}; criteria exceeded: " + + (", ".join(criteria_union) if criteria_union else "none") + + ( + ", and every cell passes the physical-validity gate." + if not validity_failures + else "; some cells FAIL the physical-validity gate." + ) + ) + # Zeros this fixture asserts are genuine measurements, not missing data. # The validator rejects any other exact zero and any stale entry here, so # an unexpected zero in a future run fails the gate instead of passing as @@ -590,13 +642,9 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "DART 7 main, this commit, one sphere sliding to rolling on " "a static ground box at v0=1 m/s, mu=0.35, dt=2 ms, 1 s " "horizon, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers, " - "launch angles 0-90 deg in 15 deg steps. Outcome actually " - "observed: lateral drift reaches 2.1e-3 m against a 1e-4 m " - "isotropy tolerance, with exact nulls at 0, 45, and 90 deg " - "and antisymmetry about 45 deg to within 1e-13 of peak, in " - "both contact solvers; the drift criterion is the only one of " - "the three tolerances exceeded, and every cell passes the " - "physical-validity gate. Says nothing about other speeds, " + "launch angles 0-90 deg in 15 deg steps. " + + outcome_sentence + + " Says nothing about other speeds, " "shapes, stacks, historical DART versions, or DART 6." ), "limitations": [ diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index 4ecfe6c632aef..60774152ec969 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -942,3 +942,38 @@ def test_visual_artifacts_are_verified_by_content(tmp_path): (fake / "real.png").write_bytes(png) packet["evidence"]["visual"] = ["real.png"] assert MODULE.packet_errors(packet, base_dir=fake) == [] + + +def test_not_applicable_lane_cannot_conclude(tmp_path): + manifest = _minimal_manifest(["CT-001"]) + lane = manifest["claims"][0]["lanes"][LANE] + lane["status"] = "not-applicable" + lane["reason"] = "does not apply here" + lane["disposition"] = "fixed" + lane["evidence"] = ["evidence/packet.json"] + errors = MODULE.manifest_errors(manifest, ["CT-001"]) + assert any("concludes nothing" in error for error in errors) + assert any("must not hold" in error for error in errors) + + +def test_metadata_keys_are_not_identities(): + packet = complete_packet() + packet["configuration"]["resolved"] = {"method_note": "not measured"} + errors = MODULE.packet_errors(packet) + assert any("no recognizable" in error for error in errors) + + +def test_raw_rows_need_measurement_content(): + packet = complete_packet() + packet["evidence"]["raw_rows"] = [{"note": "pending"}] + errors = MODULE.packet_errors(packet) + assert any("metadata-only record" in error for error in errors) + packet["evidence"]["raw_rows"] = [{"lateral_drift_m": 1.5e-3}] + assert MODULE.packet_errors(packet) == [] + + +def test_metric_group_needs_a_real_measurement(): + packet = complete_packet() + packet["metrics"]["numerical"] = {"method": "manual", "note": "not measured"} + errors = MODULE.packet_errors(packet) + assert any("only semantic annotations" in error for error in errors) From ffe03a447dbdd7cc7415a13cb80b9e3229465339 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 23:08:05 -0700 Subject: [PATCH 24/52] Regenerate CT-001 at the round-4 commit The claim boundary now comes from the run's own summary; notably this regeneration reports an exact drift null only at 0 deg (45/90 deg are near-zero but not bit-zero across both solvers), which the previous hardcoded text overstated -- exactly the staleness the derivation removes. Disposition unchanged (reproduced); target commit 5abeeb81a70 contains the current writer. --- .../evidence/CT-001-dart7-rolling-direction.json | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json index dcead5644bf35..f750f03964205 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json @@ -1198,7 +1198,7 @@ } }, "result": { - "claim_boundary": "DART 7 main, this commit, one sphere sliding to rolling on a static ground box at v0=1 m/s, mu=0.35, dt=2 ms, 1 s horizon, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers, launch angles 0-90 deg in 15 deg steps. Outcome actually observed: lateral drift reaches 2.1e-3 m against a 1e-4 m isotropy tolerance, with exact nulls at 0, 45, and 90 deg and antisymmetry about 45 deg to within 1e-13 of peak, in both contact solvers; the drift criterion is the only one of the three tolerances exceeded, and every cell passes the physical-validity gate. Says nothing about other speeds, shapes, stacks, historical DART versions, or DART 6.", + "claim_boundary": "DART 7 main, this commit, one sphere sliding to rolling on a static ground box at v0=1 m/s, mu=0.35, dt=2 ms, 1 s horizon, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers, launch angles 0-90 deg in 15 deg steps. Outcome actually observed: lateral drift reaches 2.1e-03 m against a 1e-04 m isotropy tolerance, with exact nulls at 0 deg and antisymmetry about 45 deg to within 8e-14 of peak, in BOXED_LCP and SEQUENTIAL_IMPULSE; criteria exceeded: max_abs_lateral_drift_m, and every cell passes the physical-validity gate. Says nothing about other speeds, shapes, stacks, historical DART versions, or DART 6.", "disposition": "reproduced", "limitations": [ "Resolved identity now comes from the World's own bake-time resolution, corroborated by per-solver trajectory-hash differences.", @@ -1262,7 +1262,7 @@ }, "target": { "branch": "main", - "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", + "commit": "5abeeb81a701d353027e3027ae227287f8cac71f", "commit_role": "Source state measured: the library and fixture were run at this commit, which is HEAD at capture time. The packet and its writer land in a later commit, so re-running the recorded command requires the child commit that adds the writer; the measured behavior belongs to this one.", "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, From b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 23:32:09 -0700 Subject: [PATCH 25/52] Close the Codex round-5 validator bypasses Identity keys match whole tokens; asserted deterministic repeats require the recorded deterministic_repeats_identical verification flag; measurement windows must be structured numeric objects; visual artifacts are validated as structurally complete media containers (a bare signature or truncated file is rejected; full decoding would need an image dependency and is a recorded boundary); fetch hints must match the durable PR-ref fetch command form (reachability is guaranteed by GitHub PR head refs and checked at write time via --freshness -- a recorded boundary, since a hard existence check would break post-squash-merge CI). Every remaining writer's claim boundary is now derived or exact-pattern guarded: CT-002 asserts its settle-energy pattern and formats the quoted joules from the run; CT-003 aborts if a rerun observes the violations its prose says were not observed; CT-004 asserts finiteness and per-family convergence; CT-005 asserts an unsaturated, non-improving controller and formats the quoted RMS errors; CT-007 asserts the hold/fail pattern and formats closures and descent from the run. The CT-001 provenance no longer claims the ResolvedSolverConfiguration binding is unexposed. Tests: 97 -> 101 validator cases. Packets whose content changes (CT-001/002/005/007) are regenerated at this commit in the follow-up commit. --- .../verification.md | 23 ++++ scripts/check_citation_evidence.py | 112 ++++++++++++------ ...citation_ct001_rolling_direction_packet.py | 25 ++-- ...ite_citation_ct002_dense_contact_packet.py | 61 +++++++++- ...e_citation_ct003_elastic_contact_packet.py | 10 ++ ...itation_ct004_articulated_energy_packet.py | 10 ++ ...write_citation_ct005_pd_tracking_packet.py | 29 ++++- ...e_citation_ct007_high_mass_ratio_packet.py | 51 +++++++- tests/test_check_citation_evidence.py | 40 ++++++- 9 files changed, 297 insertions(+), 64 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index f29e93517c89e..1fa4a8d0829a7 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -538,3 +538,26 @@ Also from the 3444 side: not-applicable lanes cannot publish dispositions or hold evidence, and the Windows check-lint variant now includes the citation gate. Tests: 93 -> 97. CT-001 regenerated at the round-4 commit in the follow-up commit. + +## Codex review round 5 (PR #3445) — 2026-08-16 + +Three findings here plus four on #3444, all addressed on both branches: +identity keys now match whole tokens (`methodology` no longer counts via +its `method` substring); every writer's claim boundary is now either +derived from the run's own measurements or guarded by an exact-pattern +SystemExit — CT-002 (settle-energy pattern + derived joules), CT-003 +(abort if violations are observed while the prose says they were not), +CT-004 (finiteness + per-family convergence), CT-005 (no saturation, no +dt-improvement, derived RMS values), CT-007 (hold/fail pattern with +derived closures and descent) join CT-001/CT-011; the CT-001 +resolved_provenance no longer claims the ResolvedSolverConfiguration +binding is unexposed. From #3444: asserted deterministic repeats require +the recorded `deterministic_repeats_identical: true` verification flag; +measurement windows must be structured numeric objects (prose strings +rejected); visual artifacts are checked as structurally complete +containers (begin AND end markers plus minimum size — a bare eight-byte +signature is rejected; full decoding would need an image dependency and +is a recorded boundary); fetch hints must be the durable PR-ref fetch +command form (reachability itself is guaranteed by GitHub PR head refs +and checked at write time via --freshness — a recorded boundary, since a +hard existence check would break post-squash-merge CI). Tests: 97 -> 101. diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index d060cf87bdaf9..35b9d7bb8b340 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -123,8 +123,22 @@ # configuration.requested/resolved must name a recognizable identity, not an # arbitrary placeholder object. Keys that merely mention an identity word in # a metadata role (method_note, backend_reason, ...) do not count. -IDENTITY_KEY_RE = re.compile( - r"solver|method|detector|integrator|integration|backend|family", re.I +FETCH_HINT_RE = re.compile(r"^git fetch origin pull/3445/head\b") +IDENTITY_KEY_TOKENS = frozenset( + { + "solver", + "solvers", + "method", + "methods", + "detector", + "detectors", + "integrator", + "integration", + "backend", + "backends", + "family", + "families", + } ) IDENTITY_METADATA_SUFFIXES = ( "_note", @@ -141,9 +155,13 @@ def _is_identity_key(key: str) -> bool: - return bool(IDENTITY_KEY_RE.search(key)) and not key.endswith( - IDENTITY_METADATA_SUFFIXES - ) + """Whole-token identity match: `contact_solver_method` counts, + `methodology` (substring only) and `method_note` (metadata role) do + not.""" + if key.endswith(IDENTITY_METADATA_SUFFIXES): + return False + tokens = re.split(r"[^a-zA-Z0-9]+", key.lower()) + return any(token in IDENTITY_KEY_TOKENS for token in tokens) def _has_identity_key(value: object) -> bool: @@ -234,42 +252,53 @@ def _resolve_evidence_path(raw_path: str, base_dir: "Path | None") -> "Path | No return None -_VISUAL_SIGNATURES = { - ".png": (b"\x89PNG\r\n\x1a\n",), - ".apng": (b"\x89PNG\r\n\x1a\n",), - ".jpg": (b"\xff\xd8\xff",), - ".jpeg": (b"\xff\xd8\xff",), - ".gif": (b"GIF87a", b"GIF89a"), - ".webm": (b"\x1a\x45\xdf\xa3",), -} - - def _visual_content_issue(path: "Path") -> "str | None": """Why a resolved visual artifact's bytes do not match its claimed type. - A prose file renamed to `capture.png` satisfies a suffix check; the - magic-signature check makes the artifact itself carry the evidence. + A prose file renamed to `capture.png` satisfies a suffix check, and a + bare eight-byte signature satisfies a header check; requiring the + container's structural begin AND end markers (plus a minimum size) means + the artifact must at least be a complete container of its claimed type. + Full decoding would need an image dependency; that boundary is recorded + in the dev-task verification log. """ suffix = path.suffix.lower() try: with path.open("rb") as stream: - header = stream.read(256) + data = stream.read(8 * 1024 * 1024) except OSError: return "could not be read for media-signature verification" mismatch = ( - "does not carry the media signature its suffix claims; a renamed " - "non-media file is not visual evidence" + "is not a structurally complete media file of its claimed type; a " + "renamed or truncated artifact is not visual evidence" ) - if suffix in _VISUAL_SIGNATURES: - if not any(header.startswith(sig) for sig in _VISUAL_SIGNATURES[suffix]): - return mismatch - return None + if len(data) < 64: + return mismatch + header = data[:256] + if suffix in (".png", ".apng"): + ok = ( + header.startswith(b"\x89PNG\r\n\x1a\n") + and b"IHDR" in data[:64] + and data.rstrip().endswith(b"IEND\xaeB`\x82") + ) + return None if ok else mismatch + if suffix in (".jpg", ".jpeg"): + ok = header.startswith(b"\xff\xd8\xff") and data.rstrip().endswith(b"\xff\xd9") + return None if ok else mismatch + if suffix == ".gif": + ok = header.startswith((b"GIF87a", b"GIF89a")) and data.rstrip().endswith( + b"\x3b" + ) + return None if ok else mismatch if suffix == ".webp": return None if header[:4] == b"RIFF" and header[8:12] == b"WEBP" else mismatch if suffix == ".mp4": return None if header[4:8] == b"ftyp" else mismatch + if suffix == ".webm": + return None if header.startswith(b"\x1a\x45\xdf\xa3") else mismatch if suffix == ".svg": - return None if b"" in data + return None if ok else mismatch return None @@ -515,11 +544,17 @@ def packet_errors( commit = target.get("commit") if not (isinstance(commit, str) and COMMIT_RE.match(commit)): errors.append("target.commit must be a 40-hex commit hash") - if not _is_nonempty_str(target.get("fetch_hint")): + fetch_hint = target.get("fetch_hint") + if not ( + _is_nonempty_str(fetch_hint) and FETCH_HINT_RE.match(fetch_hint.strip()) + ): errors.append( - "target.fetch_hint must record how a clean checkout fetches " - "target.commit (e.g. 'git fetch origin pull//head'); a " - "commit that later becomes unreachable is not reproducible" + "target.fetch_hint must be the durable PR-ref fetch command " + f"(matching {FETCH_HINT_RE.pattern!r}); arbitrary prose does " + "not make target.commit reachable from a clean checkout. " + "Reachability itself is guaranteed by GitHub PR head refs " + "surviving squash-merge and is checked at packet-writing " + "time via --freshness" ) scene = packet.get("scene") @@ -604,6 +639,13 @@ def packet_errors( has_repeats = ( isinstance(repeats, int) and not isinstance(repeats, bool) and repeats >= 2 ) + if has_repeats and ensemble.get("deterministic_repeats_identical") is not True: + errors.append( + "ensemble.deterministic_repeats is asserted without " + "deterministic_repeats_identical: true; a repeat count the " + "writer did not verify bit-identical is a claim, not evidence" + ) + has_repeats = False # A sweep or seed list only counts as an ensemble when its entries are # valid and mutually distinct; [null, null] or a duplicated point is # one run wearing an ensemble's clothes. @@ -649,12 +691,6 @@ def packet_errors( window = ensemble.get("measurement_window") if "measurement_window" not in ensemble: errors.append("ensemble.measurement_window is required") - elif isinstance(window, str): - if not window.strip(): - errors.append( - "ensemble.measurement_window must record an actual " - "window, not an empty value" - ) elif isinstance(window, dict) and window: bad_values = sorted( key for key, item in window.items() if not _is_finite_number(item) @@ -672,9 +708,9 @@ def packet_errors( ) else: errors.append( - "ensemble.measurement_window must be a non-empty string or a " - "non-empty object of finite numeric bounds; a truthy " - "placeholder does not record when measurements were collected" + "ensemble.measurement_window must be a non-empty object of " + "finite numeric bounds; prose or truthy placeholders do not " + "record when measurements were collected" ) metrics = packet.get("metrics") diff --git a/scripts/write_citation_ct001_rolling_direction_packet.py b/scripts/write_citation_ct001_rolling_direction_packet.py index fc6d96d16d663..4395a9f627b3a 100644 --- a/scripts/write_citation_ct001_rolling_direction_packet.py +++ b/scripts/write_citation_ct001_rolling_direction_packet.py @@ -20,9 +20,9 @@ Each (contact solver, angle) cell runs twice and must be bit-identical (deterministic repeats). The packet records requested and resolved solver -identity; `ResolvedSolverConfiguration` is not yet Python-exposed, so the -resolved identity comes from World property readback plus step-profile stage -names, and that gap is a recorded limitation feeding PLAN-123 WS4. +identity from `World.resolved_configuration` (the bake-time +`ResolvedSolverConfiguration`, bound to Python in this branch) plus World +property readback and step-profile stage names. Usage (after `pixi run build`): @@ -522,17 +522,16 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "by_contact_solver_method": resolved_by_method, }, "resolved_provenance": ( - "World property readback (contact_solver_method, " + "World.resolved_configuration (the bake-time " + "ResolvedSolverConfiguration bound to Python in this " + "branch), recorded per cell as world_resolution, plus World " + "property readback (contact_solver_method, " "rigid_body_solver, gravity, time_step) after " - "enter_simulation_mode, with contact solver, timestep, and gravity " - "each asserted equal to the request and " - "stable across every repeat and sweep point. The independent " - "evidence that the selection changed behavior is that the " - "per-angle trajectory hashes differ between the two contact " - "solvers at every angle; the recorded WorldStepProfile stage " - "names are identical for both methods and therefore do not " - "discriminate. World::getResolvedConfiguration() is not yet " - "exposed to Python (PLAN-123 WS4 follow-up)." + "enter_simulation_mode, each asserted equal to the request " + "and stable across every repeat and sweep point. The " + "independent evidence that the selection changed behavior " + "is that the per-angle trajectory hashes differ between the " + "two contact solvers at every angle." ), "detector": ( "DART 7 native World collision pipeline (the World step API " diff --git a/scripts/write_citation_ct002_dense_contact_packet.py b/scripts/write_citation_ct002_dense_contact_packet.py index 8a1e3911dace0..b86ba5a0f8b3d 100644 --- a/scripts/write_citation_ct002_dense_contact_packet.py +++ b/scripts/write_citation_ct002_dense_contact_packet.py @@ -397,6 +397,60 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: # residual, so it does not reproduce the claim. disposition = "reproduced" if unstable_cells else "unresolved" + # The claim boundary asserts one specific measured pattern; regenerating + # its prose against different measurements would publish a false + # conclusion, so assert the pattern and derive the quoted numbers. + si_gains = { + row["timestep_s"]: row["max_total_energy_gain_after_settle_j"] + for row in rows + if row["contact_solver_method"] == "SEQUENTIAL_IMPULSE" + } + lcp_gains = { + row["timestep_s"]: row["max_total_energy_gain_after_settle_j"] + for row in rows + if row["contact_solver_method"] == "BOXED_LCP" + } + energy_tolerance_j = stability_tolerance["max_total_energy_gain_after_settle_j"] + boundary_expectations = { + "only sequential impulse unstable": ( + {cell["contact_solver_method"] for cell in unstable_cells} + == {"SEQUENTIAL_IMPULSE"} + ), + "sequential impulse above tolerance at every timestep": all( + gain > energy_tolerance_j for gain in si_gains.values() + ), + "boxed lcp within tolerance at every timestep": all( + gain <= energy_tolerance_j for gain in lcp_gains.values() + ), + "residual speed classified dt-linear": not any( + "not dt-linear" in reason + for cell in cell_findings + for reason in cell["instability_reasons"] + ), + "instability reasons are settle-energy only": all( + set(cell["instability_reasons"]) + <= { + "kinetic energy injected after settle window", + "total mechanical energy increased after settle", + } + for cell in unstable_cells + ), + } + failed_expectations = sorted( + name for name, ok in boundary_expectations.items() if not ok + ) + if failed_expectations: + raise SystemExit( + "CT-002: measured outcomes no longer match the claim boundary's " + f"asserted pattern ({failed_expectations}); rewrite the boundary " + "from the new findings instead of regenerating stale prose." + ) + si_gain_text = " and ".join( + f"{si_gains[dt]:.1e} J at {dt * 1000:g} ms" for dt in sorted(si_gains) + ) + lcp_gain_text = " and ".join(f"{lcp_gains[dt]:.1e} J" for dt in sorted(lcp_gains)) + initial_energy_j = rows[0]["initial_total_energy_j"] + command = ( "PYTHONPATH=build/default/cpp/Release/python pixi run python " "scripts/write_citation_ct002_dense_contact_packet.py" @@ -578,9 +632,10 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "instability. What does reproduce is narrower and " "solver-specific: after the pile settles, the largest " "single-step increase in total mechanical energy under " - "SEQUENTIAL_IMPULSE is about 1.0e-3 J at 2 ms and 1.3e-3 J " - "at 4 ms against a 1.8e-4 J tolerance on a 180 J scene, " - "while BOXED_LCP stays at or near zero (0.0 and 1.3e-6 J). " + f"SEQUENTIAL_IMPULSE is {si_gain_text} " + f"against a {energy_tolerance_j:.1e} J tolerance on a " + f"{initial_energy_j:.0f} J scene, " + f"while BOXED_LCP stays at or near zero ({lcp_gain_text}). " "The metric is a maximum over settle-window steps, so it " "does not distinguish one anomalous step from sustained " "pumping. That is a small, " diff --git a/scripts/write_citation_ct003_elastic_contact_packet.py b/scripts/write_citation_ct003_elastic_contact_packet.py index 1f1f824fc0d94..59170328b14dd 100644 --- a/scripts/write_citation_ct003_elastic_contact_packet.py +++ b/scripts/write_citation_ct003_elastic_contact_packet.py @@ -311,6 +311,16 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: ] violating_cells = [cell for cell in violating_cells if cell["reasons"]] disposition = "reproduced" if violating_cells else "unresolved" + if violating_cells: + # The claim boundary asserts the cited failure mode was NOT observed; + # a rerun that observes violations must not pair `reproduced` with + # that prose. Abort and force a conscious rewrite. + raise SystemExit( + "CT-003: this run observed energy-envelope or finiteness " + f"violations ({violating_cells}); the claim boundary asserts " + "the failure mode was not observed. Rewrite the boundary, " + "limitations, and disposition rationale from the new findings." + ) command = ( "PYTHONPATH=build/default/cpp/Release/python pixi run python " diff --git a/scripts/write_citation_ct004_articulated_energy_packet.py b/scripts/write_citation_ct004_articulated_energy_packet.py index f57007e6111b6..4bc73235c1ea0 100644 --- a/scripts/write_citation_ct004_articulated_energy_packet.py +++ b/scripts/write_citation_ct004_articulated_energy_packet.py @@ -321,6 +321,16 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: # `unresolved` until matched-cost timing lands, following the CT-003 # precedent for "the cited behavior was not observed". disposition = "unresolved" + if not all_finite or set(converging) != set(family_summary): + # The claim boundary asserts both families stay finite with drift + # shrinking as the timestep shrinks; abort rather than regenerate + # that prose against contradicting measurements. + raise SystemExit( + "CT-004: finiteness or per-family convergence no longer matches " + f"the claim boundary (all_finite={all_finite}, " + f"converging={converging}); rewrite the boundary from the new " + "findings." + ) convergence_finding = { "all_cells_finite": all_finite, "families_with_shrinking_drift": converging, diff --git a/scripts/write_citation_ct005_pd_tracking_packet.py b/scripts/write_citation_ct005_pd_tracking_packet.py index b46abedc722e7..9d6d02d8f7023 100644 --- a/scripts/write_citation_ct005_pd_tracking_packet.py +++ b/scripts/write_citation_ct005_pd_tracking_packet.py @@ -388,6 +388,30 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: # absent and the row cannot be promoted -- the same reasoning that keeps # CT-004 unresolved. The accuracy result is published as its own finding. disposition = "unresolved" + # The claim boundary asserts: stable and unsaturated in every cell, and + # RMS tracking error NOT improving under timestep refinement. Assert the + # pattern and derive the quoted numbers from this run. + if any(row["torque_saturated"] for row in rows) or not all( + row["finite"] for row in rows + ): + raise SystemExit( + "CT-005: a cell saturated or went non-finite; the claim " + "boundary asserts a stable, unsaturated controller everywhere. " + "Rewrite the boundary from the new findings." + ) + if improving: + raise SystemExit( + "CT-005: tracking error now improves with timestep refinement " + f"for {improving}; the claim boundary asserts the opposite " + "(controller-limited error). Rewrite the boundary from the new " + "findings." + ) + rms_at_largest = max( + stats["rms_error_at_largest_timestep"] for stats in family_summary.values() + ) + rms_at_smallest = max( + stats["rms_error_at_smallest_timestep"] for stats in family_summary.values() + ) error_spread = { family: ( stats["rms_error_at_smallest_timestep"] @@ -568,8 +592,9 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "both multibody integration families. What is established: " "the controller stays stable and unsaturated in every " "cell, and RMS tracking error does NOT improve as the " - "timestep is refined eightfold (about 2.1e-3 rad at 4 ms " - "versus 2.4e-3 rad at 0.5 ms, against a 0.25 rad reference " + f"timestep is refined eightfold (about {rms_at_largest:.1e} " + f"rad at 4 ms versus {rms_at_smallest:.1e} rad at 0.5 ms, " + "against a 0.25 rad reference " "amplitude), while control work rises monotonically as the " "timestep shrinks -- the error here is controller-limited, " "not integration-limited. What is " diff --git a/scripts/write_citation_ct007_high_mass_ratio_packet.py b/scripts/write_citation_ct007_high_mass_ratio_packet.py index b4010307fbacc..d4294a63182f2 100644 --- a/scripts/write_citation_ct007_high_mass_ratio_packet.py +++ b/scripts/write_citation_ct007_high_mass_ratio_packet.py @@ -366,6 +366,44 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: # row stays unresolved and the measurement is published as WS5 input. disposition = "unresolved" + # The claim boundary quotes this run's failure pattern and closures; + # assert the pattern and derive the numbers so a regeneration under + # different behavior cannot keep stale prose. + si_finding = baseline_finding["SEQUENTIAL_IMPULSE"] + lcp_finding = baseline_finding["BOXED_LCP"] + si_closures = { + float(ratio): value + for ratio, value in method_summary["SEQUENTIAL_IMPULSE"][ + "relative_gap_closure_by_ratio" + ].items() + } + lcp_closures = { + float(ratio): value + for ratio, value in method_summary["BOXED_LCP"][ + "relative_gap_closure_by_ratio" + ].items() + } + si_fails = [float(r) for r in si_finding["fails_at_mass_ratios"]] + si_holds = sorted(set(si_closures) - set(si_fails)) + if not si_fails or not si_holds or not lcp_finding["holds_across_swept_range"]: + raise SystemExit( + "CT-007: the failure pattern changed (sequential impulse " + f"fails at {si_fails}, boxed-LCP holds=" + f"{lcp_finding['holds_across_swept_range']}); the claim boundary " + "describes SI holding at low ratios, failing at high ones, and " + "boxed-LCP holding throughout. Rewrite it from the new findings." + ) + si_holds_text = " and ".join(f"{r:g}" for r in si_holds) + si_fails_text = " and ".join(f"{r:g}" for r in si_fails) + si_hold_closures = " and ".join(f"{si_closures[r]:.1e}" for r in si_holds) + heavy_descent_m = ( + 2.0 + * float(parameters["box_half_extent_m"]) + * max(si_closures[r] for r in si_fails) + ) + lcp_values = sorted(lcp_closures.values()) + lcp_range_text = f"{lcp_values[0]:.1e} to {lcp_values[-1]:.1e}" + command = ( "PYTHONPATH=build/default/cpp/Release/python pixi run python " "scripts/write_citation_ct007_high_mass_ratio_packet.py" @@ -521,13 +559,14 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "whether an exact cone would help. What this establishes is " "the baseline arm the WS5 GO/NO-GO needs, and it is not a " "null result: SEQUENTIAL_IMPULSE -- the World's default " - "contact solver -- holds the stack at mass ratios 1 and 10 " - "(relative closure 1.4e-4 and 1.0e-3) but fails completely " - "at 100 and 1000, where the heavy box descends a full box " - "height (0.20002 m) while the light box beneath it moves by " + f"contact solver -- holds the stack at mass ratios " + f"{si_holds_text} (relative closure {si_hold_closures}) but " + f"fails completely at {si_fails_text}, where the heavy box " + f"descends a full box height ({heavy_descent_m:.5f} m) while " + "the light box beneath it moves by " "microns, coming to rest fully interpenetrated at near-zero " - "velocity. BOXED_LCP holds across all four decades (closure " - "3.3e-5 to 3.0e-4). Both outcomes are bit-identical across " + f"velocity. BOXED_LCP holds across the swept range (closure " + f"{lcp_range_text}). Both outcomes are bit-identical across " "repeats. Says nothing about deeper stacks, other shapes or " "materials, other timesteps, historical DART versions, or " "DART 6." diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index 60774152ec969..d25f9c6e8c82f 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -68,6 +68,7 @@ def complete_packet() -> dict: "ensemble": { "kind": "deterministic-repeats", "deterministic_repeats": 2, + "deterministic_repeats_identical": True, "measurement_window": {"start_s": 0.0, "end_s": 1.0}, }, "metrics": { @@ -923,7 +924,8 @@ def test_measurement_window_placeholders_fail(): packet["ensemble"]["measurement_window"] = {"warmup_steps": 250} assert MODULE.packet_errors(packet) == [] packet["ensemble"]["measurement_window"] = "full 1 s horizon" - assert MODULE.packet_errors(packet) == [] + errors = MODULE.packet_errors(packet) + assert any("measurement_window" in error for error in errors) def test_visual_artifacts_are_verified_by_content(tmp_path): @@ -933,7 +935,7 @@ def test_visual_artifacts_are_verified_by_content(tmp_path): packet = complete_packet() packet["evidence"]["visual"] = ["capture.png"] errors = MODULE.packet_errors(packet, base_dir=fake) - assert any("media signature" in error for error in errors) + assert any("structurally complete" in error for error in errors) png = b"\x89PNG\r\n\x1a\n" + bytes.fromhex( "0000000d49484452000000010000000108060000001f15c489" "0000000a49444154789c63000100000500010d0a2db4" @@ -977,3 +979,37 @@ def test_metric_group_needs_a_real_measurement(): packet["metrics"]["numerical"] = {"method": "manual", "note": "not measured"} errors = MODULE.packet_errors(packet) assert any("only semantic annotations" in error for error in errors) + + +def test_identity_tokens_are_whole_words(): + packet = complete_packet() + packet["configuration"]["resolved"] = {"methodology": "pending"} + errors = MODULE.packet_errors(packet) + assert any("no recognizable" in error for error in errors) + + +def test_asserted_repeats_need_recorded_verification(): + packet = complete_packet() + del packet["ensemble"]["deterministic_repeats_identical"] + errors = MODULE.packet_errors(packet) + assert any("did not verify bit-identical" in error for error in errors) + + +def test_fetch_hint_must_be_the_durable_pr_ref_command(): + packet = complete_packet() + packet["target"]["fetch_hint"] = "not a command" + errors = MODULE.packet_errors(packet) + assert any("durable PR-ref fetch command" in error for error in errors) + packet["target"]["fetch_hint"] = "git fetch origin pull/999/head" + errors = MODULE.packet_errors(packet) + assert any("durable PR-ref fetch command" in error for error in errors) + + +def test_signature_only_media_is_rejected(tmp_path): + root = tmp_path / "design" + root.mkdir() + (root / "stub.png").write_bytes(b"\x89PNG\r\n\x1a\n") + packet = complete_packet() + packet["evidence"]["visual"] = ["stub.png"] + errors = MODULE.packet_errors(packet, base_dir=root) + assert any("structurally complete" in error for error in errors) From 15e5a3af368c04153635dd449646bc784cdd0675 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 23:39:40 -0700 Subject: [PATCH 26/52] Regenerate CT-001/002/005/007 at the round-5 commit Their boundaries now format quantitative prose from each run's own summary under exact-pattern guards; measured patterns unchanged, so all dispositions are unchanged. Target commits name b3fe1c9eeff, whose tree contains every current writer. --- .../evidence/CT-001-dart7-rolling-direction.json | 4 ++-- .../evidence/CT-002-dart7-dense-inelastic-contact.json | 4 ++-- .../evidence/CT-005-dart7-pd-tracking.json | 4 ++-- .../evidence/CT-007-dart7-high-mass-ratio.json | 4 ++-- 4 files changed, 8 insertions(+), 8 deletions(-) diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json index f750f03964205..cd0111d29c39a 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json @@ -99,7 +99,7 @@ } } }, - "resolved_provenance": "World property readback (contact_solver_method, rigid_body_solver, gravity, time_step) after enter_simulation_mode, with contact solver, timestep, and gravity each asserted equal to the request and stable across every repeat and sweep point. The independent evidence that the selection changed behavior is that the per-angle trajectory hashes differ between the two contact solvers at every angle; the recorded WorldStepProfile stage names are identical for both methods and therefore do not discriminate. World::getResolvedConfiguration() is not yet exposed to Python (PLAN-123 WS4 follow-up).", + "resolved_provenance": "World.resolved_configuration (the bake-time ResolvedSolverConfiguration bound to Python in this branch), recorded per cell as world_resolution, plus World property readback (contact_solver_method, rigid_body_solver, gravity, time_step) after enter_simulation_mode, each asserted equal to the request and stable across every repeat and sweep point. The independent evidence that the selection changed behavior is that the per-angle trajectory hashes differ between the two contact solvers at every angle.", "substeps": 1, "timestep": 0.002 }, @@ -1262,7 +1262,7 @@ }, "target": { "branch": "main", - "commit": "5abeeb81a701d353027e3027ae227287f8cac71f", + "commit": "b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c", "commit_role": "Source state measured: the library and fixture were run at this commit, which is HEAD at capture time. The packet and its writer land in a later commit, so re-running the recorded command requires the child commit that adds the writer; the measured behavior belongs to this one.", "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json index f01e87696eb96..4c04019c6c62c 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json @@ -611,7 +611,7 @@ } }, "result": { - "claim_boundary": "DART 7 main, this commit, a bounded (not source-exact) 6x6x6 sphere-grid drop with restitution 0 and mu=0.8, 2 s horizon, timesteps 2 and 4 ms, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers. Outcome actually observed: no cell failed (4 of 4 finite, no fall-through, penetration within tolerance), and the one cell above the settle-speed tolerance has a final speed exactly proportional to dt (speed/dt identical across timesteps), which is converging integrator residual and is explicitly NOT counted as instability. What does reproduce is narrower and solver-specific: after the pile settles, the largest single-step increase in total mechanical energy under SEQUENTIAL_IMPULSE is about 1.0e-3 J at 2 ms and 1.3e-3 J at 4 ms against a 1.8e-4 J tolerance on a 180 J scene, while BOXED_LCP stays at or near zero (0.0 and 1.3e-6 J). The metric is a maximum over settle-window steps, so it does not distinguish one anomalous step from sustained pumping. That is a small, non-divergent, non-physical energy gain in a resting inelastic pile, not a blow-up. The poor-scaling limb of the cited claim is not instrumented at all (performance is typed unsupported). Says nothing about the original SimBenchmark scene parameters, historical DART versions, other densities/materials, or DART 6.", + "claim_boundary": "DART 7 main, this commit, a bounded (not source-exact) 6x6x6 sphere-grid drop with restitution 0 and mu=0.8, 2 s horizon, timesteps 2 and 4 ms, SEQUENTIAL_IMPULSE and BOXED_LCP contact solvers. Outcome actually observed: no cell failed (4 of 4 finite, no fall-through, penetration within tolerance), and the one cell above the settle-speed tolerance has a final speed exactly proportional to dt (speed/dt identical across timesteps), which is converging integrator residual and is explicitly NOT counted as instability. What does reproduce is narrower and solver-specific: after the pile settles, the largest single-step increase in total mechanical energy under SEQUENTIAL_IMPULSE is 1.0e-03 J at 2 ms and 1.3e-03 J at 4 ms against a 1.8e-04 J tolerance on a 180 J scene, while BOXED_LCP stays at or near zero (0.0e+00 J and 1.3e-06 J). The metric is a maximum over settle-window steps, so it does not distinguish one anomalous step from sustained pumping. That is a small, non-divergent, non-physical energy gain in a resting inelastic pile, not a blow-up. The poor-scaling limb of the cited claim is not instrumented at all (performance is typed unsupported). Says nothing about the original SimBenchmark scene parameters, historical DART versions, other densities/materials, or DART 6.", "disposition": "reproduced", "limitations": [ "Bounded reconstruction: the original SimBenchmark asset, material, and timestep grid are not reproduced exactly; sourcing the exact historical setup is future corpus work.", @@ -671,7 +671,7 @@ }, "target": { "branch": "main", - "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", + "commit": "b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c", "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, "title": "Dense 6x6x6 inelastic contact stability (DART 7 first packet)" diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json index 42384b9ab9a69..1534c9848ac6e 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json @@ -921,7 +921,7 @@ } }, "result": { - "claim_boundary": "DART 7 main, this commit, a 4-link PD-tracked chain with no contact, fixed gains, a 2 s horizon, timesteps 0.5-4 ms, and both multibody integration families. What is established: the controller stays stable and unsaturated in every cell, and RMS tracking error does NOT improve as the timestep is refined eightfold (about 2.1e-3 rad at 4 ms versus 2.4e-3 rad at 0.5 ms, against a 0.25 rad reference amplitude), while control work rises monotonically as the timestep shrinks -- the error here is controller-limited, not integration-limited. What is NOT established: the cited speed/accuracy tradeoff, because step cost is not measured at all; constraint error, because this scene has no constraints to violate; and any ranking between the integration families. Says nothing about contacting or legged control, gain tuning, historical DART versions, or DART 6.", + "claim_boundary": "DART 7 main, this commit, a 4-link PD-tracked chain with no contact, fixed gains, a 2 s horizon, timesteps 0.5-4 ms, and both multibody integration families. What is established: the controller stays stable and unsaturated in every cell, and RMS tracking error does NOT improve as the timestep is refined eightfold (about 2.1e-03 rad at 4 ms versus 2.4e-03 rad at 0.5 ms, against a 0.25 rad reference amplitude), while control work rises monotonically as the timestep shrinks -- the error here is controller-limited, not integration-limited. What is NOT established: the cited speed/accuracy tradeoff, because step cost is not measured at all; constraint error, because this scene has no constraints to violate; and any ranking between the integration families. Says nothing about contacting or legged control, gain tuning, historical DART versions, or DART 6.", "disposition": "unresolved", "limitations": [ "No contact: a controlled robot's hard cases are contact transitions, which this fixture deliberately excludes so the tracking signal is unambiguous.", @@ -976,7 +976,7 @@ }, "target": { "branch": "main", - "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", + "commit": "b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json index 97ad5d2950bca..5285f69d1d9b2 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json @@ -950,7 +950,7 @@ } }, "result": { - "claim_boundary": "DART 7 main, this commit, a two-box stack with a 1 kg lower box, mass ratios 1 to 1000, 2 ms timestep, 2 s horizon, SEQUENTIAL_IMPULSE and BOXED_LCP. The cited claim is comparative and its exact-cone arm does not exist on this branch, so the row is NOT reproduced and nothing here says whether an exact cone would help. What this establishes is the baseline arm the WS5 GO/NO-GO needs, and it is not a null result: SEQUENTIAL_IMPULSE -- the World's default contact solver -- holds the stack at mass ratios 1 and 10 (relative closure 1.4e-4 and 1.0e-3) but fails completely at 100 and 1000, where the heavy box descends a full box height (0.20002 m) while the light box beneath it moves by microns, coming to rest fully interpenetrated at near-zero velocity. BOXED_LCP holds across all four decades (closure 3.3e-5 to 3.0e-4). Both outcomes are bit-identical across repeats. Says nothing about deeper stacks, other shapes or materials, other timesteps, historical DART versions, or DART 6.", + "claim_boundary": "DART 7 main, this commit, a two-box stack with a 1 kg lower box, mass ratios 1 to 1000, 2 ms timestep, 2 s horizon, SEQUENTIAL_IMPULSE and BOXED_LCP. The cited claim is comparative and its exact-cone arm does not exist on this branch, so the row is NOT reproduced and nothing here says whether an exact cone would help. What this establishes is the baseline arm the WS5 GO/NO-GO needs, and it is not a null result: SEQUENTIAL_IMPULSE -- the World's default contact solver -- holds the stack at mass ratios 1 and 10 (relative closure 1.4e-04 and 1.0e-03) but fails completely at 100 and 1000, where the heavy box descends a full box height (0.20002 m) while the light box beneath it moves by microns, coming to rest fully interpenetrated at near-zero velocity. BOXED_LCP holds across the swept range (closure -3.0e-04 to 2.0e-04). Both outcomes are bit-identical across repeats. Says nothing about deeper stacks, other shapes or materials, other timesteps, historical DART versions, or DART 6.", "disposition": "unresolved", "limitations": [ "The sequential-impulse failure is a geometric observation at one timestep. This packet does not establish the mechanism; a fixed iteration budget is the obvious suspect, and confirming it needs the per-solve residual WS4 has not exposed yet.", @@ -1007,7 +1007,7 @@ }, "target": { "branch": "main", - "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", + "commit": "b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", "fetch_hint": "git fetch origin pull/3445/head && git checkout " }, From dd02028d2e0e4fff016e9aba6d8491a8b6d1d641 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sat, 15 Aug 2026 23:55:40 -0700 Subject: [PATCH 27/52] Close the Codex round-6 validator bypasses Placeholder identity values (unknown, n/a, not measured) are rejected; fetch hints must fullmatch the exact fetch-and-checkout command (prefix or suffix variants rejected); the media check reads the file header and tail separately so legitimate files above the old 8 MiB cap are no longer falsely reported truncated; measurement windows must name their bounds (start_s/end_s or warmup_steps/continuation_steps); negative controls are pinned defect-by-defect via .expected-errors.json sidecars (sidecar added for the tracked control) so one regressed check fails the control instead of hiding behind unrelated survivors. Tests: 101 -> 104 validator cases. No packet content changed in this round (all real hints/windows/identities already complied). --- .../verification.md | 15 ++++ ...entionally-incomplete.expected-errors.json | 20 +++++ scripts/check_citation_evidence.py | 89 ++++++++++++++++--- tests/test_check_citation_evidence.py | 50 ++++++++++- 4 files changed, 160 insertions(+), 14 deletions(-) create mode 100644 docs/plans/123-citation-driven-simulation-trust/evidence/negative-controls/CT-002-dart7-intentionally-incomplete.expected-errors.json diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 1fa4a8d0829a7..ab799ae967838 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -561,3 +561,18 @@ is a recorded boundary); fetch hints must be the durable PR-ref fetch command form (reachability itself is guaranteed by GitHub PR head refs and checked at write time via --freshness — a recorded boundary, since a hard existence check would break post-squash-merge CI). Tests: 97 -> 101. + +## Codex review round 6 (PR #3445) — 2026-08-16 + +Six findings across both PRs, all fixed on both branches: placeholder +identity values ("unknown", "n/a", "not measured") are rejected; the +fetch hint must fullmatch the exact fetch-and-checkout command (prefix +or suffix variants rejected); the media check reads the file header and +TAIL separately so legitimate files above the old 8 MiB cap are no +longer falsely reported truncated; measurement windows must NAME their +bounds (start_s/end_s or warmup_steps/continuation_steps); and every +negative control now carries an .expected-errors.json sidecar pinning +EACH seeded defect — if any single validator check regresses, its +pinned error disappears and the control fails, instead of hiding behind +three unrelated survivors. Sidecars added for both tracked controls. +Tests: 101 -> 104 (main), 81 -> 84 (6.20). diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/negative-controls/CT-002-dart7-intentionally-incomplete.expected-errors.json b/docs/plans/123-citation-driven-simulation-trust/evidence/negative-controls/CT-002-dart7-intentionally-incomplete.expected-errors.json new file mode 100644 index 0000000000000..7f3f764a17e02 --- /dev/null +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/negative-controls/CT-002-dart7-intentionally-incomplete.expected-errors.json @@ -0,0 +1,20 @@ +[ + "target.commit must be a 40-hex commit hash", + "target.fetch_hint must be the durable PR-ref fetch command", + "scene.digest must match sha256:<64 hex>", + "scene.parameters must publish the non-empty parameter object", + "configuration.resolved must be a non-empty object", + "configuration.resolved_provenance must name", + "configuration.fallback_policy must be a non-empty string", + "single runs are not evidence", + "ensemble.measurement_window is required", + "metrics.physical must record a non-empty measurement 'method'", + "reports exact zero at ['energy_drift_j', 'max_penetration_m']", + "metrics.numerical is unsupported but has no non-empty reason", + "reports exact zero at ['wall_time_ms']", + "evidence.commands must be a non-empty list of commands", + "evidence must carry raw_rows inline or non-empty raw_paths", + "evidence.visual must list visual artifacts or be typed", + "result.claim_boundary must be a non-empty string", + "result.limitations must be a non-empty list" +] diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 35b9d7bb8b340..2e3d98c2d9f32 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -69,6 +69,20 @@ UNSUPPORTED_SENTINELS = frozenset( {"", "-", "--", "n/a", "na", "none", "null", "tbd", "unknown", "unsupported"} ) + +IDENTITY_PLACEHOLDER_VALUES = frozenset( + UNSUPPORTED_SENTINELS + | {"not measured", "pending", "todo", "missing", "unspecified"} +) + + +def _is_identity_value(value: object) -> bool: + return ( + _is_nonempty_str(value) + and value.strip().lower() not in IDENTITY_PLACEHOLDER_VALUES + ) + + LANE_KEYS = ("dart7", "dart6") BRANCH_BY_LANE = {"dart7": "main", "dart6": "release-6.20"} FIRST_WAVE_FAMILY_CAP = 6 @@ -123,7 +137,9 @@ # configuration.requested/resolved must name a recognizable identity, not an # arbitrary placeholder object. Keys that merely mention an identity word in # a metadata role (method_note, backend_reason, ...) do not count. -FETCH_HINT_RE = re.compile(r"^git fetch origin pull/3445/head\b") +FETCH_HINT_RE = re.compile( + r"^git fetch origin pull/3445/head" r" && git checkout $" +) IDENTITY_KEY_TOKENS = frozenset( { "solver", @@ -170,7 +186,7 @@ def _has_identity_key(value: object) -> bool: {"method_note": "..."} have none.""" if isinstance(value, dict): for key, item in value.items(): - if _is_identity_key(str(key)) and _is_nonempty_str(item): + if _is_identity_key(str(key)) and _is_identity_value(item): return True if _has_identity_key(item): return True @@ -259,34 +275,42 @@ def _visual_content_issue(path: "Path") -> "str | None": bare eight-byte signature satisfies a header check; requiring the container's structural begin AND end markers (plus a minimum size) means the artifact must at least be a complete container of its claimed type. - Full decoding would need an image dependency; that boundary is recorded - in the dev-task verification log. + The header and the file TAIL are read separately so large legitimate + files are not falsely reported truncated. Full decoding would need an + image dependency; that boundary is recorded in the dev-task verification + log. """ suffix = path.suffix.lower() try: + size = path.stat().st_size with path.open("rb") as stream: - data = stream.read(8 * 1024 * 1024) + header = stream.read(4096) + if size > 4096: + stream.seek(-min(size, 4096), 2) + tail = stream.read(4096) + else: + tail = header except OSError: return "could not be read for media-signature verification" mismatch = ( "is not a structurally complete media file of its claimed type; a " "renamed or truncated artifact is not visual evidence" ) - if len(data) < 64: + if size < 64: return mismatch - header = data[:256] + stripped_tail = tail.rstrip() if suffix in (".png", ".apng"): ok = ( header.startswith(b"\x89PNG\r\n\x1a\n") - and b"IHDR" in data[:64] - and data.rstrip().endswith(b"IEND\xaeB`\x82") + and b"IHDR" in header[:64] + and stripped_tail.endswith(b"IEND\xaeB`\x82") ) return None if ok else mismatch if suffix in (".jpg", ".jpeg"): - ok = header.startswith(b"\xff\xd8\xff") and data.rstrip().endswith(b"\xff\xd9") + ok = header.startswith(b"\xff\xd8\xff") and stripped_tail.endswith(b"\xff\xd9") return None if ok else mismatch if suffix == ".gif": - ok = header.startswith((b"GIF87a", b"GIF89a")) and data.rstrip().endswith( + ok = header.startswith((b"GIF87a", b"GIF89a")) and stripped_tail.endswith( b"\x3b" ) return None if ok else mismatch @@ -297,7 +321,7 @@ def _visual_content_issue(path: "Path") -> "str | None": if suffix == ".webm": return None if header.startswith(b"\x1a\x45\xdf\xa3") else mismatch if suffix == ".svg": - ok = b"" in data + ok = b"" in stripped_tail return None if ok else mismatch return None @@ -706,6 +730,15 @@ def packet_errors( errors.append( "ensemble.measurement_window start_s must not exceed end_s" ) + has_time_bounds = {"start_s", "end_s"} <= set(window) + has_step_bounds = {"warmup_steps", "continuation_steps"} <= set(window) + if not (has_time_bounds or has_step_bounds): + errors.append( + "ensemble.measurement_window must name its bounds " + "(start_s/end_s or warmup_steps/continuation_steps); " + "unnamed numbers do not record when measurements were " + "collected" + ) else: errors.append( "ensemble.measurement_window must be a non-empty object of " @@ -1260,6 +1293,8 @@ def validate_tree( "fails closed" ) for packet_path in negative_paths: + if packet_path.name.endswith(".expected-errors.json"): + continue packet = _load_json(packet_path, errors) if packet is None: continue @@ -1270,6 +1305,36 @@ def validate_tree( f"{len(issues)} validation error(s); it must stay clearly " "incomplete (>= 3) or the fail-closed proof is vacuous" ) + # A sidecar pins each SEEDED defect individually: if a validator + # check regresses, its expected error disappears and this fails, + # instead of hiding behind three unrelated survivors. + sidecar = packet_path.with_name( + packet_path.name[: -len(".json")] + ".expected-errors.json" + ) + if sidecar.is_file(): + expected = _load_json(sidecar, errors) + if not ( + isinstance(expected, list) + and expected + and all(_is_nonempty_str(item) for item in expected) + ): + errors.append( + f"{sidecar}: must be a non-empty list of expected error " + "substrings" + ) + else: + for needle in expected: + if not any(needle in issue for issue in issues): + errors.append( + f"{packet_path}: seeded defect no longer " + f"detected (no error contains {needle!r}); a " + "validator check regressed" + ) + else: + errors.append( + f"{packet_path}: negative control has no " + ".expected-errors.json sidecar pinning its seeded defects" + ) return errors diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index d25f9c6e8c82f..bda3cb1d4b39d 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -40,7 +40,9 @@ def complete_packet() -> dict: "target": { "branch": "main", "commit": "0" * 40, - "fetch_hint": "git fetch origin pull/3445/head", + "fetch_hint": ( + "git fetch origin pull/3445/head && git checkout " + ), }, "scene": { "id": "test_scene", @@ -356,6 +358,9 @@ def _write_tree(tmp_path, *, packet=None, negative=None, manifest=None): (negative_dir / "incomplete.json").write_text( json.dumps(negative), encoding="utf-8" ) + (negative_dir / "incomplete.expected-errors.json").write_text( + json.dumps(["missing required top-level keys"]), encoding="utf-8" + ) return plan_dir @@ -921,8 +926,14 @@ def test_measurement_window_placeholders_fail(): errors = MODULE.packet_errors(packet) assert any("measurement_window" in error for error in errors), bad packet = complete_packet() - packet["ensemble"]["measurement_window"] = {"warmup_steps": 250} + packet["ensemble"]["measurement_window"] = { + "warmup_steps": 250, + "continuation_steps": 100, + } assert MODULE.packet_errors(packet) == [] + packet["ensemble"]["measurement_window"] = {"foo": 1.0} + errors = MODULE.packet_errors(packet) + assert any("must name its bounds" in error for error in errors) packet["ensemble"]["measurement_window"] = "full 1 s horizon" errors = MODULE.packet_errors(packet) assert any("measurement_window" in error for error in errors) @@ -1013,3 +1024,38 @@ def test_signature_only_media_is_rejected(tmp_path): packet["evidence"]["visual"] = ["stub.png"] errors = MODULE.packet_errors(packet, base_dir=root) assert any("structurally complete" in error for error in errors) + + +def test_identity_placeholder_values_are_rejected(): + for bad in ("unknown", "n/a", "not measured", "pending"): + packet = complete_packet() + packet["configuration"]["resolved"] = {"solver": bad} + errors = MODULE.packet_errors(packet) + assert any("no recognizable" in error for error in errors), bad + + +def test_fetch_hint_suffix_or_prefix_variants_fail(): + for bad in ( + "git fetch origin pull/3445/head", + "git fetch origin pull/3445/head; echo no-checkout", + "git fetch origin pull/3445/head-wrong && git checkout ", + ): + packet = complete_packet() + packet["target"]["fetch_hint"] = bad + errors = MODULE.packet_errors(packet) + assert any("durable PR-ref fetch command" in error for error in errors), bad + + +def test_negative_control_sidecar_pins_each_seeded_defect(tmp_path): + tree_dir = _write_tree(tmp_path, packet=None, negative={"schema": "x"}) + sidecar = ( + tree_dir / "evidence" / "negative-controls" / "incomplete.expected-errors.json" + ) + sidecar.write_text( + json.dumps(["an error text that no check produces"]), encoding="utf-8" + ) + errors = MODULE.validate_tree(tree_dir) + assert any("seeded defect no longer detected" in error for error in errors) + sidecar.unlink() + errors = MODULE.validate_tree(tree_dir) + assert any("no .expected-errors.json sidecar" in error for error in errors) From 8a778f0e7575ff1b9b4468a233838325e96e071a Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 00:12:47 -0700 Subject: [PATCH 28/52] Close the Codex round-7 validator bypasses Fetch hints are now the runnable command embedding the packet's own target commit, validated to check out exactly target.commit (the placeholder form was not executable); a sidecar orphaned by deleting its control packet no longer counts as a negative control; step-based window bounds must be sane non-negative integers; evidence commands must be the reproducible repository form (VAR=value prefixes plus 'pixi run ...'); deterministic-repeat claims must be bound to hash-bearing recorded evidence; bookkeeping-only numeric rows (seed/index/id) do not count as measurements; the fetch helper documents PR-ownership semantics with a DART_CITATION_PR override for forward work. All packet hints migrated in place to the runnable form embedding their recorded commits -- content-equal to the updated writers' emissions, no simulation data touched (boundary recorded in the verification log). Tests: 104 -> 110 validator cases. --- .../verification.md | 21 ++++ .../CT-001-dart7-rolling-direction.json | 2 +- .../CT-002-dart7-dense-inelastic-contact.json | 2 +- .../CT-003-dart7-dense-elastic-contact.json | 2 +- ...004-dart7-articulated-energy-momentum.json | 2 +- .../evidence/CT-005-dart7-pd-tracking.json | 2 +- .../CT-007-dart7-high-mass-ratio.json | 2 +- .../CT-011-dart7-restore-equivalence.json | 2 +- ...entionally-incomplete.expected-errors.json | 2 +- scripts/check_citation_evidence.py | 103 +++++++++++++++--- scripts/citation_packet_utils.py | 18 ++- ...citation_ct001_rolling_direction_packet.py | 4 +- ...ite_citation_ct002_dense_contact_packet.py | 4 +- ...e_citation_ct003_elastic_contact_packet.py | 4 +- ...itation_ct004_articulated_energy_packet.py | 4 +- ...write_citation_ct005_pd_tracking_packet.py | 4 +- ...e_citation_ct007_high_mass_ratio_packet.py | 4 +- ...tation_ct011_restore_equivalence_packet.py | 4 +- tests/test_check_citation_evidence.py | 80 +++++++++++++- 19 files changed, 216 insertions(+), 50 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index ab799ae967838..bbb465a02db9d 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -576,3 +576,24 @@ EACH seeded defect — if any single validator check regresses, its pinned error disappears and the control fails, instead of hiding behind three unrelated survivors. Sidecars added for both tracked controls. Tests: 101 -> 104 (main), 81 -> 84 (6.20). + +## Codex review round 7 (PR #3445) — 2026-08-16 + +Seven findings across both PRs, all addressed on both branches: the +fetch hint is now the RUNNABLE command embedding the packet's own target +commit (validated to check out exactly target.commit; the placeholder +form was not executable); a sidecar left behind after its control packet +is deleted no longer counts as a negative control; step-based window +bounds must be sane non-negative integers; evidence commands must be +the reproducible repository form (VAR=value prefixes + 'pixi run ...'); +deterministic-repeat claims must be bound to hash-bearing recorded +evidence; bookkeeping-only numeric rows (seed/index/id) no longer count +as measurements; and the fetch helper documents PR-ownership semantics +with a DART_CITATION_PR override for forward work from a later PR. +Existing packets' hints were migrated in place to the runnable form +embedding their recorded commits — content-equal to what the updated +writers emit, with no simulation data touched; the recorded boundary is +that the hint FORMAT is validator-versioned metadata, so re-running the +recorded command at an older target commit reproduces the measurements +but formats this one metadata field per that commit's writer. Tests: +104 -> 110 (main), 84 -> 90 (6.20). diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json index cd0111d29c39a..3d6ad63047830 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json @@ -1264,7 +1264,7 @@ "branch": "main", "commit": "b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c", "commit_role": "Source state measured: the library and fixture were run at this commit, which is HEAD at capture time. The packet and its writer land in a later commit, so re-running the recorded command requires the child commit that adds the writer; the measured behavior belongs to this one.", - "fetch_hint": "git fetch origin pull/3445/head && git checkout " + "fetch_hint": "git fetch origin pull/3445/head && git checkout b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c" }, "title": "Rolling-direction friction dependence (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json index 4c04019c6c62c..dbdc01c37a2f7 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json @@ -672,7 +672,7 @@ "target": { "branch": "main", "commit": "b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c", - "fetch_hint": "git fetch origin pull/3445/head && git checkout " + "fetch_hint": "git fetch origin pull/3445/head && git checkout b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c" }, "title": "Dense 6x6x6 inelastic contact stability (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json index e76fa8fd7b7de..4aef59205ffa7 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json @@ -575,7 +575,7 @@ "target": { "branch": "main", "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", - "fetch_hint": "git fetch origin pull/3445/head && git checkout " + "fetch_hint": "git fetch origin pull/3445/head && git checkout f4e9e5c11d5d3b443693cde9142bf317e2115cd3" }, "title": "Dense elastic contact energy envelope (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json index f3feb12714ec7..d9413312ffc84 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json @@ -979,7 +979,7 @@ "branch": "main", "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", - "fetch_hint": "git fetch origin pull/3445/head && git checkout " + "fetch_hint": "git fetch origin pull/3445/head && git checkout f4e9e5c11d5d3b443693cde9142bf317e2115cd3" }, "title": "Articulated energy drift versus timestep across integration families (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json index 1534c9848ac6e..b10e62ec8e37e 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json @@ -978,7 +978,7 @@ "branch": "main", "commit": "b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", - "fetch_hint": "git fetch origin pull/3445/head && git checkout " + "fetch_hint": "git fetch origin pull/3445/head && git checkout b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c" }, "title": "PD-control tracking error and control work versus timestep (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json index 5285f69d1d9b2..8c2f33ab264e0 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json @@ -1009,7 +1009,7 @@ "branch": "main", "commit": "b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", - "fetch_hint": "git fetch origin pull/3445/head && git checkout " + "fetch_hint": "git fetch origin pull/3445/head && git checkout b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c" }, "title": "High-mass-ratio stack conditioning baseline for the exact-cone GO/NO-GO (DART 7)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json index 18e7099565f64..a12b014c654ba 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json @@ -374,7 +374,7 @@ "branch": "main", "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", - "fetch_hint": "git fetch origin pull/3445/head && git checkout " + "fetch_hint": "git fetch origin pull/3445/head && git checkout f4e9e5c11d5d3b443693cde9142bf317e2115cd3" }, "title": "State-vector restore equivalence under contact (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/negative-controls/CT-002-dart7-intentionally-incomplete.expected-errors.json b/docs/plans/123-citation-driven-simulation-trust/evidence/negative-controls/CT-002-dart7-intentionally-incomplete.expected-errors.json index 7f3f764a17e02..014474b972065 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/negative-controls/CT-002-dart7-intentionally-incomplete.expected-errors.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/negative-controls/CT-002-dart7-intentionally-incomplete.expected-errors.json @@ -1,6 +1,6 @@ [ "target.commit must be a 40-hex commit hash", - "target.fetch_hint must be the durable PR-ref fetch command", + "target.fetch_hint must be the runnable durable PR-ref", "scene.digest must match sha256:<64 hex>", "scene.parameters must publish the non-empty parameter object", "configuration.resolved must be a non-empty object", diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 2e3d98c2d9f32..0afaeee7173d7 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -70,6 +70,28 @@ {"", "-", "--", "n/a", "na", "none", "null", "tbd", "unknown", "unsupported"} ) +COMMAND_RE = re.compile(r"^(?:[A-Za-z_][A-Za-z0-9_]*=\S+\s+)*pixi run\s+\S.*$") +NUMERIC_BOOKKEEPING_KEYS = frozenset( + {"seed", "seeds", "index", "idx", "id", "ids", "repeat", "repeats", "run"} +) + + +def _has_hash_leaf(value: object) -> bool: + """True when any nested key names a hash/sha256 with a non-empty value.""" + if isinstance(value, dict): + for key, item in value.items(): + key_l = str(key).lower() + if ("sha256" in key_l or "hash" in key_l) and ( + _is_nonempty_str(item) or (isinstance(item, (list, dict)) and item) + ): + return True + if _has_hash_leaf(item): + return True + elif isinstance(value, list): + return any(_has_hash_leaf(item) for item in value) + return False + + IDENTITY_PLACEHOLDER_VALUES = frozenset( UNSUPPORTED_SENTINELS | {"not measured", "pending", "todo", "missing", "unspecified"} @@ -138,7 +160,7 @@ def _is_identity_value(value: object) -> bool: # arbitrary placeholder object. Keys that merely mention an identity word in # a metadata role (method_note, backend_reason, ...) do not count. FETCH_HINT_RE = re.compile( - r"^git fetch origin pull/3445/head" r" && git checkout $" + r"^git fetch origin pull/3445/head && git checkout ([0-9a-f]{40})$" ) IDENTITY_KEY_TOKENS = frozenset( { @@ -569,16 +591,26 @@ def packet_errors( if not (isinstance(commit, str) and COMMIT_RE.match(commit)): errors.append("target.commit must be a 40-hex commit hash") fetch_hint = target.get("fetch_hint") - if not ( - _is_nonempty_str(fetch_hint) and FETCH_HINT_RE.match(fetch_hint.strip()) - ): + hint_match = ( + FETCH_HINT_RE.match(fetch_hint.strip()) + if _is_nonempty_str(fetch_hint) + else None + ) + if hint_match is None: errors.append( - "target.fetch_hint must be the durable PR-ref fetch command " - f"(matching {FETCH_HINT_RE.pattern!r}); arbitrary prose does " - "not make target.commit reachable from a clean checkout. " - "Reachability itself is guaranteed by GitHub PR head refs " - "surviving squash-merge and is checked at packet-writing " - "time via --freshness" + "target.fetch_hint must be the runnable durable PR-ref " + f"command (matching {FETCH_HINT_RE.pattern!r}); arbitrary " + "prose or placeholder arguments cannot be executed to reach " + "target.commit. Reachability itself is guaranteed by GitHub " + "PR head refs surviving squash-merge and is checked at " + "packet-writing time via --freshness" + ) + elif isinstance(commit, str) and hint_match.group(1) != commit: + errors.append( + "target.fetch_hint checks out " + f"{hint_match.group(1)[:12]}... but target.commit is " + f"{commit[:12]}...; the hint must reproduce THIS packet's " + "target" ) scene = packet.get("scene") @@ -663,6 +695,14 @@ def packet_errors( has_repeats = ( isinstance(repeats, int) and not isinstance(repeats, bool) and repeats >= 2 ) + if has_repeats and not _has_hash_leaf(packet.get("evidence")): + errors.append( + "ensemble.deterministic_repeats is asserted but the " + "evidence carries no *hash*/sha256 field binding the " + "repeats to recorded trajectories; an unverifiable repeat " + "claim is not an ensemble" + ) + has_repeats = False if has_repeats and ensemble.get("deterministic_repeats_identical") is not True: errors.append( "ensemble.deterministic_repeats is asserted without " @@ -732,6 +772,22 @@ def packet_errors( ) has_time_bounds = {"start_s", "end_s"} <= set(window) has_step_bounds = {"warmup_steps", "continuation_steps"} <= set(window) + if has_step_bounds: + warmup = window.get("warmup_steps") + continuation = window.get("continuation_steps") + if not ( + isinstance(warmup, int) + and not isinstance(warmup, bool) + and warmup >= 0 + and isinstance(continuation, int) + and not isinstance(continuation, bool) + and continuation >= 1 + ): + errors.append( + "ensemble.measurement_window step bounds must be " + "non-negative integers with continuation_steps >= 1" + ) + has_step_bounds = False if not (has_time_bounds or has_step_bounds): errors.append( "ensemble.measurement_window must name its bounds " @@ -769,6 +825,16 @@ def packet_errors( and all(_is_nonempty_str(command) for command in commands) ): errors.append("evidence.commands must be a non-empty list of commands") + else: + for index, command in enumerate(commands): + if not COMMAND_RE.match(command.strip()): + errors.append( + f"evidence.commands[{index}] must be the " + "reproducible repository form (optional VAR=value " + "prefixes followed by 'pixi run ...'); arbitrary " + "shell strings are not the promised reproduction " + "path" + ) raw_paths = evidence.get("raw_paths") raw_rows = evidence.get("raw_rows") has_paths = isinstance(raw_paths, list) and bool(raw_paths) @@ -785,13 +851,15 @@ def packet_errors( ) continue if not any( - _is_finite_number(leaf) or isinstance(leaf, bool) - for _, leaf in _metric_leaves(row) + (_is_finite_number(leaf) or isinstance(leaf, bool)) + and re.sub(r"(\[\d+\])+$", "", path.rsplit(".", 1)[-1]) + not in NUMERIC_BOOKKEEPING_KEYS + for path, leaf in _metric_leaves(row) ): errors.append( f"evidence.raw_rows[{index}] carries no numeric or " - "boolean measurement; a metadata-only record is not " - "raw evidence" + "boolean measurement beyond bookkeeping (seed/index/" + "id/...); a metadata-only record is not raw evidence" ) if has_paths: for index, raw_path in enumerate(raw_paths): @@ -1286,6 +1354,11 @@ def validate_tree( negative_paths = ( sorted(negative_dir.rglob("*.json")) if negative_dir.is_dir() else [] ) + negative_paths = [ + path + for path in negative_paths + if not path.name.endswith(".expected-errors.json") + ] if not negative_paths: errors.append( f"{negative_dir}: at least one intentionally incomplete " @@ -1293,8 +1366,6 @@ def validate_tree( "fails closed" ) for packet_path in negative_paths: - if packet_path.name.endswith(".expected-errors.json"): - continue packet = _load_json(packet_path, errors) if packet is None: continue diff --git a/scripts/citation_packet_utils.py b/scripts/citation_packet_utils.py index 6e3269984cffb..3383cad897866 100644 --- a/scripts/citation_packet_utils.py +++ b/scripts/citation_packet_utils.py @@ -118,12 +118,18 @@ def solver_iterations_by_method( CITATION_PR_NUMBER = 3445 -def target_fetch_hint() -> str: - """How a clean checkout reaches a packet's target.commit forever.""" - return ( - f"git fetch origin pull/{CITATION_PR_NUMBER}/head && " - "git checkout " - ) +def target_fetch_hint(commit: str) -> str: + """The runnable command reaching `commit` from a clean checkout forever. + + CITATION_PR_NUMBER is the PR that owns this evidence tree revision. When + a later PR takes ownership, update the constant together with the + validator's FETCH_HINT_RE; for forward work from another PR, override + with the DART_CITATION_PR environment variable. + """ + import os as _os + + pr = _os.environ.get("DART_CITATION_PR", str(CITATION_PR_NUMBER)) + return f"git fetch origin pull/{pr}/head && git checkout {commit}" def packet_content_digest(packet: dict[str, Any]) -> str: diff --git a/scripts/write_citation_ct001_rolling_direction_packet.py b/scripts/write_citation_ct001_rolling_direction_packet.py index 4395a9f627b3a..715f541b93d69 100644 --- a/scripts/write_citation_ct001_rolling_direction_packet.py +++ b/scripts/write_citation_ct001_rolling_direction_packet.py @@ -493,8 +493,8 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: }, "target": { "branch": "main", - "commit": git_head(), - "fetch_hint": target_fetch_hint(), + "commit": (head_commit := git_head()), + "fetch_hint": target_fetch_hint(head_commit), "commit_role": ( "Source state measured: the library and fixture were run at " "this commit, which is HEAD at capture time. The packet and " diff --git a/scripts/write_citation_ct002_dense_contact_packet.py b/scripts/write_citation_ct002_dense_contact_packet.py index b86ba5a0f8b3d..0115cb12e03ed 100644 --- a/scripts/write_citation_ct002_dense_contact_packet.py +++ b/scripts/write_citation_ct002_dense_contact_packet.py @@ -468,8 +468,8 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: }, "target": { "branch": "main", - "commit": git_head(), - "fetch_hint": target_fetch_hint(), + "commit": (head_commit := git_head()), + "fetch_hint": target_fetch_hint(head_commit), }, "scene": { "id": parameters["scene_id"], diff --git a/scripts/write_citation_ct003_elastic_contact_packet.py b/scripts/write_citation_ct003_elastic_contact_packet.py index 59170328b14dd..4fb34d8a9a3c8 100644 --- a/scripts/write_citation_ct003_elastic_contact_packet.py +++ b/scripts/write_citation_ct003_elastic_contact_packet.py @@ -338,8 +338,8 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: }, "target": { "branch": "main", - "commit": git_head(), - "fetch_hint": target_fetch_hint(), + "commit": (head_commit := git_head()), + "fetch_hint": target_fetch_hint(head_commit), }, "scene": { "id": parameters["scene_id"], diff --git a/scripts/write_citation_ct004_articulated_energy_packet.py b/scripts/write_citation_ct004_articulated_energy_packet.py index 4bc73235c1ea0..ba5381f4bc8be 100644 --- a/scripts/write_citation_ct004_articulated_energy_packet.py +++ b/scripts/write_citation_ct004_articulated_energy_packet.py @@ -363,8 +363,8 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: }, "target": { "branch": "main", - "commit": git_head(), - "fetch_hint": target_fetch_hint(), + "commit": (head_commit := git_head()), + "fetch_hint": target_fetch_hint(head_commit), "commit_role": ( "Source state measured: the library and fixture ran at this " "commit, which is HEAD at capture time. The packet and its " diff --git a/scripts/write_citation_ct005_pd_tracking_packet.py b/scripts/write_citation_ct005_pd_tracking_packet.py index 9d6d02d8f7023..0e40381f3e1c5 100644 --- a/scripts/write_citation_ct005_pd_tracking_packet.py +++ b/scripts/write_citation_ct005_pd_tracking_packet.py @@ -457,8 +457,8 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: }, "target": { "branch": "main", - "commit": git_head(), - "fetch_hint": target_fetch_hint(), + "commit": (head_commit := git_head()), + "fetch_hint": target_fetch_hint(head_commit), "commit_role": ( "Source state measured: the library and fixture ran at this " "commit, which is HEAD at capture time. The packet and its " diff --git a/scripts/write_citation_ct007_high_mass_ratio_packet.py b/scripts/write_citation_ct007_high_mass_ratio_packet.py index d4294a63182f2..e4f2f40247cba 100644 --- a/scripts/write_citation_ct007_high_mass_ratio_packet.py +++ b/scripts/write_citation_ct007_high_mass_ratio_packet.py @@ -424,8 +424,8 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: }, "target": { "branch": "main", - "commit": git_head(), - "fetch_hint": target_fetch_hint(), + "commit": (head_commit := git_head()), + "fetch_hint": target_fetch_hint(head_commit), "commit_role": ( "Source state measured: the library and fixture ran at this " "commit, which is HEAD at capture time. The packet and its " diff --git a/scripts/write_citation_ct011_restore_equivalence_packet.py b/scripts/write_citation_ct011_restore_equivalence_packet.py index 55304677cd0db..184412b5c19f6 100644 --- a/scripts/write_citation_ct011_restore_equivalence_packet.py +++ b/scripts/write_citation_ct011_restore_equivalence_packet.py @@ -412,8 +412,8 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: }, "target": { "branch": "main", - "commit": git_head(), - "fetch_hint": target_fetch_hint(), + "commit": (head_commit := git_head()), + "fetch_hint": target_fetch_hint(head_commit), "commit_role": ( "Source state measured: the library and fixture ran at this " "commit, which is HEAD at capture time. The packet and its " diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index bda3cb1d4b39d..cdf6141af80ef 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -41,7 +41,7 @@ def complete_packet() -> dict: "branch": "main", "commit": "0" * 40, "fetch_hint": ( - "git fetch origin pull/3445/head && git checkout " + "git fetch origin pull/3445/head && git checkout " + "0" * 40 ), }, "scene": { @@ -91,7 +91,13 @@ def complete_packet() -> dict: }, "evidence": { "commands": ["pixi run python scripts/example.py"], - "raw_rows": [{"angle_deg": 0.0, "lateral_drift_m": 0.0}], + "raw_rows": [ + { + "angle_deg": 0.0, + "lateral_drift_m": 0.0, + "trajectory_sha256": "d" * 64, + } + ], "visual": { "status": "not-applicable", "reason": "numeric oracle only", @@ -484,6 +490,11 @@ def test_dangling_raw_path_fails(tmp_path): assert any("does not resolve" in error for error in errors) (tmp_path / "real.csv").write_text("x", encoding="utf-8") packet["evidence"]["raw_paths"] = ["real.csv"] + # raw_paths carry no hash leaf, so the repeats claim needs one recorded + # elsewhere; a sweep ensemble sidesteps that requirement here. + del packet["ensemble"]["deterministic_repeats"] + del packet["ensemble"]["deterministic_repeats_identical"] + packet["ensemble"]["sweep"] = [{"angle_deg": 0.0}, {"angle_deg": 15.0}] assert MODULE.packet_errors(packet, base_dir=tmp_path) == [] @@ -981,7 +992,9 @@ def test_raw_rows_need_measurement_content(): packet["evidence"]["raw_rows"] = [{"note": "pending"}] errors = MODULE.packet_errors(packet) assert any("metadata-only record" in error for error in errors) - packet["evidence"]["raw_rows"] = [{"lateral_drift_m": 1.5e-3}] + packet["evidence"]["raw_rows"] = [ + {"lateral_drift_m": 1.5e-3, "trajectory_sha256": "f" * 64} + ] assert MODULE.packet_errors(packet) == [] @@ -1010,10 +1023,10 @@ def test_fetch_hint_must_be_the_durable_pr_ref_command(): packet = complete_packet() packet["target"]["fetch_hint"] = "not a command" errors = MODULE.packet_errors(packet) - assert any("durable PR-ref fetch command" in error for error in errors) + assert any("runnable durable PR-ref" in error for error in errors) packet["target"]["fetch_hint"] = "git fetch origin pull/999/head" errors = MODULE.packet_errors(packet) - assert any("durable PR-ref fetch command" in error for error in errors) + assert any("runnable durable PR-ref" in error for error in errors) def test_signature_only_media_is_rejected(tmp_path): @@ -1043,7 +1056,7 @@ def test_fetch_hint_suffix_or_prefix_variants_fail(): packet = complete_packet() packet["target"]["fetch_hint"] = bad errors = MODULE.packet_errors(packet) - assert any("durable PR-ref fetch command" in error for error in errors), bad + assert any("runnable durable PR-ref" in error for error in errors), bad def test_negative_control_sidecar_pins_each_seeded_defect(tmp_path): @@ -1059,3 +1072,58 @@ def test_negative_control_sidecar_pins_each_seeded_defect(tmp_path): sidecar.unlink() errors = MODULE.validate_tree(tree_dir) assert any("no .expected-errors.json sidecar" in error for error in errors) + + +def test_fetch_hint_must_check_out_the_target_commit(): + packet = complete_packet() + packet["target"]["fetch_hint"] = ( + "git fetch origin pull/3445/head && git checkout " + "1" * 40 + ) + errors = MODULE.packet_errors(packet) + assert any("must reproduce THIS packet's target" in error for error in errors) + + +def test_orphan_sidecar_does_not_count_as_a_control(tmp_path): + tree_dir = _write_tree(tmp_path, packet=None, negative={"schema": "x"}) + (tree_dir / "evidence" / "negative-controls" / "incomplete.json").unlink() + errors = MODULE.validate_tree(tree_dir) + assert any("at least one intentionally incomplete" in error for error in errors) + + +def test_step_window_bounds_must_be_sane_integers(): + packet = complete_packet() + packet["ensemble"]["measurement_window"] = { + "warmup_steps": -1.5, + "continuation_steps": -2, + } + errors = MODULE.packet_errors(packet) + assert any("non-negative integers" in error for error in errors) + + +def test_commands_must_be_the_reproducible_pixi_form(): + for bad in ("pending", "echo success", "bash -c 'anything'"): + packet = complete_packet() + packet["evidence"]["commands"] = [bad] + errors = MODULE.packet_errors(packet) + assert any("reproducible repository form" in error for error in errors), bad + packet = complete_packet() + packet["evidence"]["commands"] = [ + "PYTHONPATH=build/x pixi run python scripts/write.py" + ] + assert MODULE.packet_errors(packet) == [] + + +def test_bookkeeping_only_rows_fail(): + packet = complete_packet() + packet["evidence"]["raw_rows"] = [{"seed": 1, "trajectory_sha256": "e" * 64}] + errors = MODULE.packet_errors(packet) + assert any("beyond bookkeeping" in error for error in errors) + + +def test_repeats_need_hash_bearing_evidence(): + packet = complete_packet() + packet["evidence"]["raw_rows"] = [{"angle_deg": 0.0, "lateral_drift_m": 0.5}] + errors = MODULE.packet_errors(packet) + assert any( + "binding the repeats to recorded trajectories" in error for error in errors + ) From f005f1fcd909540762155f7670c2403e0847058c Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 00:27:50 -0700 Subject: [PATCH 29/52] Close the Codex round-8 validator bypasses A lane's published disposition must match its packet on OPEN lanes too (the program keeps current-best dispositions on open lanes; they may be null but may not contradict the evidence -- the null-until-closed framing was resolved this way and recorded); reproduction commands reject shell tails (;, |, &, backticks, dollar); duplicate JSON object keys are rejected at load since consumer-dependent first/last-wins is not portable machine evidence; raw_paths must name raw-data artifacts (csv/tsv/json/jsonl/ndjson/npz/npy/parquet); hash leaves must be actual hex digests so 'pending' cannot bind a repeat claim; source URLs must be retrievable http(s) URLs. Tests: 110 -> 116 validator cases. --- .../verification.md | 15 ++++ scripts/check_citation_evidence.py | 74 +++++++++++++++---- tests/test_check_citation_evidence.py | 74 ++++++++++++++++++- 3 files changed, 145 insertions(+), 18 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index bbb465a02db9d..3e76696069d92 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -597,3 +597,18 @@ that the hint FORMAT is validator-versioned metadata, so re-running the recorded command at an older target commit reproduces the measurements but formats this one metadata field per that commit's writer. Tests: 104 -> 110 (main), 84 -> 90 (6.20). + +## Codex review round 8 (PR #3445) — 2026-08-16 + +Six findings, all fixed on both branches: a lane's published disposition +must match its packet's disposition on OPEN lanes too (the program's +semantics keep dispositions on open lanes as current-best; they may be +null but may not contradict — Codex's null-until-closed framing was +resolved this way); reproduction commands reject shell tails +(;, |, &, backticks, $); duplicate JSON object keys are rejected at load +(consumer-dependent first/last-wins is not portable evidence); +raw_paths must name raw-data artifacts (csv/tsv/json/jsonl/ndjson/ +npz/npy/parquet — prose or code files rejected); hash leaves must be +actual hex digests, so "hash": "pending" cannot bind a repeat claim; +and source URLs must be retrievable http(s) URLs. Tests: 110 -> 116 +(main), 90 -> 96 (6.20). diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 0afaeee7173d7..27d837e2a50a4 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -70,7 +70,13 @@ {"", "-", "--", "n/a", "na", "none", "null", "tbd", "unknown", "unsupported"} ) -COMMAND_RE = re.compile(r"^(?:[A-Za-z_][A-Za-z0-9_]*=\S+\s+)*pixi run\s+\S.*$") +COMMAND_RE = re.compile( + r"^(?:[A-Za-z_][A-Za-z0-9_]*=[^\s;|&`$]+\s+)*pixi run\s+[^;|&`$]+$" +) +HASH_VALUE_RE = re.compile(r"^(sha256:)?[0-9a-f]{32,}$") +RAW_DATA_SUFFIXES = frozenset( + {".csv", ".tsv", ".json", ".jsonl", ".ndjson", ".npz", ".npy", ".parquet"} +) NUMERIC_BOOKKEEPING_KEYS = frozenset( {"seed", "seeds", "index", "idx", "id", "ids", "repeat", "repeats", "run"} ) @@ -82,7 +88,8 @@ def _has_hash_leaf(value: object) -> bool: for key, item in value.items(): key_l = str(key).lower() if ("sha256" in key_l or "hash" in key_l) and ( - _is_nonempty_str(item) or (isinstance(item, (list, dict)) and item) + (_is_nonempty_str(item) and HASH_VALUE_RE.match(item.strip())) + or (isinstance(item, (list, dict)) and _has_hash_leaf(item)) ): return True if _has_hash_leaf(item): @@ -575,8 +582,12 @@ def packet_errors( if not isinstance(source, dict): errors.append("source must be an object") else: - if not _is_nonempty_str(source.get("url")): - errors.append("source.url must be a non-empty string") + url = source.get("url") + if not (_is_nonempty_str(url) and re.match(r"^https?://\S+$", url.strip())): + errors.append( + "source.url must be a retrievable http(s) URL; a placeholder " + "does not bind the claim to its source" + ) if not _is_nonempty_str(source.get("claim")): errors.append("source.claim must be a non-empty string") @@ -869,6 +880,17 @@ def packet_errors( ) continue issue = _evidence_path_issue(raw_path, base_dir) + if issue is not None and "must be a relative path" in issue: + errors.append(f"evidence.raw_paths[{index}] {issue}") + continue + if PurePosixPath(raw_path).suffix.lower() not in RAW_DATA_SUFFIXES: + errors.append( + f"evidence.raw_paths[{index}] {raw_path!r} must name " + "a raw-data artifact " + f"({', '.join(sorted(RAW_DATA_SUFFIXES))}); prose or " + "code files are not raw evidence" + ) + continue if issue is not None: errors.append(f"evidence.raw_paths[{index}] {issue}") visual = evidence.get("visual") @@ -1102,11 +1124,23 @@ def _reject_nonstandard_constant(constant: str) -> None: raise ValueError(f"non-standard JSON constant {constant!r}") +def _reject_duplicate_keys(pairs: list) -> dict: + """Different JSON consumers disagree on duplicate keys (first vs last + wins), so a packet carrying one is not portable machine evidence.""" + seen: dict = {} + for key, value in pairs: + if key in seen: + raise ValueError(f"duplicate JSON object key {key!r}") + seen[key] = value + return seen + + def _load_json(path: Path, errors: list[str]) -> object | None: try: return json.loads( path.read_text(encoding="utf-8"), parse_constant=_reject_nonstandard_constant, + object_pairs_hook=_reject_duplicate_keys, ) except (OSError, ValueError) as error: # json.JSONDecodeError subclasses ValueError; parse_constant raises @@ -1154,7 +1188,7 @@ def validate_tree( manifest = _load_json(manifest_path, errors) known_ids: set[str] = set() lane_evidence: dict[str, tuple[str, str]] = {} - closed_lane_dispositions: dict[str, object] = {} + lane_dispositions: dict[str, tuple[str, object]] = {} if manifest is not None: manifest_issues = manifest_errors(manifest, corpus_ids) errors.extend(f"{manifest_path}: {issue}" for issue in manifest_issues) @@ -1218,8 +1252,10 @@ def validate_tree( str(claim_id), str(lane_name), ) - if lane.get("status") == "closed": - closed_lane_dispositions[rel] = lane.get("disposition") + lane_dispositions[rel] = ( + str(lane.get("status")), + lane.get("disposition"), + ) # Every lane-referenced path is validated as a packet, wherever it sits. # Enumerating only `evidence/*.json` would let a lane close a row with a @@ -1302,7 +1338,8 @@ def validate_tree( f"{packet_path}: target.branch {branch!r} does not match " f"lane {lane_name} branch {expected_branch!r}" ) - if rel in closed_lane_dispositions: + lane_status, lane_disposition = lane_dispositions.get(rel, (None, None)) + if lane_status == "closed": passes = ( packet.get("review", {}).get("passes") if isinstance(packet.get("review"), dict) @@ -1326,19 +1363,24 @@ def validate_tree( "two INDEPENDENT review passes (distinct " "reviewers); a duplicated reviewer is one review" ) - lane_disposition = closed_lane_dispositions[rel] + # A lane's published disposition -- open OR closed -- must be the + # one its evidence records; open lanes may defer (null) but may + # not contradict. + if lane_status in ("closed", "in-progress", "audit-required"): packet_disposition = ( packet.get("result", {}).get("disposition") if isinstance(packet.get("result"), dict) else None ) - if lane_disposition != packet_disposition: - errors.append( - f"{packet_path}: the closing lane records disposition " - f"{lane_disposition!r} but the packet's result is " - f"{packet_disposition!r}; the manifest cannot publish " - "a conclusion its evidence does not support" - ) + if lane_status == "closed" or lane_disposition is not None: + if lane_disposition != packet_disposition: + errors.append( + f"{packet_path}: the lane records disposition " + f"{lane_disposition!r} but the packet's result " + f"is {packet_disposition!r}; the manifest cannot " + "publish a conclusion its evidence does not " + "support" + ) if freshness_head is not None: commit = ( packet.get("target", {}).get("commit") diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index cdf6141af80ef..6aa3e01be8c54 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -837,8 +837,8 @@ def test_non_string_lane_evidence_entry_fails(tmp_path): def test_raw_path_pointing_at_a_directory_fails(tmp_path): packet = complete_packet() del packet["evidence"]["raw_rows"] - (tmp_path / "somedir").mkdir() - packet["evidence"]["raw_paths"] = ["somedir"] + (tmp_path / "somedir.csv").mkdir() + packet["evidence"]["raw_paths"] = ["somedir.csv"] errors = MODULE.packet_errors(packet, base_dir=tmp_path) assert any("does not resolve" in error for error in errors) @@ -1127,3 +1127,73 @@ def test_repeats_need_hash_bearing_evidence(): assert any( "binding the repeats to recorded trajectories" in error for error in errors ) + + +def test_open_lane_disposition_must_match_packet(tmp_path): + packet = copy.deepcopy(complete_packet()) # result.disposition: unresolved + manifest = _minimal_manifest(["CT-001"]) + lane = manifest["claims"][0]["lanes"][LANE] + lane["status"] = "in-progress" + lane["disposition"] = "fixed" + lane["evidence"] = ["evidence/packet.json"] + tree_dir = _write_tree( + tmp_path, packet=packet, negative={"schema": "x"}, manifest=manifest + ) + errors = MODULE.validate_tree(tree_dir) + assert any( + "cannot publish a conclusion its evidence does not support" in error + for error in errors + ) + + +def test_commands_with_shell_tails_fail(): + for bad in ( + "pixi run test; false", + "pixi run test && rm -rf /", + "pixi run test | tee log", + "pixi run test `id`", + ): + packet = complete_packet() + packet["evidence"]["commands"] = [bad] + errors = MODULE.packet_errors(packet) + assert any("reproducible repository form" in error for error in errors), bad + + +def test_duplicate_json_keys_fail_at_load(tmp_path): + tree_dir = _write_tree(tmp_path, packet=None, negative={"schema": "x"}) + (tree_dir / "evidence" / "packet.json").write_text( + '{"schema": "a", "schema": "b"}', encoding="utf-8" + ) + manifest = _minimal_manifest(["CT-001"]) + manifest["claims"][0]["lanes"][LANE]["status"] = "in-progress" + manifest["claims"][0]["lanes"][LANE]["evidence"] = ["evidence/packet.json"] + (tree_dir / "claims-manifest.json").write_text( + json.dumps(manifest), encoding="utf-8" + ) + errors = MODULE.validate_tree(tree_dir) + assert any("duplicate JSON object key" in error for error in errors) + + +def test_prose_raw_paths_fail(tmp_path): + (tmp_path / "AGENTS.md").write_text("prose", encoding="utf-8") + packet = complete_packet() + del packet["evidence"]["raw_rows"] + packet["evidence"]["raw_paths"] = ["AGENTS.md"] + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any("must name a raw-data artifact" in error for error in errors) + + +def test_placeholder_hash_values_do_not_bind_repeats(): + packet = complete_packet() + packet["evidence"]["raw_rows"] = [{"lateral_drift_m": 0.5, "hash": "pending"}] + errors = MODULE.packet_errors(packet) + assert any( + "binding the repeats to recorded trajectories" in error for error in errors + ) + + +def test_placeholder_source_urls_fail(): + packet = complete_packet() + packet["source"]["url"] = "pending" + errors = MODULE.packet_errors(packet) + assert any("retrievable http(s) URL" in error for error in errors) From c23d7dc596ac23a6ee9fc7c16ea6108a23074920 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 00:40:53 -0700 Subject: [PATCH 30/52] Close the Codex round-9 validator bypasses One genuine defect in the round-8 hardening fixed: COMMAND_RE admitted embedded newlines (negated classes included \n and \$ matched before a trailing newline), letting a two-line payload pass as one command -- now newline-excluding with \Z anchoring, plus tests. Also: repeat claims above 2 require a recorded per-repeat hash list of that length (two verified repeats may rely on the identical-flag plus a trajectory digest -- recorded boundary); scene.parameters must carry at least one numeric/boolean value; raw-data artifacts are parsed structurally (JSON variants parse, CSV/TSV delimited, NumPy/parquet magic bytes -- deep tabular semantics are a recorded boundary). Tests: 116 -> 120. --- .../verification.md | 16 +++ scripts/check_citation_evidence.py | 104 +++++++++++++++++- tests/test_check_citation_evidence.py | 47 +++++++- 3 files changed, 165 insertions(+), 2 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 3e76696069d92..924d14a78b13e 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -612,3 +612,19 @@ npz/npy/parquet — prose or code files rejected); hash leaves must be actual hex digests, so "hash": "pending" cannot bind a repeat claim; and source URLs must be retrievable http(s) URLs. Tests: 110 -> 116 (main), 90 -> 96 (6.20). + +## Codex review round 9 (PR #3445) — 2026-08-16 + +Five findings (one reported on both PRs), all fixed on both branches. +One was a GENUINE defect in the round-8 hardening: COMMAND_RE's negated +character classes admitted embedded newlines and `$` matched before a +trailing newline, so a two-line payload passed as one command — fixed +with newline-excluding classes and \\Z anchoring, with tests. The rest: +repeat claims above 2 now require a recorded per-repeat hash list of +that length (two verified repeats may rely on the identical-flag plus a +trajectory digest — recorded boundary); scene.parameters must carry at +least one numeric/boolean value (metadata-only objects bind nothing); +raw-data artifacts are parsed structurally (JSON variants must parse, +CSV/TSV need delimited lines, NumPy/parquet magic bytes — deep semantic +validation of tabular contents is a recorded boundary). Tests: +116 -> 120 (main), 96 -> 100 (6.20). diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 27d837e2a50a4..52e95b10e2b28 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -71,7 +71,7 @@ ) COMMAND_RE = re.compile( - r"^(?:[A-Za-z_][A-Za-z0-9_]*=[^\s;|&`$]+\s+)*pixi run\s+[^;|&`$]+$" + r"^(?:[A-Za-z_][A-Za-z0-9_]*=[^\s;|&`$]+[ \t]+)*" r"pixi run[ \t]+[^;|&`$\n\r]+\Z" ) HASH_VALUE_RE = re.compile(r"^(sha256:)?[0-9a-f]{32,}$") RAW_DATA_SUFFIXES = frozenset( @@ -82,6 +82,28 @@ ) +def _has_hash_list(value: object, length: int) -> bool: + """True when a hash-named key holds a list of >= `length` digest values.""" + if isinstance(value, dict): + for key, item in value.items(): + key_l = str(key).lower() + if ( + ("sha256" in key_l or "hash" in key_l) + and isinstance(item, list) + and len(item) >= length + and all( + _is_nonempty_str(entry) and HASH_VALUE_RE.match(entry.strip()) + for entry in item + ) + ): + return True + if _has_hash_list(item, length): + return True + elif isinstance(value, list): + return any(_has_hash_list(item, length) for item in value) + return False + + def _has_hash_leaf(value: object) -> bool: """True when any nested key names a hash/sha256 with a non-empty value.""" if isinstance(value, dict): @@ -297,6 +319,53 @@ def _resolve_evidence_path(raw_path: str, base_dir: "Path | None") -> "Path | No return None +def _raw_data_content_issue(path: "Path") -> "str | None": + """Why a raw-data artifact's bytes do not match its claimed format. + + Mirrors the visual check: a prose file renamed to `rows.csv` must not + satisfy the raw-evidence requirement. JSON variants must parse; CSV/TSV + need delimited tabular lines; NumPy/parquet containers must carry their + magic bytes. Deep semantic validation of tabular contents is a recorded + boundary. + """ + suffix = path.suffix.lower() + try: + with path.open("rb") as stream: + head = stream.read(1 * 1024 * 1024) + except OSError: + return "could not be read for format verification" + mismatch = ( + "does not parse as its claimed raw-data format; a renamed prose " + "file is not raw evidence" + ) + if suffix == ".json": + try: + json.loads(path.read_text(encoding="utf-8")) + except OSError, ValueError: + return mismatch + return None + if suffix in (".jsonl", ".ndjson"): + try: + for line in path.read_text(encoding="utf-8").splitlines(): + if line.strip(): + json.loads(line) + except OSError, ValueError: + return mismatch + return None + if suffix in (".csv", ".tsv"): + delimiter = b"," if suffix == ".csv" else b"\t" + first_lines = head.splitlines()[:2] + if not first_lines or not any(delimiter in line for line in first_lines): + return mismatch + return None + if suffix == ".npy": + return None if head.startswith(b"\x93NUMPY") else mismatch + if suffix in (".npz", ".parquet"): + ok = head.startswith(b"PK") if suffix == ".npz" else head[:4] == b"PAR1" + return None if ok else mismatch + return None + + def _visual_content_issue(path: "Path") -> "str | None": """Why a resolved visual artifact's bytes do not match its claimed type. @@ -634,6 +703,16 @@ def packet_errors( if not (isinstance(digest, str) and SCENE_DIGEST_RE.match(digest)): errors.append("scene.digest must match sha256:<64 hex>") parameters = scene.get("parameters") + if isinstance(parameters, dict) and parameters: + if not any( + _is_finite_number(leaf) or isinstance(leaf, bool) + for _, leaf in _metric_leaves(parameters) + ): + errors.append( + "scene.parameters carries no numeric or boolean values; " + "metadata-only parameters do not describe a scene the " + "digest could bind" + ) if not (isinstance(parameters, dict) and parameters): errors.append( "scene.parameters must publish the non-empty parameter " @@ -706,6 +785,20 @@ def packet_errors( has_repeats = ( isinstance(repeats, int) and not isinstance(repeats, bool) and repeats >= 2 ) + if ( + has_repeats + and isinstance(repeats, int) + and repeats > 2 + and not _has_hash_list(packet.get("evidence"), repeats) + ): + errors.append( + f"ensemble.deterministic_repeats={repeats} (> 2) requires a " + "recorded per-repeat hash list of that length somewhere in " + "evidence; two verified repeats may rely on the " + "deterministic_repeats_identical flag plus a trajectory " + "digest, larger claims must show their repeats" + ) + has_repeats = False if has_repeats and not _has_hash_leaf(packet.get("evidence")): errors.append( "ensemble.deterministic_repeats is asserted but the " @@ -893,6 +986,15 @@ def packet_errors( continue if issue is not None: errors.append(f"evidence.raw_paths[{index}] {issue}") + continue + resolved = _resolve_evidence_path(raw_path, base_dir) + if resolved is not None: + content_issue = _raw_data_content_issue(resolved) + if content_issue is not None: + errors.append( + f"evidence.raw_paths[{index}] {raw_path!r} " + f"{content_issue}" + ) visual = evidence.get("visual") if isinstance(visual, dict): if visual.get("status") != "not-applicable" or not _is_nonempty_str( diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index 6aa3e01be8c54..ebdb2773d3a99 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -488,7 +488,7 @@ def test_dangling_raw_path_fails(tmp_path): packet["evidence"]["raw_paths"] = ["does/not/exist.csv"] errors = MODULE.packet_errors(packet, base_dir=tmp_path) assert any("does not resolve" in error for error in errors) - (tmp_path / "real.csv").write_text("x", encoding="utf-8") + (tmp_path / "real.csv").write_text("a,b\n1,2\n", encoding="utf-8") packet["evidence"]["raw_paths"] = ["real.csv"] # raw_paths carry no hash leaf, so the repeats claim needs one recorded # elsewhere; a sweep ensemble sidesteps that requirement here. @@ -1197,3 +1197,48 @@ def test_placeholder_source_urls_fail(): packet["source"]["url"] = "pending" errors = MODULE.packet_errors(packet) assert any("retrievable http(s) URL" in error for error in errors) + + +def test_commands_with_embedded_newlines_fail(): + for bad in ("pixi run check\nfalse", "pixi run check\necho hacked\n"): + packet = complete_packet() + packet["evidence"]["commands"] = [bad] + errors = MODULE.packet_errors(packet) + assert any("reproducible repository form" in error for error in errors), bad + + +def test_large_repeat_claims_need_per_repeat_hash_lists(): + packet = complete_packet() + packet["ensemble"]["deterministic_repeats"] = 1000 + errors = MODULE.packet_errors(packet) + assert any("must show their repeats" in error for error in errors) + packet["evidence"]["repeat_hashes_sha256"] = ["a" * 64] * 1000 + assert MODULE.packet_errors(packet) == [] + + +def test_metadata_only_scene_parameters_fail(): + packet = complete_packet() + params = {"note": "pending"} + packet["scene"]["parameters"] = params + packet["scene"]["digest"] = ( + "sha256:" + + hashlib.sha256( + json.dumps(params, sort_keys=True, separators=(",", ":")).encode() + ).hexdigest() + ) + errors = MODULE.packet_errors(packet) + assert any("metadata-only parameters" in error for error in errors) + + +def test_prose_inside_raw_data_files_fails(tmp_path): + (tmp_path / "rows.csv").write_text("totally unstructured prose", encoding="utf-8") + packet = complete_packet() + del packet["ensemble"]["deterministic_repeats"] + del packet["ensemble"]["deterministic_repeats_identical"] + packet["ensemble"]["sweep"] = [{"angle_deg": 0.0}, {"angle_deg": 15.0}] + del packet["evidence"]["raw_rows"] + packet["evidence"]["raw_paths"] = ["rows.csv"] + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any( + "does not parse as its claimed raw-data format" in error for error in errors + ) From 3fe6fbfbfcdf090a8f265d7758088f50957987b0 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 00:55:47 -0700 Subject: [PATCH 31/52] Close the Codex round-10 validator bypasses and fix the CT-011 instrument The substantive finding: CT-011's comparisons hashed World.state_vector -- the translational-only state whose incompleteness is the packet's own finding -- so bit-exact arms could not see rotational divergence (the ballistic control especially, where rotation never feeds translation). Every arm now hashes the FULL rigid-body state (pose including orientation, linear and angular velocity, per body in name order), the docstring and claim boundary say so, and the packet is regenerated at this commit in the follow-up commit under the exact-pattern guard. Also: the documented DART_CITATION_PR override emitted hints the validator rejects and is removed (ownership transfer updates the writer constant and FETCH_HINT_RE together); declared sweep/seed point counts must be matched by recorded raw rows; path-based evidence must pin artifact bytes via evidence.artifact_digests (sha256 verified against the file). Tests: 120 -> 122 validator cases. --- .../verification.md | 20 +++++++ scripts/check_citation_evidence.py | 54 +++++++++++++++++++ scripts/citation_packet_utils.py | 14 ++--- ...tation_ct011_restore_equivalence_packet.py | 36 +++++++++++-- tests/test_check_citation_evidence.py | 40 +++++++++++++- 5 files changed, 151 insertions(+), 13 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 924d14a78b13e..3b18ece86b776 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -628,3 +628,23 @@ raw-data artifacts are parsed structurally (JSON variants must parse, CSV/TSV need delimited lines, NumPy/parquet magic bytes — deep semantic validation of tabular contents is a recorded boundary). Tests: 116 -> 120 (main), 96 -> 100 (6.20). + +## Codex review round 10 (PR #3445) — 2026-08-16 + +Four findings, two genuine. The substantive one: CT-011's comparison +instrument hashed `World.state_vector` — the translational-only state +whose incompleteness is the packet's own finding — so "bit-exact" arms +could not see rotational divergence (the ballistic control especially, +where rotation never feeds translation). The writer now hashes the FULL +rigid-body state (pose including orientation, linear and angular +velocity, per body in name order) in every arm, the boundary says so, +and the packet is regenerated under the stronger instrument (the +exact-pattern guard aborts if any arm outcome changed). Second genuine +defect: the documented DART_CITATION_PR override emitted hints the +branch validator rejects — removed; ownership transfer updates the +writer constant and FETCH_HINT_RE together. Also: declared sweep/seed +point counts must be matched by recorded raw rows, and path-based +evidence must pin artifact bytes via evidence.artifact_digests +(sha256 verified against the file), so swapping a referenced artifact +invalidates the packet and its digest-bound reviews. Tests: 120 -> 122 +(main), 100 -> 102 (6.20). diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 52e95b10e2b28..b86e77eb77d8e 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -773,6 +773,7 @@ def packet_errors( if not _is_nonempty_str(configuration.get("fallback_policy")): errors.append("configuration.fallback_policy must be a non-empty string") + declared_points = 0 ensemble = packet.get("ensemble") if not isinstance(ensemble, dict): errors.append("ensemble must be an object") @@ -836,6 +837,8 @@ def packet_errors( "ensemble.sweep must contain at least two DISTINCT points" ) has_sweep = False + if has_sweep: + declared_points = max(declared_points, len(canonical_points)) has_seeds = isinstance(seeds, list) and len(seeds) >= 2 if has_seeds: for index, entry in enumerate(seeds): @@ -851,6 +854,8 @@ def packet_errors( if has_seeds and len({repr(entry) for entry in seeds}) < 2: errors.append("ensemble.seeds must contain at least two DISTINCT seeds") has_seeds = False + if has_seeds: + declared_points = max(declared_points, len(seeds)) if not (has_repeats or has_sweep or has_seeds): errors.append( "ensemble must record deterministic_repeats >= 2, a sweep of " @@ -943,6 +948,12 @@ def packet_errors( raw_rows = evidence.get("raw_rows") has_paths = isinstance(raw_paths, list) and bool(raw_paths) has_rows = isinstance(raw_rows, list) and bool(raw_rows) + if has_rows and declared_points > len(raw_rows): + errors.append( + f"ensemble declares {declared_points} sweep/seed points but " + f"evidence.raw_rows records only {len(raw_rows)} rows; every " + "declared point needs at least one recorded sample" + ) if not (has_paths or has_rows): errors.append("evidence must carry raw_rows inline or non-empty raw_paths") if has_rows: @@ -995,6 +1006,11 @@ def packet_errors( f"evidence.raw_paths[{index}] {raw_path!r} " f"{content_issue}" ) + referenced_paths: list[str] = [] + if has_paths: + referenced_paths.extend( + raw_path for raw_path in raw_paths if _is_nonempty_str(raw_path) + ) visual = evidence.get("visual") if isinstance(visual, dict): if visual.get("status") != "not-applicable" or not _is_nonempty_str( @@ -1046,6 +1062,44 @@ def packet_errors( "evidence.visual must list visual artifacts or be typed " "not-applicable with a reason" ) + if isinstance(visual, list): + for item in visual: + if _is_nonempty_str(item): + referenced_paths.append(item) + elif isinstance(item, dict) and _is_nonempty_str(item.get("path")): + referenced_paths.append(item["path"]) + if referenced_paths: + digests = evidence.get("artifact_digests") + if not isinstance(digests, dict): + errors.append( + "evidence.artifact_digests must map every referenced " + "artifact path to its sha256; without it, review " + "digests do not bind the referenced bytes" + ) + else: + for ref in referenced_paths: + recorded = digests.get(ref) + if not ( + isinstance(recorded, str) and SCENE_DIGEST_RE.match(recorded) + ): + errors.append( + f"evidence.artifact_digests[{ref!r}] must record " + "sha256:<64 hex> for the referenced artifact" + ) + continue + resolved = _resolve_evidence_path(ref, base_dir) + if resolved is not None: + actual = ( + "sha256:" + + hashlib.sha256(resolved.read_bytes()).hexdigest() + ) + if actual != recorded: + errors.append( + f"evidence.artifact_digests[{ref!r}] does " + "not match the referenced file's bytes; a " + "swapped artifact invalidates the packet " + "and its reviews" + ) result = packet.get("result") if not isinstance(result, dict): diff --git a/scripts/citation_packet_utils.py b/scripts/citation_packet_utils.py index 3383cad897866..bd99631aea047 100644 --- a/scripts/citation_packet_utils.py +++ b/scripts/citation_packet_utils.py @@ -122,14 +122,14 @@ def target_fetch_hint(commit: str) -> str: """The runnable command reaching `commit` from a clean checkout forever. CITATION_PR_NUMBER is the PR that owns this evidence tree revision. When - a later PR takes ownership, update the constant together with the - validator's FETCH_HINT_RE; for forward work from another PR, override - with the DART_CITATION_PR environment variable. + a later PR takes ownership, update this constant TOGETHER WITH the + validator's FETCH_HINT_RE in the same change; there is deliberately no + env override, because a hint the branch validator rejects would only + manufacture invalid packets. """ - import os as _os - - pr = _os.environ.get("DART_CITATION_PR", str(CITATION_PR_NUMBER)) - return f"git fetch origin pull/{pr}/head && git checkout {commit}" + return ( + f"git fetch origin pull/{CITATION_PR_NUMBER}/head && " f"git checkout {commit}" + ) def packet_content_digest(packet: dict[str, Any]) -> str: diff --git a/scripts/write_citation_ct011_restore_equivalence_packet.py b/scripts/write_citation_ct011_restore_equivalence_packet.py index 184412b5c19f6..72f2a353968cd 100644 --- a/scripts/write_citation_ct011_restore_equivalence_packet.py +++ b/scripts/write_citation_ct011_restore_equivalence_packet.py @@ -9,7 +9,10 @@ The fixture is a five-sphere pile settling on a ground box, run per contact solver, with a ballistic (no-ground) control scene. Arms, each hashed over -the full state vector for every post-restore step: +the FULL rigid-body state (pose including orientation, linear and angular +velocity, per body) for every post-restore step -- hashing the translational +`state_vector` itself would blind the comparison to exactly the rotational +divergence this packet documents: - `continuation`: keep stepping from the snapshot point (baseline); - `inplace_restore` x2: restore the snapshot into the same world twice, with @@ -132,11 +135,31 @@ def build_world( return world +def full_state_array(world: Any) -> np.ndarray: + """Full rigid-body state: pose (4x4) + linear + angular velocity per + dynamic body, bodies in name order. + + `World.state_vector` is translational by design (the very CT-011 + finding), so hashing it would blind these comparisons to rotational + divergence -- especially in the ballistic control, where rotation never + feeds translation. This observable closes that hole. + """ + parts: list[np.ndarray] = [] + for name in sorted(world.get_rigid_body_names()): + body = world.get_rigid_body(name) + if body.is_static: + continue + parts.append(np.asarray(body.transform, dtype=float).reshape(-1)) + parts.append(np.asarray(body.linear_velocity, dtype=float)) + parts.append(np.asarray(body.angular_velocity, dtype=float)) + return np.concatenate(parts) + + def hash_continuation(world: Any, steps: int) -> str: digest = hashlib.sha256() for _ in range(steps): world.step() - digest.update(np.ascontiguousarray(world.state_vector).tobytes()) + digest.update(np.ascontiguousarray(full_state_array(world)).tobytes()) return digest.hexdigest() @@ -147,7 +170,7 @@ def divergence_profile( max_delta = 0.0 for index in range(steps): world.step() - delta = float(np.max(np.abs(np.asarray(world.state_vector) - reference[index]))) + delta = float(np.max(np.abs(full_state_array(world) - reference[index]))) if delta > 0.0 and first is None: first = index max_delta = max(max_delta, delta) @@ -191,7 +214,7 @@ def run_protocol(method_name: str, parameters: dict[str, Any]) -> dict[str, Any] reference_digest = hashlib.sha256() for _ in range(steps): world.step() - state = np.array(world.state_vector, copy=True) + state = full_state_array(world) reference_states.append(state) reference_digest.update(np.ascontiguousarray(state).tobytes()) continuation_hash = reference_digest.hexdigest() @@ -538,7 +561,10 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: "claim_boundary": ( "DART 7 main, this commit, a five-sphere pile with a 0.5 s " "contact-rich warm-up, restored via World.state_vector, " - "both contact solvers. Established: restore is NOT a " + "both contact solvers, all arms compared over the FULL " + "rigid-body state (pose including orientation, linear and " + "angular velocity) so rotational divergence is visible even " + "where it does not feed translation. Established: restore is NOT a " "function of the restored state once the world has contact " "history -- the in-place continuation differs from the " "original at the first step, and two in-place restores of " diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index ebdb2773d3a99..fa8b87018d9aa 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -491,10 +491,15 @@ def test_dangling_raw_path_fails(tmp_path): (tmp_path / "real.csv").write_text("a,b\n1,2\n", encoding="utf-8") packet["evidence"]["raw_paths"] = ["real.csv"] # raw_paths carry no hash leaf, so the repeats claim needs one recorded - # elsewhere; a sweep ensemble sidesteps that requirement here. + # elsewhere; a sweep ensemble sidesteps that requirement here, and + # path-based evidence must pin its artifact bytes. del packet["ensemble"]["deterministic_repeats"] del packet["ensemble"]["deterministic_repeats_identical"] packet["ensemble"]["sweep"] = [{"angle_deg": 0.0}, {"angle_deg": 15.0}] + packet["evidence"]["artifact_digests"] = { + "real.csv": "sha256:" + + hashlib.sha256((tmp_path / "real.csv").read_bytes()).hexdigest() + } assert MODULE.packet_errors(packet, base_dir=tmp_path) == [] @@ -734,6 +739,10 @@ def test_ensemble_sweep_and_seed_entries_must_be_valid_and_distinct(): assert any(needle in error for error in errors), (bad, errors) good = copy.deepcopy(base) good["ensemble"]["sweep"] = [{"angle_deg": 0.0}, {"angle_deg": 15.0}] + good["evidence"]["raw_rows"] = [ + {"angle_deg": 0.0, "lateral_drift_m": 0.0, "trajectory_sha256": "d" * 64}, + {"angle_deg": 15.0, "lateral_drift_m": 0.1, "trajectory_sha256": "e" * 64}, + ] assert MODULE.packet_errors(good) == [] @@ -965,6 +974,9 @@ def test_visual_artifacts_are_verified_by_content(tmp_path): ) (fake / "real.png").write_bytes(png) packet["evidence"]["visual"] = ["real.png"] + packet["evidence"]["artifact_digests"] = { + "real.png": "sha256:" + hashlib.sha256(png).hexdigest() + } assert MODULE.packet_errors(packet, base_dir=fake) == [] @@ -1242,3 +1254,29 @@ def test_prose_inside_raw_data_files_fails(tmp_path): assert any( "does not parse as its claimed raw-data format" in error for error in errors ) + + +def test_declared_sweep_points_need_matching_rows(): + packet = complete_packet() + del packet["ensemble"]["deterministic_repeats"] + del packet["ensemble"]["deterministic_repeats_identical"] + packet["ensemble"]["sweep"] = [{"angle_deg": 0.0}, {"angle_deg": 15.0}] + errors = MODULE.packet_errors(packet) + assert any( + "every declared point needs at least one recorded sample" in e for e in errors + ) + + +def test_swapped_artifacts_invalidate_the_packet(tmp_path): + (tmp_path / "real.csv").write_text("a,b\n1,2\n", encoding="utf-8") + packet = complete_packet() + del packet["ensemble"]["deterministic_repeats"] + del packet["ensemble"]["deterministic_repeats_identical"] + packet["ensemble"]["sweep"] = [{"angle_deg": 0.0}, {"angle_deg": 15.0}] + del packet["evidence"]["raw_rows"] + packet["evidence"]["raw_paths"] = ["real.csv"] + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any("artifact_digests must map" in e for e in errors) + packet["evidence"]["artifact_digests"] = {"real.csv": "sha256:" + "0" * 64} + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any("does not match the referenced file's bytes" in e for e in errors) From c7cdb59b0ec803bc5f0b06bb19aa947f059b7162 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 00:56:52 -0700 Subject: [PATCH 32/52] Regenerate CT-011 under the full-state instrument Every arm's outcome is unchanged with rotational state now visible: in-place restores diverge, fresh/same-history/pre-contact arms match, and the ballistic control is exact over the FULL rigid-body state -- free-flight spheres develop no spin, so its exactness is now a real full-state result rather than a translational artifact. The exact-pattern guard passed; disposition unchanged (unresolved). --- .../CT-011-dart7-restore-equivalence.json | 54 +++++++++---------- 1 file changed, 27 insertions(+), 27 deletions(-) diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json index a12b014c654ba..a708f3886ada1 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json @@ -114,11 +114,11 @@ "raw_rows": [ { "ballistic_control_sha256": [ - "faf3d3ac5fd6ed8337fa1ea5c6061b576ab938c2be4200f8783ae1b7511142c9", - "faf3d3ac5fd6ed8337fa1ea5c6061b576ab938c2be4200f8783ae1b7511142c9" + "16a1ec7d64ac5045e2eb17b75ae0515285dc9ef6f71e35a5e51700f68dc902fb", + "16a1ec7d64ac5045e2eb17b75ae0515285dc9ef6f71e35a5e51700f68dc902fb" ], "contact_solver_method": "SEQUENTIAL_IMPULSE", - "continuation_sha256": "0e4b9d59c31f6fa79978a4a1c4893e5adff0a010866c6a58b6e72db0f0b7aa61", + "continuation_sha256": "26aeeada76d96c05e831b49bc015344dff3f89696f31b414673063165444d506", "findings": { "ballistic_restore_exact": true, "fresh_restore_matches_continuation": false, @@ -129,18 +129,18 @@ "same_history_restores_match": true }, "fresh_restore_sha256": [ - "15cf3d42f26b04b3cb1242c75c276a113c35344f9463c7dd2349c1a34667f70b", - "15cf3d42f26b04b3cb1242c75c276a113c35344f9463c7dd2349c1a34667f70b" + "f6d8ffae48a9d9f8a60521d53d667efec3f8b1bb6c810b334fcd68c6b25fcdfa", + "f6d8ffae48a9d9f8a60521d53d667efec3f8b1bb6c810b334fcd68c6b25fcdfa" ], "inplace_divergence_vs_continuation": { "first_divergent_step": 0, - "max_abs_state_delta": 0.004756158618442807 + "max_abs_state_delta": 1.8014933537668798 }, "inplace_restore_sha256": [ - "5879f6fde1ccb32093a641e2720b45a147580a93c1079a981c19888a52263d21", - "cc166727c022a4f0cd763d38351ea19f1177c9906c5e5ace5555abfd02b9d22d" + "6ee86bd7cfb895d5ec008e753be0d5208aa84889b6adef165c002605da8f3149", + "43afd72ea859ba81fddeb163788ada6b3d11cbe162244a15f93d4c8374b1695c" ], - "precontact_history_restore_sha256": "15cf3d42f26b04b3cb1242c75c276a113c35344f9463c7dd2349c1a34667f70b", + "precontact_history_restore_sha256": "f6d8ffae48a9d9f8a60521d53d667efec3f8b1bb6c810b334fcd68c6b25fcdfa", "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", "time_step_s": 0.002, @@ -176,17 +176,17 @@ } }, "same_history_restore_sha256": [ - "0e4b9d59c31f6fa79978a4a1c4893e5adff0a010866c6a58b6e72db0f0b7aa61", - "0e4b9d59c31f6fa79978a4a1c4893e5adff0a010866c6a58b6e72db0f0b7aa61" + "26aeeada76d96c05e831b49bc015344dff3f89696f31b414673063165444d506", + "26aeeada76d96c05e831b49bc015344dff3f89696f31b414673063165444d506" ] }, { "ballistic_control_sha256": [ - "faf3d3ac5fd6ed8337fa1ea5c6061b576ab938c2be4200f8783ae1b7511142c9", - "faf3d3ac5fd6ed8337fa1ea5c6061b576ab938c2be4200f8783ae1b7511142c9" + "16a1ec7d64ac5045e2eb17b75ae0515285dc9ef6f71e35a5e51700f68dc902fb", + "16a1ec7d64ac5045e2eb17b75ae0515285dc9ef6f71e35a5e51700f68dc902fb" ], "contact_solver_method": "BOXED_LCP", - "continuation_sha256": "3c71899bd799aa0e56388fa7e85ec47549b4a707fda1f03a2809e168fcfb7ac6", + "continuation_sha256": "538fd382678f25ae2fb3b8d841470797336ed283c69bea3c793daa6aa1b6f34d", "findings": { "ballistic_restore_exact": true, "fresh_restore_matches_continuation": false, @@ -197,18 +197,18 @@ "same_history_restores_match": true }, "fresh_restore_sha256": [ - "0614045a6cd87e6d01c45366defb286360d98691664faf8368530329a7167fbe", - "0614045a6cd87e6d01c45366defb286360d98691664faf8368530329a7167fbe" + "e86200daaecf1f7d3a621ca6e438f92cb8ebfef0f366b3a5da46e2eb7ad57bac", + "e86200daaecf1f7d3a621ca6e438f92cb8ebfef0f366b3a5da46e2eb7ad57bac" ], "inplace_divergence_vs_continuation": { "first_divergent_step": 0, - "max_abs_state_delta": 0.012694905840379356 + "max_abs_state_delta": 1.7561984148535297 }, "inplace_restore_sha256": [ - "452b08b7a2ba4943f9bc475849b3f9f37d4c62a5fbb10f8e1aa78b0bd198a3e1", - "a1dd32085b5ae5b6b25b48cc30f01b7402160c9be2e44fabf82670d450f04ede" + "59eb48d602ffd56eebd03a499414ff0d1db3d1c8f10a50cbd22c9c5f9c01e91e", + "919bb27b12ae8bfe0e5ca6c01cfdd71b0a0df73f12ce786e64d8eb22824e9d9f" ], - "precontact_history_restore_sha256": "0614045a6cd87e6d01c45366defb286360d98691664faf8368530329a7167fbe", + "precontact_history_restore_sha256": "e86200daaecf1f7d3a621ca6e438f92cb8ebfef0f366b3a5da46e2eb7ad57bac", "resolved": { "contact_solver_method": "BOXED_LCP", "time_step_s": 0.002, @@ -244,8 +244,8 @@ } }, "same_history_restore_sha256": [ - "3c71899bd799aa0e56388fa7e85ec47549b4a707fda1f03a2809e168fcfb7ac6", - "3c71899bd799aa0e56388fa7e85ec47549b4a707fda1f03a2809e168fcfb7ac6" + "538fd382678f25ae2fb3b8d841470797336ed283c69bea3c793daa6aa1b6f34d", + "538fd382678f25ae2fb3b8d841470797336ed283c69bea3c793daa6aa1b6f34d" ] } ], @@ -282,11 +282,11 @@ "inplace_divergence": { "BOXED_LCP": { "first_divergent_step": 0, - "max_abs_state_delta": 0.012694905840379356 + "max_abs_state_delta": 1.7561984148535297 }, "SEQUENTIAL_IMPULSE": { "first_divergent_step": 0, - "max_abs_state_delta": 0.004756158618442807 + "max_abs_state_delta": 1.8014933537668798 } }, "measured_zero_fields": [ @@ -318,7 +318,7 @@ } }, "result": { - "claim_boundary": "DART 7 main, this commit, a five-sphere pile with a 0.5 s contact-rich warm-up, restored via World.state_vector, both contact solvers. Established: restore is NOT a function of the restored state once the world has contact history -- the in-place continuation differs from the original at the first step, and two in-place restores of the same snapshot differ from each other when different history precedes them -- while a freshly built world restoring the same vector is bit-exact and repeatable, identical histories give identical continuations, pre-contact history is harmless, and the ballistic control is exact. Everything is deterministic given full history; nothing here is nondeterminism. Root cause: the state vector is translational by design and silently omits orientation and angular velocity, so the restore is partial. The corpus row's reset cost, overhead, and concurrency halves are not measured. Says nothing about Skeleton-based state APIs, clone(), replay restore, historical DART versions, or DART 6.", + "claim_boundary": "DART 7 main, this commit, a five-sphere pile with a 0.5 s contact-rich warm-up, restored via World.state_vector, both contact solvers, all arms compared over the FULL rigid-body state (pose including orientation, linear and angular velocity) so rotational divergence is visible even where it does not feed translation. Established: restore is NOT a function of the restored state once the world has contact history -- the in-place continuation differs from the original at the first step, and two in-place restores of the same snapshot differ from each other when different history precedes them -- while a freshly built world restoring the same vector is bit-exact and repeatable, identical histories give identical continuations, pre-contact history is harmless, and the ballistic control is exact. Everything is deterministic given full history; nothing here is nondeterminism. Root cause: the state vector is translational by design and silently omits orientation and angular velocity, so the restore is partial. The corpus row's reset cost, overhead, and concurrency halves are not measured. Says nothing about Skeleton-based state APIs, clone(), replay restore, historical DART versions, or DART 6.", "disposition": "unresolved", "limitations": [ "The mechanism is identified: World.state_vector is a translational state by design (per rigid body it carries position and linear velocity only -- 30 entries for the five-sphere scene -- as documented for the differentiable rigid-body path), so orientation and angular velocity are never captured or restored. A restored world keeps whatever rotational state it already had, which is exactly the history dependence measured here: after the warm-up the spheres carry finite spin (about 7 rad/s on s0) and rolled orientations, a fresh world has identity orientations and zero spin, and friction couples that rotational state into the continuation. Contact history matters only because contact is what makes spheres spin.", @@ -372,9 +372,9 @@ }, "target": { "branch": "main", - "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", + "commit": "3fe6fbfbfcdf090a8f265d7758088f50957987b0", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", - "fetch_hint": "git fetch origin pull/3445/head && git checkout f4e9e5c11d5d3b443693cde9142bf317e2115cd3" + "fetch_hint": "git fetch origin pull/3445/head && git checkout 3fe6fbfbfcdf090a8f265d7758088f50957987b0" }, "title": "State-vector restore equivalence under contact (DART 7 first packet)" } From dd68714d6a7a928b3af784e08369469b0e945216 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 01:12:43 -0700 Subject: [PATCH 33/52] Close the Codex round-11 bypasses and generalize the instrument fix Two genuine defects: writers could attribute a dirty tree's results to HEAD (git_head() now aborts on tracked modifications, excluding the evidence outputs they themselves write), and CT-007's finiteness and repeat hashes inspected translation only. The CT-011 lesson is generalized: the instrument sweep found the same rotational blindness in CT-002/003 and a NaN-blindness in CT-001 (NaN comparisons are False, so its analytic-speed validity gates would silently pass a NaN trajectory) -- every writer now checks and hashes the FULL rigid-body state; CT-004/005 operate in joint space, already complete for their scenes. Hardening: declared sweep points must be observed by rows carrying their exact coordinates; declared seeds must appear in rows under a seed field; JSON raw artifacts must parse to at least one real measurement leaf. Tests: 122 -> 125 validator cases. Affected packets (CT-001/002/003/007) are regenerated at this commit in the follow-up commit. --- .../verification.md | 19 +++++ scripts/check_citation_evidence.py | 69 +++++++++++++++++-- ...citation_ct001_rolling_direction_packet.py | 42 +++++++++++ ...ite_citation_ct002_dense_contact_packet.py | 40 ++++++++++- ...e_citation_ct003_elastic_contact_packet.py | 41 ++++++++++- ...itation_ct004_articulated_energy_packet.py | 29 ++++++++ ...write_citation_ct005_pd_tracking_packet.py | 29 ++++++++ ...e_citation_ct007_high_mass_ratio_packet.py | 46 ++++++++++++- ...tation_ct011_restore_equivalence_packet.py | 29 ++++++++ tests/test_check_citation_evidence.py | 44 ++++++++++++ 10 files changed, 376 insertions(+), 12 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 3b18ece86b776..f7555c20eee57 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -648,3 +648,22 @@ evidence must pin artifact bytes via evidence.artifact_digests (sha256 verified against the file), so swapping a referenced artifact invalidates the packet and its digest-bound reviews. Tests: 120 -> 122 (main), 100 -> 102 (6.20). + +## Codex review round 11 (PR #3445) — 2026-08-16 + +Five findings, two genuine, all fixed on both branches. Genuine #1: +writers could attribute a dirty tree's results to HEAD — git_head() now +aborts on tracked modifications (evidence outputs excluded, since they +are the writers' own products). Genuine #2: CT-007's finiteness and +repeat hashes inspected translation only — the CT-011 lesson +generalized; the instrument sweep found the same blindness in +CT-002/003 (final-state checks and hashes) and a NaN-blindness in both +branches' CT-001 (NaN comparisons are False, so the analytic-speed +validity gates would silently PASS a NaN trajectory) — every writer now +checks and hashes the FULL rigid-body state (CT-004/005 operate in +joint space, which is already complete for their scenes). Hardening: +declared sweep points must be OBSERVED by rows carrying their exact +coordinates (count parity is not enough), declared seeds must appear in +rows under a seed field, and JSON raw artifacts must parse to content +with at least one real measurement leaf. Affected packets regenerated +at the round-11 commit. Tests: 122 -> 125 (main), 102 -> 105 (6.20). diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index b86e77eb77d8e..2b627340fc69c 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -82,6 +82,18 @@ ) +def _has_measurement_leaf(value: object) -> bool: + """True when the value contains a numeric/boolean leaf beyond + bookkeeping keys (the raw-rows measurement rule, applied to parsed + artifacts).""" + return any( + (_is_finite_number(leaf) or isinstance(leaf, bool)) + and re.sub(r"(\[\d+\])+$", "", path.rsplit(".", 1)[-1]) + not in NUMERIC_BOOKKEEPING_KEYS + for path, leaf in _metric_leaves(value) + ) + + def _has_hash_list(value: object, length: int) -> bool: """True when a hash-named key holds a list of >= `length` digest values.""" if isinstance(value, dict): @@ -338,19 +350,29 @@ def _raw_data_content_issue(path: "Path") -> "str | None": "does not parse as its claimed raw-data format; a renamed prose " "file is not raw evidence" ) + no_content = ( + "parses but carries no numeric or boolean measurement content; a " + "structurally empty artifact is not raw evidence" + ) if suffix == ".json": try: - json.loads(path.read_text(encoding="utf-8")) + parsed = json.loads(path.read_text(encoding="utf-8")) except OSError, ValueError: return mismatch + if not _has_measurement_leaf(parsed): + return no_content return None if suffix in (".jsonl", ".ndjson"): try: - for line in path.read_text(encoding="utf-8").splitlines(): - if line.strip(): - json.loads(line) + parsed_lines = [ + json.loads(line) + for line in path.read_text(encoding="utf-8").splitlines() + if line.strip() + ] except OSError, ValueError: return mismatch + if not parsed_lines or not _has_measurement_leaf(parsed_lines): + return no_content return None if suffix in (".csv", ".tsv"): delimiter = b"," if suffix == ".csv" else b"\t" @@ -774,6 +796,8 @@ def packet_errors( errors.append("configuration.fallback_policy must be a non-empty string") declared_points = 0 + sweep = None + seeds = None ensemble = packet.get("ensemble") if not isinstance(ensemble, dict): errors.append("ensemble must be an object") @@ -954,6 +978,43 @@ def packet_errors( f"evidence.raw_rows records only {len(raw_rows)} rows; every " "declared point needs at least one recorded sample" ) + # Count parity is not enough: each declared object point must be + # OBSERVED by a row carrying its exact coordinates, or one point's + # measurements could be repeated to stand in for the others. + if has_rows and isinstance(sweep, list): + for point in sweep: + if not (isinstance(point, dict) and point): + continue + if not any( + isinstance(row, dict) + and all(row.get(key) == value for key, value in point.items()) + for row in raw_rows + ): + errors.append( + f"ensemble.sweep point {point} has no matching row " + "in evidence.raw_rows recording those coordinates; " + "a declared configuration without an observation is " + "not swept" + ) + if has_rows and isinstance(seeds, list): + for seed in seeds: + if not ( + (isinstance(seed, int) and not isinstance(seed, bool)) + or _is_nonempty_str(seed) + ): + continue + if not any( + isinstance(row, dict) + and any( + "seed" in str(key).lower() and row[key] == seed for key in row + ) + for row in raw_rows + ): + errors.append( + f"ensemble seed {seed!r} has no row recording it " + "under a seed field; a declared seed without an " + "observation is not an ensemble member" + ) if not (has_paths or has_rows): errors.append("evidence must carry raw_rows inline or non-empty raw_paths") if has_rows: diff --git a/scripts/write_citation_ct001_rolling_direction_packet.py b/scripts/write_citation_ct001_rolling_direction_packet.py index 715f541b93d69..37b1292ff3639 100644 --- a/scripts/write_citation_ct001_rolling_direction_packet.py +++ b/scripts/write_citation_ct001_rolling_direction_packet.py @@ -163,11 +163,22 @@ def run_single( stage_names: list[str] = [] lateral_axis = np.array([-direction[1], direction[0], 0.0]) + finite = True for step_index in range(step_count): world.step() position = np.asarray(sphere.translation, dtype=float) velocity = np.asarray(sphere.linear_velocity, dtype=float) angular = np.asarray(sphere.angular_velocity, dtype=float) + # NaN comparisons are False, so the analytic-speed validity gates + # would silently PASS a NaN trajectory; track finiteness explicitly + # over the full observed state. + if not ( + np.all(np.isfinite(position)) + and np.all(np.isfinite(velocity)) + and np.all(np.isfinite(angular)) + and np.all(np.isfinite(np.asarray(sphere.transform, dtype=float))) + ): + finite = False trajectory.update(position.tobytes()) trajectory.update(velocity.tobytes()) trajectory.update(angular.tobytes()) @@ -227,6 +238,7 @@ def run_single( "angle_deg": angle_deg, "contact_solver_method": method_name, "resolved": resolved, + "finite": finite, "trajectory_sha256": trajectory.hexdigest(), "along_travel_m": along, "lateral_drift_m": lateral, @@ -292,6 +304,35 @@ def antisymmetry_residual(rows: list[dict[str, Any]]) -> float: def git_head() -> str: + """HEAD commit, refusing to attribute a dirty tree's results to it. + + A packet's target.commit claims the recorded results come from that + commit's code; uncommitted modifications to tracked files would make + that attribution false, so generation aborts instead. (The packet + output file itself is checked before being overwritten, so an + unmodified existing packet does not block regeneration.) + """ + dirty = subprocess.run( + [ + "git", + "status", + "--porcelain", + "--untracked-files=no", + "--", + ".", + ":(exclude)docs/plans/123-citation-driven-simulation-trust/evidence", + ":(exclude)docs/design/dart6_citation_driven_contact_trust/evidence", + ], + cwd=REPO_ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + if dirty: + raise SystemExit( + "refusing to generate evidence from a dirty tree; commit or " + "stash these tracked modifications first:\n" + dirty + ) return subprocess.run( ["git", "rev-parse", "HEAD"], cwd=REPO_ROOT, @@ -397,6 +438,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: f"{row['contact_solver_method']} angle {row['angle_deg']}: {reason}" for row in rows for reason, bad in ( + ("non-finite state", not row["finite"]), ( "final speed departs from the analytic rolling speed 5/7 v0", abs( diff --git a/scripts/write_citation_ct002_dense_contact_packet.py b/scripts/write_citation_ct002_dense_contact_packet.py index 0115cb12e03ed..85872fb460105 100644 --- a/scripts/write_citation_ct002_dense_contact_packet.py +++ b/scripts/write_citation_ct002_dense_contact_packet.py @@ -203,13 +203,20 @@ def run_single( for body in bodies: velocity = np.asarray(body.linear_velocity, dtype=float) position = np.asarray(body.translation, dtype=float) - if not (np.all(np.isfinite(velocity)) and np.all(np.isfinite(position))): + transform = np.asarray(body.transform, dtype=float) + angular = np.asarray(body.angular_velocity, dtype=float) + if not ( + np.all(np.isfinite(velocity)) + and np.all(np.isfinite(transform)) + and np.all(np.isfinite(angular)) + ): non_finite = True break speeds.append(float(np.linalg.norm(velocity))) min_height = min(min_height, float(position[2])) - state_hash.update(position.tobytes()) + state_hash.update(transform.tobytes()) state_hash.update(velocity.tobytes()) + state_hash.update(angular.tobytes()) final_metrics = world.compute_step_metrics() return { @@ -694,6 +701,35 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: def git_head() -> str: + """HEAD commit, refusing to attribute a dirty tree's results to it. + + A packet's target.commit claims the recorded results come from that + commit's code; uncommitted modifications to tracked files would make + that attribution false, so generation aborts instead. (The packet + output file itself is checked before being overwritten, so an + unmodified existing packet does not block regeneration.) + """ + dirty = subprocess.run( + [ + "git", + "status", + "--porcelain", + "--untracked-files=no", + "--", + ".", + ":(exclude)docs/plans/123-citation-driven-simulation-trust/evidence", + ":(exclude)docs/design/dart6_citation_driven_contact_trust/evidence", + ], + cwd=REPO_ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + if dirty: + raise SystemExit( + "refusing to generate evidence from a dirty tree; commit or " + "stash these tracked modifications first:\n" + dirty + ) return subprocess.run( ["git", "rev-parse", "HEAD"], cwd=REPO_ROOT, diff --git a/scripts/write_citation_ct003_elastic_contact_packet.py b/scripts/write_citation_ct003_elastic_contact_packet.py index 4fb34d8a9a3c8..8aa7a8cfa1b14 100644 --- a/scripts/write_citation_ct003_elastic_contact_packet.py +++ b/scripts/write_citation_ct003_elastic_contact_packet.py @@ -192,12 +192,18 @@ def run_single( if not non_finite: for body in bodies: velocity = np.asarray(body.linear_velocity, dtype=float) - position = np.asarray(body.translation, dtype=float) - if not (np.all(np.isfinite(velocity)) and np.all(np.isfinite(position))): + transform = np.asarray(body.transform, dtype=float) + angular = np.asarray(body.angular_velocity, dtype=float) + if not ( + np.all(np.isfinite(velocity)) + and np.all(np.isfinite(transform)) + and np.all(np.isfinite(angular)) + ): non_finite = True break - state_hash.update(position.tobytes()) + state_hash.update(transform.tobytes()) state_hash.update(velocity.tobytes()) + state_hash.update(angular.tobytes()) final_metrics = world.compute_step_metrics() return { @@ -226,6 +232,35 @@ def run_single( def git_head() -> str: + """HEAD commit, refusing to attribute a dirty tree's results to it. + + A packet's target.commit claims the recorded results come from that + commit's code; uncommitted modifications to tracked files would make + that attribution false, so generation aborts instead. (The packet + output file itself is checked before being overwritten, so an + unmodified existing packet does not block regeneration.) + """ + dirty = subprocess.run( + [ + "git", + "status", + "--porcelain", + "--untracked-files=no", + "--", + ".", + ":(exclude)docs/plans/123-citation-driven-simulation-trust/evidence", + ":(exclude)docs/design/dart6_citation_driven_contact_trust/evidence", + ], + cwd=REPO_ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + if dirty: + raise SystemExit( + "refusing to generate evidence from a dirty tree; commit or " + "stash these tracked modifications first:\n" + dirty + ) return subprocess.run( ["git", "rev-parse", "HEAD"], cwd=REPO_ROOT, diff --git a/scripts/write_citation_ct004_articulated_energy_packet.py b/scripts/write_citation_ct004_articulated_energy_packet.py index ba5381f4bc8be..0b87c89780ca3 100644 --- a/scripts/write_citation_ct004_articulated_energy_packet.py +++ b/scripts/write_citation_ct004_articulated_energy_packet.py @@ -189,6 +189,35 @@ def run_single( def git_head() -> str: + """HEAD commit, refusing to attribute a dirty tree's results to it. + + A packet's target.commit claims the recorded results come from that + commit's code; uncommitted modifications to tracked files would make + that attribution false, so generation aborts instead. (The packet + output file itself is checked before being overwritten, so an + unmodified existing packet does not block regeneration.) + """ + dirty = subprocess.run( + [ + "git", + "status", + "--porcelain", + "--untracked-files=no", + "--", + ".", + ":(exclude)docs/plans/123-citation-driven-simulation-trust/evidence", + ":(exclude)docs/design/dart6_citation_driven_contact_trust/evidence", + ], + cwd=REPO_ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + if dirty: + raise SystemExit( + "refusing to generate evidence from a dirty tree; commit or " + "stash these tracked modifications first:\n" + dirty + ) return subprocess.run( ["git", "rev-parse", "HEAD"], cwd=REPO_ROOT, diff --git a/scripts/write_citation_ct005_pd_tracking_packet.py b/scripts/write_citation_ct005_pd_tracking_packet.py index 0e40381f3e1c5..3aab41ba34ed9 100644 --- a/scripts/write_citation_ct005_pd_tracking_packet.py +++ b/scripts/write_citation_ct005_pd_tracking_packet.py @@ -287,6 +287,35 @@ def run_single( def git_head() -> str: + """HEAD commit, refusing to attribute a dirty tree's results to it. + + A packet's target.commit claims the recorded results come from that + commit's code; uncommitted modifications to tracked files would make + that attribution false, so generation aborts instead. (The packet + output file itself is checked before being overwritten, so an + unmodified existing packet does not block regeneration.) + """ + dirty = subprocess.run( + [ + "git", + "status", + "--porcelain", + "--untracked-files=no", + "--", + ".", + ":(exclude)docs/plans/123-citation-driven-simulation-trust/evidence", + ":(exclude)docs/design/dart6_citation_driven_contact_trust/evidence", + ], + cwd=REPO_ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + if dirty: + raise SystemExit( + "refusing to generate evidence from a dirty tree; commit or " + "stash these tracked modifications first:\n" + dirty + ) return subprocess.run( ["git", "rev-parse", "HEAD"], cwd=REPO_ROOT, diff --git a/scripts/write_citation_ct007_high_mass_ratio_packet.py b/scripts/write_citation_ct007_high_mass_ratio_packet.py index e4f2f40247cba..3dbc5d4a8c581 100644 --- a/scripts/write_citation_ct007_high_mass_ratio_packet.py +++ b/scripts/write_citation_ct007_high_mass_ratio_packet.py @@ -163,8 +163,16 @@ def run_single( max_iterations = max(max_iterations, int(metrics.last_step_iterations)) max_contacts = max(max_contacts, int(metrics.active_contact_count)) for body in boxes: - position = np.asarray(body.translation, dtype=float) - if not np.all(np.isfinite(position)): + # Full-state finiteness: a blow-up confined to orientation or + # angular velocity must not publish as a finite cell. + state = np.concatenate( + [ + np.asarray(body.transform, dtype=float).reshape(-1), + np.asarray(body.linear_velocity, dtype=float), + np.asarray(body.angular_velocity, dtype=float), + ] + ) + if not np.all(np.isfinite(state)): non_finite = True break if non_finite: @@ -194,8 +202,11 @@ def run_single( velocity = np.asarray(body.linear_velocity, dtype=float) positions.append(position) speeds.append(float(np.linalg.norm(velocity))) - state_hash.update(position.tobytes()) + # Full-state hash so rotational differences are visible in the + # repeat comparison, not just translation. + state_hash.update(np.asarray(body.transform, dtype=float).tobytes()) state_hash.update(velocity.tobytes()) + state_hash.update(np.asarray(body.angular_velocity, dtype=float).tobytes()) lower_sink = lower_rest_z - float(positions[0][2]) upper_sink = upper_rest_z - float(positions[1][2]) @@ -222,6 +233,35 @@ def run_single( def git_head() -> str: + """HEAD commit, refusing to attribute a dirty tree's results to it. + + A packet's target.commit claims the recorded results come from that + commit's code; uncommitted modifications to tracked files would make + that attribution false, so generation aborts instead. (The packet + output file itself is checked before being overwritten, so an + unmodified existing packet does not block regeneration.) + """ + dirty = subprocess.run( + [ + "git", + "status", + "--porcelain", + "--untracked-files=no", + "--", + ".", + ":(exclude)docs/plans/123-citation-driven-simulation-trust/evidence", + ":(exclude)docs/design/dart6_citation_driven_contact_trust/evidence", + ], + cwd=REPO_ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + if dirty: + raise SystemExit( + "refusing to generate evidence from a dirty tree; commit or " + "stash these tracked modifications first:\n" + dirty + ) return subprocess.run( ["git", "rev-parse", "HEAD"], cwd=REPO_ROOT, diff --git a/scripts/write_citation_ct011_restore_equivalence_packet.py b/scripts/write_citation_ct011_restore_equivalence_packet.py index 72f2a353968cd..89b1ac95a1d38 100644 --- a/scripts/write_citation_ct011_restore_equivalence_packet.py +++ b/scripts/write_citation_ct011_restore_equivalence_packet.py @@ -303,6 +303,35 @@ def run_protocol(method_name: str, parameters: dict[str, Any]) -> dict[str, Any] def git_head() -> str: + """HEAD commit, refusing to attribute a dirty tree's results to it. + + A packet's target.commit claims the recorded results come from that + commit's code; uncommitted modifications to tracked files would make + that attribution false, so generation aborts instead. (The packet + output file itself is checked before being overwritten, so an + unmodified existing packet does not block regeneration.) + """ + dirty = subprocess.run( + [ + "git", + "status", + "--porcelain", + "--untracked-files=no", + "--", + ".", + ":(exclude)docs/plans/123-citation-driven-simulation-trust/evidence", + ":(exclude)docs/design/dart6_citation_driven_contact_trust/evidence", + ], + cwd=REPO_ROOT, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + if dirty: + raise SystemExit( + "refusing to generate evidence from a dirty tree; commit or " + "stash these tracked modifications first:\n" + dirty + ) return subprocess.run( ["git", "rev-parse", "HEAD"], cwd=REPO_ROOT, diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index fa8b87018d9aa..b6cb664e6e9f1 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -1280,3 +1280,47 @@ def test_swapped_artifacts_invalidate_the_packet(tmp_path): packet["evidence"]["artifact_digests"] = {"real.csv": "sha256:" + "0" * 64} errors = MODULE.packet_errors(packet, base_dir=tmp_path) assert any("does not match the referenced file's bytes" in e for e in errors) + + +def test_sweep_points_must_be_observed_by_rows(): + packet = complete_packet() + del packet["ensemble"]["deterministic_repeats"] + del packet["ensemble"]["deterministic_repeats_identical"] + packet["ensemble"]["sweep"] = [{"angle_deg": 0.0}, {"angle_deg": 15.0}] + packet["evidence"]["raw_rows"] = [ + {"angle_deg": 0.0, "lateral_drift_m": 0.0, "trajectory_sha256": "d" * 64}, + {"angle_deg": 0.0, "lateral_drift_m": 0.0, "trajectory_sha256": "d" * 64}, + ] + errors = MODULE.packet_errors(packet) + assert any("has no matching row" in error for error in errors) + + +def test_declared_seeds_must_be_observed_by_rows(): + packet = complete_packet() + del packet["ensemble"]["deterministic_repeats"] + del packet["ensemble"]["deterministic_repeats_identical"] + packet["ensemble"]["seeds"] = [7, 11] + packet["evidence"]["raw_rows"] = [ + {"seed": 7, "lateral_drift_m": 0.1, "trajectory_sha256": "d" * 64}, + {"seed": 7, "lateral_drift_m": 0.1, "trajectory_sha256": "d" * 64}, + ] + errors = MODULE.packet_errors(packet) + assert any("has no row recording it" in error for error in errors) + + +def test_structurally_empty_json_artifacts_fail(tmp_path): + (tmp_path / "rows.json").write_text('"just prose"', encoding="utf-8") + packet = complete_packet() + del packet["ensemble"]["deterministic_repeats"] + del packet["ensemble"]["deterministic_repeats_identical"] + packet["ensemble"]["sweep"] = [{"angle_deg": 0.0}, {"angle_deg": 15.0}] + del packet["evidence"]["raw_rows"] + packet["evidence"]["raw_paths"] = ["rows.json"] + packet["evidence"]["artifact_digests"] = { + "rows.json": "sha256:" + + hashlib.sha256((tmp_path / "rows.json").read_bytes()).hexdigest() + } + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any( + "carries no numeric or boolean measurement content" in error for error in errors + ) From 25d4d522e1e2e569f1005128d31c9866b184e10b Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 01:25:08 -0700 Subject: [PATCH 34/52] Regenerate CT-001/002/003/007 under the round-11 instruments Full-state finiteness and hashing (and CT-001's explicit NaN gate) change no measured pattern: all cells remain finite, dispositions unchanged (CT-001/002 reproduced; CT-003/007 unresolved). The clean-tree guard was exercised by this regeneration itself. --- .../CT-001-dart7-rolling-direction.json | 18 ++++++- .../CT-002-dart7-dense-inelastic-contact.json | 28 +++++----- .../CT-003-dart7-dense-elastic-contact.json | 28 +++++----- .../CT-007-dart7-high-mass-ratio.json | 52 +++++++++---------- 4 files changed, 70 insertions(+), 56 deletions(-) diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json index 3d6ad63047830..7b0325697116b 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json @@ -187,6 +187,7 @@ "rigid_body_position", "kinematics" ], + "finite": true, "heading_error_deg": 0.0, "lateral_drift_m": 0.0, "max_active_contacts": 1, @@ -252,6 +253,7 @@ "rigid_body_position", "kinematics" ], + "finite": true, "heading_error_deg": -3.1169435878342416e-14, "lateral_drift_m": -0.0021005566494608774, "max_active_contacts": 1, @@ -317,6 +319,7 @@ "rigid_body_position", "kinematics" ], + "finite": true, "heading_error_deg": -5.343331864858703e-14, "lateral_drift_m": -0.001883289782513231, "max_active_contacts": 1, @@ -382,6 +385,7 @@ "rigid_body_position", "kinematics" ], + "finite": true, "heading_error_deg": 0.0, "lateral_drift_m": 5.551115123125783e-17, "max_active_contacts": 1, @@ -447,6 +451,7 @@ "rigid_body_position", "kinematics" ], + "finite": true, "heading_error_deg": 5.343331864858703e-14, "lateral_drift_m": 0.0018832897825133976, "max_active_contacts": 1, @@ -512,6 +517,7 @@ "rigid_body_position", "kinematics" ], + "finite": true, "heading_error_deg": 3.1169435878342416e-14, "lateral_drift_m": 0.0021005566494608774, "max_active_contacts": 1, @@ -577,6 +583,7 @@ "rigid_body_position", "kinematics" ], + "finite": true, "heading_error_deg": 6.426647569961019e-30, "lateral_drift_m": 7.105154224747257e-19, "max_active_contacts": 1, @@ -642,6 +649,7 @@ "rigid_body_position", "kinematics" ], + "finite": true, "heading_error_deg": 0.0, "lateral_drift_m": 0.0, "max_active_contacts": 1, @@ -707,6 +715,7 @@ "rigid_body_position", "kinematics" ], + "finite": true, "heading_error_deg": -3.1169435878342416e-14, "lateral_drift_m": -0.002100556554215066, "max_active_contacts": 1, @@ -772,6 +781,7 @@ "rigid_body_position", "kinematics" ], + "finite": true, "heading_error_deg": -5.343331864858703e-14, "lateral_drift_m": -0.0018832896891917694, "max_active_contacts": 1, @@ -837,6 +847,7 @@ "rigid_body_position", "kinematics" ], + "finite": true, "heading_error_deg": -4.4527765540489165e-15, "lateral_drift_m": 5.551115123125783e-17, "max_active_contacts": 1, @@ -902,6 +913,7 @@ "rigid_body_position", "kinematics" ], + "finite": true, "heading_error_deg": 4.8980542094538094e-14, "lateral_drift_m": 0.0018832896891918804, "max_active_contacts": 1, @@ -967,6 +979,7 @@ "rigid_body_position", "kinematics" ], + "finite": true, "heading_error_deg": 3.1169435878342416e-14, "lateral_drift_m": 0.002100556554215066, "max_active_contacts": 1, @@ -1032,6 +1045,7 @@ "rigid_body_position", "kinematics" ], + "finite": true, "heading_error_deg": 6.426647569961019e-30, "lateral_drift_m": 7.105150776790931e-19, "max_active_contacts": 1, @@ -1262,9 +1276,9 @@ }, "target": { "branch": "main", - "commit": "b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c", + "commit": "dd68714d6a7a928b3af784e08369469b0e945216", "commit_role": "Source state measured: the library and fixture were run at this commit, which is HEAD at capture time. The packet and its writer land in a later commit, so re-running the recorded command requires the child commit that adds the writer; the measured behavior belongs to this one.", - "fetch_hint": "git fetch origin pull/3445/head && git checkout b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c" + "fetch_hint": "git fetch origin pull/3445/head && git checkout dd68714d6a7a928b3af784e08369469b0e945216" }, "title": "Rolling-direction friction dependence (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json index dbdc01c37a2f7..c744175c9b548 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json @@ -225,7 +225,7 @@ "final_kinetic_energy_j": 0.0073483693432820196, "final_max_speed_mps": 0.04032772547075842, "final_min_height_m": 0.048578028173510944, - "final_state_sha256": "d3f6fdd984ba5e6eabaa2841a1bc305f6237a6e224039a079177ec6ca400eb04", + "final_state_sha256": "5d4c0c3f7dca99eeb91fcd4e6ccc18a8f279d2239be70bbc116e7566e2c5f251", "finite": true, "initial_total_energy_j": 180.1116000000002, "max_active_contacts": 216, @@ -235,8 +235,8 @@ "max_total_energy_gain_after_settle_j": 0.0009981350274301803, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "d3f6fdd984ba5e6eabaa2841a1bc305f6237a6e224039a079177ec6ca400eb04", - "d3f6fdd984ba5e6eabaa2841a1bc305f6237a6e224039a079177ec6ca400eb04" + "5d4c0c3f7dca99eeb91fcd4e6ccc18a8f279d2239be70bbc116e7566e2c5f251", + "5d4c0c3f7dca99eeb91fcd4e6ccc18a8f279d2239be70bbc116e7566e2c5f251" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -286,7 +286,7 @@ "final_kinetic_energy_j": 0.029393477373128078, "final_max_speed_mps": 0.08065545094151684, "final_min_height_m": 0.04459709802896138, - "final_state_sha256": "bed22ef86fbea67a6176944a75e64973b4d0ab7f14c0d2604f61d67823b05f9d", + "final_state_sha256": "a57da239b087400119b423d4365a6e7748ab5b892f107978bf7e3ac8f9952292", "finite": true, "initial_total_energy_j": 180.1116000000002, "max_active_contacts": 216, @@ -296,8 +296,8 @@ "max_total_energy_gain_after_settle_j": 0.001308211355691924, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "bed22ef86fbea67a6176944a75e64973b4d0ab7f14c0d2604f61d67823b05f9d", - "bed22ef86fbea67a6176944a75e64973b4d0ab7f14c0d2604f61d67823b05f9d" + "a57da239b087400119b423d4365a6e7748ab5b892f107978bf7e3ac8f9952292", + "a57da239b087400119b423d4365a6e7748ab5b892f107978bf7e3ac8f9952292" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -347,7 +347,7 @@ "final_kinetic_energy_j": 1.4819720438078604e-27, "final_max_speed_mps": 2.3901020052008448e-14, "final_min_height_m": 0.04989999999999998, - "final_state_sha256": "5260253337e1348e1b0f89954a78b9d01a5715d982d77fe0384818189a8f9e33", + "final_state_sha256": "66e2142f4b817762eeff5c34e35e3ad4f79a2f7fd09a3fc0f6e9513d014e7b56", "finite": true, "initial_total_energy_j": 180.1116000000002, "max_active_contacts": 216, @@ -357,8 +357,8 @@ "max_total_energy_gain_after_settle_j": 0.0, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "5260253337e1348e1b0f89954a78b9d01a5715d982d77fe0384818189a8f9e33", - "5260253337e1348e1b0f89954a78b9d01a5715d982d77fe0384818189a8f9e33" + "66e2142f4b817762eeff5c34e35e3ad4f79a2f7fd09a3fc0f6e9513d014e7b56", + "66e2142f4b817762eeff5c34e35e3ad4f79a2f7fd09a3fc0f6e9513d014e7b56" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -408,7 +408,7 @@ "final_kinetic_energy_j": 3.674601533738957e-28, "final_max_speed_mps": 1.1914080833008711e-14, "final_min_height_m": 0.04989999999999998, - "final_state_sha256": "1a94d8a3a56f6cf1088d78188cbc7982f71e7b0287d0b2335c18dfac375ca3b8", + "final_state_sha256": "c73cf1f3ede55e68bb58581c948738d6471403bb75c99ad2b71f7e95f0a924dc", "finite": true, "initial_total_energy_j": 180.1116000000002, "max_active_contacts": 216, @@ -418,8 +418,8 @@ "max_total_energy_gain_after_settle_j": 1.342411593441284e-06, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "1a94d8a3a56f6cf1088d78188cbc7982f71e7b0287d0b2335c18dfac375ca3b8", - "1a94d8a3a56f6cf1088d78188cbc7982f71e7b0287d0b2335c18dfac375ca3b8" + "c73cf1f3ede55e68bb58581c948738d6471403bb75c99ad2b71f7e95f0a924dc", + "c73cf1f3ede55e68bb58581c948738d6471403bb75c99ad2b71f7e95f0a924dc" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -671,8 +671,8 @@ }, "target": { "branch": "main", - "commit": "b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c", - "fetch_hint": "git fetch origin pull/3445/head && git checkout b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c" + "commit": "dd68714d6a7a928b3af784e08369469b0e945216", + "fetch_hint": "git fetch origin pull/3445/head && git checkout dd68714d6a7a928b3af784e08369469b0e945216" }, "title": "Dense 6x6x6 inelastic contact stability (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json index 4aef59205ffa7..f3c8a19b3f33a 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json @@ -223,7 +223,7 @@ "contact_solver_method": "SEQUENTIAL_IMPULSE", "energy_envelope_excess_j": 0.0, "envelope_set_by_initial_state": true, - "final_state_sha256": "f1d5fac4ff2da8ee3e862cc55b49ea2362ecefd6a6527464dab5352f784c949c", + "final_state_sha256": "785ee993fda9dd80b394f5964edd34844bbf69dccfac7061f673698b6c6e5d6e", "final_total_energy_j": 63.44696118939535, "finite": true, "initial_total_energy_j": 180.1116000000002, @@ -234,8 +234,8 @@ "max_total_energy_j": 180.1116000000002, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "f1d5fac4ff2da8ee3e862cc55b49ea2362ecefd6a6527464dab5352f784c949c", - "f1d5fac4ff2da8ee3e862cc55b49ea2362ecefd6a6527464dab5352f784c949c" + "785ee993fda9dd80b394f5964edd34844bbf69dccfac7061f673698b6c6e5d6e", + "785ee993fda9dd80b394f5964edd34844bbf69dccfac7061f673698b6c6e5d6e" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -284,7 +284,7 @@ "contact_solver_method": "SEQUENTIAL_IMPULSE", "energy_envelope_excess_j": 0.0, "envelope_set_by_initial_state": true, - "final_state_sha256": "6cc4601838f1c5213104137dabb8e8064e376090d6cb22068335c12c87a8b125", + "final_state_sha256": "3e111883d46bcd87f0296a824fde812b8f026e884fae7b47072f343873d79d46", "final_total_energy_j": 63.25766139362141, "finite": true, "initial_total_energy_j": 180.1116000000002, @@ -295,8 +295,8 @@ "max_total_energy_j": 180.1116000000002, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "6cc4601838f1c5213104137dabb8e8064e376090d6cb22068335c12c87a8b125", - "6cc4601838f1c5213104137dabb8e8064e376090d6cb22068335c12c87a8b125" + "3e111883d46bcd87f0296a824fde812b8f026e884fae7b47072f343873d79d46", + "3e111883d46bcd87f0296a824fde812b8f026e884fae7b47072f343873d79d46" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -345,7 +345,7 @@ "contact_solver_method": "BOXED_LCP", "energy_envelope_excess_j": 0.0, "envelope_set_by_initial_state": true, - "final_state_sha256": "31c88c2d34b7e738c731aaa2e3e5a4dc9ebca29de00e901dc7296ebddb3402c1", + "final_state_sha256": "31c731bd98621b0db2c88edf07cb7f3d429cc7c53b0036c199e3bcf640c49670", "final_total_energy_j": 63.58479412413587, "finite": true, "initial_total_energy_j": 180.1116000000002, @@ -356,8 +356,8 @@ "max_total_energy_j": 180.1116000000002, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "31c88c2d34b7e738c731aaa2e3e5a4dc9ebca29de00e901dc7296ebddb3402c1", - "31c88c2d34b7e738c731aaa2e3e5a4dc9ebca29de00e901dc7296ebddb3402c1" + "31c731bd98621b0db2c88edf07cb7f3d429cc7c53b0036c199e3bcf640c49670", + "31c731bd98621b0db2c88edf07cb7f3d429cc7c53b0036c199e3bcf640c49670" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -406,7 +406,7 @@ "contact_solver_method": "BOXED_LCP", "energy_envelope_excess_j": 0.0, "envelope_set_by_initial_state": true, - "final_state_sha256": "2438f7c534de26f786af0b223b92f0898d8fcc5618698326c5e9f14be6936513", + "final_state_sha256": "5fceb3e62646e056c3ba3d84a7460f672bb6f1fb1fa041178be399451837f142", "final_total_energy_j": 63.746668542792605, "finite": true, "initial_total_energy_j": 180.1116000000002, @@ -417,8 +417,8 @@ "max_total_energy_j": 180.1116000000002, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "2438f7c534de26f786af0b223b92f0898d8fcc5618698326c5e9f14be6936513", - "2438f7c534de26f786af0b223b92f0898d8fcc5618698326c5e9f14be6936513" + "5fceb3e62646e056c3ba3d84a7460f672bb6f1fb1fa041178be399451837f142", + "5fceb3e62646e056c3ba3d84a7460f672bb6f1fb1fa041178be399451837f142" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -574,8 +574,8 @@ }, "target": { "branch": "main", - "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", - "fetch_hint": "git fetch origin pull/3445/head && git checkout f4e9e5c11d5d3b443693cde9142bf317e2115cd3" + "commit": "dd68714d6a7a928b3af784e08369469b0e945216", + "fetch_hint": "git fetch origin pull/3445/head && git checkout dd68714d6a7a928b3af784e08369469b0e945216" }, "title": "Dense elastic contact energy envelope (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json index 8c2f33ab264e0..dcd2dd00ca4dd 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json @@ -391,7 +391,7 @@ { "contact_solver_method": "SEQUENTIAL_IMPULSE", "final_max_speed_mps": 0.00045895830534830895, - "final_state_sha256": "0d8e9e8806ccf8792179d1d7a7ab1803a91339b44302827bf36e481d64e3de1c", + "final_state_sha256": "cc33a80165e5e0a58e8b10bb10869981a784e80dcae8ea4c32d968928a8f3ce6", "finite": true, "gap_closure_m": 2.8186383662115455e-05, "lower_sink_m": 3.8875436905949634e-05, @@ -401,8 +401,8 @@ "max_solver_iterations": 8, "relative_gap_closure": 0.00014093191831057728, "repeat_state_sha256": [ - "0d8e9e8806ccf8792179d1d7a7ab1803a91339b44302827bf36e481d64e3de1c", - "0d8e9e8806ccf8792179d1d7a7ab1803a91339b44302827bf36e481d64e3de1c" + "cc33a80165e5e0a58e8b10bb10869981a784e80dcae8ea4c32d968928a8f3ce6", + "cc33a80165e5e0a58e8b10bb10869981a784e80dcae8ea4c32d968928a8f3ce6" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -448,7 +448,7 @@ { "contact_solver_method": "SEQUENTIAL_IMPULSE", "final_max_speed_mps": 0.029482304187420837, - "final_state_sha256": "546e71ddc989dab2389455d59f9735021eac98f9b8a54fccaf980777d01b8345", + "final_state_sha256": "28c654f5bbda220b08138deccc24b3b71d316c85e528915cccefc2ee5e9f4ffb", "finite": true, "gap_closure_m": 0.00020330330581502798, "lower_sink_m": 0.0003313631340894213, @@ -458,8 +458,8 @@ "max_solver_iterations": 8, "relative_gap_closure": 0.00101651652907514, "repeat_state_sha256": [ - "546e71ddc989dab2389455d59f9735021eac98f9b8a54fccaf980777d01b8345", - "546e71ddc989dab2389455d59f9735021eac98f9b8a54fccaf980777d01b8345" + "28c654f5bbda220b08138deccc24b3b71d316c85e528915cccefc2ee5e9f4ffb", + "28c654f5bbda220b08138deccc24b3b71d316c85e528915cccefc2ee5e9f4ffb" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -505,7 +505,7 @@ { "contact_solver_method": "SEQUENTIAL_IMPULSE", "final_max_speed_mps": 4.830545337362943e-07, - "final_state_sha256": "f6872ff128b5b9d6bd9bbd1fc3adee6e270f78b5f69db09dc2f296a07c391ce7", + "final_state_sha256": "88751ed0d540613946ff1961b182136cb8d80977d021060e10cd499dbf341975", "finite": true, "gap_closure_m": 0.2000161894973165, "lower_sink_m": 2.403341138421111e-06, @@ -515,8 +515,8 @@ "max_solver_iterations": 8, "relative_gap_closure": 1.0000809474865824, "repeat_state_sha256": [ - "f6872ff128b5b9d6bd9bbd1fc3adee6e270f78b5f69db09dc2f296a07c391ce7", - "f6872ff128b5b9d6bd9bbd1fc3adee6e270f78b5f69db09dc2f296a07c391ce7" + "88751ed0d540613946ff1961b182136cb8d80977d021060e10cd499dbf341975", + "88751ed0d540613946ff1961b182136cb8d80977d021060e10cd499dbf341975" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -562,7 +562,7 @@ { "contact_solver_method": "SEQUENTIAL_IMPULSE", "final_max_speed_mps": 3.542993305133921e-06, - "final_state_sha256": "d0c911c7ae66e79c3b76279726147f29ab73c6e1b261e2661ce52ea22ee2567e", + "final_state_sha256": "c7eb04fb30595972cdd4a82be8086c27520aad419e02a22e8fb92b2484dc8c0b", "finite": true, "gap_closure_m": 0.2000041526819762, "lower_sink_m": 5.348353172895948e-06, @@ -572,8 +572,8 @@ "max_solver_iterations": 8, "relative_gap_closure": 1.0000207634098808, "repeat_state_sha256": [ - "d0c911c7ae66e79c3b76279726147f29ab73c6e1b261e2661ce52ea22ee2567e", - "d0c911c7ae66e79c3b76279726147f29ab73c6e1b261e2661ce52ea22ee2567e" + "c7eb04fb30595972cdd4a82be8086c27520aad419e02a22e8fb92b2484dc8c0b", + "c7eb04fb30595972cdd4a82be8086c27520aad419e02a22e8fb92b2484dc8c0b" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -619,7 +619,7 @@ { "contact_solver_method": "BOXED_LCP", "final_max_speed_mps": 1.4938402596549068e-11, - "final_state_sha256": "d84ea1f3527a71440ea5adec7ba0f7b7e6441595a2ed019131cd0065904fefbb", + "final_state_sha256": "0b1b397ea8cc02c519448933ba052996198f4ec818306633b4854203cf538c80", "finite": true, "gap_closure_m": 3.925327088483144e-05, "lower_sink_m": 1.3457496714358586e-06, @@ -629,8 +629,8 @@ "max_solver_iterations": 0, "relative_gap_closure": 0.00019626635442415719, "repeat_state_sha256": [ - "d84ea1f3527a71440ea5adec7ba0f7b7e6441595a2ed019131cd0065904fefbb", - "d84ea1f3527a71440ea5adec7ba0f7b7e6441595a2ed019131cd0065904fefbb" + "0b1b397ea8cc02c519448933ba052996198f4ec818306633b4854203cf538c80", + "0b1b397ea8cc02c519448933ba052996198f4ec818306633b4854203cf538c80" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -676,7 +676,7 @@ { "contact_solver_method": "BOXED_LCP", "final_max_speed_mps": 8.665822654413277e-06, - "final_state_sha256": "ac4dc0f23c4de5fbbdb30f19506cb6eb798962e06bc9d429593a7f324e95deaf", + "final_state_sha256": "98b2aee62fa0aff364ad8cf622c02a0ebf5e2fcbed9ee01043d9c233f529b471", "finite": true, "gap_closure_m": 3.957920874031462e-05, "lower_sink_m": 3.093141870136318e-07, @@ -686,8 +686,8 @@ "max_solver_iterations": 0, "relative_gap_closure": 0.0001978960437015731, "repeat_state_sha256": [ - "ac4dc0f23c4de5fbbdb30f19506cb6eb798962e06bc9d429593a7f324e95deaf", - "ac4dc0f23c4de5fbbdb30f19506cb6eb798962e06bc9d429593a7f324e95deaf" + "98b2aee62fa0aff364ad8cf622c02a0ebf5e2fcbed9ee01043d9c233f529b471", + "98b2aee62fa0aff364ad8cf622c02a0ebf5e2fcbed9ee01043d9c233f529b471" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -733,7 +733,7 @@ { "contact_solver_method": "BOXED_LCP", "final_max_speed_mps": 0.0008169729485216048, - "final_state_sha256": "43ac32aeec725c269d2b042c4ebae272c40d5f5bbf817e010c1f0353740e385f", + "final_state_sha256": "b78d23a138a9a9e2c7786b9a318cb49e4f0ab61504005ff3cd589b3bcd93240c", "finite": true, "gap_closure_m": 6.527479483597887e-06, "lower_sink_m": 3.401292682585211e-05, @@ -743,8 +743,8 @@ "max_solver_iterations": 0, "relative_gap_closure": 3.263739741798943e-05, "repeat_state_sha256": [ - "43ac32aeec725c269d2b042c4ebae272c40d5f5bbf817e010c1f0353740e385f", - "43ac32aeec725c269d2b042c4ebae272c40d5f5bbf817e010c1f0353740e385f" + "b78d23a138a9a9e2c7786b9a318cb49e4f0ab61504005ff3cd589b3bcd93240c", + "b78d23a138a9a9e2c7786b9a318cb49e4f0ab61504005ff3cd589b3bcd93240c" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -790,7 +790,7 @@ { "contact_solver_method": "BOXED_LCP", "final_max_speed_mps": 0.041688097943890993, - "final_state_sha256": "2ced2859d56f20af8c566c3c5192580f8e5d768e9acff317f5097839f7829fa1", + "final_state_sha256": "89fd0ea4211562eedc94ea2c17363c8d48e1f8c8b7b0a2ab8c8ad4a3537ff2f7", "finite": true, "gap_closure_m": -6.027789048593246e-05, "lower_sink_m": 0.00011855439148550362, @@ -800,8 +800,8 @@ "max_solver_iterations": 0, "relative_gap_closure": -0.0003013894524296623, "repeat_state_sha256": [ - "2ced2859d56f20af8c566c3c5192580f8e5d768e9acff317f5097839f7829fa1", - "2ced2859d56f20af8c566c3c5192580f8e5d768e9acff317f5097839f7829fa1" + "89fd0ea4211562eedc94ea2c17363c8d48e1f8c8b7b0a2ab8c8ad4a3537ff2f7", + "89fd0ea4211562eedc94ea2c17363c8d48e1f8c8b7b0a2ab8c8ad4a3537ff2f7" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -1007,9 +1007,9 @@ }, "target": { "branch": "main", - "commit": "b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c", + "commit": "dd68714d6a7a928b3af784e08369469b0e945216", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", - "fetch_hint": "git fetch origin pull/3445/head && git checkout b3fe1c9eeff0f79aaf5eedb6d59bc655a3032a1c" + "fetch_hint": "git fetch origin pull/3445/head && git checkout dd68714d6a7a928b3af784e08369469b0e945216" }, "title": "High-mass-ratio stack conditioning baseline for the exact-cone GO/NO-GO (DART 7)" } From 46a0ed20f858d958ab7e62c27f8d1f7018121988 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 01:39:08 -0700 Subject: [PATCH 35/52] Close the Codex round-12 findings (loop checkpoint) Scalar sweep points must be observed by rows carrying that value; sweep/seed ensembles require inline rows so per-point coverage is checkable; host provenance is required with an explicit performance_valid boolean, enforcing what the design doc promised; CSV/TSV artifacts need at least one numeric data cell; WebP/MP4 containers are checked for size-field consistency. artifact_digests now count as the hash-bearing repeat binding they genuinely are, resolving an emergent cross-rule contradiction. Tests: 125 -> 130 validator cases. The review loop pauses at this checkpoint per the session criterion (round 12 was the first with no genuine defect); all posted findings through round 12 are addressed. --- .../verification.md | 23 +++++ scripts/check_citation_evidence.py | 81 ++++++++++++++--- tests/test_check_citation_evidence.py | 86 +++++++++++++++++-- 3 files changed, 174 insertions(+), 16 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index f7555c20eee57..6fd89fcc87b45 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -667,3 +667,26 @@ coordinates (count parity is not enough), declared seeds must appear in rows under a seed field, and JSON raw artifacts must parse to content with at least one real measurement leaf. Affected packets regenerated at the round-11 commit. Tests: 122 -> 125 (main), 102 -> 105 (6.20). + +## Codex review round 12 (PR #3445) — 2026-08-16, loop checkpoint + +Five findings, all coherent completions of earlier rules, none a code +defect; all fixed on both branches: scalar sweep points must be +observed by rows carrying that value; sweep/seed ensembles require +inline rows (path-only evidence cannot demonstrate per-point coverage); +`host` is now REQUIRED with an explicit `performance_valid` boolean +(the design doc promised host-validity recording; the validator now +enforces it); CSV/TSV artifacts need at least one numeric data cell; +WebP/MP4 containers are checked for size-field consistency, not just +signatures. One emergent contradiction between rounds was resolved: +requiring rows for sweeps plus hash-bearing evidence for repeats would +have made path-based evidence impossible under any ensemble — +artifact_digests now count as the hash-bearing binding they genuinely +are. Tests: 125 -> 130 (main), 105 -> 110 (6.20). + +LOOP CHECKPOINT after 12 rounds (findings per round: +29-7-5-6-7-6-7-6-5-4-5-5): rounds 1-11 each contained genuine +defects or contract gaps and materially strengthened the program; +round 12 was the first with none. Per the criterion announced in +session, the review loop pauses here for the maintainer's decision; +every posted finding through round 12 is addressed and pushed. diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 2b627340fc69c..253e73489b880 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -121,11 +121,17 @@ def _has_hash_leaf(value: object) -> bool: if isinstance(value, dict): for key, item in value.items(): key_l = str(key).lower() - if ("sha256" in key_l or "hash" in key_l) and ( - (_is_nonempty_str(item) and HASH_VALUE_RE.match(item.strip())) - or (isinstance(item, (list, dict)) and _has_hash_leaf(item)) - ): - return True + if "sha256" in key_l or "hash" in key_l or "digest" in key_l: + if _is_nonempty_str(item) and HASH_VALUE_RE.match(item.strip()): + return True + if isinstance(item, dict) and any( + _is_nonempty_str(entry) + and HASH_VALUE_RE.match(entry.strip().removeprefix("sha256:")) + for entry in item.values() + ): + return True + if isinstance(item, (list, dict)) and _has_hash_leaf(item): + return True if _has_hash_leaf(item): return True elif isinstance(value, list): @@ -273,7 +279,7 @@ def _has_identity_key(value: object) -> bool: "review", "host", } -REQUIRED_PACKET_KEYS = PACKET_TOP_LEVEL_KEYS - {"host"} +REQUIRED_PACKET_KEYS = set(PACKET_TOP_LEVEL_KEYS) METRIC_GROUPS = ("physical", "numerical", "performance", "allocation") GIT_QUERY_ERRORS = (OSError, subprocess.CalledProcessError) @@ -376,9 +382,22 @@ def _raw_data_content_issue(path: "Path") -> "str | None": return None if suffix in (".csv", ".tsv"): delimiter = b"," if suffix == ".csv" else b"\t" - first_lines = head.splitlines()[:2] - if not first_lines or not any(delimiter in line for line in first_lines): + lines = [line for line in head.splitlines() if line.strip()] + if len(lines) < 2 or not any(delimiter in line for line in lines[:2]): return mismatch + has_numeric_cell = False + for line in lines[1:50]: + for cell in line.split(delimiter): + try: + float(cell.strip()) + except ValueError: + continue + has_numeric_cell = True + break + if has_numeric_cell: + break + if not has_numeric_cell: + return no_content return None if suffix == ".npy": return None if head.startswith(b"\x93NUMPY") else mismatch @@ -435,9 +454,19 @@ def _visual_content_issue(path: "Path") -> "str | None": ) return None if ok else mismatch if suffix == ".webp": - return None if header[:4] == b"RIFF" and header[8:12] == b"WEBP" else mismatch + if not (header[:4] == b"RIFF" and header[8:12] == b"WEBP"): + return mismatch + riff_size = int.from_bytes(header[4:8], "little") + if riff_size + 8 != size: + return mismatch + return None if suffix == ".mp4": - return None if header[4:8] == b"ftyp" else mismatch + if header[4:8] != b"ftyp": + return mismatch + box_size = int.from_bytes(header[0:4], "big") + if not (8 <= box_size <= size): + return mismatch + return None if suffix == ".webm": return None if header.startswith(b"\x1a\x45\xdf\xa3") else mismatch if suffix == ".svg": @@ -798,6 +827,8 @@ def packet_errors( declared_points = 0 sweep = None seeds = None + has_sweep = False + has_seeds = False ensemble = packet.get("ensemble") if not isinstance(ensemble, dict): errors.append("ensemble must be an object") @@ -983,6 +1014,17 @@ def packet_errors( # measurements could be repeated to stand in for the others. if has_rows and isinstance(sweep, list): for point in sweep: + if _is_finite_number(point) or _is_nonempty_str(point): + if not any( + isinstance(row, dict) and point in row.values() + for row in raw_rows + ): + errors.append( + f"ensemble.sweep point {point!r} has no row " + "carrying that value; a declared configuration " + "without an observation is not swept" + ) + continue if not (isinstance(point, dict) and point): continue if not any( @@ -1017,6 +1059,12 @@ def packet_errors( ) if not (has_paths or has_rows): errors.append("evidence must carry raw_rows inline or non-empty raw_paths") + if (has_sweep or has_seeds) and not has_rows: + errors.append( + "a sweep/seed ensemble requires inline evidence.raw_rows so " + "each declared point's observation is checkable; path-only " + "raw evidence cannot demonstrate per-point coverage" + ) if has_rows: for index, row in enumerate(raw_rows): if not (isinstance(row, dict) and row): @@ -1181,6 +1229,19 @@ def packet_errors( "has at least one honest limitation" ) + host = packet.get("host") + if not isinstance(host, dict): + errors.append("host must be an object recording provenance") + else: + if not _is_nonempty_str(host.get("platform")): + errors.append("host.platform must be a non-empty string") + if not isinstance(host.get("performance_valid"), bool): + errors.append( + "host.performance_valid must be an explicit boolean; timing " + "or allocation numbers from an uncontrolled host must not " + "pass silently" + ) + review = packet.get("review") if not isinstance(review, dict): errors.append("review must be an object") diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index b6cb664e6e9f1..78a6ea0c0f4ce 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -109,6 +109,11 @@ def complete_packet() -> dict: "limitations": ["Single fixture."], }, "review": {"passes": []}, + "host": { + "platform": "test-host", + "python": "3.14", + "performance_valid": False, + }, } @@ -490,12 +495,8 @@ def test_dangling_raw_path_fails(tmp_path): assert any("does not resolve" in error for error in errors) (tmp_path / "real.csv").write_text("a,b\n1,2\n", encoding="utf-8") packet["evidence"]["raw_paths"] = ["real.csv"] - # raw_paths carry no hash leaf, so the repeats claim needs one recorded - # elsewhere; a sweep ensemble sidesteps that requirement here, and - # path-based evidence must pin its artifact bytes. - del packet["ensemble"]["deterministic_repeats"] - del packet["ensemble"]["deterministic_repeats_identical"] - packet["ensemble"]["sweep"] = [{"angle_deg": 0.0}, {"angle_deg": 15.0}] + # Path-based evidence pins its artifact bytes; those digests also carry + # the hash-bearing evidence the repeats claim binds to. packet["evidence"]["artifact_digests"] = { "real.csv": "sha256:" + hashlib.sha256((tmp_path / "real.csv").read_bytes()).hexdigest() @@ -1324,3 +1325,76 @@ def test_structurally_empty_json_artifacts_fail(tmp_path): assert any( "carries no numeric or boolean measurement content" in error for error in errors ) + + +def test_scalar_sweep_points_must_be_observed(): + packet = complete_packet() + del packet["ensemble"]["deterministic_repeats"] + del packet["ensemble"]["deterministic_repeats_identical"] + packet["ensemble"]["sweep"] = [0.0, 15.0] + packet["evidence"]["raw_rows"] = [ + {"angle_deg": 0.0, "lateral_drift_m": 0.0, "trajectory_sha256": "d" * 64}, + {"angle_deg": 99.0, "lateral_drift_m": 0.0, "trajectory_sha256": "d" * 64}, + ] + errors = MODULE.packet_errors(packet) + assert any("has no row carrying that value" in error for error in errors) + + +def test_sweep_ensembles_require_rows(tmp_path): + (tmp_path / "rows.csv").write_text("a,b\n1,2\n", encoding="utf-8") + packet = complete_packet() + del packet["ensemble"]["deterministic_repeats"] + del packet["ensemble"]["deterministic_repeats_identical"] + packet["ensemble"]["sweep"] = [{"angle_deg": 0.0}, {"angle_deg": 15.0}] + del packet["evidence"]["raw_rows"] + packet["evidence"]["raw_paths"] = ["rows.csv"] + packet["evidence"]["artifact_digests"] = { + "rows.csv": "sha256:" + + hashlib.sha256((tmp_path / "rows.csv").read_bytes()).hexdigest() + } + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any("requires inline evidence.raw_rows" in error for error in errors) + + +def test_host_provenance_is_required(): + packet = complete_packet() + del packet["host"] + errors = MODULE.packet_errors(packet) + assert any("missing required top-level keys" in error for error in errors) + packet = complete_packet() + del packet["host"]["performance_valid"] + errors = MODULE.packet_errors(packet) + assert any( + "performance_valid must be an explicit boolean" in error for error in errors + ) + + +def test_metadata_only_csv_fails(tmp_path): + (tmp_path / "rows.csv").write_text("note,status\nfoo,pending\n", encoding="utf-8") + packet = complete_packet() + del packet["evidence"]["raw_rows"] + packet["evidence"]["raw_paths"] = ["rows.csv"] + packet["evidence"]["artifact_digests"] = { + "rows.csv": "sha256:" + + hashlib.sha256((tmp_path / "rows.csv").read_bytes()).hexdigest() + } + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any( + "carries no numeric or boolean measurement content" in error for error in errors + ) + + +def test_signature_only_webp_and_mp4_fail(tmp_path): + root = tmp_path / "d" + root.mkdir() + (root / "clip.webp").write_bytes( + b"RIFF" + b"\x00\x01\x02\x03" + b"WEBP" + b"p" * 60 + ) + packet = complete_packet() + packet["evidence"]["visual"] = ["clip.webp"] + packet["evidence"]["artifact_digests"] = { + "clip.webp": "sha256:" + + hashlib.sha256((root / "clip.webp").read_bytes()).hexdigest() + } + errors = MODULE.packet_errors(packet, base_dir=root) + assert any("structurally complete" in error for error in errors) From c117b9e37fa1a783318c46e93daebbe80e8009f1 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 03:46:29 -0700 Subject: [PATCH 36/52] Record the pre-merge test-all gate at the loop checkpoint --- .../citation_driven_simulation_trust/verification.md | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 6fd89fcc87b45..38f12cb38b989 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -690,3 +690,9 @@ defects or contract gaps and materially strengthened the program; round 12 was the first with none. Per the criterion announced in session, the review loop pauses here for the maintainer's decision; every posted finding through round 12 is addressed and pushed. + +## Pre-merge gate at the loop checkpoint — 2026-08-16 + +Full `pixi run test-all` on the checkpoint head `46a0ed20f85`: exit 0, +all phases pass. This is the pre-merge evidence for PR #3445; the +merge decision (or a continue-the-loop call) rests with the maintainer. From 706e65dc60eaaf531703f78e7ae89793ca7ce954 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 11:06:11 -0700 Subject: [PATCH 37/52] Close the Codex round-13 findings Structural parsing for binary raw artifacts (NPY header grammar via stdlib, NPZ zip membership, parquet header+footer); WebM must carry its DocType in the EBML header; reviewer identities are NFKC-normalized before the distinct-reviewer closure count; sweep points must be objects naming their coordinates -- bare scalars are rejected, replacing the value-appears-anywhere heuristic that unrelated measurements could satisfy. Tests: 130 -> 133 validator cases. --- .../verification.md | 15 +++ scripts/check_citation_evidence.py | 93 ++++++++++++++----- tests/test_check_citation_evidence.py | 59 +++++++++++- 3 files changed, 143 insertions(+), 24 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 38f12cb38b989..8f31a14605743 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -696,3 +696,18 @@ every posted finding through round 12 is addressed and pushed. Full `pixi run test-all` on the checkpoint head `46a0ed20f85`: exit 0, all phases pass. This is the pre-merge evidence for PR #3445; the merge decision (or a continue-the-loop call) rests with the maintainer. + +## Codex review round 13 (PR #3445) — 2026-08-16, loop resumed by maintainer + +Loop resumed on the maintainer's instruction. Five findings, all fixed +on both branches: NPY headers are parsed structurally (magic, version, +header length, literal dict with a valid shape tuple — stdlib only); +NPZ must be a readable zip containing at least one .npy member; parquet +needs its PAR1 footer as well as header; WebM must carry its DocType in +the EBML header (signature-plus-padding rejected); reviewer identities +are NFKC-normalized before the distinct count (Jos\u00e9 vs +Jose+combining-accent is one reviewer); and scalar sweep points are now +rejected outright — a sweep point must be an object naming its +coordinates, replacing the value-appears-anywhere heuristic that +unrelated measurements could satisfy. Tests: 130 -> 133 (main), +110 -> 113 (6.20). diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 253e73489b880..a0042b47b1d83 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -39,6 +39,7 @@ import re import subprocess import sys +import unicodedata from pathlib import Path, PurePosixPath REPO_ROOT = Path(__file__).resolve().parents[1] @@ -400,10 +401,68 @@ def _raw_data_content_issue(path: "Path") -> "str | None": return no_content return None if suffix == ".npy": - return None if head.startswith(b"\x93NUMPY") else mismatch - if suffix in (".npz", ".parquet"): - ok = head.startswith(b"PK") if suffix == ".npz" else head[:4] == b"PAR1" - return None if ok else mismatch + return _npy_header_issue(head, mismatch) + if suffix == ".npz": + try: + import io + import zipfile + + with path.open("rb") as stream: + archive = zipfile.ZipFile(io.BytesIO(stream.read())) + names = archive.namelist() + except OSError, zipfile.BadZipFile: + return mismatch + if not names or not any(name.endswith(".npy") for name in names): + return mismatch + return None + if suffix == ".parquet": + try: + with path.open("rb") as stream: + stream.seek(-4, 2) + tail4 = stream.read(4) + except OSError: + return mismatch + if head[:4] != b"PAR1" or tail4 != b"PAR1": + return mismatch + return None + return None + + +def _npy_header_issue(head: bytes, mismatch: str) -> "str | None": + """Parse the NPY header structurally (stdlib only): magic, version, + header length, and a literal dict with descr/shape/fortran_order whose + shape is a tuple of non-negative ints.""" + import ast as _ast + import struct + + if not head.startswith(b"\x93NUMPY") or len(head) < 10: + return mismatch + major = head[6] + if major == 1: + (header_len,) = struct.unpack("= 0 + for dim in header["shape"] + ) + ): + return mismatch return None @@ -468,7 +527,8 @@ def _visual_content_issue(path: "Path") -> "str | None": return mismatch return None if suffix == ".webm": - return None if header.startswith(b"\x1a\x45\xdf\xa3") else mismatch + ok = header.startswith(b"\x1a\x45\xdf\xa3") and b"webm" in header + return None if ok else mismatch if suffix == ".svg": ok = b"" in stripped_tail return None if ok else mismatch @@ -879,12 +939,12 @@ def packet_errors( for index, entry in enumerate(sweep): if isinstance(entry, dict) and entry: canonical_points.append(json.dumps(entry, sort_keys=True)) - elif _is_finite_number(entry) or _is_nonempty_str(entry): - canonical_points.append(json.dumps(entry)) else: errors.append( - f"ensemble.sweep[{index}] must be a non-empty object, " - "finite number, or non-empty string sweep point" + f"ensemble.sweep[{index}] must be a non-empty object " + "naming its coordinates (e.g. {'angle_deg': 15.0}); " + "a bare scalar cannot be matched to the row that " + "observed it" ) has_sweep = False if has_sweep and len(set(canonical_points)) < 2: @@ -1014,17 +1074,6 @@ def packet_errors( # measurements could be repeated to stand in for the others. if has_rows and isinstance(sweep, list): for point in sweep: - if _is_finite_number(point) or _is_nonempty_str(point): - if not any( - isinstance(row, dict) and point in row.values() - for row in raw_rows - ): - errors.append( - f"ensemble.sweep point {point!r} has no row " - "carrying that value; a declared configuration " - "without an observation is not swept" - ) - continue if not (isinstance(point, dict) and point): continue if not any( @@ -1630,7 +1679,9 @@ def validate_tree( ) else: reviewers = { - entry["reviewer"].strip().casefold() + unicodedata.normalize( + "NFKC", entry["reviewer"].strip() + ).casefold() for entry in passes if isinstance(entry, dict) and _is_nonempty_str(entry.get("reviewer")) diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index 78a6ea0c0f4ce..8d30be9a06453 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -1327,17 +1327,17 @@ def test_structurally_empty_json_artifacts_fail(tmp_path): ) -def test_scalar_sweep_points_must_be_observed(): +def test_scalar_sweep_points_are_rejected(): packet = complete_packet() del packet["ensemble"]["deterministic_repeats"] del packet["ensemble"]["deterministic_repeats_identical"] packet["ensemble"]["sweep"] = [0.0, 15.0] packet["evidence"]["raw_rows"] = [ {"angle_deg": 0.0, "lateral_drift_m": 0.0, "trajectory_sha256": "d" * 64}, - {"angle_deg": 99.0, "lateral_drift_m": 0.0, "trajectory_sha256": "d" * 64}, + {"angle_deg": 15.0, "lateral_drift_m": 0.0, "trajectory_sha256": "d" * 64}, ] errors = MODULE.packet_errors(packet) - assert any("has no row carrying that value" in error for error in errors) + assert any("naming its coordinates" in error for error in errors) def test_sweep_ensembles_require_rows(tmp_path): @@ -1398,3 +1398,56 @@ def test_signature_only_webp_and_mp4_fail(tmp_path): } errors = MODULE.packet_errors(packet, base_dir=root) assert any("structurally complete" in error for error in errors) + + +def test_unicode_variant_reviewers_count_once(tmp_path): + packet = copy.deepcopy(complete_packet()) + packet["review"]["passes"] = [ + _bound_pass(packet, "Jos\u00e9"), + _bound_pass(packet, "Jose\u0301"), + ] + manifest = _minimal_manifest(["CT-001"]) + lane = manifest["claims"][0]["lanes"][LANE] + lane["status"] = "closed" + lane["disposition"] = "unresolved" + lane["evidence"] = ["evidence/packet.json"] + tree_dir = _write_tree( + tmp_path, packet=packet, negative={"schema": "x"}, manifest=manifest + ) + errors = MODULE.validate_tree(tree_dir) + assert any("INDEPENDENT" in error for error in errors) + + +def test_stub_binary_artifacts_fail(tmp_path): + import zipfile as _zip + + empty_zip = tmp_path / "empty.npz" + with _zip.ZipFile(empty_zip, "w"): + pass + stub_npy = tmp_path / "stub.npy" + stub_npy.write_bytes(b"\x93NUMPY" + b"x" * 100) + for name in ("empty.npz", "stub.npy"): + packet = complete_packet() + del packet["evidence"]["raw_rows"] + packet["evidence"]["raw_paths"] = [name] + packet["evidence"]["artifact_digests"] = { + name: "sha256:" + hashlib.sha256((tmp_path / name).read_bytes()).hexdigest() + } + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any( + "does not parse as its claimed raw-data format" in error for error in errors + ), name + + +def test_stub_webm_fails(tmp_path): + root = tmp_path / "d" + root.mkdir() + (root / "clip.webm").write_bytes(b"\x1a\x45\xdf\xa3" + b"p" * 96) + packet = complete_packet() + packet["evidence"]["visual"] = ["clip.webm"] + packet["evidence"]["artifact_digests"] = { + "clip.webm": "sha256:" + + hashlib.sha256((root / "clip.webm").read_bytes()).hexdigest() + } + errors = MODULE.packet_errors(packet, base_dir=root) + assert any("structurally complete" in error for error in errors) From 654193f85be94b2e5d9f45cee1093a4715b16baa Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 11:33:49 -0700 Subject: [PATCH 38/52] Close the Codex round-14 findings The resolved-configuration binding and helper docs no longer say 'actually ran': they state BAKE-TIME resolution -- the configuration that will run when a domain is exercised -- and direct corroboration of execution to WorldStepProfile stage names and trajectory evidence (activity-aware per-domain execution tracking is recorded as WS4 follow-up alongside the per-solve residual). CT-002's repeat gate hashed only the final state, so repeats reaching the same endpoint via different transients would have passed while publishing path-dependent metrics from one path -- CT-002/003/007 now hash the full rigid-body state EVERY step (CT-001/011 already did); those packets are regenerated at this commit in the follow-up commit. evidence/raw/ is reserved for raw artifacts (excluded from packet discovery, un-referenceable by lanes). Tests: 133 -> 134 validator cases. --- .../verification.md | 18 +++++++++++++++++ python/dartpy/simulation/module_compute.cpp | 6 +++++- python/dartpy/simulation/module_world.cpp | 14 ++++++++----- scripts/check_citation_evidence.py | 11 ++++++++++ scripts/citation_packet_utils.py | 5 ++++- ...ite_citation_ct002_dense_contact_packet.py | 8 ++++++++ ...e_citation_ct003_elastic_contact_packet.py | 7 +++++++ ...e_citation_ct007_high_mass_ratio_packet.py | 7 +++++++ tests/test_check_citation_evidence.py | 20 +++++++++++++++++++ 9 files changed, 89 insertions(+), 7 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 8f31a14605743..e4f934f9532d3 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -711,3 +711,21 @@ rejected outright — a sweep point must be an object naming its coordinates, replacing the value-appears-anywhere heuristic that unrelated measurements could satisfy. Tests: 130 -> 133 (main), 110 -> 113 (6.20). + +## Codex review round 14 (PR #3445) — 2026-08-16 + +Three findings, two substantive, all fixed: the resolved-configuration +binding and helper docs no longer say "actually ran" — they state +BAKE-TIME resolution (the configuration that will run when a domain is +exercised) and direct corroboration of execution to WorldStepProfile +stage names and trajectory evidence; activity-aware per-domain +execution tracking is recorded as WS4 follow-up work alongside the +per-solve residual. CT-002's repeat gate hashed only the FINAL state, +so repeats reaching the same endpoint via different transients would +have passed while publishing path-dependent metrics (peak penetration, +settle energy) from one path — CT-002/003/007 now hash the full +rigid-body state EVERY step (CT-001/011 already did); the three +packets are regenerated at the round-14 commit. From #3444: +evidence/raw/ is now reserved for raw artifacts — excluded from packet +discovery, and lanes may not reference into it. Tests: 133 -> 134 +(main), 113 -> 114 (6.20). diff --git a/python/dartpy/simulation/module_compute.cpp b/python/dartpy/simulation/module_compute.cpp index 9d5388df31811..7e2ca59294634 100644 --- a/python/dartpy/simulation/module_compute.cpp +++ b/python/dartpy/simulation/module_compute.cpp @@ -314,7 +314,11 @@ void defSimPartCompute(nb::module_& m) .def_ro( "resolved", &sim::compute::ResolvedConfigurationNote::resolved, - "Method family the step actually ran.") + "Method family the bake resolved this domain to. This is the " + "configuration that WILL run when the domain is exercised; it " + "does not by itself prove the domain was active in any step -- " + "corroborate execution via WorldStepProfile stage names or " + "trajectory evidence.") .def_ro( "reason", &sim::compute::ResolvedConfigurationNote::reason, diff --git a/python/dartpy/simulation/module_world.cpp b/python/dartpy/simulation/module_world.cpp index a7169873c3568..ee61f3652b210 100644 --- a/python/dartpy/simulation/module_world.cpp +++ b/python/dartpy/simulation/module_world.cpp @@ -576,11 +576,15 @@ void defSimPartWorld(nb::module_& m) &sim::World::getResolvedConfiguration, nb::rv_policy::reference_internal, "Per-domain solver families this World resolved at " - "enter_simulation_mode: what was requested, what actually ran, and " - "why. Empty until the World enters simulation mode, and re-recorded " - "on every rebake. Use this rather than echoing back the requested " - "option, so a benchmark or evidence packet cannot report a method " - "that did not run.") + "enter_simulation_mode: what was requested, what the bake " + "resolved it to, and why. Empty until the World enters " + "simulation mode, and re-recorded on every rebake. This is " + "BAKE-TIME resolution -- the configuration that will run when a " + "domain is exercised -- not proof that a domain was active in " + "any step; corroborate execution via WorldStepProfile stage " + "names or trajectory evidence. Use it rather than echoing back " + "the requested option, so a benchmark or evidence packet cannot " + "report a method that was never even configured.") .def( "compute_step_metrics", &sim::World::computeStepMetrics, diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index a0042b47b1d83..a9d7e55ba4629 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -1566,6 +1566,14 @@ def validate_tree( ) continue rel = normalized + if normalized.startswith("evidence/raw/"): + errors.append( + f"{manifest_path}: {claim_id}.lanes." + f"{lane_name}.evidence entry {rel!r} points " + "into evidence/raw/, which holds raw " + "artifacts, not packets" + ) + continue if True: if rel in lane_evidence: owner = lane_evidence[rel] @@ -1616,11 +1624,14 @@ def validate_tree( "fails closed and can never be a claim's evidence" ) + raw_artifacts_dir = evidence_dir / "raw" packet_paths = ( sorted( path for path in evidence_dir.rglob("*.json") if negative_dir.resolve() not in path.resolve().parents + and raw_artifacts_dir.resolve() not in path.resolve().parents + and not path.name.endswith(".expected-errors.json") ) if evidence_dir.is_dir() else [] diff --git a/scripts/citation_packet_utils.py b/scripts/citation_packet_utils.py index bd99631aea047..e684e9588817b 100644 --- a/scripts/citation_packet_utils.py +++ b/scripts/citation_packet_utils.py @@ -209,7 +209,10 @@ def record_review_pass(packet_path: "Any", reviewer: str, summary: str) -> None: def world_resolved_configuration(world: Any) -> dict[str, Any]: """Record the World's own bake-time solver resolution. - This is the authoritative answer to "which method actually ran": the + This is the authoritative answer to "which method the bake resolved + each domain to" -- the configuration that runs when a domain is + exercised, corroborated in packets by step-profile stage names and + trajectory evidence rather than trusted alone: the World reports, per domain, what was requested, what it resolved to, and why. Echoing back the requested option cannot distinguish a method that ran from one that was silently substituted. diff --git a/scripts/write_citation_ct002_dense_contact_packet.py b/scripts/write_citation_ct002_dense_contact_packet.py index 85872fb460105..4b742d5da3d0a 100644 --- a/scripts/write_citation_ct002_dense_contact_packet.py +++ b/scripts/write_citation_ct002_dense_contact_packet.py @@ -175,6 +175,14 @@ def run_single( for step_index in range(step_count): world.step() + # Per-step full-state trajectory hashing: repeats that settle to the + # same final state via different transient paths must NOT pass the + # determinism gate, because the published metrics (peak penetration, + # settle energy gains) are path-dependent. + for body in bodies: + state_hash.update(np.asarray(body.transform, dtype=float).tobytes()) + state_hash.update(np.asarray(body.linear_velocity, dtype=float).tobytes()) + state_hash.update(np.asarray(body.angular_velocity, dtype=float).tobytes()) metrics = world.compute_step_metrics() kinetic = float(metrics.kinetic_energy) total_energy = float(metrics.total_energy) diff --git a/scripts/write_citation_ct003_elastic_contact_packet.py b/scripts/write_citation_ct003_elastic_contact_packet.py index 8aa7a8cfa1b14..0666cb16e4c98 100644 --- a/scripts/write_citation_ct003_elastic_contact_packet.py +++ b/scripts/write_citation_ct003_elastic_contact_packet.py @@ -173,6 +173,13 @@ def run_single( for _ in range(step_count): world.step() + # Per-step full-state trajectory hashing: repeats that reach the + # same final state via different transient paths must NOT pass the + # determinism gate, because the published metrics are path-dependent. + for body in bodies: + state_hash.update(np.asarray(body.transform, dtype=float).tobytes()) + state_hash.update(np.asarray(body.linear_velocity, dtype=float).tobytes()) + state_hash.update(np.asarray(body.angular_velocity, dtype=float).tobytes()) metrics = world.compute_step_metrics() total_energy = float(metrics.total_energy) if not math.isfinite(total_energy): diff --git a/scripts/write_citation_ct007_high_mass_ratio_packet.py b/scripts/write_citation_ct007_high_mass_ratio_packet.py index 3dbc5d4a8c581..c36aad4823f1b 100644 --- a/scripts/write_citation_ct007_high_mass_ratio_packet.py +++ b/scripts/write_citation_ct007_high_mass_ratio_packet.py @@ -158,6 +158,13 @@ def run_single( for _ in range(step_count): world.step() + # Per-step full-state trajectory hashing: repeats that reach the + # same final state via different transient paths must NOT pass the + # determinism gate, because the published metrics are path-dependent. + for body in boxes: + state_hash.update(np.asarray(body.transform, dtype=float).tobytes()) + state_hash.update(np.asarray(body.linear_velocity, dtype=float).tobytes()) + state_hash.update(np.asarray(body.angular_velocity, dtype=float).tobytes()) metrics = world.compute_step_metrics() max_penetration = max(max_penetration, float(metrics.max_penetration_depth)) max_iterations = max(max_iterations, int(metrics.last_step_iterations)) diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index 8d30be9a06453..d83ece6ef1dd5 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -1451,3 +1451,23 @@ def test_stub_webm_fails(tmp_path): } errors = MODULE.packet_errors(packet, base_dir=root) assert any("structurally complete" in error for error in errors) + + +def test_raw_artifacts_under_evidence_raw_are_not_packets(tmp_path): + tree_dir = _write_tree(tmp_path, packet=None, negative={"schema": "x"}) + raw_dir = tree_dir / "evidence" / "raw" + raw_dir.mkdir() + (raw_dir / "rows.json").write_text( + json.dumps([{"lateral_drift_m": 0.5}]), encoding="utf-8" + ) + errors = MODULE.validate_tree(tree_dir) + assert not any("rows.json" in error for error in errors) + manifest = _minimal_manifest(["CT-001"]) + lane = manifest["claims"][0]["lanes"][LANE] + lane["status"] = "in-progress" + lane["evidence"] = ["evidence/raw/rows.json"] + (tree_dir / "claims-manifest.json").write_text( + json.dumps(manifest), encoding="utf-8" + ) + errors = MODULE.validate_tree(tree_dir) + assert any("holds raw artifacts, not packets" in error for error in errors) From 4499cf8036855d31b2f3d5c4efafbcf65aceec47 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 11:46:04 -0700 Subject: [PATCH 39/52] Regenerate CT-002/003/007 under per-step trajectory hashing Deterministic repeats remain bit-identical along the WHOLE path, not just at the endpoint, so the path-dependent metrics (peak penetration, settle energy gains) are now covered by the determinism gate. Dispositions unchanged (CT-002 reproduced; CT-003/007 unresolved). --- .../CT-002-dart7-dense-inelastic-contact.json | 28 +++++----- .../CT-003-dart7-dense-elastic-contact.json | 28 +++++----- .../CT-007-dart7-high-mass-ratio.json | 52 +++++++++---------- 3 files changed, 54 insertions(+), 54 deletions(-) diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json index c744175c9b548..54692076cddf9 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json @@ -225,7 +225,7 @@ "final_kinetic_energy_j": 0.0073483693432820196, "final_max_speed_mps": 0.04032772547075842, "final_min_height_m": 0.048578028173510944, - "final_state_sha256": "5d4c0c3f7dca99eeb91fcd4e6ccc18a8f279d2239be70bbc116e7566e2c5f251", + "final_state_sha256": "cf98f2db55323032686ac102d9b2a8ce3434a684961144139337a2ab2f8bfa86", "finite": true, "initial_total_energy_j": 180.1116000000002, "max_active_contacts": 216, @@ -235,8 +235,8 @@ "max_total_energy_gain_after_settle_j": 0.0009981350274301803, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "5d4c0c3f7dca99eeb91fcd4e6ccc18a8f279d2239be70bbc116e7566e2c5f251", - "5d4c0c3f7dca99eeb91fcd4e6ccc18a8f279d2239be70bbc116e7566e2c5f251" + "cf98f2db55323032686ac102d9b2a8ce3434a684961144139337a2ab2f8bfa86", + "cf98f2db55323032686ac102d9b2a8ce3434a684961144139337a2ab2f8bfa86" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -286,7 +286,7 @@ "final_kinetic_energy_j": 0.029393477373128078, "final_max_speed_mps": 0.08065545094151684, "final_min_height_m": 0.04459709802896138, - "final_state_sha256": "a57da239b087400119b423d4365a6e7748ab5b892f107978bf7e3ac8f9952292", + "final_state_sha256": "5e1abe9c3bc5d7faa494bb849b3841f90dc765c56f52a86e3f463dc7f0ef6c72", "finite": true, "initial_total_energy_j": 180.1116000000002, "max_active_contacts": 216, @@ -296,8 +296,8 @@ "max_total_energy_gain_after_settle_j": 0.001308211355691924, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "a57da239b087400119b423d4365a6e7748ab5b892f107978bf7e3ac8f9952292", - "a57da239b087400119b423d4365a6e7748ab5b892f107978bf7e3ac8f9952292" + "5e1abe9c3bc5d7faa494bb849b3841f90dc765c56f52a86e3f463dc7f0ef6c72", + "5e1abe9c3bc5d7faa494bb849b3841f90dc765c56f52a86e3f463dc7f0ef6c72" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -347,7 +347,7 @@ "final_kinetic_energy_j": 1.4819720438078604e-27, "final_max_speed_mps": 2.3901020052008448e-14, "final_min_height_m": 0.04989999999999998, - "final_state_sha256": "66e2142f4b817762eeff5c34e35e3ad4f79a2f7fd09a3fc0f6e9513d014e7b56", + "final_state_sha256": "2315ba41bfe626fc491f77231495ba8c523170dd7eb2e5aae3b7f9133f224e7a", "finite": true, "initial_total_energy_j": 180.1116000000002, "max_active_contacts": 216, @@ -357,8 +357,8 @@ "max_total_energy_gain_after_settle_j": 0.0, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "66e2142f4b817762eeff5c34e35e3ad4f79a2f7fd09a3fc0f6e9513d014e7b56", - "66e2142f4b817762eeff5c34e35e3ad4f79a2f7fd09a3fc0f6e9513d014e7b56" + "2315ba41bfe626fc491f77231495ba8c523170dd7eb2e5aae3b7f9133f224e7a", + "2315ba41bfe626fc491f77231495ba8c523170dd7eb2e5aae3b7f9133f224e7a" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -408,7 +408,7 @@ "final_kinetic_energy_j": 3.674601533738957e-28, "final_max_speed_mps": 1.1914080833008711e-14, "final_min_height_m": 0.04989999999999998, - "final_state_sha256": "c73cf1f3ede55e68bb58581c948738d6471403bb75c99ad2b71f7e95f0a924dc", + "final_state_sha256": "078454f244beabf5c11719f6700c9651aa8a9adc327f5410fb5dcfacb6efe06a", "finite": true, "initial_total_energy_j": 180.1116000000002, "max_active_contacts": 216, @@ -418,8 +418,8 @@ "max_total_energy_gain_after_settle_j": 1.342411593441284e-06, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "c73cf1f3ede55e68bb58581c948738d6471403bb75c99ad2b71f7e95f0a924dc", - "c73cf1f3ede55e68bb58581c948738d6471403bb75c99ad2b71f7e95f0a924dc" + "078454f244beabf5c11719f6700c9651aa8a9adc327f5410fb5dcfacb6efe06a", + "078454f244beabf5c11719f6700c9651aa8a9adc327f5410fb5dcfacb6efe06a" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -671,8 +671,8 @@ }, "target": { "branch": "main", - "commit": "dd68714d6a7a928b3af784e08369469b0e945216", - "fetch_hint": "git fetch origin pull/3445/head && git checkout dd68714d6a7a928b3af784e08369469b0e945216" + "commit": "654193f85be94b2e5d9f45cee1093a4715b16baa", + "fetch_hint": "git fetch origin pull/3445/head && git checkout 654193f85be94b2e5d9f45cee1093a4715b16baa" }, "title": "Dense 6x6x6 inelastic contact stability (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json index f3c8a19b3f33a..6782d61048acd 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json @@ -223,7 +223,7 @@ "contact_solver_method": "SEQUENTIAL_IMPULSE", "energy_envelope_excess_j": 0.0, "envelope_set_by_initial_state": true, - "final_state_sha256": "785ee993fda9dd80b394f5964edd34844bbf69dccfac7061f673698b6c6e5d6e", + "final_state_sha256": "0fefb9f04ad592f14b99ea48690698e2f8cb86e9a91680782ed4b7138019083e", "final_total_energy_j": 63.44696118939535, "finite": true, "initial_total_energy_j": 180.1116000000002, @@ -234,8 +234,8 @@ "max_total_energy_j": 180.1116000000002, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "785ee993fda9dd80b394f5964edd34844bbf69dccfac7061f673698b6c6e5d6e", - "785ee993fda9dd80b394f5964edd34844bbf69dccfac7061f673698b6c6e5d6e" + "0fefb9f04ad592f14b99ea48690698e2f8cb86e9a91680782ed4b7138019083e", + "0fefb9f04ad592f14b99ea48690698e2f8cb86e9a91680782ed4b7138019083e" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -284,7 +284,7 @@ "contact_solver_method": "SEQUENTIAL_IMPULSE", "energy_envelope_excess_j": 0.0, "envelope_set_by_initial_state": true, - "final_state_sha256": "3e111883d46bcd87f0296a824fde812b8f026e884fae7b47072f343873d79d46", + "final_state_sha256": "dcaf261f1e5efdb9d037e72d43e6c6931d77f891079ab502d9332a0ebb5adeb0", "final_total_energy_j": 63.25766139362141, "finite": true, "initial_total_energy_j": 180.1116000000002, @@ -295,8 +295,8 @@ "max_total_energy_j": 180.1116000000002, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "3e111883d46bcd87f0296a824fde812b8f026e884fae7b47072f343873d79d46", - "3e111883d46bcd87f0296a824fde812b8f026e884fae7b47072f343873d79d46" + "dcaf261f1e5efdb9d037e72d43e6c6931d77f891079ab502d9332a0ebb5adeb0", + "dcaf261f1e5efdb9d037e72d43e6c6931d77f891079ab502d9332a0ebb5adeb0" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -345,7 +345,7 @@ "contact_solver_method": "BOXED_LCP", "energy_envelope_excess_j": 0.0, "envelope_set_by_initial_state": true, - "final_state_sha256": "31c731bd98621b0db2c88edf07cb7f3d429cc7c53b0036c199e3bcf640c49670", + "final_state_sha256": "aa8f86db2ed20270f2775810e277122c4f207c33ba74acb1ad0ed0985cfd497b", "final_total_energy_j": 63.58479412413587, "finite": true, "initial_total_energy_j": 180.1116000000002, @@ -356,8 +356,8 @@ "max_total_energy_j": 180.1116000000002, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "31c731bd98621b0db2c88edf07cb7f3d429cc7c53b0036c199e3bcf640c49670", - "31c731bd98621b0db2c88edf07cb7f3d429cc7c53b0036c199e3bcf640c49670" + "aa8f86db2ed20270f2775810e277122c4f207c33ba74acb1ad0ed0985cfd497b", + "aa8f86db2ed20270f2775810e277122c4f207c33ba74acb1ad0ed0985cfd497b" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -406,7 +406,7 @@ "contact_solver_method": "BOXED_LCP", "energy_envelope_excess_j": 0.0, "envelope_set_by_initial_state": true, - "final_state_sha256": "5fceb3e62646e056c3ba3d84a7460f672bb6f1fb1fa041178be399451837f142", + "final_state_sha256": "699fc4cb41bca1b43c3df6db2347a8b32946602e261744ee3b3ac26d5a1db06f", "final_total_energy_j": 63.746668542792605, "finite": true, "initial_total_energy_j": 180.1116000000002, @@ -417,8 +417,8 @@ "max_total_energy_j": 180.1116000000002, "raw_last_step_residual_max": 0.0, "repeat_state_sha256": [ - "5fceb3e62646e056c3ba3d84a7460f672bb6f1fb1fa041178be399451837f142", - "5fceb3e62646e056c3ba3d84a7460f672bb6f1fb1fa041178be399451837f142" + "699fc4cb41bca1b43c3df6db2347a8b32946602e261744ee3b3ac26d5a1db06f", + "699fc4cb41bca1b43c3df6db2347a8b32946602e261744ee3b3ac26d5a1db06f" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -574,8 +574,8 @@ }, "target": { "branch": "main", - "commit": "dd68714d6a7a928b3af784e08369469b0e945216", - "fetch_hint": "git fetch origin pull/3445/head && git checkout dd68714d6a7a928b3af784e08369469b0e945216" + "commit": "654193f85be94b2e5d9f45cee1093a4715b16baa", + "fetch_hint": "git fetch origin pull/3445/head && git checkout 654193f85be94b2e5d9f45cee1093a4715b16baa" }, "title": "Dense elastic contact energy envelope (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json index dcd2dd00ca4dd..01ce03269c6e4 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json @@ -391,7 +391,7 @@ { "contact_solver_method": "SEQUENTIAL_IMPULSE", "final_max_speed_mps": 0.00045895830534830895, - "final_state_sha256": "cc33a80165e5e0a58e8b10bb10869981a784e80dcae8ea4c32d968928a8f3ce6", + "final_state_sha256": "b0b12d128066d56e2f71250f7c2d313ed2435fcb1830fc3a68389f9464b1323e", "finite": true, "gap_closure_m": 2.8186383662115455e-05, "lower_sink_m": 3.8875436905949634e-05, @@ -401,8 +401,8 @@ "max_solver_iterations": 8, "relative_gap_closure": 0.00014093191831057728, "repeat_state_sha256": [ - "cc33a80165e5e0a58e8b10bb10869981a784e80dcae8ea4c32d968928a8f3ce6", - "cc33a80165e5e0a58e8b10bb10869981a784e80dcae8ea4c32d968928a8f3ce6" + "b0b12d128066d56e2f71250f7c2d313ed2435fcb1830fc3a68389f9464b1323e", + "b0b12d128066d56e2f71250f7c2d313ed2435fcb1830fc3a68389f9464b1323e" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -448,7 +448,7 @@ { "contact_solver_method": "SEQUENTIAL_IMPULSE", "final_max_speed_mps": 0.029482304187420837, - "final_state_sha256": "28c654f5bbda220b08138deccc24b3b71d316c85e528915cccefc2ee5e9f4ffb", + "final_state_sha256": "604e554951a6265fb4c98b8e79d72af6ee5bf2379b9ebfff7a86aa9ca49b63ac", "finite": true, "gap_closure_m": 0.00020330330581502798, "lower_sink_m": 0.0003313631340894213, @@ -458,8 +458,8 @@ "max_solver_iterations": 8, "relative_gap_closure": 0.00101651652907514, "repeat_state_sha256": [ - "28c654f5bbda220b08138deccc24b3b71d316c85e528915cccefc2ee5e9f4ffb", - "28c654f5bbda220b08138deccc24b3b71d316c85e528915cccefc2ee5e9f4ffb" + "604e554951a6265fb4c98b8e79d72af6ee5bf2379b9ebfff7a86aa9ca49b63ac", + "604e554951a6265fb4c98b8e79d72af6ee5bf2379b9ebfff7a86aa9ca49b63ac" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -505,7 +505,7 @@ { "contact_solver_method": "SEQUENTIAL_IMPULSE", "final_max_speed_mps": 4.830545337362943e-07, - "final_state_sha256": "88751ed0d540613946ff1961b182136cb8d80977d021060e10cd499dbf341975", + "final_state_sha256": "b01e5cf5b6aea2120746a5a005e6743cdc14ebd61849db3e5c7cb0bac1201194", "finite": true, "gap_closure_m": 0.2000161894973165, "lower_sink_m": 2.403341138421111e-06, @@ -515,8 +515,8 @@ "max_solver_iterations": 8, "relative_gap_closure": 1.0000809474865824, "repeat_state_sha256": [ - "88751ed0d540613946ff1961b182136cb8d80977d021060e10cd499dbf341975", - "88751ed0d540613946ff1961b182136cb8d80977d021060e10cd499dbf341975" + "b01e5cf5b6aea2120746a5a005e6743cdc14ebd61849db3e5c7cb0bac1201194", + "b01e5cf5b6aea2120746a5a005e6743cdc14ebd61849db3e5c7cb0bac1201194" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -562,7 +562,7 @@ { "contact_solver_method": "SEQUENTIAL_IMPULSE", "final_max_speed_mps": 3.542993305133921e-06, - "final_state_sha256": "c7eb04fb30595972cdd4a82be8086c27520aad419e02a22e8fb92b2484dc8c0b", + "final_state_sha256": "cbe313663ff3a086901bdb6629b6c5e69430cde7c6c0a953064516e9a35c4bf1", "finite": true, "gap_closure_m": 0.2000041526819762, "lower_sink_m": 5.348353172895948e-06, @@ -572,8 +572,8 @@ "max_solver_iterations": 8, "relative_gap_closure": 1.0000207634098808, "repeat_state_sha256": [ - "c7eb04fb30595972cdd4a82be8086c27520aad419e02a22e8fb92b2484dc8c0b", - "c7eb04fb30595972cdd4a82be8086c27520aad419e02a22e8fb92b2484dc8c0b" + "cbe313663ff3a086901bdb6629b6c5e69430cde7c6c0a953064516e9a35c4bf1", + "cbe313663ff3a086901bdb6629b6c5e69430cde7c6c0a953064516e9a35c4bf1" ], "resolved": { "contact_solver_method": "SEQUENTIAL_IMPULSE", @@ -619,7 +619,7 @@ { "contact_solver_method": "BOXED_LCP", "final_max_speed_mps": 1.4938402596549068e-11, - "final_state_sha256": "0b1b397ea8cc02c519448933ba052996198f4ec818306633b4854203cf538c80", + "final_state_sha256": "0f3e93ee597d0900b1752ef9c7196d7fe859aa292ee4b7bc1cb3b13742641e67", "finite": true, "gap_closure_m": 3.925327088483144e-05, "lower_sink_m": 1.3457496714358586e-06, @@ -629,8 +629,8 @@ "max_solver_iterations": 0, "relative_gap_closure": 0.00019626635442415719, "repeat_state_sha256": [ - "0b1b397ea8cc02c519448933ba052996198f4ec818306633b4854203cf538c80", - "0b1b397ea8cc02c519448933ba052996198f4ec818306633b4854203cf538c80" + "0f3e93ee597d0900b1752ef9c7196d7fe859aa292ee4b7bc1cb3b13742641e67", + "0f3e93ee597d0900b1752ef9c7196d7fe859aa292ee4b7bc1cb3b13742641e67" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -676,7 +676,7 @@ { "contact_solver_method": "BOXED_LCP", "final_max_speed_mps": 8.665822654413277e-06, - "final_state_sha256": "98b2aee62fa0aff364ad8cf622c02a0ebf5e2fcbed9ee01043d9c233f529b471", + "final_state_sha256": "27327611123e8f800304c77e0b78104c79e2098eee69068eddd2749870f0ccef", "finite": true, "gap_closure_m": 3.957920874031462e-05, "lower_sink_m": 3.093141870136318e-07, @@ -686,8 +686,8 @@ "max_solver_iterations": 0, "relative_gap_closure": 0.0001978960437015731, "repeat_state_sha256": [ - "98b2aee62fa0aff364ad8cf622c02a0ebf5e2fcbed9ee01043d9c233f529b471", - "98b2aee62fa0aff364ad8cf622c02a0ebf5e2fcbed9ee01043d9c233f529b471" + "27327611123e8f800304c77e0b78104c79e2098eee69068eddd2749870f0ccef", + "27327611123e8f800304c77e0b78104c79e2098eee69068eddd2749870f0ccef" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -733,7 +733,7 @@ { "contact_solver_method": "BOXED_LCP", "final_max_speed_mps": 0.0008169729485216048, - "final_state_sha256": "b78d23a138a9a9e2c7786b9a318cb49e4f0ab61504005ff3cd589b3bcd93240c", + "final_state_sha256": "2089a6b074450753cba04f55015c4c8c544a5b60ae79b8d3cd944573cff51bab", "finite": true, "gap_closure_m": 6.527479483597887e-06, "lower_sink_m": 3.401292682585211e-05, @@ -743,8 +743,8 @@ "max_solver_iterations": 0, "relative_gap_closure": 3.263739741798943e-05, "repeat_state_sha256": [ - "b78d23a138a9a9e2c7786b9a318cb49e4f0ab61504005ff3cd589b3bcd93240c", - "b78d23a138a9a9e2c7786b9a318cb49e4f0ab61504005ff3cd589b3bcd93240c" + "2089a6b074450753cba04f55015c4c8c544a5b60ae79b8d3cd944573cff51bab", + "2089a6b074450753cba04f55015c4c8c544a5b60ae79b8d3cd944573cff51bab" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -790,7 +790,7 @@ { "contact_solver_method": "BOXED_LCP", "final_max_speed_mps": 0.041688097943890993, - "final_state_sha256": "89fd0ea4211562eedc94ea2c17363c8d48e1f8c8b7b0a2ab8c8ad4a3537ff2f7", + "final_state_sha256": "15628287ee93ea8632372d54bb0aefe8fb90e2c60c4b4f9a859ed028c4bfcabc", "finite": true, "gap_closure_m": -6.027789048593246e-05, "lower_sink_m": 0.00011855439148550362, @@ -800,8 +800,8 @@ "max_solver_iterations": 0, "relative_gap_closure": -0.0003013894524296623, "repeat_state_sha256": [ - "89fd0ea4211562eedc94ea2c17363c8d48e1f8c8b7b0a2ab8c8ad4a3537ff2f7", - "89fd0ea4211562eedc94ea2c17363c8d48e1f8c8b7b0a2ab8c8ad4a3537ff2f7" + "15628287ee93ea8632372d54bb0aefe8fb90e2c60c4b4f9a859ed028c4bfcabc", + "15628287ee93ea8632372d54bb0aefe8fb90e2c60c4b4f9a859ed028c4bfcabc" ], "resolved": { "contact_solver_method": "BOXED_LCP", @@ -1007,9 +1007,9 @@ }, "target": { "branch": "main", - "commit": "dd68714d6a7a928b3af784e08369469b0e945216", + "commit": "654193f85be94b2e5d9f45cee1093a4715b16baa", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", - "fetch_hint": "git fetch origin pull/3445/head && git checkout dd68714d6a7a928b3af784e08369469b0e945216" + "fetch_hint": "git fetch origin pull/3445/head && git checkout 654193f85be94b2e5d9f45cee1093a4715b16baa" }, "title": "High-mass-ratio stack conditioning baseline for the exact-cone GO/NO-GO (DART 7)" } From 08afdd72802726ecab988ea01b2381c99d08f315 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 12:02:04 -0700 Subject: [PATCH 40/52] Close the Codex round-15 findings Overlapping sweep points must be matched to DISTINCT rows -- a backtracking system-of-distinct-representatives search replaces the independent any() matching that one row could satisfy twice; NPY arrays with a zero dimension or header-only payloads are rejected; NPZ members ending in .npy are themselves parsed as valid non-empty arrays rather than trusted by name. Tests: 134 -> 137 validator cases. --- .../verification.md | 10 +++ scripts/check_citation_evidence.py | 61 +++++++++++++++---- tests/test_check_citation_evidence.py | 57 +++++++++++++++++ 3 files changed, 117 insertions(+), 11 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index e4f934f9532d3..e556d57851449 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -729,3 +729,13 @@ packets are regenerated at the round-14 commit. From #3444: evidence/raw/ is now reserved for raw artifacts — excluded from packet discovery, and lanes may not reference into it. Tests: 133 -> 134 (main), 113 -> 114 (6.20). + +## Codex review round 15 (PR #3445) — 2026-08-16 + +Three findings, all fixed on both branches: overlapping sweep points +must be matched to DISTINCT rows (a backtracking system-of-distinct- +representatives search replaces the independent any() matching one row +could satisfy twice); NPY arrays with a zero dimension or header-only +payloads are rejected; NPZ members ending in .npy are themselves parsed +as valid non-empty arrays rather than trusted by name. Tests: 134 -> 137 +(main), 114 -> 117 (6.20). diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index a9d7e55ba4629..1db148d5680c1 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -83,6 +83,26 @@ ) +def _distinct_assignment_exists(match_sets: "list[set[int]]") -> bool: + """Backtracking search for a system of distinct representatives: every + declared point must map to its OWN row.""" + ordered = sorted(match_sets, key=len) + used: set[int] = set() + + def assign(index: int) -> bool: + if index == len(ordered): + return True + for row_index in ordered[index]: + if row_index not in used: + used.add(row_index) + if assign(index + 1): + return True + used.remove(row_index) + return False + + return assign(0) + + def _has_measurement_leaf(value: object) -> bool: """True when the value contains a numeric/boolean leaf beyond bookkeeping keys (the raw-rows measurement rule, applied to parsed @@ -410,9 +430,14 @@ def _raw_data_content_issue(path: "Path") -> "str | None": with path.open("rb") as stream: archive = zipfile.ZipFile(io.BytesIO(stream.read())) names = archive.namelist() - except OSError, zipfile.BadZipFile: - return mismatch - if not names or not any(name.endswith(".npy") for name in names): + npy_members = [name for name in names if name.endswith(".npy")] + if not npy_members: + return mismatch + for name in npy_members: + member_bytes = archive.read(name) + if _npy_header_issue(member_bytes, mismatch) is not None: + return mismatch + except OSError, zipfile.BadZipFile, KeyError: return mismatch return None if suffix == ".parquet": @@ -458,11 +483,14 @@ def _npy_header_issue(head: bytes, mismatch: str) -> "str | None": and {"descr", "shape", "fortran_order"} <= set(header) and isinstance(header["shape"], tuple) and all( - isinstance(dim, int) and not isinstance(dim, bool) and dim >= 0 + isinstance(dim, int) and not isinstance(dim, bool) and dim >= 1 for dim in header["shape"] ) ): return mismatch + if len(head) <= header_start + header_len: + # Header-only file: a declared array with no payload bytes. + return mismatch return None @@ -1073,20 +1101,31 @@ def packet_errors( # OBSERVED by a row carrying its exact coordinates, or one point's # measurements could be repeated to stand in for the others. if has_rows and isinstance(sweep, list): - for point in sweep: - if not (isinstance(point, dict) and point): - continue - if not any( - isinstance(row, dict) + object_points = [ + point for point in sweep if isinstance(point, dict) and point + ] + match_sets = [] + for point in object_points: + matches = { + index + for index, row in enumerate(raw_rows) + if isinstance(row, dict) and all(row.get(key) == value for key, value in point.items()) - for row in raw_rows - ): + } + if not matches: errors.append( f"ensemble.sweep point {point} has no matching row " "in evidence.raw_rows recording those coordinates; " "a declared configuration without an observation is " "not swept" ) + match_sets.append(matches) + if all(match_sets) and not _distinct_assignment_exists(match_sets): + errors.append( + "ensemble.sweep points cannot be matched to DISTINCT " + "rows; overlapping coordinates must each have their own " + "recorded observation" + ) if has_rows and isinstance(seeds, list): for seed in seeds: if not ( diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index d83ece6ef1dd5..de3c7fafbfcf3 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -1471,3 +1471,60 @@ def test_raw_artifacts_under_evidence_raw_are_not_packets(tmp_path): ) errors = MODULE.validate_tree(tree_dir) assert any("holds raw artifacts, not packets" in error for error in errors) + + +def test_overlapping_sweep_points_need_distinct_rows(): + packet = complete_packet() + del packet["ensemble"]["deterministic_repeats"] + del packet["ensemble"]["deterministic_repeats_identical"] + packet["ensemble"]["sweep"] = [ + {"angle_deg": 0.0}, + {"angle_deg": 0.0, "detector": "fcl"}, + ] + packet["evidence"]["raw_rows"] = [ + { + "angle_deg": 0.0, + "detector": "fcl", + "lateral_drift_m": 0.1, + "trajectory_sha256": "d" * 64, + }, + {"unrelated_metric": 1.0, "trajectory_sha256": "e" * 64}, + ] + errors = MODULE.packet_errors(packet) + assert any("DISTINCT" in error and "own" in error for error in errors) + + +def test_empty_npy_arrays_fail(tmp_path): + header = b"{'descr': ' Date: Sun, 16 Aug 2026 12:18:15 -0700 Subject: [PATCH 41/52] Close the Codex round-16 findings Reproduction command lists begin with 'pixi run build' (writers emit it; existing packets migrated in place, content-equal to future regenerations); measured performance is forbidden while host.performance_valid is false; parquet stubs are rejected via footer metadata-length sanity; declared seeds need DISTINCT observed rows; time windows must be non-empty intervals with non-negative start; CT-007 gains a whole-stack fall-through criterion (lower-box sink beyond a half extent counts as failure -- current cells pass by three orders of margin, so the criterion guards future runs). Tests: 137 -> 141 validator cases. CT-007 is regenerated at this commit in the follow-up commit. --- .../verification.md | 20 +++++++ .../CT-001-dart7-rolling-direction.json | 1 + .../CT-002-dart7-dense-inelastic-contact.json | 1 + .../CT-003-dart7-dense-elastic-contact.json | 1 + ...004-dart7-articulated-energy-momentum.json | 1 + .../evidence/CT-005-dart7-pd-tracking.json | 1 + .../CT-007-dart7-high-mass-ratio.json | 1 + .../CT-011-dart7-restore-equivalence.json | 1 + scripts/check_citation_evidence.py | 55 ++++++++++++++++-- ...citation_ct001_rolling_direction_packet.py | 2 +- ...ite_citation_ct002_dense_contact_packet.py | 2 +- ...e_citation_ct003_elastic_contact_packet.py | 2 +- ...itation_ct004_articulated_energy_packet.py | 2 +- ...write_citation_ct005_pd_tracking_packet.py | 2 +- ...e_citation_ct007_high_mass_ratio_packet.py | 9 ++- ...tation_ct011_restore_equivalence_packet.py | 2 +- tests/test_check_citation_evidence.py | 56 +++++++++++++++++++ 17 files changed, 146 insertions(+), 13 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index e556d57851449..6faa6c0f5bcb5 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -739,3 +739,23 @@ could satisfy twice); NPY arrays with a zero dimension or header-only payloads are rejected; NPZ members ending in .npy are themselves parsed as valid non-empty arrays rather than trusted by name. Tests: 134 -> 137 (main), 114 -> 117 (6.20). + +## Codex review round 16 (PR #3445) — 2026-08-16 + +Seven findings across both PRs (the 3445 review was cut against the +pre-round-15 head after a worktree slip delayed that push; its findings +were all fresh regardless), all fixed: reproduction command lists now +begin with `pixi run build` (writers emit it; existing packets migrated +in place, content-equal); measured performance is forbidden while +host.performance_valid is false; parquet stubs are rejected via footer +metadata-length sanity; declared seeds need DISTINCT observed rows +(same system-of-distinct-representatives as sweep points); time windows +must be non-empty intervals with non-negative start; CT-007 gains a +whole-stack fall-through criterion (lower-box sink beyond a half +extent counts as failure — current cells all pass by 3 orders of +margin, so the pattern is unchanged and the criterion guards future +runs); and the 6.20 design doc's packet-contract list is aligned with +the enforced schema (model/license provenance applies when external +assets are used; procedural scenes carry their construction in +scene.parameters). CT-007 regenerated at the round-16 commit. Tests: +137 -> 141 (main), 117 -> 121 (6.20). diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json index 7b0325697116b..771bb981e3516 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json @@ -173,6 +173,7 @@ }, "evidence": { "commands": [ + "pixi run build", "PYTHONPATH=build/default/cpp/Release/python pixi run python scripts/write_citation_ct001_rolling_direction_packet.py" ], "raw_rows": [ diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json index 54692076cddf9..a7101c9c6060e 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-002-dart7-dense-inelastic-contact.json @@ -217,6 +217,7 @@ }, "evidence": { "commands": [ + "pixi run build", "PYTHONPATH=build/default/cpp/Release/python pixi run python scripts/write_citation_ct002_dense_contact_packet.py" ], "raw_rows": [ diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json index 6782d61048acd..cfe92fd17715b 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-003-dart7-dense-elastic-contact.json @@ -216,6 +216,7 @@ }, "evidence": { "commands": [ + "pixi run build", "PYTHONPATH=build/default/cpp/Release/python pixi run python scripts/write_citation_ct003_elastic_contact_packet.py" ], "raw_rows": [ diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json index d9413312ffc84..b2d8da9add3f7 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json @@ -394,6 +394,7 @@ }, "evidence": { "commands": [ + "pixi run build", "PYTHONPATH=build/default/cpp/Release/python pixi run python scripts/write_citation_ct004_articulated_energy_packet.py" ], "raw_rows": [ diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json index b10e62ec8e37e..a26fc95576d6c 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-005-dart7-pd-tracking.json @@ -386,6 +386,7 @@ }, "evidence": { "commands": [ + "pixi run build", "PYTHONPATH=build/default/cpp/Release/python pixi run python scripts/write_citation_ct005_pd_tracking_packet.py" ], "raw_rows": [ diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json index 01ce03269c6e4..316b6f9254b3b 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json @@ -385,6 +385,7 @@ }, "evidence": { "commands": [ + "pixi run build", "PYTHONPATH=build/default/cpp/Release/python pixi run python scripts/write_citation_ct007_high_mass_ratio_packet.py" ], "raw_rows": [ diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json index a708f3886ada1..41931b9516414 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json @@ -109,6 +109,7 @@ }, "evidence": { "commands": [ + "pixi run build", "PYTHONPATH=build/default/cpp/Release/python pixi run python scripts/write_citation_ct011_restore_equivalence_packet.py" ], "raw_rows": [ diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 1db148d5680c1..b7c87b8f42f7c 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -449,6 +449,16 @@ def _raw_data_content_issue(path: "Path") -> "str | None": return mismatch if head[:4] != b"PAR1" or tail4 != b"PAR1": return mismatch + try: + with path.open("rb") as stream: + stream.seek(-8, 2) + footer = stream.read(8) + except OSError: + return mismatch + footer_len = int.from_bytes(footer[:4], "little") + file_size = path.stat().st_size + if not (0 < footer_len <= file_size - 12): + return mismatch return None return None @@ -1018,9 +1028,16 @@ def packet_errors( ) start = window.get("start_s") end = window.get("end_s") - if _is_finite_number(start) and _is_finite_number(end) and start > end: + if ( + _is_finite_number(start) + and _is_finite_number(end) + and (start < 0 or end <= start) + ): errors.append( - "ensemble.measurement_window start_s must not exceed end_s" + "ensemble.measurement_window must be a non-empty " + "interval with start_s >= 0 and end_s > start_s; a " + "zero-duration or negative window contains no " + "measurements" ) has_time_bounds = {"start_s", "end_s"} <= set(window) has_step_bounds = {"warmup_steps", "continuation_steps"} <= set(window) @@ -1127,24 +1144,35 @@ def packet_errors( "recorded observation" ) if has_rows and isinstance(seeds, list): + seed_match_sets = [] for seed in seeds: if not ( (isinstance(seed, int) and not isinstance(seed, bool)) or _is_nonempty_str(seed) ): continue - if not any( - isinstance(row, dict) + matches = { + index + for index, row in enumerate(raw_rows) + if isinstance(row, dict) and any( "seed" in str(key).lower() and row[key] == seed for key in row ) - for row in raw_rows - ): + } + if not matches: errors.append( f"ensemble seed {seed!r} has no row recording it " "under a seed field; a declared seed without an " "observation is not an ensemble member" ) + seed_match_sets.append(matches) + if all(seed_match_sets) and not _distinct_assignment_exists( + seed_match_sets + ): + errors.append( + "ensemble seeds cannot be matched to DISTINCT rows; " + "every declared seed needs its own recorded observation" + ) if not (has_paths or has_rows): errors.append("evidence must carry raw_rows inline or non-empty raw_paths") if (has_sweep or has_seeds) and not has_rows: @@ -1329,6 +1357,21 @@ def packet_errors( "or allocation numbers from an uncontrolled host must not " "pass silently" ) + elif host.get("performance_valid") is False: + performance_group = ( + packet.get("metrics", {}).get("performance") + if isinstance(packet.get("metrics"), dict) + else None + ) + if ( + isinstance(performance_group, dict) + and performance_group.get("status") != "unsupported" + ): + errors.append( + "metrics.performance publishes measurements while " + "host.performance_valid is false; timing from an " + "uncontrolled host must be typed unsupported" + ) review = packet.get("review") if not isinstance(review, dict): diff --git a/scripts/write_citation_ct001_rolling_direction_packet.py b/scripts/write_citation_ct001_rolling_direction_packet.py index 37b1292ff3639..c87cb646203dc 100644 --- a/scripts/write_citation_ct001_rolling_direction_packet.py +++ b/scripts/write_citation_ct001_rolling_direction_packet.py @@ -665,7 +665,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: }, }, "evidence": { - "commands": [command], + "commands": ["pixi run build", command], "raw_rows": rows, "visual": { "status": "not-applicable", diff --git a/scripts/write_citation_ct002_dense_contact_packet.py b/scripts/write_citation_ct002_dense_contact_packet.py index 4b742d5da3d0a..28ce2ed58193a 100644 --- a/scripts/write_citation_ct002_dense_contact_packet.py +++ b/scripts/write_citation_ct002_dense_contact_packet.py @@ -620,7 +620,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: }, }, "evidence": { - "commands": [command], + "commands": ["pixi run build", command], "raw_rows": rows, "visual": { "status": "not-applicable", diff --git a/scripts/write_citation_ct003_elastic_contact_packet.py b/scripts/write_citation_ct003_elastic_contact_packet.py index 0666cb16e4c98..b951489a3f7c9 100644 --- a/scripts/write_citation_ct003_elastic_contact_packet.py +++ b/scripts/write_citation_ct003_elastic_contact_packet.py @@ -498,7 +498,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: }, }, "evidence": { - "commands": [command], + "commands": ["pixi run build", command], "raw_rows": rows, "visual": { "status": "not-applicable", diff --git a/scripts/write_citation_ct004_articulated_energy_packet.py b/scripts/write_citation_ct004_articulated_energy_packet.py index 0b87c89780ca3..1e19c053ff942 100644 --- a/scripts/write_citation_ct004_articulated_energy_packet.py +++ b/scripts/write_citation_ct004_articulated_energy_packet.py @@ -507,7 +507,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: }, }, "evidence": { - "commands": [command], + "commands": ["pixi run build", command], "raw_rows": rows, "visual": { "status": "not-applicable", diff --git a/scripts/write_citation_ct005_pd_tracking_packet.py b/scripts/write_citation_ct005_pd_tracking_packet.py index 3aab41ba34ed9..fa5cf63607691 100644 --- a/scripts/write_citation_ct005_pd_tracking_packet.py +++ b/scripts/write_citation_ct005_pd_tracking_packet.py @@ -603,7 +603,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: }, }, "evidence": { - "commands": [command], + "commands": ["pixi run build", command], "raw_rows": rows, "visual": { "status": "not-applicable", diff --git a/scripts/write_citation_ct007_high_mass_ratio_packet.py b/scripts/write_citation_ct007_high_mass_ratio_packet.py index c36aad4823f1b..ece53c0239a4d 100644 --- a/scripts/write_citation_ct007_high_mass_ratio_packet.py +++ b/scripts/write_citation_ct007_high_mass_ratio_packet.py @@ -366,6 +366,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: # A stack whose boxes close more than a tenth of a box height has not been # held apart in any useful sense. collapse_threshold = 0.1 + half = float(parameters["box_half_extent_m"]) degraded_cells = [ { "contact_solver_method": row["contact_solver_method"], @@ -376,6 +377,9 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: for row in rows if not row["finite"] or (row["relative_gap_closure"] or 0.0) > collapse_threshold + # Whole-stack fall-through keeps the gap closure near zero while + # both boxes leave the ground; the sink of the LOWER box catches it. + or (row["lower_sink_m"] or 0.0) > half ] # Characterize each method's failure onset: the smallest swept ratio at @@ -389,6 +393,9 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: and ( not row["finite"] or (row["relative_gap_closure"] or 0.0) > collapse_threshold + # Whole-stack fall-through keeps the gap closure near zero while + # both boxes leave the ground; the sink of the LOWER box catches it. + or (row["lower_sink_m"] or 0.0) > half ) ) baseline_finding[method] = { @@ -585,7 +592,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: }, }, "evidence": { - "commands": [command], + "commands": ["pixi run build", command], "raw_rows": rows, "visual": { "status": "not-applicable", diff --git a/scripts/write_citation_ct011_restore_equivalence_packet.py b/scripts/write_citation_ct011_restore_equivalence_packet.py index 89b1ac95a1d38..b22bb8f503bfd 100644 --- a/scripts/write_citation_ct011_restore_equivalence_packet.py +++ b/scripts/write_citation_ct011_restore_equivalence_packet.py @@ -575,7 +575,7 @@ def build_packet(output_path: Path | None = None) -> dict[str, Any]: }, }, "evidence": { - "commands": [command], + "commands": ["pixi run build", command], "raw_rows": rows, "visual": { "status": "not-applicable", diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index de3c7fafbfcf3..0636afcf06857 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -1528,3 +1528,59 @@ def test_npz_members_must_be_valid_arrays(tmp_path): assert any( "does not parse as its claimed raw-data format" in error for error in errors ) + + +def test_seeds_need_distinct_rows(): + packet = complete_packet() + del packet["ensemble"]["deterministic_repeats"] + del packet["ensemble"]["deterministic_repeats_identical"] + packet["ensemble"]["seeds"] = [7, 11] + packet["evidence"]["raw_rows"] = [ + { + "seed": 7, + "other_seed": 11, + "metric": 1.0, + "trajectory_sha256": "d" * 64, + }, + {"metric": 2.0, "trajectory_sha256": "e" * 64}, + ] + errors = MODULE.packet_errors(packet) + assert any("DISTINCT rows" in error for error in errors) + + +def test_zero_duration_windows_fail(): + for bad in ( + {"start_s": 1.0, "end_s": 1.0}, + {"start_s": -2.0, "end_s": -1.0}, + ): + packet = complete_packet() + packet["ensemble"]["measurement_window"] = bad + errors = MODULE.packet_errors(packet) + assert any( + "non-empty" in error and "interval" in error for error in errors + ), bad + + +def test_invalid_host_forbids_measured_performance(): + packet = complete_packet() + packet["metrics"]["performance"] = { + "method": "wall clock", + "step_time_ms": 1.2, + } + errors = MODULE.packet_errors(packet) + assert any("uncontrolled host" in error for error in errors) + + +def test_parquet_stub_fails(tmp_path): + (tmp_path / "rows.parquet").write_bytes(b"PAR1PAR1") + packet = complete_packet() + del packet["evidence"]["raw_rows"] + packet["evidence"]["raw_paths"] = ["rows.parquet"] + packet["evidence"]["artifact_digests"] = { + "rows.parquet": "sha256:" + + hashlib.sha256((tmp_path / "rows.parquet").read_bytes()).hexdigest() + } + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any( + "does not parse as its claimed raw-data format" in error for error in errors + ) From 962b49bf4c3d648d5d8a59d8f24d431dbef3db54 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 12:19:20 -0700 Subject: [PATCH 42/52] Regenerate CT-007 under the fall-through criterion The hold/fail pattern is unchanged (sequential impulse fails at mass ratios 100 and 1000, boxed-LCP holds throughout) and every cell passes the new whole-stack fall-through check by three orders of margin. --- .../evidence/CT-007-dart7-high-mass-ratio.json | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json index 316b6f9254b3b..2798bfb239888 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-007-dart7-high-mass-ratio.json @@ -1008,9 +1008,9 @@ }, "target": { "branch": "main", - "commit": "654193f85be94b2e5d9f45cee1093a4715b16baa", + "commit": "4f7f28b9eabc2e2a4a05b16ae77905d4fd0716c1", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", - "fetch_hint": "git fetch origin pull/3445/head && git checkout 654193f85be94b2e5d9f45cee1093a4715b16baa" + "fetch_hint": "git fetch origin pull/3445/head && git checkout 4f7f28b9eabc2e2a4a05b16ae77905d4fd0716c1" }, "title": "High-mass-ratio stack conditioning baseline for the exact-cone GO/NO-GO (DART 7)" } From 4272dd17dc4991eadff5dc0c5d19316c5000acd2 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 12:39:40 -0700 Subject: [PATCH 43/52] Close the Codex round-17 findings The validator now requires a recognized build step in evidence.commands (writer-side emission alone could not stop a future packet from omitting it); NPY descr dtypes are restricted to numeric/boolean/complex -- string and object arrays carry no measurement, structured record dtypes are a recorded boundary; a measured-with-method group whose every quantity is a typed-unsupported marker is rejected: an unmeasured group must be typed unsupported as a WHOLE, not dressed as measured. Tests: 141 -> 143 validator cases. --- .../verification.md | 12 ++++++ scripts/check_citation_evidence.py | 28 ++++++++++++ tests/test_check_citation_evidence.py | 43 ++++++++++++++++++- 3 files changed, 81 insertions(+), 2 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 6faa6c0f5bcb5..171b76206ee25 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -759,3 +759,15 @@ the enforced schema (model/license provenance applies when external assets are used; procedural scenes carry their construction in scene.parameters). CT-007 regenerated at the round-16 commit. Tests: 137 -> 141 (main), 117 -> 121 (6.20). + +## Codex review round 17 (PR #3445) — 2026-08-16 + +Three findings, all fixed on both branches: the validator now requires +a recognized build step in evidence.commands (the writer-side emission +alone could not stop a future packet from omitting it); NPY descr +dtypes are restricted to numeric/boolean/complex (string and object +arrays carry no measurement; structured record dtypes are a recorded +boundary); and a measured-with-method group whose every quantity is a +typed-unsupported marker is rejected — an unmeasured group must be +typed unsupported as a WHOLE, not dressed as measured. Tests: +141 -> 143 (main), 121 -> 123 (6.20). diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index b7c87b8f42f7c..3770bc115653a 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -496,7 +496,11 @@ def _npy_header_issue(head: bytes, mismatch: str) -> "str | None": isinstance(dim, int) and not isinstance(dim, bool) and dim >= 1 for dim in header["shape"] ) + and isinstance(header["descr"], str) + and re.match(r"^[<>|=]?[bifuc][0-9]+$", header["descr"]) ): + # String/object/structured dtypes carry no numeric measurement; + # structured record dtypes are a recorded boundary. return mismatch if len(head) <= header_start + header_len: # Header-only file: a declared array with no payload bytes. @@ -675,6 +679,7 @@ def metric_group_errors(name: str, group: object) -> list[str]: observed_zero_fields: list[str] = [] leaf_count = 0 measurement_leaves = 0 + real_measurement_leaves = 0 for key in value_keys: leaves = _metric_leaves(group[key], key) leaf_count += len(leaves) @@ -718,6 +723,7 @@ def metric_group_errors(name: str, group: object) -> list[str]: ) elif isinstance(leaf, (int, float)) and not isinstance(leaf, bool): measurement_leaves += 1 + real_measurement_leaves += 1 if not math.isfinite(leaf): errors.append(f"metrics.{name}.{path} contains a non-finite number") elif leaf == 0: @@ -728,6 +734,7 @@ def metric_group_errors(name: str, group: object) -> list[str]: # here keeps the type chain exhaustive so nothing falls # through unvalidated. measurement_leaves += 1 + real_measurement_leaves += 1 else: errors.append( f"metrics.{name}.{path} has unrecognized leaf type " @@ -742,6 +749,18 @@ def metric_group_errors(name: str, group: object) -> list[str]: "group needs at least one numeric/boolean measurement or " "typed-unsupported marker" ) + if ( + value_keys + and leaf_count > 0 + and real_measurement_leaves == 0 + and (measurement_leaves > 0) + ): + errors.append( + f"metrics.{name} records a method but every quantity is typed " + "unsupported; type the WHOLE group " + "{'status': 'unsupported', 'reason': ...} instead of dressing " + "an unmeasured group as measured-with-method" + ) if value_keys and leaf_count == 0: errors.append( f"metrics.{name} has a method but only empty containers; that is " @@ -1104,6 +1123,15 @@ def packet_errors( "shell strings are not the promised reproduction " "path" ) + if not any( + re.match(r"^pixi run build\b", command.strip()) for command in commands + ): + errors.append( + "evidence.commands must include the build step " + "('pixi run build ...') before execution; a clean " + "checkout from target.fetch_hint has no artifacts, and " + "an existing checkout may hold stale ones" + ) raw_paths = evidence.get("raw_paths") raw_rows = evidence.get("raw_rows") has_paths = isinstance(raw_paths, list) and bool(raw_paths) diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index 0636afcf06857..dbb3f43e2213b 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -90,7 +90,10 @@ def complete_packet() -> dict: }, }, "evidence": { - "commands": ["pixi run python scripts/example.py"], + "commands": [ + "pixi run build", + "pixi run python scripts/example.py", + ], "raw_rows": [ { "angle_deg": 0.0, @@ -1121,9 +1124,15 @@ def test_commands_must_be_the_reproducible_pixi_form(): assert any("reproducible repository form" in error for error in errors), bad packet = complete_packet() packet["evidence"]["commands"] = [ - "PYTHONPATH=build/x pixi run python scripts/write.py" + "pixi run build", + "PYTHONPATH=build/x pixi run python scripts/write.py", ] assert MODULE.packet_errors(packet) == [] + packet["evidence"]["commands"] = [ + "PYTHONPATH=build/x pixi run python scripts/write.py" + ] + errors = MODULE.packet_errors(packet) + assert any("must include the build step" in error for error in errors) def test_bookkeeping_only_rows_fail(): @@ -1584,3 +1593,33 @@ def test_parquet_stub_fails(tmp_path): assert any( "does not parse as its claimed raw-data format" in error for error in errors ) + + +def test_all_unsupported_measured_groups_fail(): + packet = complete_packet() + packet["metrics"]["numerical"] = { + "method": "not actually measured", + "value": {"status": "unsupported", "reason": "instrument absent"}, + } + errors = MODULE.packet_errors(packet) + assert any("dressing" in error for error in errors) + + +def test_string_dtype_npy_fails(tmp_path): + header = b"{'descr': ' Date: Sun, 16 Aug 2026 12:55:04 -0700 Subject: [PATCH 44/52] Close the Codex round-18 findings One genuine documentation bug fixed: the dashboard overclaimed that all six first-wave families have a main packet -- CT-006 heel-strike has none (audit-required, gated on WS3); corrected to five of six. Validator: allocation measurements also require a performance-valid host; sweep points may not declare null coordinates (absent row fields would spuriously match them); seed fields match by whole token; the build step must PRECEDE the evidence command. Tests: 143 -> 147 validator cases. --- .../verification.md | 13 +++++ docs/plans/dashboard.md | 5 +- scripts/check_citation_evidence.py | 53 ++++++++++++++++--- tests/test_check_citation_evidence.py | 43 +++++++++++++++ 4 files changed, 106 insertions(+), 8 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 171b76206ee25..ef0669abca702 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -771,3 +771,16 @@ boundary); and a measured-with-method group whose every quantity is a typed-unsupported marker is rejected — an unmeasured group must be typed unsupported as a WHOLE, not dressed as measured. Tests: 141 -> 143 (main), 121 -> 123 (6.20). + +## Codex review round 18 (PR #3445) — 2026-08-16 + +Five findings, one a genuine documentation bug, all fixed: the +dashboard overclaimed that all six first-wave families have a `main` +packet — CT-006 heel-strike has none (audit-required, gated on WS3); +corrected to five of six. Validator: allocation measurements also +require a performance-valid host (the round-16 rule covered only +timing); sweep points may not declare null coordinates (absent row +fields would spuriously match them); seed fields match by whole token +(`unseeded_metric` no longer counts); the build step must PRECEDE the +evidence command, not merely appear somewhere in the list. Tests: +143 -> 147 (main), 123 -> 127 (6.20). diff --git a/docs/plans/dashboard.md b/docs/plans/dashboard.md index c54ac4708ce95..e32ac56fda858 100644 --- a/docs/plans/dashboard.md +++ b/docs/plans/dashboard.md @@ -51,8 +51,9 @@ its own line so status updates remain git-history friendly. - Status: Active - Horizon: Now - Dimension: Algorithm extensibility -- Next step: all six first-wave families now have at least one `main` - packet (plus CT-001 on `release-6.20`). Two findings need maintainer +- Next step: five of the six first-wave families have a `main` packet + (plus CT-001 on `release-6.20`); CT-006 heel-strike stays open, gated + on WS3 contact semantics. Two findings need maintainer decisions: CT-007 (the default SEQUENTIAL_IMPULSE solver lets a heavy box sink fully through a light one at mass ratios >= 100 while BOXED_LCP holds) and CT-011 (state-vector restore is not a function of the restored diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 3770bc115653a..83ef725854ab4 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -83,6 +83,13 @@ ) +def _is_seed_key(key: str) -> bool: + """Whole-token seed match: `seed`, `rng_seed`, `seed_value` count; + `unseeded_metric` does not.""" + tokens = re.split(r"[^a-zA-Z0-9]+", key.lower()) + return "seed" in tokens or "seeds" in tokens + + def _distinct_assignment_exists(match_sets: "list[set[int]]") -> bool: """Backtracking search for a system of distinct representatives: every declared point must map to its OWN row.""" @@ -1123,15 +1130,27 @@ def packet_errors( "shell strings are not the promised reproduction " "path" ) - if not any( - re.match(r"^pixi run build\b", command.strip()) for command in commands - ): + build_indices = [ + index + for index, command in enumerate(commands) + if re.match(r"^pixi run build\b", command.strip()) + ] + non_build_indices = [ + index for index in range(len(commands)) if index not in build_indices + ] + if not build_indices: errors.append( "evidence.commands must include the build step " "('pixi run build ...') before execution; a clean " "checkout from target.fetch_hint has no artifacts, and " "an existing checkout may hold stale ones" ) + elif non_build_indices and min(build_indices) > min(non_build_indices): + errors.append( + "evidence.commands must run the build step BEFORE the " + "evidence command; building afterwards reproduces " + "nothing" + ) raw_paths = evidence.get("raw_paths") raw_rows = evidence.get("raw_rows") has_paths = isinstance(raw_paths, list) and bool(raw_paths) @@ -1151,6 +1170,16 @@ def packet_errors( ] match_sets = [] for point in object_points: + null_coords = sorted( + key for key, value in point.items() if value is None + ) + if null_coords: + errors.append( + f"ensemble.sweep point {point} declares null " + f"coordinates at {null_coords}; absent row fields " + "would spuriously match them" + ) + continue matches = { index for index, row in enumerate(raw_rows) @@ -1183,9 +1212,7 @@ def packet_errors( index for index, row in enumerate(raw_rows) if isinstance(row, dict) - and any( - "seed" in str(key).lower() and row[key] == seed for key in row - ) + and any(_is_seed_key(str(key)) and row[key] == seed for key in row) } if not matches: errors.append( @@ -1400,6 +1427,20 @@ def packet_errors( "host.performance_valid is false; timing from an " "uncontrolled host must be typed unsupported" ) + allocation_group = ( + packet.get("metrics", {}).get("allocation") + if isinstance(packet.get("metrics"), dict) + else None + ) + if ( + isinstance(allocation_group, dict) + and allocation_group.get("status") != "unsupported" + ): + errors.append( + "metrics.allocation publishes measurements while " + "host.performance_valid is false; allocation counts " + "from an uncontrolled host must be typed unsupported" + ) review = packet.get("review") if not isinstance(review, dict): diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index dbb3f43e2213b..8055aad407762 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -1623,3 +1623,46 @@ def test_string_dtype_npy_fails(tmp_path): assert any( "does not parse as its claimed raw-data format" in error for error in errors ) + + +def test_allocation_needs_a_valid_host_too(): + packet = complete_packet() + packet["metrics"]["allocation"] = {"method": "counter", "allocs": 12.0} + errors = MODULE.packet_errors(packet) + assert any("allocation counts" in error for error in errors) + + +def test_null_sweep_coordinates_fail(): + packet = complete_packet() + del packet["ensemble"]["deterministic_repeats"] + del packet["ensemble"]["deterministic_repeats_identical"] + packet["ensemble"]["sweep"] = [{"angle_deg": None}, {"other": None}] + packet["evidence"]["raw_rows"] = [ + {"metric": 1.0, "trajectory_sha256": "d" * 64}, + {"metric": 2.0, "trajectory_sha256": "e" * 64}, + ] + errors = MODULE.packet_errors(packet) + assert any("null" in error and "coordinates" in error for error in errors) + + +def test_seed_fields_match_by_token(): + packet = complete_packet() + del packet["ensemble"]["deterministic_repeats"] + del packet["ensemble"]["deterministic_repeats_identical"] + packet["ensemble"]["seeds"] = [7, 11] + packet["evidence"]["raw_rows"] = [ + {"unseeded_metric": 7.0, "trajectory_sha256": "d" * 64}, + {"unseeded_metric": 11.0, "trajectory_sha256": "e" * 64}, + ] + errors = MODULE.packet_errors(packet) + assert any("has no row recording it" in error for error in errors) + + +def test_build_must_precede_the_evidence_command(): + packet = complete_packet() + packet["evidence"]["commands"] = [ + "pixi run python scripts/example.py", + "pixi run build", + ] + errors = MODULE.packet_errors(packet) + assert any("BEFORE the evidence command" in error for error in errors) From 5ee6d30aeb9ecf91d8824408efa7ef59b53b1262 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 13:10:57 -0700 Subject: [PATCH 45/52] Close the Codex round-19 findings One genuine false positive fixed: env-prefixed build commands (DART_PARALLEL_JOBS=4 pixi run build) passed COMMAND_RE but were not recognized as the build step, rejecting valid evidence -- fixed with a prefix-aware BUILD_COMMAND_RE and a regression test. NPY payloads are verified against the declared array size; artifact digests no longer count as repeat verification (repeat claims bind to trajectory hashes in rows or the ensemble; digests prove file identity only); bookkeeping keys are matched case-insensitively by whole token. Tests: 147 -> 151 validator cases. --- .../verification.md | 15 +++++ scripts/check_citation_evidence.py | 55 +++++++++++++++---- tests/test_check_citation_evidence.py | 54 +++++++++++++++++- 3 files changed, 112 insertions(+), 12 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index ef0669abca702..7bea63f091dc9 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -784,3 +784,18 @@ fields would spuriously match them); seed fields match by whole token (`unseeded_metric` no longer counts); the build step must PRECEDE the evidence command, not merely appear somewhere in the list. Tests: 143 -> 147 (main), 123 -> 127 (6.20). + +## Codex review round 19 (PR #3445) — 2026-08-16 + +Four findings, one a genuine false-positive bug in the round-17 build +gate (env-prefixed build commands like `DART_PARALLEL_JOBS=4 pixi run +build` passed COMMAND_RE but were not recognized as the build step, +rejecting valid evidence) — fixed with a prefix-aware BUILD_COMMAND_RE +and a regression test. Also: NPY payloads are verified against the +declared array size (a million-element declaration with one payload +byte is rejected); artifact digests no longer count as repeat +verification — repeat claims bind to trajectory hashes in rows or the +ensemble, with artifact_digests proving file identity only; and +bookkeeping keys are matched case-insensitively by whole token (`Seed`, +`run_id`, `Repeat-2` are bookkeeping; `contact_count_max` remains a +measurement). Tests: 147 -> 151 (main), 127 -> 131 (6.20). diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 83ef725854ab4..1efde1b8fad5b 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -71,6 +71,9 @@ {"", "-", "--", "n/a", "na", "none", "null", "tbd", "unknown", "unsupported"} ) +BUILD_COMMAND_RE = re.compile( + r"^(?:[A-Za-z_][A-Za-z0-9_]*=[^\s;|&`$]+[ \t]+)*pixi run build\b" +) COMMAND_RE = re.compile( r"^(?:[A-Za-z_][A-Za-z0-9_]*=[^\s;|&`$]+[ \t]+)*" r"pixi run[ \t]+[^;|&`$\n\r]+\Z" ) @@ -110,14 +113,24 @@ def assign(index: int) -> bool: return assign(0) +def _is_bookkeeping_key(terminal: str) -> bool: + """Case-insensitive whole-token bookkeeping match: `Seed`, `run_id`, + and `repeat-2` are bookkeeping; `contact_count_max` is a measurement.""" + tokens = [ + token + for token in re.split(r"[^a-zA-Z0-9]+", terminal.lower()) + if token and not token.isdigit() + ] + return bool(tokens) and all(token in NUMERIC_BOOKKEEPING_KEYS for token in tokens) + + def _has_measurement_leaf(value: object) -> bool: """True when the value contains a numeric/boolean leaf beyond bookkeeping keys (the raw-rows measurement rule, applied to parsed artifacts).""" return any( (_is_finite_number(leaf) or isinstance(leaf, bool)) - and re.sub(r"(\[\d+\])+$", "", path.rsplit(".", 1)[-1]) - not in NUMERIC_BOOKKEEPING_KEYS + and not _is_bookkeeping_key(re.sub(r"(\[\d+\])+$", "", path.rsplit(".", 1)[-1])) for path, leaf in _metric_leaves(value) ) @@ -428,7 +441,7 @@ def _raw_data_content_issue(path: "Path") -> "str | None": return no_content return None if suffix == ".npy": - return _npy_header_issue(head, mismatch) + return _npy_header_issue(head, mismatch, path.stat().st_size) if suffix == ".npz": try: import io @@ -470,7 +483,9 @@ def _raw_data_content_issue(path: "Path") -> "str | None": return None -def _npy_header_issue(head: bytes, mismatch: str) -> "str | None": +def _npy_header_issue( + head: bytes, mismatch: str, total_size: "int | None" = None +) -> "str | None": """Parse the NPY header structurally (stdlib only): magic, version, header length, and a literal dict with descr/shape/fortran_order whose shape is a tuple of non-negative ints.""" @@ -509,8 +524,15 @@ def _npy_header_issue(head: bytes, mismatch: str) -> "str | None": # String/object/structured dtypes carry no numeric measurement; # structured record dtypes are a recorded boundary. return mismatch - if len(head) <= header_start + header_len: - # Header-only file: a declared array with no payload bytes. + if total_size is None: + total_size = len(head) + itemsize_match = re.search(r"([0-9]+)$", header["descr"]) + itemsize = int(itemsize_match.group(1)) if itemsize_match else 1 + expected_payload = itemsize + for dim in header["shape"]: + expected_payload *= dim + if total_size < header_start + header_len + expected_payload: + # Truncated payload: the declared array does not fit in the file. return mismatch return None @@ -979,7 +1001,19 @@ def packet_errors( "digest, larger claims must show their repeats" ) has_repeats = False - if has_repeats and not _has_hash_leaf(packet.get("evidence")): + evidence_for_repeats = ( + { + key: value + for key, value in packet.get("evidence", {}).items() + if key != "artifact_digests" + } + if isinstance(packet.get("evidence"), dict) + else packet.get("evidence") + ) + repeat_hash_sources = [evidence_for_repeats, ensemble] + if has_repeats and not any( + _has_hash_leaf(source) for source in repeat_hash_sources + ): errors.append( "ensemble.deterministic_repeats is asserted but the " "evidence carries no *hash*/sha256 field binding the " @@ -1133,7 +1167,7 @@ def packet_errors( build_indices = [ index for index, command in enumerate(commands) - if re.match(r"^pixi run build\b", command.strip()) + if BUILD_COMMAND_RE.match(command.strip()) ] non_build_indices = [ index for index in range(len(commands)) if index not in build_indices @@ -1247,8 +1281,9 @@ def packet_errors( continue if not any( (_is_finite_number(leaf) or isinstance(leaf, bool)) - and re.sub(r"(\[\d+\])+$", "", path.rsplit(".", 1)[-1]) - not in NUMERIC_BOOKKEEPING_KEYS + and not _is_bookkeeping_key( + re.sub(r"(\[\d+\])+$", "", path.rsplit(".", 1)[-1]) + ) for path, leaf in _metric_leaves(row) ): errors.append( diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index 8055aad407762..b4c25c77ab57e 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -498,12 +498,14 @@ def test_dangling_raw_path_fails(tmp_path): assert any("does not resolve" in error for error in errors) (tmp_path / "real.csv").write_text("a,b\n1,2\n", encoding="utf-8") packet["evidence"]["raw_paths"] = ["real.csv"] - # Path-based evidence pins its artifact bytes; those digests also carry - # the hash-bearing evidence the repeats claim binds to. + # Path-based evidence pins its artifact bytes; repeat verification is + # bound separately by an explicit trajectory digest (artifact digests + # prove file identity, not repeat determinism). packet["evidence"]["artifact_digests"] = { "real.csv": "sha256:" + hashlib.sha256((tmp_path / "real.csv").read_bytes()).hexdigest() } + packet["ensemble"]["repeat_trajectory_sha256"] = "a" * 64 assert MODULE.packet_errors(packet, base_dir=tmp_path) == [] @@ -1666,3 +1668,51 @@ def test_build_must_precede_the_evidence_command(): ] errors = MODULE.packet_errors(packet) assert any("BEFORE the evidence command" in error for error in errors) + + +def test_artifact_digests_do_not_prove_repeats(): + packet = complete_packet() + packet["evidence"]["raw_rows"] = [{"angle_deg": 0.0, "lateral_drift_m": 0.5}] + packet["evidence"]["artifact_digests"] = {"x.csv": "sha256:" + "a" * 64} + errors = MODULE.packet_errors(packet) + assert any( + "binding the repeats to recorded trajectories" in error for error in errors + ) + + +def test_env_prefixed_build_commands_are_recognized(): + packet = complete_packet() + packet["evidence"]["commands"] = [ + "DART_PARALLEL_JOBS=4 pixi run build", + "pixi run python scripts/example.py", + ] + assert MODULE.packet_errors(packet) == [] + + +def test_truncated_npy_payloads_fail(tmp_path): + header = b"{'descr': ' Date: Sun, 16 Aug 2026 13:25:29 -0700 Subject: [PATCH 46/52] Close the Codex round-20 findings A commands list consisting only of build steps is rejected -- the reproduction sequence must also RUN the evidence command; the build match requires the exact task name (arguments allowed, invented suffixed tasks like build-nothing not); zero-width NPY dtypes are rejected; numerically equivalent sweep points (1 vs 1.0) deduplicate in the distinct-point check, matching the row matcher's equality semantics. Tests: 151 -> 155 validator cases. --- .../verification.md | 11 +++++ scripts/check_citation_evidence.py | 31 ++++++++++-- tests/test_check_citation_evidence.py | 49 +++++++++++++++++++ 3 files changed, 87 insertions(+), 4 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 7bea63f091dc9..ba857f96940f2 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -799,3 +799,14 @@ ensemble, with artifact_digests proving file identity only; and bookkeeping keys are matched case-insensitively by whole token (`Seed`, `run_id`, `Repeat-2` are bookkeeping; `contact_count_max` remains a measurement). Tests: 147 -> 151 (main), 127 -> 131 (6.20). + +## Codex review round 20 (PR #3445) — 2026-08-16 + +Five findings (the build-task one reported on both PRs), all fixed: +a commands list consisting only of build steps is rejected (a +reproduction sequence must also RUN the evidence command); the build +match no longer accepts invented suffixed tasks like `build-nothing` +(exact task name, arguments allowed); zero-width NPY dtypes (` 155 (main), 131 -> 135 (6.20). diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 1efde1b8fad5b..900d022fc53dd 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -72,7 +72,8 @@ ) BUILD_COMMAND_RE = re.compile( - r"^(?:[A-Za-z_][A-Za-z0-9_]*=[^\s;|&`$]+[ \t]+)*pixi run build\b" + r"^(?:[A-Za-z_][A-Za-z0-9_]*=[^\s;|&`$]+[ \t]+)*pixi run build" + r"(?:[ \t][^;|&`$\n\r]*)?\Z" ) COMMAND_RE = re.compile( r"^(?:[A-Za-z_][A-Za-z0-9_]*=[^\s;|&`$]+[ \t]+)*" r"pixi run[ \t]+[^;|&`$\n\r]+\Z" @@ -519,7 +520,7 @@ def _npy_header_issue( for dim in header["shape"] ) and isinstance(header["descr"], str) - and re.match(r"^[<>|=]?[bifuc][0-9]+$", header["descr"]) + and re.match(r"^[<>|=]?[bifuc][1-9][0-9]*$", header["descr"]) ): # String/object/structured dtypes carry no numeric measurement; # structured record dtypes are a recorded boundary. @@ -1034,9 +1035,25 @@ def packet_errors( has_sweep = isinstance(sweep, list) and len(sweep) >= 2 if has_sweep: canonical_points: list[str] = [] + + def _canonical_value(value: object) -> object: + # 1 and 1.0 are the same coordinate; the row matcher's == + # treats them as equal, so the distinct-point check must too. + if isinstance(value, int) and not isinstance(value, bool): + return float(value) + return value + for index, entry in enumerate(sweep): if isinstance(entry, dict) and entry: - canonical_points.append(json.dumps(entry, sort_keys=True)) + canonical_points.append( + json.dumps( + { + key: _canonical_value(value) + for key, value in entry.items() + }, + sort_keys=True, + ) + ) else: errors.append( f"ensemble.sweep[{index}] must be a non-empty object " @@ -1179,7 +1196,13 @@ def packet_errors( "checkout from target.fetch_hint has no artifacts, and " "an existing checkout may hold stale ones" ) - elif non_build_indices and min(build_indices) > min(non_build_indices): + elif not non_build_indices: + errors.append( + "evidence.commands contains only build steps; a " + "reproduction sequence must also RUN the evidence " + "command" + ) + elif min(build_indices) > min(non_build_indices): errors.append( "evidence.commands must run the build step BEFORE the " "evidence command; building afterwards reproduces " diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index b4c25c77ab57e..4d6cda4f86b08 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -1716,3 +1716,52 @@ def test_cased_bookkeeping_keys_are_still_bookkeeping(): packet["evidence"]["raw_rows"] = [row] errors = MODULE.packet_errors(packet) assert any("beyond bookkeeping" in error for error in errors), row + + +def test_build_only_command_lists_fail(): + packet = complete_packet() + packet["evidence"]["commands"] = ["pixi run build"] + errors = MODULE.packet_errors(packet) + assert any("must also RUN the evidence command" in error for error in errors) + + +def test_build_suffix_tasks_do_not_count(): + packet = complete_packet() + packet["evidence"]["commands"] = [ + "pixi run build-nothing", + "pixi run python scripts/example.py", + ] + errors = MODULE.packet_errors(packet) + assert any("must include the build step" in error for error in errors) + + +def test_zero_width_npy_dtypes_fail(tmp_path): + header = b"{'descr': ' Date: Sun, 16 Aug 2026 14:01:30 -0700 Subject: [PATCH 47/52] Close the Codex round-21 findings One genuine parity gap fixed: main's CT-001 energy-injection gate started from previous_energy = None, skipping the first contact solve -- the fix its release-6.20 twin received in round 2 was never mirrored; now seeded from the configured pre-step state, packet regenerated in the follow-up commit. Also: identity metadata suffixes excluded case-insensitively; configurations must name a solver/method/integrator identity (a detector alone is not a configuration; requested method sweeps count as identity lists); a non-build command must run a repository harness; CSV numeric cells must be finite; Parquet dropped from supported raw formats (stdlib cannot decode it -- fail-closed means unsupported). Tests: 155 -> 159 validator cases. --- .../verification.md | 16 ++++ scripts/check_citation_evidence.py | 92 ++++++++++++------- ...citation_ct001_rolling_direction_packet.py | 8 +- tests/test_check_citation_evidence.py | 48 +++++++++- 4 files changed, 126 insertions(+), 38 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index ba857f96940f2..06cf5a087c388 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -810,3 +810,19 @@ match no longer accepts invented suffixed tasks like `build-nothing` are rejected; and numerically equivalent sweep points (1 vs 1.0) deduplicate in the distinct-point check, matching the row matcher's equality semantics. Tests: 151 -> 155 (main), 131 -> 135 (6.20). + +## Codex review round 21 (PR #3445) — 2026-08-16 + +Six findings, one a genuine parity gap: main's CT-001 energy-injection +gate still started from `previous_energy = None`, skipping the first +contact solve — the very fix its release-6.20 twin received in round 2 +but that was never mirrored; seeded from the configured pre-step state +and the packet regenerated. Also: identity metadata suffixes are +excluded case-insensitively; a configuration must name a +solver/method/integrator identity (a detector alone is not a +configuration; requested method SWEEPS count as identity lists); a +non-build command must run a repository harness (build+lint sequences +rejected); CSV numeric cells must be finite; and Parquet is dropped +from supported raw formats — stdlib cannot decode it, and the +fail-closed answer to an unvalidatable format is to not accept it. +Tests: 155 -> 159 (main), 135 -> 139 (6.20). diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 900d022fc53dd..f4add72ad7c12 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -80,7 +80,7 @@ ) HASH_VALUE_RE = re.compile(r"^(sha256:)?[0-9a-f]{32,}$") RAW_DATA_SUFFIXES = frozenset( - {".csv", ".tsv", ".json", ".jsonl", ".ndjson", ".npz", ".npy", ".parquet"} + {".csv", ".tsv", ".json", ".jsonl", ".ndjson", ".npz", ".npy"} ) NUMERIC_BOOKKEEPING_KEYS = frozenset( {"seed", "seeds", "index", "idx", "id", "ids", "repeat", "repeats", "run"} @@ -188,6 +188,13 @@ def _has_hash_leaf(value: object) -> bool: def _is_identity_value(value: object) -> bool: + if isinstance(value, list): + # A requested sweep of methods IS the requested identity. + return bool(value) and all( + _is_nonempty_str(entry) + and entry.strip().lower() not in IDENTITY_PLACEHOLDER_VALUES + for entry in value + ) return ( _is_nonempty_str(value) and value.strip().lower() not in IDENTITY_PLACEHOLDER_VALUES @@ -283,26 +290,52 @@ def _is_identity_value(value: object) -> bool: def _is_identity_key(key: str) -> bool: """Whole-token identity match: `contact_solver_method` counts, - `methodology` (substring only) and `method_note` (metadata role) do - not.""" - if key.endswith(IDENTITY_METADATA_SUFFIXES): + `methodology` (substring only) and `method_note`/`Method_Note` + (metadata role, any case) do not.""" + lowered = key.lower() + if lowered.endswith(IDENTITY_METADATA_SUFFIXES): return False - tokens = re.split(r"[^a-zA-Z0-9]+", key.lower()) + tokens = re.split(r"[^a-zA-Z0-9]+", lowered) return any(token in IDENTITY_KEY_TOKENS for token in tokens) -def _has_identity_key(value: object) -> bool: +SOLVER_CATEGORY_TOKENS = frozenset( + { + "solver", + "solvers", + "method", + "methods", + "integrator", + "integration", + "family", + "families", + } +) + + +def _has_identity_key(value: object, tokens: "frozenset | None" = None) -> bool: """True when any key at any depth names a solver/method/... identity with a non-empty string value; {"placeholder": null} and - {"method_note": "..."} have none.""" + {"method_note": "..."} have none. `tokens` restricts the accepted + identity categories.""" if isinstance(value, dict): for key, item in value.items(): - if _is_identity_key(str(key)) and _is_identity_value(item): + if ( + _is_identity_key(str(key)) + and _is_identity_value(item) + and ( + tokens is None + or any( + token in tokens + for token in re.split(r"[^a-zA-Z0-9]+", str(key).lower()) + ) + ) + ): return True - if _has_identity_key(item): + if _has_identity_key(item, tokens): return True elif isinstance(value, list): - return any(_has_identity_key(item) for item in value) + return any(_has_identity_key(item, tokens) for item in value) return False @@ -431,9 +464,11 @@ def _raw_data_content_issue(path: "Path") -> "str | None": for line in lines[1:50]: for cell in line.split(delimiter): try: - float(cell.strip()) + cell_value = float(cell.strip()) except ValueError: continue + if not math.isfinite(cell_value): + continue has_numeric_cell = True break if has_numeric_cell: @@ -461,26 +496,6 @@ def _raw_data_content_issue(path: "Path") -> "str | None": except OSError, zipfile.BadZipFile, KeyError: return mismatch return None - if suffix == ".parquet": - try: - with path.open("rb") as stream: - stream.seek(-4, 2) - tail4 = stream.read(4) - except OSError: - return mismatch - if head[:4] != b"PAR1" or tail4 != b"PAR1": - return mismatch - try: - with path.open("rb") as stream: - stream.seek(-8, 2) - footer = stream.read(8) - except OSError: - return mismatch - footer_len = int.from_bytes(footer[:4], "little") - file_size = path.stat().st_size - if not (0 < footer_len <= file_size - 12): - return mismatch - return None return None @@ -958,6 +973,12 @@ def packet_errors( "solver/method/detector/integrator/backend identity field; " "an arbitrary placeholder object is not a configuration" ) + elif not _has_identity_key(value, SOLVER_CATEGORY_TOKENS): + errors.append( + f"configuration.{side} names no solver/method/integrator " + "identity; a detector alone does not record which solver " + "ran" + ) if not _is_nonempty_str(configuration.get("resolved_provenance")): errors.append( "configuration.resolved_provenance must name how the resolved " @@ -1202,6 +1223,15 @@ def _canonical_value(value: object) -> object: "reproduction sequence must also RUN the evidence " "command" ) + elif not any( + re.search(r"python[ \t].*scripts/", commands[index]) + for index in non_build_indices + ): + errors.append( + "no evidence.commands entry runs a repository harness " + "(python ... scripts/...); build or maintenance tasks " + "alone do not regenerate evidence" + ) elif min(build_indices) > min(non_build_indices): errors.append( "evidence.commands must run the build step BEFORE the " diff --git a/scripts/write_citation_ct001_rolling_direction_packet.py b/scripts/write_citation_ct001_rolling_direction_packet.py index c87cb646203dc..5a55b5a4cb97c 100644 --- a/scripts/write_citation_ct001_rolling_direction_packet.py +++ b/scripts/write_citation_ct001_rolling_direction_packet.py @@ -159,7 +159,10 @@ def run_single( max_iterations = 0 max_residual = 0.0 contact_count_max = 0 - previous_energy = None + # Seed from the configured pre-step state so the first contact solve is + # inside the energy-injection gate (the sphere starts touching the + # ground); mirrors the release-6.20 writer's fix. + previous_energy = float(world.compute_step_metrics().kinetic_energy) stage_names: list[str] = [] lateral_axis = np.array([-direction[1], direction[0], 0.0]) @@ -185,8 +188,7 @@ def run_single( metrics = world.compute_step_metrics() kinetic = float(metrics.kinetic_energy) - if previous_energy is not None: - max_energy_gain = max(max_energy_gain, kinetic - previous_energy) + max_energy_gain = max(max_energy_gain, kinetic - previous_energy) previous_energy = kinetic max_penetration = max(max_penetration, float(metrics.max_penetration_depth)) max_iterations = max(max_iterations, int(metrics.last_step_iterations)) diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index 4d6cda4f86b08..f897e87912b41 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -1582,7 +1582,9 @@ def test_invalid_host_forbids_measured_performance(): assert any("uncontrolled host" in error for error in errors) -def test_parquet_stub_fails(tmp_path): +def test_parquet_is_an_unsupported_raw_format(tmp_path): + # Parquet cannot be decoded with the stdlib, so the fail-closed answer + # is to not accept it as raw evidence at all. (tmp_path / "rows.parquet").write_bytes(b"PAR1PAR1") packet = complete_packet() del packet["evidence"]["raw_rows"] @@ -1591,10 +1593,9 @@ def test_parquet_stub_fails(tmp_path): "rows.parquet": "sha256:" + hashlib.sha256((tmp_path / "rows.parquet").read_bytes()).hexdigest() } + packet["ensemble"]["repeat_trajectory_sha256"] = "a" * 64 errors = MODULE.packet_errors(packet, base_dir=tmp_path) - assert any( - "does not parse as its claimed raw-data format" in error for error in errors - ) + assert any("must name a raw-data artifact" in error for error in errors) def test_all_unsupported_measured_groups_fail(): @@ -1765,3 +1766,42 @@ def test_numeric_equivalent_sweep_points_deduplicate(): ] errors = MODULE.packet_errors(packet) assert any("DISTINCT points" in error for error in errors) + + +def test_cased_metadata_identity_keys_are_rejected(): + packet = complete_packet() + packet["configuration"]["resolved"] = {"Method_Note": "arbitrary prose"} + errors = MODULE.packet_errors(packet) + assert any("no recognizable" in error for error in errors) + + +def test_detector_alone_is_not_a_configuration(): + packet = complete_packet() + packet["configuration"]["resolved"] = {"detector": "fcl"} + errors = MODULE.packet_errors(packet) + assert any( + "a detector alone does not record which solver ran" in error for error in errors + ) + + +def test_maintenance_tasks_do_not_execute_evidence(): + packet = complete_packet() + packet["evidence"]["commands"] = ["pixi run build", "pixi run lint"] + errors = MODULE.packet_errors(packet) + assert any("repository harness" in error for error in errors) + + +def test_non_finite_csv_cells_do_not_count(tmp_path): + (tmp_path / "rows.csv").write_text("metric,other\nNaN,text\n", encoding="utf-8") + packet = complete_packet() + del packet["evidence"]["raw_rows"] + packet["evidence"]["raw_paths"] = ["rows.csv"] + packet["evidence"]["artifact_digests"] = { + "rows.csv": "sha256:" + + hashlib.sha256((tmp_path / "rows.csv").read_bytes()).hexdigest() + } + packet["ensemble"]["repeat_trajectory_sha256"] = "a" * 64 + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any( + "carries no numeric or boolean measurement content" in error for error in errors + ) From 009e5072c8c6674d2388e00735255d1d5fc9aeca Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 14:02:47 -0700 Subject: [PATCH 48/52] Regenerate CT-001 with the seeded energy-injection gate The first contact solve is now inside the gate; every cell still passes physical validity and the disposition is unchanged (reproduced). --- .../evidence/CT-001-dart7-rolling-direction.json | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json index 771bb981e3516..88e174bce4c5e 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-001-dart7-rolling-direction.json @@ -1277,9 +1277,9 @@ }, "target": { "branch": "main", - "commit": "dd68714d6a7a928b3af784e08369469b0e945216", + "commit": "1cce306ff466e7fd6ea00118e9577bf4bf9b5665", "commit_role": "Source state measured: the library and fixture were run at this commit, which is HEAD at capture time. The packet and its writer land in a later commit, so re-running the recorded command requires the child commit that adds the writer; the measured behavior belongs to this one.", - "fetch_hint": "git fetch origin pull/3445/head && git checkout dd68714d6a7a928b3af784e08369469b0e945216" + "fetch_hint": "git fetch origin pull/3445/head && git checkout 1cce306ff466e7fd6ea00118e9577bf4bf9b5665" }, "title": "Rolling-direction friction dependence (DART 7 first packet)" } From b5fe2472e848bc293e8600098f39581d32cae459 Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 14:19:25 -0700 Subject: [PATCH 49/52] Close the Codex round-22 findings Target commits must exist in the repository object store (CI lint now fetches full history; shallow clones error explicitly), evidence commands may only reference harness scripts that exist, NPY/NPZ payloads must carry at least one finite element (unsupported float widths are rejected as unverifiable), and CSV finite-cell scanning classifies columns by header so bookkeeping-only tables cannot close a lane. Tests: 159 -> 169. --- .github/workflows/ci_lint.yml | 5 + .../verification.md | 21 ++ scripts/check_citation_evidence.py | 208 ++++++++++++++++-- tests/test_check_citation_evidence.py | 151 ++++++++++++- 4 files changed, 359 insertions(+), 26 deletions(-) diff --git a/.github/workflows/ci_lint.yml b/.github/workflows/ci_lint.yml index c90570bbe7eb2..922476cbed0da 100644 --- a/.github/workflows/ci_lint.yml +++ b/.github/workflows/ci_lint.yml @@ -46,6 +46,11 @@ jobs: - name: Checkout uses: actions/checkout@v7 + with: + # check-citation-evidence verifies every packet's target commit + # against the local object store; a shallow clone cannot answer + # whether a historical commit exists. + fetch-depth: 0 - name: Setup pixi (CI) uses: ./.github/actions/setup-pixi-ci diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 06cf5a087c388..b69019c9da680 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -826,3 +826,24 @@ rejected); CSV numeric cells must be finite; and Parquet is dropped from supported raw formats — stdlib cannot decode it, and the fail-closed answer to an unvalidatable format is to not accept it. Tests: 155 -> 159 (main), 135 -> 139 (6.20). + +## Codex review round 22 (PR #3445) — 2026-08-16 + +Three findings, all closing "well-shaped but nonexistent" gaps. A +40-hex target commit is now verified against the repository object +store (a hash that names no commit attributes measurements to no +source revision); the CI lint checkout fetches full history so the +check can answer for historical targets, and shallow clones get an +explicit unshallow error instead of a silent pass. Every +`scripts/...` path referenced by an evidence command must be a real +file in the scripts/ tree — `pixi run python scripts/does_not_exist.py` +is no longer a valid reproduction sequence. NPY/NPZ payloads are now +value-checked with the stdlib: float/complex arrays must contain at +least one finite element (all-NaN/Inf artifacts are rejected the way +inline rows and CSV cells already were), f2/f4/f8 and c8/c16 decode, +and other float widths are rejected as unverifiable. The 6.20 round-22 +CSV finding is mirrored here too: with a header row, the finite-cell +requirement applies only to non-bookkeeping columns, so `seed,note` +metadata cannot close a lane. Network reachability of the PR ref +remains with `--freshness`; the default gate is offline by design. +Tests: 159 -> 169 (main), 139 -> 149 (6.20). diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index f4add72ad7c12..a3bec9d722848 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -32,6 +32,7 @@ from __future__ import annotations import argparse +import functools import hashlib import json import math @@ -136,6 +137,34 @@ def _has_measurement_leaf(value: object) -> bool: ) +@functools.lru_cache(maxsize=None) +def _git_object_status(commit: str) -> str: + """'exists', 'missing', 'shallow', or 'unavailable' for a commit hash in + THIS repository's object store. + + Offline by design: the default gate proves the recorded checkout points + at a commit that actually exists here; network reachability through the + PR ref stays with `--freshness`. + """ + + def _git(*argv: str) -> "subprocess.CompletedProcess[bytes]": + return subprocess.run( + ["git", "-C", str(REPO_ROOT), *argv], + capture_output=True, + check=False, + ) + + if _git("cat-file", "-e", f"{commit}^{{commit}}").returncode == 0: + return "exists" + inside = _git("rev-parse", "--is-inside-work-tree") + if inside.returncode != 0 or inside.stdout.strip() != b"true": + return "unavailable" + shallow = _git("rev-parse", "--is-shallow-repository") + if shallow.returncode == 0 and shallow.stdout.strip() == b"true": + return "shallow" + return "missing" + + def _has_hash_list(value: object, length: int) -> bool: """True when a hash-named key holds a list of >= `length` digest values.""" if isinstance(value, dict): @@ -417,9 +446,10 @@ def _raw_data_content_issue(path: "Path") -> "str | None": Mirrors the visual check: a prose file renamed to `rows.csv` must not satisfy the raw-evidence requirement. JSON variants must parse; CSV/TSV - need delimited tabular lines; NumPy/parquet containers must carry their - magic bytes. Deep semantic validation of tabular contents is a recorded - boundary. + need delimited tabular lines with a finite value in a non-bookkeeping + column; NumPy containers must parse structurally AND carry at least one + finite (or boolean/integer) element. Deep semantic validation of + tabular contents is a recorded boundary. """ suffix = path.suffix.lower() try: @@ -460,14 +490,46 @@ def _raw_data_content_issue(path: "Path") -> "str | None": lines = [line for line in head.splitlines() if line.strip()] if len(lines) < 2 or not any(delimiter in line for line in lines[:2]): return mismatch + + def _cell_float(cell: bytes) -> "float | None": + try: + return float(cell.strip()) + except ValueError: + return None + + first_row = lines[0].split(delimiter) + if any(_cell_float(cell) is None for cell in first_row): + # Header row: the finite-cell requirement applies only to + # columns that are not run bookkeeping (seed/index/repeat/...), + # so `seed,note` metadata cannot masquerade as a measurement. + header_names = [ + cell.strip().decode("utf-8", "replace") for cell in first_row + ] + measurement_columns = { + column + for column, name in enumerate(header_names) + if not (_is_bookkeeping_key(name) or _is_seed_key(name)) + } + if not measurement_columns: + return ( + "carries only bookkeeping columns " + f"({', '.join(header_names)}); run metadata without a " + "measurement column is not raw evidence" + ) + data_lines = lines[1:50] + else: + # Headerless numeric table: every column is a candidate + # measurement (there is no name to classify). + measurement_columns = None + data_lines = lines[0:50] has_numeric_cell = False - for line in lines[1:50]: - for cell in line.split(delimiter): - try: - cell_value = float(cell.strip()) - except ValueError: - continue - if not math.isfinite(cell_value): + for line in data_lines: + for column, cell in enumerate(line.split(delimiter)): + if measurement_columns is not None: + if column not in measurement_columns: + continue + cell_value = _cell_float(cell) + if cell_value is None or not math.isfinite(cell_value): continue has_numeric_cell = True break @@ -477,7 +539,29 @@ def _raw_data_content_issue(path: "Path") -> "str | None": return no_content return None if suffix == ".npy": - return _npy_header_issue(head, mismatch, path.stat().st_size) + parsed = _npy_parsed_header(head, mismatch, path.stat().st_size) + if isinstance(parsed, str): + return parsed + payload_offset, descr, payload_length = parsed + + def _payload_chunks(): + with path.open("rb") as stream: + stream.seek(payload_offset) + remaining = payload_length + while remaining > 0: + block = stream.read(min(1024 * 1024, remaining)) + if not block: + return + remaining -= len(block) + yield block + + try: + value_issue = _npy_payload_value_issue(descr, _payload_chunks()) + except OSError: + return "could not be read for format verification" + if value_issue is not None: + return value_issue[1] + return None if suffix == ".npz": try: import io @@ -489,22 +573,38 @@ def _raw_data_content_issue(path: "Path") -> "str | None": npy_members = [name for name in names if name.endswith(".npy")] if not npy_members: return mismatch + any_finite = False + first_value_issue: "str | None" = None for name in npy_members: member_bytes = archive.read(name) - if _npy_header_issue(member_bytes, mismatch) is not None: + parsed = _npy_parsed_header(member_bytes, mismatch, len(member_bytes)) + if isinstance(parsed, str): return mismatch + payload_offset, descr, payload_length = parsed + payload = member_bytes[payload_offset : payload_offset + payload_length] + value_issue = _npy_payload_value_issue(descr, [payload]) + if value_issue is None: + any_finite = True + elif value_issue[0] == "unsupported": + return value_issue[1] + elif first_value_issue is None: + first_value_issue = value_issue[1] + if not any_finite: + return first_value_issue or mismatch except OSError, zipfile.BadZipFile, KeyError: return mismatch return None return None -def _npy_header_issue( +def _npy_parsed_header( head: bytes, mismatch: str, total_size: "int | None" = None -) -> "str | None": +) -> "str | tuple[int, str, int]": """Parse the NPY header structurally (stdlib only): magic, version, header length, and a literal dict with descr/shape/fortran_order whose - shape is a tuple of non-negative ints.""" + shape is a tuple of non-negative ints. Returns the mismatch message on + failure, else (payload offset, descr, payload byte length) for value + inspection.""" import ast as _ast import struct @@ -550,7 +650,47 @@ def _npy_header_issue( if total_size < header_start + header_len + expected_payload: # Truncated payload: the declared array does not fit in the file. return mismatch - return None + return (header_start + header_len, header["descr"], expected_payload) + + +def _npy_payload_value_issue(descr: str, chunks: "object") -> "tuple[str, str] | None": + """('unsupported' | 'nonfinite', message) when an NPY payload proves no + finite measurement; None when at least one finite (or boolean/integer) + element is present. `descr` was already validated by the header parse.""" + import struct + + match = re.match(r"^([<>|=]?)([bifuc])([0-9]+)$", descr) + assert match is not None + byte_order, kind, itemsize_text = match.groups() + itemsize = int(itemsize_text) + if kind in ("b", "i", "u"): + # Booleans and integers cannot encode NaN/Inf; the non-empty shape + # already enforced by the header check makes them observations. + return None + scalar_width = itemsize if kind == "f" else itemsize // 2 + scalar_code = {2: "e", 4: "f", 8: "d"}.get(scalar_width) + if scalar_code is None or (kind == "c" and itemsize % 2 != 0): + return ( + "unsupported", + f"uses dtype {descr!r} whose element values the standard " + "library cannot decode; store measurements as f2/f4/f8 (or " + "c8/c16) so their finiteness stays checkable", + ) + prefix = byte_order if byte_order in ("<", ">") else "=" + element = struct.Struct(prefix + scalar_code * (1 if kind == "f" else 2)) + carry = b"" + for chunk in chunks: + data = carry + chunk if carry else chunk + usable = len(data) - (len(data) % element.size) + for parts in element.iter_unpack(data[:usable]): + if all(math.isfinite(part) for part in parts): + return None + carry = data[usable:] + return ( + "nonfinite", + "contains no finite element; an all-NaN/Inf array records no " + "measurement and is not raw evidence", + ) def _visual_content_issue(path: "Path") -> "str | None": @@ -905,6 +1045,29 @@ def packet_errors( f"{commit[:12]}...; the hint must reproduce THIS packet's " "target" ) + if isinstance(commit, str) and COMMIT_RE.match(commit): + object_status = _git_object_status(commit) + if object_status == "missing": + errors.append( + f"target.commit {commit[:12]}... does not exist in this " + "repository's object store; a well-shaped hash for a " + "nonexistent commit attributes the measurements to no " + "source revision (regenerate the packet at a real, " + "pushed commit)" + ) + elif object_status == "shallow": + errors.append( + f"target.commit {commit[:12]}... cannot be verified in " + "a shallow clone; fetch full history " + "(git fetch --unshallow) so recorded provenance stays " + "checkable" + ) + elif object_status == "unavailable": + errors.append( + "target.commit cannot be verified because no git " + "object store is available; run the checker from a " + "repository checkout" + ) scene = packet.get("scene") if not isinstance(scene, dict): @@ -1202,6 +1365,19 @@ def _canonical_value(value: object) -> object: "shell strings are not the promised reproduction " "path" ) + continue + for script_ref in re.findall(r"scripts/[A-Za-z0-9_.\-/]+", command): + script_path = (REPO_ROOT / script_ref).resolve() + if not ( + script_path.is_file() + and script_path.is_relative_to(REPO_ROOT / "scripts") + ): + errors.append( + f"evidence.commands[{index}] references " + f"{script_ref}, which is not a file in this " + "repository's scripts/ tree; a reproduction " + "sequence must invoke a harness that exists" + ) build_indices = [ index for index, command in enumerate(commands) diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index f897e87912b41..ca8808d9dbd5b 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -8,6 +8,7 @@ import hashlib import importlib.util import json +import subprocess import sys from pathlib import Path @@ -17,6 +18,18 @@ SCRIPT = ROOT / "scripts" / "check_citation_evidence.py" PLAN_DIR = ROOT / "docs" / "plans" / "123-citation-driven-simulation-trust" +# The validator verifies target commits against the real object store and +# harness paths against the real scripts/ tree, so the passing fixture must +# use a commit and harness that actually exist. +HEAD_COMMIT = subprocess.run( + ["git", "-C", str(ROOT), "rev-parse", "HEAD"], + capture_output=True, + text=True, + check=True, +).stdout.strip() +HARNESS = "scripts/write_citation_ct001_rolling_direction_packet.py" +assert (ROOT / HARNESS).is_file() + def _load_module(): spec = importlib.util.spec_from_file_location("check_citation_evidence", SCRIPT) @@ -39,9 +52,9 @@ def complete_packet() -> dict: "source": {"url": "https://example.org/claim", "claim": "A claim."}, "target": { "branch": "main", - "commit": "0" * 40, + "commit": HEAD_COMMIT, "fetch_hint": ( - "git fetch origin pull/3445/head && git checkout " + "0" * 40 + "git fetch origin pull/3445/head && git checkout " + HEAD_COMMIT ), }, "scene": { @@ -92,7 +105,7 @@ def complete_packet() -> dict: "evidence": { "commands": [ "pixi run build", - "pixi run python scripts/example.py", + "pixi run python " + HARNESS, ], "raw_rows": [ { @@ -1127,12 +1140,10 @@ def test_commands_must_be_the_reproducible_pixi_form(): packet = complete_packet() packet["evidence"]["commands"] = [ "pixi run build", - "PYTHONPATH=build/x pixi run python scripts/write.py", + "PYTHONPATH=build/x pixi run python " + HARNESS, ] assert MODULE.packet_errors(packet) == [] - packet["evidence"]["commands"] = [ - "PYTHONPATH=build/x pixi run python scripts/write.py" - ] + packet["evidence"]["commands"] = ["PYTHONPATH=build/x pixi run python " + HARNESS] errors = MODULE.packet_errors(packet) assert any("must include the build step" in error for error in errors) @@ -1664,7 +1675,7 @@ def test_seed_fields_match_by_token(): def test_build_must_precede_the_evidence_command(): packet = complete_packet() packet["evidence"]["commands"] = [ - "pixi run python scripts/example.py", + "pixi run python " + HARNESS, "pixi run build", ] errors = MODULE.packet_errors(packet) @@ -1685,7 +1696,7 @@ def test_env_prefixed_build_commands_are_recognized(): packet = complete_packet() packet["evidence"]["commands"] = [ "DART_PARALLEL_JOBS=4 pixi run build", - "pixi run python scripts/example.py", + "pixi run python " + HARNESS, ] assert MODULE.packet_errors(packet) == [] @@ -1730,7 +1741,7 @@ def test_build_suffix_tasks_do_not_count(): packet = complete_packet() packet["evidence"]["commands"] = [ "pixi run build-nothing", - "pixi run python scripts/example.py", + "pixi run python " + HARNESS, ] errors = MODULE.packet_errors(packet) assert any("must include the build step" in error for error in errors) @@ -1805,3 +1816,123 @@ def test_non_finite_csv_cells_do_not_count(tmp_path): assert any( "carries no numeric or boolean measurement content" in error for error in errors ) + + +def _npy_bytes(descr: str, shape: str, payload: bytes) -> bytes: + header = ( + b"{'descr': '" + descr.encode() + b"', 'fortran_order': False, " + b"'shape': (" + shape.encode() + b",), }" + ) + header += b" " * (63 - len(header) % 64) + b"\n" + return b"\x93NUMPY\x01\x00" + len(header).to_bytes(2, "little") + header + payload + + +def _path_packet(tmp_path, name: str, data: bytes) -> dict: + (tmp_path / name).write_bytes(data) + packet = complete_packet() + del packet["evidence"]["raw_rows"] + packet["evidence"]["raw_paths"] = [name] + packet["evidence"]["artifact_digests"] = { + name: "sha256:" + hashlib.sha256(data).hexdigest() + } + packet["ensemble"]["repeat_trajectory_sha256"] = "a" * 64 + return packet + + +def test_nonexistent_target_commits_fail(): + packet = complete_packet() + packet["target"]["commit"] = "f" * 40 + packet["target"]["fetch_hint"] = ( + "git fetch origin pull/3445/head && git checkout " + "f" * 40 + ) + errors = MODULE.packet_errors(packet) + assert any( + "does not exist in this repository's object store" in error for error in errors + ) + + +def test_missing_harness_scripts_fail(): + packet = complete_packet() + packet["evidence"]["commands"] = [ + "pixi run build", + "pixi run python scripts/does_not_exist.py", + ] + errors = MODULE.packet_errors(packet) + assert any("must invoke a harness that exists" in error for error in errors) + + +def test_all_nan_npy_payloads_fail(tmp_path): + import struct + + payload = struct.pack(" bytes: + buffer = io.BytesIO() + with zipfile.ZipFile(buffer, "w") as archive: + for name, data in members.items(): + archive.writestr(name, data) + return buffer.getvalue() + + packet = _path_packet(tmp_path, "nan.npz", _zip({"a.npy": nan_member})) + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any("contains no finite element" in error for error in errors) + + packet = _path_packet( + tmp_path, "mixed.npz", _zip({"a.npy": nan_member, "b.npy": fin_member}) + ) + assert MODULE.packet_errors(packet, base_dir=tmp_path) == [] + + +def test_bookkeeping_only_csv_columns_fail(tmp_path): + packet = _path_packet(tmp_path, "meta.csv", b"seed,run\n7,1\n8,2\n") + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any("run metadata without a measurement column" in error for error in errors) + + packet = _path_packet(tmp_path, "notes.csv", b"seed,note\n7,control\n8,x\n") + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any( + "carries no numeric or boolean measurement content" in error for error in errors + ) + + +def test_csv_measurement_columns_still_pass(tmp_path): + packet = _path_packet(tmp_path, "rows.csv", b"seed,drift_m\n7,0.25\n8,0.50\n") + assert MODULE.packet_errors(packet, base_dir=tmp_path) == [] + + +def test_headerless_numeric_csv_still_passes(tmp_path): + packet = _path_packet(tmp_path, "grid.csv", b"1.0,2.0\n3.0,4.0\n") + assert MODULE.packet_errors(packet, base_dir=tmp_path) == [] From f3e987cf083e2e69d36a617ee6969188a21f5a5d Mon Sep 17 00:00:00 2001 From: Jeongseok Lee Date: Sun, 16 Aug 2026 14:43:54 -0700 Subject: [PATCH 50/52] Close the Codex round-23 findings Packets are source-bound to the manifest (claims with lane evidence pin canonical source_url/source_claim and packets must match), --fetch-target-refs restores squash-merged target commits before the CI object-store check, fetch hints may name any evidence PR, NPY dtype widths are whitelisted per kind, and the harness command must invoke an evidence writer. Tests: 169 -> 175. --- .github/workflows/ci_lint.yml | 6 + .../verification.md | 21 ++++ .../claims-manifest.json | 28 +++-- scripts/check_citation_evidence.py | 117 ++++++++++++++++-- tests/test_check_citation_evidence.py | 70 ++++++++++- 5 files changed, 224 insertions(+), 18 deletions(-) diff --git a/.github/workflows/ci_lint.yml b/.github/workflows/ci_lint.yml index 922476cbed0da..19c321e4d932c 100644 --- a/.github/workflows/ci_lint.yml +++ b/.github/workflows/ci_lint.yml @@ -57,6 +57,12 @@ jobs: with: pixi-bin-path: ${{ runner.temp }}/pixi/bin/pixi + - name: Fetch citation packet target refs + # Squash merges retire topic-branch commits from branch history; + # the surviving refs/pull/N/head namespace is not covered by + # fetch-depth: 0, so fetch the refs the packets actually name. + run: pixi run python scripts/check_citation_evidence.py --fetch-target-refs + - name: Check Lint run: pixi run check-lint diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index b69019c9da680..14b9455811941 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -847,3 +847,24 @@ requirement applies only to non-bookkeeping columns, so `seed,note` metadata cannot close a lane. Network reachability of the PR ref remains with `--freshness`; the default gate is offline by design. Tests: 159 -> 169 (main), 139 -> 149 (6.20). + +## Codex review round 23 (PR #3445) — 2026-08-16 + +Two findings here plus three on the LTS lane, mirrored both ways. A +packet's source is now bound to the manifest: each claim that carries +lane evidence must pin a canonical `source_url` and `source_claim`, +and every lane-referenced packet must match them exactly — evidence +for a different cited assertion can no longer close a lane (claims +with no evidence yet may defer the binding; the corpus rows carry no +URLs to pin). The round-22 object-store check is completed for the +post-squash world: squash merges retire topic commits from branch +history, so `--fetch-target-refs` fetches the `pull/N/head` refs the +packets actually name, the lint workflow runs it before check-lint, +and the missing-object error names that remedy. From the LTS lane: +the fetch-hint regex accepts any numeric PR ref (future evidence PRs +were previously impossible to land; the sha must still match +target.commit and reachability stays with --freshness), NPY dtype +widths are whitelisted per kind (` 175 (main), 149 -> 155 (6.20). diff --git a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json index 0e83c11ef4c11..05942a09177dc 100644 --- a/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json +++ b/docs/plans/123-citation-driven-simulation-trust/claims-manifest.json @@ -33,7 +33,9 @@ "evidence": [], "notes": "Open PR #3377 owns exact-Coulomb FBF rolling scenes on release-6.20; reuse, do not duplicate." } - } + }, + "source_url": "https://leggedrobotics.github.io/SimBenchmark/", + "source_claim": "Polyhedral friction can produce direction-dependent rolling/sliding behavior." }, { "id": "CT-002", @@ -56,7 +58,9 @@ "disposition": null, "evidence": [] } - } + }, + "source_url": "https://leggedrobotics.github.io/SimBenchmark/", + "source_claim": "Dense contact may fail, become unstable, or scale poorly for some timestep/solver settings." }, { "id": "CT-003", @@ -79,7 +83,9 @@ "disposition": null, "evidence": [] } - } + }, + "source_url": "https://leggedrobotics.github.io/SimBenchmark/", + "source_claim": "Elastic dense contact may inject energy or expose solver failure." }, { "id": "CT-004", @@ -102,7 +108,9 @@ "disposition": null, "evidence": [] } - } + }, + "source_url": "https://leggedrobotics.github.io/SimBenchmark/", + "source_claim": "Articulated integration/contact accuracy must be compared at matched cost." }, { "id": "CT-005", @@ -125,7 +133,9 @@ "disposition": null, "evidence": [] } - } + }, + "source_url": "https://leggedrobotics.github.io/SimBenchmark/", + "source_claim": "Controlled robot tracking exposes whole-step speed/accuracy tradeoffs." }, { "id": "CT-006", @@ -169,7 +179,9 @@ "disposition": null, "evidence": [] } - } + }, + "source_url": "https://arxiv.org/abs/2405.17020", + "source_claim": "Exact Coulomb cones and adaptive proximal methods may improve conditioning and remove friction-pyramid anisotropy." }, { "id": "CT-008", @@ -254,7 +266,9 @@ "disposition": null, "evidence": [] } - } + }, + "source_url": "https://doi.org/10.21105/joss.06771", + "source_claim": "Research workflows need fast reset, concurrency, low overhead, and deterministic synchronous stepping around DART." }, { "id": "CT-012", diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index a3bec9d722848..7fcd0260f0288 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -86,6 +86,13 @@ NUMERIC_BOOKKEEPING_KEYS = frozenset( {"seed", "seeds", "index", "idx", "id", "ids", "repeat", "repeats", "run"} ) +NPY_DTYPE_WIDTHS = { + "b": frozenset({1}), + "i": frozenset({1, 2, 4, 8}), + "u": frozenset({1, 2, 4, 8}), + "f": frozenset({2, 4, 8, 16}), + "c": frozenset({8, 16, 32}), +} def _is_seed_key(key: str) -> bool: @@ -285,7 +292,7 @@ def _is_identity_value(value: object) -> bool: # arbitrary placeholder object. Keys that merely mention an identity word in # a metadata role (method_note, backend_reason, ...) do not count. FETCH_HINT_RE = re.compile( - r"^git fetch origin pull/3445/head && git checkout ([0-9a-f]{40})$" + r"^git fetch origin pull/([0-9]+)/head && git checkout ([0-9a-f]{40})$" ) IDENTITY_KEY_TOKENS = frozenset( { @@ -640,10 +647,13 @@ def _npy_parsed_header( # String/object/structured dtypes carry no numeric measurement; # structured record dtypes are a recorded boundary. return mismatch + kind = header["descr"].lstrip("<>|=")[0] + itemsize = int(header["descr"].lstrip("<>|=")[1:]) + if itemsize not in NPY_DTYPE_WIDTHS[kind]: + # ` object: "command" ) elif not any( - re.search(r"python[ \t].*scripts/", commands[index]) + re.search( + r"python[ \t].*scripts/write_citation[A-Za-z0-9_]*\.py", + commands[index], + ) for index in non_build_indices ): errors.append( - "no evidence.commands entry runs a repository harness " - "(python ... scripts/...); build or maintenance tasks " - "alone do not regenerate evidence" + "no evidence.commands entry runs an evidence-writer " + "harness (python ... scripts/write_citation*.py); build " + "or maintenance tasks and unrelated scripts do not " + "regenerate this packet's evidence" ) elif min(build_indices) > min(non_build_indices): errors.append( @@ -1785,6 +1800,25 @@ def manifest_errors(manifest: object, corpus_ids: list[str]) -> list[str]: errors.append(f"{claim_id}: title must be non-empty") if not _is_nonempty_str(claim.get("source")): errors.append(f"{claim_id}: source must be non-empty") + lanes_probe = claim.get("lanes") + has_lane_evidence = isinstance(lanes_probe, dict) and any( + isinstance(lane, dict) and lane.get("evidence") + for lane in lanes_probe.values() + ) + source_url = claim.get("source_url") + source_claim = claim.get("source_claim") + url_ok = _is_nonempty_str(source_url) and re.match(r"^https?://", source_url) + if source_url is not None and not url_ok: + errors.append(f"{claim_id}: source_url must be an http(s) citation URL") + if source_claim is not None and not _is_nonempty_str(source_claim): + errors.append(f"{claim_id}: source_claim must be non-empty text") + if has_lane_evidence and not (url_ok and _is_nonempty_str(source_claim)): + errors.append( + f"{claim_id}: a claim with lane evidence must pin the " + "canonical source_url and source_claim its packets bind " + "to; without them a lane can close on evidence for a " + "different cited assertion" + ) family = claim.get("first_wave_family") if family is not None and family not in families: errors.append( @@ -1931,6 +1965,7 @@ def validate_tree( known_ids: set[str] = set() lane_evidence: dict[str, tuple[str, str]] = {} lane_dispositions: dict[str, tuple[str, object]] = {} + claim_sources: dict[str, tuple[object, object]] = {} if manifest is not None: manifest_issues = manifest_errors(manifest, corpus_ids) errors.extend(f"{manifest_path}: {issue}" for issue in manifest_issues) @@ -1941,6 +1976,10 @@ def validate_tree( claim_id = claim.get("id") if isinstance(claim_id, str): known_ids.add(claim_id) + claim_sources[claim_id] = ( + claim.get("source_url"), + claim.get("source_claim"), + ) lanes = claim.get("lanes") if not isinstance(lanes, dict): continue @@ -2080,6 +2119,24 @@ def validate_tree( f"{packet_path}: claim_id {packet.get('claim_id')} does " f"not match manifest lane {claim_id}.{lane_name}" ) + canonical_url, canonical_claim = claim_sources.get(claim_id, (None, None)) + source = packet.get("source") + packet_url = source.get("url") if isinstance(source, dict) else None + packet_claim = source.get("claim") if isinstance(source, dict) else None + if canonical_url is not None and packet_url != canonical_url: + errors.append( + f"{packet_path}: source.url {packet_url!r} does not " + f"match the manifest's canonical source_url for " + f"{claim_id}; a lane must close on evidence for the " + "cited source, not a different one" + ) + if canonical_claim is not None and packet_claim != canonical_claim: + errors.append( + f"{packet_path}: source.claim does not match the " + f"manifest's canonical source_claim for {claim_id}; " + "evidence for a different assertion cannot close this " + "lane" + ) expected_branch = BRANCH_BY_LANE.get(lane_name) branch = ( packet.get("target", {}).get("branch") @@ -2207,6 +2264,25 @@ def validate_tree( return errors +def collect_target_refs(plan_dir: Path) -> "list[str]": + """Unique `pull/N/head` refs named by packet fetch hints (negative + controls included; fetching an extra ref is harmless).""" + refs: set[str] = set() + for packet_path in sorted(plan_dir.rglob("*.json")): + try: + packet = json.loads(packet_path.read_text(encoding="utf-8")) + except OSError, ValueError: + continue + if not isinstance(packet, dict): + continue + target = packet.get("target") + hint = target.get("fetch_hint") if isinstance(target, dict) else None + match = FETCH_HINT_RE.match(hint.strip()) if isinstance(hint, str) else None + if match is not None: + refs.add(f"pull/{match.group(1)}/head") + return sorted(refs) + + def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument( @@ -2220,11 +2296,32 @@ def parse_args() -> argparse.Namespace: action="store_true", help="Require every evidence packet to record the current HEAD", ) + parser.add_argument( + "--fetch-target-refs", + action="store_true", + help=( + "Fetch every packet's recorded PR head ref into the local " + "object store and exit (CI aid: keeps squash-merged target " + "commits verifiable); validation still reports any ref that " + "stays unavailable" + ), + ) return parser.parse_args() def main() -> int: args = parse_args() + if args.fetch_target_refs: + for ref in collect_target_refs(args.plan_dir): + fetched = subprocess.run( + ["git", "-C", str(REPO_ROOT), "fetch", "origin", ref], + capture_output=True, + text=True, + check=False, + ) + state = "fetched" if fetched.returncode == 0 else "unavailable" + print(f"check_citation_evidence: {ref}: {state}") + return 0 freshness_head: str | None = None if args.freshness: freshness_head = _git_head(REPO_ROOT) diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index ca8808d9dbd5b..8c8ca55c53cd2 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -299,6 +299,8 @@ def _minimal_manifest(ids): "id": claim_id, "title": f"Claim {claim_id}", "source": "somewhere", + "source_url": "https://example.org/claim", + "source_claim": "A claim.", "first_wave_family": None, "lanes": { "dart7": { @@ -1799,7 +1801,7 @@ def test_maintenance_tasks_do_not_execute_evidence(): packet = complete_packet() packet["evidence"]["commands"] = ["pixi run build", "pixi run lint"] errors = MODULE.packet_errors(packet) - assert any("repository harness" in error for error in errors) + assert any("evidence-writer" in error for error in errors) def test_non_finite_csv_cells_do_not_count(tmp_path): @@ -1936,3 +1938,69 @@ def test_csv_measurement_columns_still_pass(tmp_path): def test_headerless_numeric_csv_still_passes(tmp_path): packet = _path_packet(tmp_path, "grid.csv", b"1.0,2.0\n3.0,4.0\n") assert MODULE.packet_errors(packet, base_dir=tmp_path) == [] + + +def test_fetch_hints_may_name_other_evidence_prs(): + packet = complete_packet() + packet["target"]["fetch_hint"] = ( + "git fetch origin pull/9999/head && git checkout " + HEAD_COMMIT + ) + assert MODULE.packet_errors(packet) == [] + + +def test_invalid_npy_dtype_widths_fail(tmp_path): + packet = _path_packet(tmp_path, "odd.npy", _npy_bytes(" Date: Sun, 16 Aug 2026 15:06:30 -0700 Subject: [PATCH 51/52] Close the Codex round-24 findings Sweep coordinates are excluded from row-measurement checks, repeat binding requires the supported trajectory/state_sha256 key family, the writer command must regenerate THIS claim's packet, raw JSON loads through the strict NaN/duplicate-key hooks, NPY format versions are whitelisted, sweep points canonicalize recursively, and the CT-004 hash covers the full articulated state (CT-004/CT-011 regenerated in the follow-up commit). Tests: 175 -> 181. --- .../verification.md | 22 ++++ scripts/check_citation_evidence.py | 108 +++++++++++++----- ...itation_ct004_articulated_energy_packet.py | 21 +++- ...tation_ct011_restore_equivalence_packet.py | 19 +++ tests/test_check_citation_evidence.py | 66 ++++++++++- 5 files changed, 203 insertions(+), 33 deletions(-) diff --git a/docs/dev_tasks/citation_driven_simulation_trust/verification.md b/docs/dev_tasks/citation_driven_simulation_trust/verification.md index 14b9455811941..35e4f37c1126c 100644 --- a/docs/dev_tasks/citation_driven_simulation_trust/verification.md +++ b/docs/dev_tasks/citation_driven_simulation_trust/verification.md @@ -868,3 +868,25 @@ widths are whitelisted per kind (` 175 (main), 149 -> 155 (6.20). + +## Codex review round 24 (PR #3445) — 2026-08-16 + +Two findings here plus five on the LTS lane, mirrored both ways. Sweep +coordinates no longer count as row measurements: a row restating its +declared coordinates (plus hashes) records no outcome, so each row +must carry a numeric/boolean value beyond bookkeeping, seed keys, AND +the sweep's coordinate keys. CT-004's repeat hash covered only total +energy and angular momentum — aggregates that do not uniquely identify +a multi-link state — so the writer now hashes every joint position and +velocity at every step (with finiteness gating) and the packet is +regenerated in the follow-up commit. From the LTS lane: NPY format +versions are whitelisted to 1.0/2.0/3.0; repeat binding requires the +supported trajectory/state_sha256 key family instead of any +hash-named field (CT-011 now emits a row-level trajectory_sha256 +chaining all its arm state digests, regenerated alongside); the +evidence-writer command must name THIS claim's writer +(write_citation_*.py); raw JSON artifacts load through the same +strict NaN/duplicate-key hooks as packets; and sweep-point +deduplication canonicalizes recursively so {"config": {"x": 1}} and +{"config": {"x": 1.0}} are one point, matching the row matcher's +equality. Tests: 175 -> 181 (main), 155 -> 161 (6.20). diff --git a/scripts/check_citation_evidence.py b/scripts/check_citation_evidence.py index 7fcd0260f0288..eb887442817c1 100644 --- a/scripts/check_citation_evidence.py +++ b/scripts/check_citation_evidence.py @@ -172,13 +172,15 @@ def _git(*argv: str) -> "subprocess.CompletedProcess[bytes]": return "missing" +REPEAT_HASH_KEY_RE = re.compile(r"^(?:repeat_)?(?:final_)?(?:trajectory|state)_sha256$") + + def _has_hash_list(value: object, length: int) -> bool: - """True when a hash-named key holds a list of >= `length` digest values.""" + """True when a supported repeat-hash key holds >= `length` digests.""" if isinstance(value, dict): for key, item in value.items(): - key_l = str(key).lower() if ( - ("sha256" in key_l or "hash" in key_l) + REPEAT_HASH_KEY_RE.match(str(key).lower()) and isinstance(item, list) and len(item) >= length and all( @@ -195,21 +197,19 @@ def _has_hash_list(value: object, length: int) -> bool: def _has_hash_leaf(value: object) -> bool: - """True when any nested key names a hash/sha256 with a non-empty value.""" + """True when a SUPPORTED repeat-hash key (the trajectory/state_sha256 + family) holds a digest or a list of digests; an arbitrarily named hash + field does not bind repeats to their recorded runs.""" if isinstance(value, dict): for key, item in value.items(): - key_l = str(key).lower() - if "sha256" in key_l or "hash" in key_l or "digest" in key_l: + if REPEAT_HASH_KEY_RE.match(str(key).lower()): if _is_nonempty_str(item) and HASH_VALUE_RE.match(item.strip()): return True - if isinstance(item, dict) and any( - _is_nonempty_str(entry) - and HASH_VALUE_RE.match(entry.strip().removeprefix("sha256:")) - for entry in item.values() + if isinstance(item, list) and any( + _is_nonempty_str(entry) and HASH_VALUE_RE.match(entry.strip()) + for entry in item ): return True - if isinstance(item, (list, dict)) and _has_hash_leaf(item): - return True if _has_hash_leaf(item): return True elif isinstance(value, list): @@ -474,7 +474,11 @@ def _raw_data_content_issue(path: "Path") -> "str | None": ) if suffix == ".json": try: - parsed = json.loads(path.read_text(encoding="utf-8")) + parsed = json.loads( + path.read_text(encoding="utf-8"), + parse_constant=_reject_nonstandard_constant, + object_pairs_hook=_reject_duplicate_keys, + ) except OSError, ValueError: return mismatch if not _has_measurement_leaf(parsed): @@ -483,7 +487,11 @@ def _raw_data_content_issue(path: "Path") -> "str | None": if suffix in (".jsonl", ".ndjson"): try: parsed_lines = [ - json.loads(line) + json.loads( + line, + parse_constant=_reject_nonstandard_constant, + object_pairs_hook=_reject_duplicate_keys, + ) for line in path.read_text(encoding="utf-8").splitlines() if line.strip() ] @@ -617,7 +625,11 @@ def _npy_parsed_header( if not head.startswith(b"\x93NUMPY") or len(head) < 10: return mismatch - major = head[6] + major, minor = head[6], head[7] + if (major, minor) not in {(1, 0), (2, 0), (3, 0)}: + # numpy.load() accepts only format versions 1.0/2.0/3.0; anything + # else is not a loadable artifact. + return mismatch if major == 1: (header_len,) = struct.unpack(" object: # 1 and 1.0 are the same coordinate; the row matcher's == - # treats them as equal, so the distinct-point check must too. - if isinstance(value, int) and not isinstance(value, bool): + # treats them as equal (recursively, through nested + # containers), so the distinct-point check must too. + if isinstance(value, bool): + return value + if isinstance(value, int): return float(value) + if isinstance(value, dict): + return {key: _canonical_value(item) for key, item in value.items()} + if isinstance(value, list): + return [_canonical_value(item) for item in value] return value for index, entry in enumerate(sweep): if isinstance(entry, dict) and entry: canonical_points.append( - json.dumps( - { - key: _canonical_value(value) - for key, value in entry.items() - }, - sort_keys=True, - ) + json.dumps(_canonical_value(entry), sort_keys=True) ) else: errors.append( @@ -1423,6 +1437,26 @@ def _canonical_value(value: object) -> object: "or maintenance tasks and unrelated scripts do not " "regenerate this packet's evidence" ) + elif ( + isinstance(claim_id, str) + and CLAIM_ID_RE.match(claim_id) + and not any( + re.search( + r"python[ \t].*scripts/write_citation_" + + re.escape(claim_id.lower().replace("-", "")) + + r"[a-z0-9_]*\.py", + commands[index], + ) + for index in non_build_indices + ) + ): + errors.append( + "no evidence.commands entry runs THIS claim's evidence " + f"writer (scripts/write_citation_" + f"{claim_id.lower().replace('-', '')}*.py); another " + "claim's writer regenerates a different experiment and " + "cannot reproduce these measurements" + ) elif min(build_indices) > min(non_build_indices): errors.append( "evidence.commands must run the build step BEFORE the " @@ -1515,6 +1549,12 @@ def _canonical_value(value: object) -> object: "raw evidence cannot demonstrate per-point coverage" ) if has_rows: + coordinate_keys = { + str(key) + for point in (sweep if isinstance(sweep, list) else []) + if isinstance(point, dict) + for key in point + } for index, row in enumerate(raw_rows): if not (isinstance(row, dict) and row): errors.append( @@ -1523,17 +1563,23 @@ def _canonical_value(value: object) -> object: "not raw evidence" ) continue + + def _terminal(path: str) -> str: + return re.sub(r"(\[\d+\])+$", "", path.rsplit(".", 1)[-1]) + if not any( (_is_finite_number(leaf) or isinstance(leaf, bool)) - and not _is_bookkeeping_key( - re.sub(r"(\[\d+\])+$", "", path.rsplit(".", 1)[-1]) - ) + and not _is_bookkeeping_key(_terminal(path)) + and not _is_seed_key(_terminal(path)) + and _terminal(path) not in coordinate_keys for path, leaf in _metric_leaves(row) ): errors.append( f"evidence.raw_rows[{index}] carries no numeric or " "boolean measurement beyond bookkeeping (seed/index/" - "id/...); a metadata-only record is not raw evidence" + "id/...) and declared sweep coordinates; a " + "coordinate restated is not an outcome, and a " + "metadata-only record is not raw evidence" ) if has_paths: for index, raw_path in enumerate(raw_paths): diff --git a/scripts/write_citation_ct004_articulated_energy_packet.py b/scripts/write_citation_ct004_articulated_energy_packet.py index 1e19c053ff942..7176bd1aafd35 100644 --- a/scripts/write_citation_ct004_articulated_energy_packet.py +++ b/scripts/write_citation_ct004_articulated_energy_packet.py @@ -156,7 +156,25 @@ def run_single( metrics = world.compute_step_metrics() total_energy = float(metrics.total_energy) angular_momentum = np.asarray(metrics.angular_momentum, dtype=float) - if not (math.isfinite(total_energy) and np.all(np.isfinite(angular_momentum))): + # Total energy and angular momentum are aggregates and do not + # uniquely identify a multi-link state, so the repeat hash covers + # every joint's position and velocity at every step. + articulated_state = np.asarray( + [ + float(value) + for joint in chain.joints + for value in ( + *np.atleast_1d(joint.position), + *np.atleast_1d(joint.velocity), + ) + ], + dtype=float, + ) + if not ( + math.isfinite(total_energy) + and np.all(np.isfinite(angular_momentum)) + and np.all(np.isfinite(articulated_state)) + ): non_finite = True break max_abs_energy_drift = max( @@ -165,6 +183,7 @@ def run_single( max_abs_angular_momentum = max( max_abs_angular_momentum, float(np.linalg.norm(angular_momentum)) ) + trajectory.update(articulated_state.tobytes()) trajectory.update(np.array([total_energy]).tobytes()) trajectory.update(angular_momentum.tobytes()) diff --git a/scripts/write_citation_ct011_restore_equivalence_packet.py b/scripts/write_citation_ct011_restore_equivalence_packet.py index b22bb8f503bfd..d06f3950968c5 100644 --- a/scripts/write_citation_ct011_restore_equivalence_packet.py +++ b/scripts/write_citation_ct011_restore_equivalence_packet.py @@ -271,9 +271,28 @@ def run_protocol(method_name: str, parameters: dict[str, Any]) -> dict[str, Any] ballistic.time = ballistic_time ballistic_restored = hash_continuation(ballistic, steps) + # The row-level trajectory digest chains every arm's full-state stream + # digest in a fixed order, so the repeat-binding contract's supported + # trajectory_sha256 field commits to all of this row's recorded runs. + trajectory_digest = hashlib.sha256( + "\n".join( + [ + continuation_hash, + inplace_first, + inplace_second, + *fresh_hashes, + *same_history_hashes, + precontact_hash, + ballistic_continuation, + ballistic_restored, + ] + ).encode("utf-8") + ).hexdigest() + return { "contact_solver_method": method_name, "resolved": resolved, + "trajectory_sha256": trajectory_digest, "continuation_sha256": continuation_hash, "inplace_restore_sha256": [inplace_first, inplace_second], "fresh_restore_sha256": fresh_hashes, diff --git a/tests/test_check_citation_evidence.py b/tests/test_check_citation_evidence.py index 8c8ca55c53cd2..0259e0cf793ec 100644 --- a/tests/test_check_citation_evidence.py +++ b/tests/test_check_citation_evidence.py @@ -1249,7 +1249,7 @@ def test_large_repeat_claims_need_per_repeat_hash_lists(): packet["ensemble"]["deterministic_repeats"] = 1000 errors = MODULE.packet_errors(packet) assert any("must show their repeats" in error for error in errors) - packet["evidence"]["repeat_hashes_sha256"] = ["a" * 64] * 1000 + packet["evidence"]["repeat_trajectory_sha256"] = ["a" * 64] * 1000 assert MODULE.packet_errors(packet) == [] @@ -2004,3 +2004,67 @@ def test_collect_target_refs_lists_packet_prs(tmp_path): negative={"schema": "dart.citation_claim_evidence/v1"}, ) assert MODULE.collect_target_refs(plan_dir) == ["pull/3445/head"] + + +def test_sweep_coordinates_are_not_measurements(): + packet = complete_packet() + packet["ensemble"]["sweep"] = [{"angle_deg": 0.0}, {"angle_deg": 15.0}] + packet["evidence"]["raw_rows"] = [ + {"angle_deg": 0.0, "trajectory_sha256": "d" * 64}, + {"angle_deg": 15.0, "trajectory_sha256": "e" * 64}, + ] + errors = MODULE.packet_errors(packet) + assert any("coordinate restated is not an outcome" in error for error in errors) + + +def test_repeat_hashes_must_use_supported_fields(): + packet = complete_packet() + packet["evidence"]["raw_rows"] = [{"angle_deg": 0.0, "lateral_drift_m": 0.0}] + packet["ensemble"]["unrelated_hash"] = "a" * 64 + errors = MODULE.packet_errors(packet) + assert any("supported repeat-hash field" in error for error in errors) + + +def test_writer_commands_must_match_the_claim(): + packet = complete_packet() + packet["evidence"]["commands"] = [ + "pixi run build", + "pixi run python scripts/write_citation_ct011_restore_equivalence_packet.py", + ] + errors = MODULE.packet_errors(packet) + assert any("THIS claim's evidence writer" in error for error in errors) + + +def test_raw_json_artifacts_reject_nonstandard_constants(tmp_path): + packet = _path_packet(tmp_path, "nan.json", b'{"metric": 1.0, "bad": NaN}') + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any( + "does not parse as its claimed raw-data format" in error for error in errors + ) + + packet = _path_packet(tmp_path, "dup.json", b'{"metric": 1.0, "metric": 2.0}') + errors = MODULE.packet_errors(packet, base_dir=tmp_path) + assert any( + "does not parse as its claimed raw-data format" in error for error in errors + ) + + +def test_nested_sweep_points_canonicalize_recursively(): + packet = complete_packet() + packet["ensemble"]["sweep"] = [{"config": {"x": 1}}, {"config": {"x": 1.0}}] + packet["evidence"]["raw_rows"] = [ + {"config": {"x": 1.0}, "lateral_drift_m": 0.0, "trajectory_sha256": "d" * 64}, + {"config": {"x": 1.0}, "lateral_drift_m": 0.1, "trajectory_sha256": "e" * 64}, + ] + errors = MODULE.packet_errors(packet) + assert any("at least two DISTINCT points" in error for error in errors) + + +def test_npy_versions_are_whitelisted(tmp_path): + data = bytearray(_npy_bytes(" Date: Sun, 16 Aug 2026 15:07:44 -0700 Subject: [PATCH 52/52] Regenerate CT-004 and CT-011 under the round-24 hash contract CT-004's trajectory digest now covers every joint position and velocity per step; CT-011 rows add the trajectory_sha256 chain over their arm state digests. Dispositions unchanged (unresolved). --- ...004-dart7-articulated-energy-momentum.json | 52 +++++++++---------- .../CT-011-dart7-restore-equivalence.json | 10 ++-- 2 files changed, 32 insertions(+), 30 deletions(-) diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json index b2d8da9add3f7..657ac12da19d0 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-004-dart7-articulated-energy-momentum.json @@ -407,8 +407,8 @@ "max_abs_energy_drift_j": 1.1453730886021916, "max_abs_relative_energy_drift": 0.5674949600416408, "repeat_trajectory_sha256": [ - "33a3b41b64f4546027954c752949346e3288deb713e3993639cc54b00f1e6b5b", - "33a3b41b64f4546027954c752949346e3288deb713e3993639cc54b00f1e6b5b" + "abf119f6bcd61168c35ad8ea827dec7e924a45eb6667a1b6d35b1ef2c95c1387", + "abf119f6bcd61168c35ad8ea827dec7e924a45eb6667a1b6d35b1ef2c95c1387" ], "resolved": { "gravity_mps2": [ @@ -452,7 +452,7 @@ }, "steps": 500, "timestep_s": 0.004, - "trajectory_sha256": "33a3b41b64f4546027954c752949346e3288deb713e3993639cc54b00f1e6b5b" + "trajectory_sha256": "abf119f6bcd61168c35ad8ea827dec7e924a45eb6667a1b6d35b1ef2c95c1387" }, { "final_total_energy_j": -1.5674156373231396, @@ -463,8 +463,8 @@ "max_abs_energy_drift_j": 0.5434502873429725, "max_abs_relative_energy_drift": 0.2692618694897877, "repeat_trajectory_sha256": [ - "ca6f6cab3e3f43b60e5cf1f5d108239cd10e4a78a6287aa0d46b994c5cd9ad03", - "ca6f6cab3e3f43b60e5cf1f5d108239cd10e4a78a6287aa0d46b994c5cd9ad03" + "edc3e45ca617293d4b4ddd3745e5cceb426bbed431f150bc2822b3d686f98442", + "edc3e45ca617293d4b4ddd3745e5cceb426bbed431f150bc2822b3d686f98442" ], "resolved": { "gravity_mps2": [ @@ -508,7 +508,7 @@ }, "steps": 1000, "timestep_s": 0.002, - "trajectory_sha256": "ca6f6cab3e3f43b60e5cf1f5d108239cd10e4a78a6287aa0d46b994c5cd9ad03" + "trajectory_sha256": "edc3e45ca617293d4b4ddd3745e5cceb426bbed431f150bc2822b3d686f98442" }, { "final_total_energy_j": -1.8004734230407382, @@ -519,8 +519,8 @@ "max_abs_energy_drift_j": 0.26467289477035383, "max_abs_relative_energy_drift": 0.13113677572528915, "repeat_trajectory_sha256": [ - "a8ab82d9906be5879e174ba1b6b38f64ce0714df75d4fa426b87c70897f39f5a", - "a8ab82d9906be5879e174ba1b6b38f64ce0714df75d4fa426b87c70897f39f5a" + "29563ff0b9fe3900df36266a1acd5059996ce5356f6c50a91c4021e93d5f92b7", + "29563ff0b9fe3900df36266a1acd5059996ce5356f6c50a91c4021e93d5f92b7" ], "resolved": { "gravity_mps2": [ @@ -564,7 +564,7 @@ }, "steps": 2000, "timestep_s": 0.001, - "trajectory_sha256": "a8ab82d9906be5879e174ba1b6b38f64ce0714df75d4fa426b87c70897f39f5a" + "trajectory_sha256": "29563ff0b9fe3900df36266a1acd5059996ce5356f6c50a91c4021e93d5f92b7" }, { "final_total_energy_j": -1.911266312486568, @@ -575,8 +575,8 @@ "max_abs_energy_drift_j": 0.1306014805077389, "max_abs_relative_energy_drift": 0.0647087684350685, "repeat_trajectory_sha256": [ - "828539234f8a9e0c7a1abc644374e751a5542dee391886e1708b8a1b82120c73", - "828539234f8a9e0c7a1abc644374e751a5542dee391886e1708b8a1b82120c73" + "caeb8e822ea1b23f24beb6a6da3a9313d6c447176b3078032c36ed982cd37d53", + "caeb8e822ea1b23f24beb6a6da3a9313d6c447176b3078032c36ed982cd37d53" ], "resolved": { "gravity_mps2": [ @@ -620,7 +620,7 @@ }, "steps": 4000, "timestep_s": 0.0005, - "trajectory_sha256": "828539234f8a9e0c7a1abc644374e751a5542dee391886e1708b8a1b82120c73" + "trajectory_sha256": "caeb8e822ea1b23f24beb6a6da3a9313d6c447176b3078032c36ed982cd37d53" }, { "final_total_energy_j": -2.0437843185234734, @@ -631,8 +631,8 @@ "max_abs_energy_drift_j": 0.1755340480261065, "max_abs_relative_energy_drift": 0.08697138824179299, "repeat_trajectory_sha256": [ - "3b260b2aad289ad0ee2c5d0bb9a88fbc0887ed7ba0bed29b988078cfb9f6ec36", - "3b260b2aad289ad0ee2c5d0bb9a88fbc0887ed7ba0bed29b988078cfb9f6ec36" + "521eb95a4346c98ef2209d21824e8d624147084f0bbfc545c2565fc97497d15a", + "521eb95a4346c98ef2209d21824e8d624147084f0bbfc545c2565fc97497d15a" ], "resolved": { "gravity_mps2": [ @@ -676,7 +676,7 @@ }, "steps": 500, "timestep_s": 0.004, - "trajectory_sha256": "3b260b2aad289ad0ee2c5d0bb9a88fbc0887ed7ba0bed29b988078cfb9f6ec36" + "trajectory_sha256": "521eb95a4346c98ef2209d21824e8d624147084f0bbfc545c2565fc97497d15a" }, { "final_total_energy_j": -2.0311100411781546, @@ -687,8 +687,8 @@ "max_abs_energy_drift_j": 0.08610771893666502, "max_abs_relative_energy_drift": 0.042663562644792916, "repeat_trajectory_sha256": [ - "c320c585b91fb94fdffb008502a85970585970f4d7b6db6e591339c6ece79582", - "c320c585b91fb94fdffb008502a85970585970f4d7b6db6e591339c6ece79582" + "0e540b5812b22d2173fa7b97ef7f8afc57ff92e652d34afc263424fd7b300097", + "0e540b5812b22d2173fa7b97ef7f8afc57ff92e652d34afc263424fd7b300097" ], "resolved": { "gravity_mps2": [ @@ -732,7 +732,7 @@ }, "steps": 1000, "timestep_s": 0.002, - "trajectory_sha256": "c320c585b91fb94fdffb008502a85970585970f4d7b6db6e591339c6ece79582" + "trajectory_sha256": "0e540b5812b22d2173fa7b97ef7f8afc57ff92e652d34afc263424fd7b300097" }, { "final_total_energy_j": -2.02471920214893, @@ -743,8 +743,8 @@ "max_abs_energy_drift_j": 0.04264872954602339, "max_abs_relative_energy_drift": 0.02113105267654267, "repeat_trajectory_sha256": [ - "74505b657816a4f2795e8817e1a4e448bf75e939797992abf2d878a4d8ddf8ae", - "74505b657816a4f2795e8817e1a4e448bf75e939797992abf2d878a4d8ddf8ae" + "e278c0f674dc725652569cab497ea04f8989b2fedf47e495a5baccbbae9142d7", + "e278c0f674dc725652569cab497ea04f8989b2fedf47e495a5baccbbae9142d7" ], "resolved": { "gravity_mps2": [ @@ -788,7 +788,7 @@ }, "steps": 2000, "timestep_s": 0.001, - "trajectory_sha256": "74505b657816a4f2795e8817e1a4e448bf75e939797992abf2d878a4d8ddf8ae" + "trajectory_sha256": "e278c0f674dc725652569cab497ea04f8989b2fedf47e495a5baccbbae9142d7" }, { "final_total_energy_j": -2.0215116473513888, @@ -799,8 +799,8 @@ "max_abs_energy_drift_j": 0.02121937787327255, "max_abs_relative_energy_drift": 0.01051350875809135, "repeat_trajectory_sha256": [ - "dac8263daea9125b41a51bb9fed6ef1e4e965f79e4256844dfe826ea243a92dd", - "dac8263daea9125b41a51bb9fed6ef1e4e965f79e4256844dfe826ea243a92dd" + "ab28b12acca2a248dfa1f15d234429eef6eaf4ae7692e54e3911d3e76813268c", + "ab28b12acca2a248dfa1f15d234429eef6eaf4ae7692e54e3911d3e76813268c" ], "resolved": { "gravity_mps2": [ @@ -844,7 +844,7 @@ }, "steps": 4000, "timestep_s": 0.0005, - "trajectory_sha256": "dac8263daea9125b41a51bb9fed6ef1e4e965f79e4256844dfe826ea243a92dd" + "trajectory_sha256": "ab28b12acca2a248dfa1f15d234429eef6eaf4ae7692e54e3911d3e76813268c" } ], "visual": { @@ -978,9 +978,9 @@ }, "target": { "branch": "main", - "commit": "f4e9e5c11d5d3b443693cde9142bf317e2115cd3", + "commit": "efd0e8bb758fb10aa8d9606aaaae0b4015ece9e7", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", - "fetch_hint": "git fetch origin pull/3445/head && git checkout f4e9e5c11d5d3b443693cde9142bf317e2115cd3" + "fetch_hint": "git fetch origin pull/3445/head && git checkout efd0e8bb758fb10aa8d9606aaaae0b4015ece9e7" }, "title": "Articulated energy drift versus timestep across integration families (DART 7 first packet)" } diff --git a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json index 41931b9516414..1d2cd80c52782 100644 --- a/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json +++ b/docs/plans/123-citation-driven-simulation-trust/evidence/CT-011-dart7-restore-equivalence.json @@ -179,7 +179,8 @@ "same_history_restore_sha256": [ "26aeeada76d96c05e831b49bc015344dff3f89696f31b414673063165444d506", "26aeeada76d96c05e831b49bc015344dff3f89696f31b414673063165444d506" - ] + ], + "trajectory_sha256": "7a1aff8619cb5f4170d4573c3a21e7ed59b5b22377bf2bb13651ff4bae6c8c3c" }, { "ballistic_control_sha256": [ @@ -247,7 +248,8 @@ "same_history_restore_sha256": [ "538fd382678f25ae2fb3b8d841470797336ed283c69bea3c793daa6aa1b6f34d", "538fd382678f25ae2fb3b8d841470797336ed283c69bea3c793daa6aa1b6f34d" - ] + ], + "trajectory_sha256": "a872beabf67b865b484d53bed8c24d15513c46941e1516bdc36a5c4ddbd5579a" } ], "visual": { @@ -373,9 +375,9 @@ }, "target": { "branch": "main", - "commit": "3fe6fbfbfcdf090a8f265d7758088f50957987b0", + "commit": "efd0e8bb758fb10aa8d9606aaaae0b4015ece9e7", "commit_role": "Source state measured: the library and fixture ran at this commit, which is HEAD at capture time. The packet and its writer land in a later commit.", - "fetch_hint": "git fetch origin pull/3445/head && git checkout 3fe6fbfbfcdf090a8f265d7758088f50957987b0" + "fetch_hint": "git fetch origin pull/3445/head && git checkout efd0e8bb758fb10aa8d9606aaaae0b4015ece9e7" }, "title": "State-vector restore equivalence under contact (DART 7 first packet)" }