From 249fccd0464f752a568b26dbec1a4256955d030b Mon Sep 17 00:00:00 2001 From: XV Date: Mon, 21 Sep 2026 01:23:44 -0400 Subject: [PATCH] Remove retired SAG and Letta benchmark implementations --- README.md | 4 + config/benchmark/benchmark-v3.json | 28 - config/benchmark/locks/letta.in | 1 - config/benchmark/locks/letta.txt | 183 ------ docker/benchmark/compose.yml | 37 -- docker/benchmark/letta.Dockerfile | 15 - docker/benchmark/sag.Dockerfile | 26 - docs/log.md | 4 + docs/reference/benchmark-target-lifecycle.md | 63 ++ docs/reference/index.md | 3 + scripts/benchmark-unit.py | 27 - scripts/benchmark_contract/fixtures.py | 2 - scripts/benchmark_runner/cli.py | 4 + scripts/benchmark_runner/docker.py | 2 - scripts/benchmark_targets/letta.py | 527 ----------------- scripts/benchmark_targets/sag.py | 574 ------------------- scripts/tests/test_benchmark_adapters.py | 81 --- scripts/tests/test_benchmark_contract.py | 5 +- scripts/tests/test_benchmark_report.py | 4 +- scripts/tests/test_benchmark_runner.py | 15 + 20 files changed, 99 insertions(+), 1506 deletions(-) delete mode 100644 config/benchmark/locks/letta.in delete mode 100644 config/benchmark/locks/letta.txt delete mode 100644 docker/benchmark/letta.Dockerfile delete mode 100644 docker/benchmark/sag.Dockerfile create mode 100644 docs/reference/benchmark-target-lifecycle.md delete mode 100644 scripts/benchmark_targets/letta.py delete mode 100644 scripts/benchmark_targets/sag.py diff --git a/README.md b/README.md index 19e3af38..ea59cf78 100644 --- a/README.md +++ b/README.md @@ -2,6 +2,10 @@ # ELF +For current competitor runs and retired target versions, see the +[benchmark target lifecycle](docs/reference/benchmark-target-lifecycle.md). +Older dated benchmark reports describe their original frozen implementations. + Evidence-linked fact memory for agents. [![License](https://img.shields.io/badge/License-GPLv3-blue.svg)](https://www.gnu.org/licenses/gpl-3.0) diff --git a/config/benchmark/benchmark-v3.json b/config/benchmark/benchmark-v3.json index ed0f2c55..5ef7ead5 100644 --- a/config/benchmark/benchmark-v3.json +++ b/config/benchmark/benchmark-v3.json @@ -114,34 +114,6 @@ "service_image": "falkordb/falkordb@sha256:286d829e23ea4f927d8c06859efb12a3288d00b03746a517f4d8c22b75202015" } }, - { - "id": "letta", - "adapter": "native_archival_search", - "role": "stateful_core_and_archival_memory", - "score_eligible": true, - "suites": ["common-core-v1", "memory-lifecycle-v1"], - "image": "elf-benchmark-letta:v4", - "dockerfile": "docker/benchmark/letta.Dockerfile", - "pin": { - "kind": "container_image", - "value": "letta/letta@sha256:aa66c3eeee13d2dfc40c650d709b550237ee31bfc91942a52fa488a13fa8c102", - "client_lock": "config/benchmark/locks/letta.txt" - } - }, - { - "id": "sag", - "adapter": "native_multi_search_source_id_trace", - "role": "structure_aware_graph_and_multi_search", - "score_eligible": true, - "suites": ["common-core-v1", "knowledge-structure-v1"], - "image": "elf-benchmark-sag:v4", - "dockerfile": "docker/benchmark/sag.Dockerfile", - "native_deviations": { - "embedding_dimensions": 1024, - "reason": "The pinned unmodified SAG schema supports 1024-dimensional embeddings." - }, - "pin": {"kind": "git", "source": "https://github.com/Zleap-AI/SAG.git", "value": "84a5b8c9bd45944b8a3cd76e0ba4762ce68ccc72"} - }, { "id": "pageindex", "adapter": "native_hierarchy_construction", diff --git a/config/benchmark/locks/letta.in b/config/benchmark/locks/letta.in deleted file mode 100644 index 40f300c4..00000000 --- a/config/benchmark/locks/letta.in +++ /dev/null @@ -1 +0,0 @@ -letta-client==0.1.148 diff --git a/config/benchmark/locks/letta.txt b/config/benchmark/locks/letta.txt deleted file mode 100644 index 64f6f28b..00000000 --- a/config/benchmark/locks/letta.txt +++ /dev/null @@ -1,183 +0,0 @@ -# This file was autogenerated by uv via the following command: -# uv pip compile --python-version 3.11 --python-platform aarch64-manylinux_2_28 --generate-hashes config/benchmark/locks/letta.in -o config/benchmark/locks/letta.txt -annotated-types==0.7.0 \ - --hash=sha256:1f02e8b43a8fbbc3f3e0d4f0f4bfc8131bcb4eebe8849b8e5c773f3a1c582a53 \ - --hash=sha256:aff07c09a53a08bc8cfccb9c85b05f1aa9a2a6f23728d790723543408344ce89 - # via pydantic -anyio==4.14.2 \ - --hash=sha256:9f505dda5ac9f0c8309b5e8bd445a8c2bf7246f3ce950121e45ea15bc41d1494 \ - --hash=sha256:cfa139f3ed1a23ee8f88a145ddb5ac7605b8bbfd8592baacd7ce3d8bb4313c7f - # via httpx -certifi==2026.6.17 \ - --hash=sha256:024c88eeec92ca068db80f02b8b07c9cef7b9fe261d1d535abfd5abd6f6af432 \ - --hash=sha256:2227dcbaafe0d2f59279d1762ddddc37783ed4354594f194ffc31d20f41fc3db - # via - # httpcore - # httpx -h11==0.16.0 \ - --hash=sha256:4e35b956cf45792e4caa5885e69fba00bdbc6ffafbfa020300e549b208ee5ff1 \ - --hash=sha256:63cf8bbe7522de3bf65932fda1d9c2772064ffb3dae62d55932da54b31cb6c86 - # via httpcore -httpcore==1.0.9 \ - --hash=sha256:2d400746a40668fc9dec9810239072b40b4484b640a8c38fd654a024c7a1bf55 \ - --hash=sha256:6e34463af53fd2ab5d807f399a9b45ea31c3dfa2276f15a2c3f00afff6e176e8 - # via httpx -httpx==0.28.1 \ - --hash=sha256:75e98c5f16b0f35b567856f597f06ff2270a374470a5c2392242528e3e3e42fc \ - --hash=sha256:d909fcccc110f8c7faf814ca82a9a4d816bc5a6dbfea25d6591d6985b8ba59ad - # via letta-client -httpx-sse==0.4.0 \ - --hash=sha256:1e81a3a3070ce322add1d3529ed42eb5f70817f45ed6ec915ab753f961139721 \ - --hash=sha256:f329af6eae57eaa2bdfd962b42524764af68075ea87370a2de920af5341e318f - # via letta-client -idna==3.18 \ - --hash=sha256:7f952cbe720b688055e3f87de14f5c3e5fdaa8bc3928985c4077ca689de849a2 \ - --hash=sha256:ffb385a7e039654cef1ab9ef32c6fafe283c0c0467bba1d9029738ce4a14a848 - # via - # anyio - # httpx -letta-client==0.1.148 \ - --hash=sha256:48ba8b95427c5b86e33845695e5d3d6bd1b0fbf9e608b767318f26ca26e343f2 \ - --hash=sha256:ca0aac9faa8ebfeb313dc694ab04072ff618daf0aa9fe606f16eab881387b709 - # via -r config/benchmark/locks/letta.in -pydantic==2.13.4 \ - --hash=sha256:45a282cde31d808236fd7ea9d919b128653c8b38b393d1c4ab335c62924d9aba \ - --hash=sha256:c40756b57adaa8b1efeeced5c196f3f3b7c435f90e84ea7f443901bec8099ef6 - # via letta-client -pydantic-core==2.46.4 \ - --hash=sha256:00c603d540afdd6b80eb39f078f33ebd46211f02f33e34a32d9f053bba711de0 \ - --hash=sha256:0186750b482eefa11d7f435892b09c5c606193ef3375bcf94aa00ae6bfb66262 \ - --hash=sha256:041bde0a48fd37cf71cab1c9d56d3e8625a3793fef1f7dd232b3ff37e978ecda \ - --hash=sha256:0c563b08bca408dc7f65f700633d8442fffb2421fc47b8101377e9fd65051ff0 \ - --hash=sha256:0cbe8b01f948de4286c74cdd6c667aceb38f5c1e26f0693b3983d9d74887c65e \ - --hash=sha256:0ce40cd7b21210e99342afafbd4d0f76d784eb5b1d60f3bdc566be4983c6c73b \ - --hash=sha256:0e96592440881c74a213e5ad528e2b24d3d4f940de2766bed9010ab1d9e51594 \ - --hash=sha256:10e17cbb10a330363733efc4d7c4d0dd827ac0909b8f6a6542298fed1ea62f29 \ - --hash=sha256:133878133d271ade3d41d1bfb2a45ec38dbdbda40bc065921c6b04e4630127e2 \ - --hash=sha256:14d4edf427bdcf950a8a02d7cb44a08614388dd6e1bdcbf4f67504fa7887da9c \ - --hash=sha256:14f4c5d6db102bd796a627bbb3a17b4cf4574b9ae861d8b7c9a9661c6dd3362d \ - --hash=sha256:17299feefe090f2caa5b8e37222bb5f663e4935a8bfa6931d4102e5df1a9f398 \ - --hash=sha256:184c081504d17f1c1066e430e117142b2c77d9448a97f7b65c6ac9fd9aee238d \ - --hash=sha256:18e5ceec2ab67e6d5f1a9085e5a24c9c4e2ac4545730bfe668680bca05e555f3 \ - --hash=sha256:19e51f073cd3df251856a8a4189fbdf1de4012c3ebacfb1884f94f1eb406079f \ - --hash=sha256:1a7dd0b3ee80d90150e3495a3a13ac34dbcbfd4f012996a6a1d8900e91b5c0fb \ - --hash=sha256:1d8ba486450b14f3b1d63bc521d410ec7565e52f887b9fb671791886436a42f7 \ - --hash=sha256:2108ba5c1c1eca18030634489dc544844144ee36357f2f9f780b93e7ddbb44b5 \ - --hash=sha256:228ee9bae8bef5b1e97ec58302f80357c37199e0d0a99174e138d28e6957b9d9 \ - --hash=sha256:23ace664830ee0bfe014a0c7bc248b1f7f25ed7ad103852c317624a1083af462 \ - --hash=sha256:2412e734dcb48da14d4e4006b82b46b74f2518b8a26ee7e58c6844a6cd6d03c4 \ - --hash=sha256:29c61fc04a3d840155ff08e475a04809278972fe6aef51e2720554e96367e34b \ - --hash=sha256:2f84c03c8607173d16b5a854ec68a2f9079ae03237a54fb506d13af47e1d018d \ - --hash=sha256:3009f12e4e90b7f88b4f9adb1b0c4a3d58fe7820f3238c190047209d148026df \ - --hash=sha256:3245406455a5d98187ec35530fd772b1d799b26667980872c8d4614991e2c4a2 \ - --hash=sha256:3447661d99f75a3683a4cf5c87da72f2161964611864dbbeac7fbb118bb4bfc0 \ - --hash=sha256:372429a130e469c9cd698925ce5fc50940b7a1336b0d82038e63d5bbc4edc519 \ - --hash=sha256:395aebd9183f9d112f569aeb5b2214d1a10a33bec8456447f7fbdfa51d38d4cd \ - --hash=sha256:3a233125ac121aa3ffba9a2b59edfc4a985a76092dc8279586ab4b71390875e7 \ - --hash=sha256:3be77f45df024d789a672ae34f8b06fb346c4f9f46ea714956660ea4862e89ac \ - --hash=sha256:3bf92c5d0e00fefaab325a4d27828fe6b6e2a21848686b5b60d2d9eeb09d76c6 \ - --hash=sha256:3ecbc122d18468d06ca279dc26a8c2e2d5acb10943bb35e36ae92096dc3b5565 \ - --hash=sha256:3fb702cd90b0446a3a1c5e470bfa0dd23c0233b676a9099ddcc964fa6ca13898 \ - --hash=sha256:428e04521a40150c85216fc8b85e8d39fece235a9cf5e383761238c7fa9b96fb \ - --hash=sha256:432c179df7874eeb73307aad2df0755e1ae0efa61ff0ea89b93e194411ae3928 \ - --hash=sha256:4a05d69cba51d852c5c3e92758653245a50c0b646ced0cf05bd793ed592839d6 \ - --hash=sha256:4c63ebc82684aa89d9a3bcbd13d515b3be44250dc68dd3bd81526c1cb31286c3 \ - --hash=sha256:4fc73cb559bdb54b1134a706a2802a4cddd27a0633f5abb7e53056268751ac6a \ - --hash=sha256:4fcbe087dbc2068af7eda3aa87634eba216dbda64d1ae73c8684b621d33f6596 \ - --hash=sha256:56cb4851bcaf3d117eddcef4fe66afd750a50274b0da8e22be256d10e5611987 \ - --hash=sha256:5855698a4856556d86e8e6cd8434bc3ac0314ee8e12089ae0e143f64c6256e4e \ - --hash=sha256:5a4330cdbc57162e4b3aa303f588ba752257694c9c9be3e7ebb11b4aca659b5d \ - --hash=sha256:5b712b53160b79a5850310b912a5ef8e57e56947c8ad690c227f5c9d7e561712 \ - --hash=sha256:5d5902252db0d3cedf8d4a1bc68f70eeb430f7e4c7104c8c476753519b423008 \ - --hash=sha256:617d7e2ca7dcb8c5cf6bcb8c59b8832c94b36196bbf1cbd1bfb56ed341905edd \ - --hash=sha256:62f875393d7f270851f20523dd2e29f082bcc82292d66db2b64ea71f64b6e1c1 \ - --hash=sha256:633147d34cf4550417f12e2b1a0383973bdf5cdfde212cb09e9a581cf10820be \ - --hash=sha256:66ce7632c22d837c95301830e111ad0128a32b8207533b60896a96c4915192ea \ - --hash=sha256:6b3ace8194b0e5204818c92802dcdca7fc6d88aabbb799d7c795540d9cd6d292 \ - --hash=sha256:6f2eeda33a839975441c86a4119e1383c50b47faf0cbb5176985565c6bb02c33 \ - --hash=sha256:7027560ee92211647d0d34e3f7cd6f50da56399d26a9c8ad0da286d3869a53f3 \ - --hash=sha256:7283d57845ecf5a163403eb0702dfc220cc4fbdd18919cb5ccea4f95ee1cdab4 \ - --hash=sha256:7a5f930472650a82629163023e630d160863fce524c616f4e5186e5de9d9a49b \ - --hash=sha256:7bfb192b3f4b9e8a89b6277b6ce787564f62cfd272055f6e685726b111dc7826 \ - --hash=sha256:811ff8e9c313ab425368bcbb36e5c4ebd7108c2bbf4e4089cfbb0b01eff63fac \ - --hash=sha256:8233f2947cf85404441fd7e0085f53b10c93e0ee78611099b5c7237e36aacbf7 \ - --hash=sha256:82cf5301172168103724d49a1444d3378cb20cdee30b116a1bd6031236298a5d \ - --hash=sha256:8358a950c8909158e3df31538a7e4edc2d7265a7c54b47f0864d9e5bae9dcebf \ - --hash=sha256:85bb3611ff1802f3ee7fdd7dbff26b56f343fb432d57a4728fdd49b6ef35e2f4 \ - --hash=sha256:86e1a4418c6cd97d60c95c71164158eaf7324fae7b0923264016baa993eba6fc \ - --hash=sha256:8b9bab013d1c7a79d3501ff86d0bc9c31bf587db4551677b96bec07df78c6b15 \ - --hash=sha256:8c5dac79fa1614d1e06ca695109c6105923bd9c7d1d6c918d4e637b7e6b32fd3 \ - --hash=sha256:8d0820e8192167f80d88d64038e609c31452eeca865b4e1d9950a27a4609b00b \ - --hash=sha256:8daafc69c93ee8a0204506a3b6b30f586ef54028f52aeeeb5c4cfc5184fd5914 \ - --hash=sha256:9037063db01f09b09e237c282b6792bd4da634b5402c4e7f0c61effed7701a04 \ - --hash=sha256:905a0ed8ea6f2d61c1738835f99b699348d7857379083e5fc497fa0c967a407c \ - --hash=sha256:90884113d8b48f760e9587002789ddd741e76ab9f89518cd1e43b1f1a52ec44b \ - --hash=sha256:91a06d2e259ecfbd8c901d70c3c507900458498142b3026a296b7de4d1322cc9 \ - --hash=sha256:926c9541b14b12b1681dca8a0b75feb510b06c6341b70a8e500c2fdcff837cce \ - --hash=sha256:9401557acd873c3a7f3eb9383edef8ac4968f9510e340f4808d427e75667e7b4 \ - --hash=sha256:9551187363ffc0de2a00b2e47c25aeaeb1020b69b668762966df15fc5659dd5a \ - --hash=sha256:962ccbab7b642487b1d8b7df90ef677e03134cf1fd8880bf698649b22a69371f \ - --hash=sha256:97e7cf2be5c77b7d1a9713a05605d49460d02c6078d38d8bef3cbe323c548424 \ - --hash=sha256:9aa768456404a8bf48a4406685ac2bec8e72b62c69313734fa3b73cf33b3a894 \ - --hash=sha256:9bc519fbf2b7578398853d815009ae5e4d4603d12f4e3f91da8c06852d3da3e9 \ - --hash=sha256:9d56801be94b86a9da183e5f3766e6310752b99ff647e38b09a9500d88e46e76 \ - --hash=sha256:9f444c499b3eefd3a92e348059471ea0c3a6e303d9c1cec09fa748fd9f895201 \ - --hash=sha256:9fa8ae11da9e2b3126c6426f147e0fba88d96d65921799bb30c6abd1cb2c97fb \ - --hash=sha256:a0f62d0a58f4e7da165457e995725421e0064f2255d8eccebc49f41bbc23b109 \ - --hash=sha256:a396dcc17e5a0b164dbe026896245a4fa9ff402edca1dff0be3d53a517f74de4 \ - --hash=sha256:aaa2a54443eff1950ba5ddc6b6ccda0d9c84a364276a62f969bdf2a390650848 \ - --hash=sha256:ad785e92e6dc634c21555edc8bd6b64957ab844541bcb96a1366c202951ae526 \ - --hash=sha256:af8244b2bef6aaad6d92cda81372de7f8c8d36c9f0c3ea36e827c60e7d9467a0 \ - --hash=sha256:b078afbc25f3a1436c7a1d2cd3e322497ee99615ba97c563566fdf46aff1ee01 \ - --hash=sha256:b2f69dec1725e79a012d920df1707de5caf7ed5e08f3be4435e25803efc47458 \ - --hash=sha256:b8458003118a712e66286df6a707db01c52c0f52f7db8e4a38f0da1d3b94fc4e \ - --hash=sha256:bb63e0198ca18aad131c089b9204c23079c3afa95487e561f4c522d519e55aba \ - --hash=sha256:bfec22eab3c8cc2ceec0248aec886624116dc079afa027ecc8ad4a7e62010f8a \ - --hash=sha256:c1747f85cee84c26985853c6f3d9bd3e75da5212912443fa111c113b9c246f39 \ - --hash=sha256:c1b3f518abeca3aa13c712fd202306e145abf59a18b094a6bafb2d2bbf59192c \ - --hash=sha256:c50f2528cf200c5eed56faf3f4e22fcd5f38c157a8b78576e6ba3168ec35f000 \ - --hash=sha256:c68fcd102d71ea85c5b2dfac3f4f8476eff42a9e078fd5faefff6d145063536b \ - --hash=sha256:c7a7bd4e39e8e4c12c39cd480356842b6a8a06e41b23a55a5e3e191718838ddf \ - --hash=sha256:c94f0688e7b8d0a67abf40e57a7eaaecd17cc9586706a31b76c031f63df052b4 \ - --hash=sha256:cbaf13819775b7f769bf4a1f066cb6df7a28d4480081a589828ef190226881cd \ - --hash=sha256:cd2213145bcc2ba85884d0ac63d222fece9209678f77b9b4d76f054c561adb28 \ - --hash=sha256:ce5c1d2a8b27468f433ca974829c44060b8097eedc39933e3c206a90ee49c4a9 \ - --hash=sha256:d396ec2b979760aaf3218e76c24e65bd0aca24983298653b3a9d7a45f9e47b30 \ - --hash=sha256:d51026d73fcfd93610abc7b27789c26b313920fcfb20e27462d74a7f8b06e983 \ - --hash=sha256:d80ee3d731373b24cebbc10d689ca4ee1875caf0d5703a245db18efd4dd37fc1 \ - --hash=sha256:d995260fdf4e1db774581b4900e0f832abe3c7c84996726bbc161b19c8f29e76 \ - --hash=sha256:da4b951fe36dc7c3a1ccb4e3cd1747c3542b8c9ceede8fc86cae054e764485f5 \ - --hash=sha256:daa27d92c36f24388fe3ad306b174781c747627f134452e4f128ea00ce1fe8c4 \ - --hash=sha256:db06ffe51636ffe9ca531fe9023dd64bdd794be8754cb5df57c5498ae5b518a7 \ - --hash=sha256:e0d65b8c354be7fb5f720c3caa8bc940bc2d20ce749c8e06135f07f8ed95dd7c \ - --hash=sha256:e68b7a074f65a2fd746c52a7ce6142ab7006074ac269ace0c25cd8ba171f8066 \ - --hash=sha256:e739fee756ba1010f8bcccb534252e85a35fe45ae92c295a06059ce58b74ccd3 \ - --hash=sha256:e846ae7835bf0703ae43f534ab79a867146dadd59dc9ca5c8b53d5c8f7c9ef02 \ - --hash=sha256:e9c26f834c65f5752f3f06cb08cb86a913ceb7274d0db6e267808a708b46bc89 \ - --hash=sha256:ea793e075b70290d89d8142074262885d3f7da19634845135751bd6344f73b50 \ - --hash=sha256:f027324c56cd5406ca49c124b0db10e56c69064fec039acc571c29020cc87c76 \ - --hash=sha256:f13a646d65d09fbf1bc6b3a9635d30095c8e7e5cc419ff35ecc563c5fd04cd49 \ - --hash=sha256:f47286a97f0bc9b8859519809077b91b2cefe4ae47fcbf5e466a009c1c5d742b \ - --hash=sha256:f747929cf940cddb5b3668a390056ddd5ba2e5010615ea2dcf4f9c4f3ab8791d \ - --hash=sha256:f99626688942fb746e545232e7726926f3be91b5975f8b55327665fafda991c7 \ - --hash=sha256:f9fa868638bf362d3d138ea55829cefb3d5f4b0d7f142234382a15e2485dbec4 \ - --hash=sha256:fbdb89b3e1c94a30cc5edfce477c6e6a5dc4d8f84665b455c27582f211a1c72c \ - --hash=sha256:fc010ab034c8c7452522748bf937df58020d256ccae0874463d1f4d01758af8e \ - --hash=sha256:fc3e9034a63de20e15e8ade85358bc6efc614008cab72898b4b4952bea0509ff \ - --hash=sha256:fd8b3d9fd264be37976686c7f65cd52a83f5e84f4bfd2adf9c1d469676bbb6ae - # via - # letta-client - # pydantic -typing-extensions==4.16.0 \ - --hash=sha256:481caa481374e813c1b176ada14e97f1f67a4539ce9cfeb3f350d78d6370c2e8 \ - --hash=sha256:dc983d19a509c94dba722ee6abd33940f7c05a89e243c47e907eb4db6f1a43e5 - # via - # anyio - # letta-client - # pydantic - # pydantic-core - # typing-inspection -typing-inspection==0.4.2 \ - --hash=sha256:4ed1cacbdc298c220f1bd249ed5287caa16f34d44ef4e9c3d0cbad5b521545e7 \ - --hash=sha256:ba561c48a67c5958007083d386c3295464928b01faa735ab8547c5692e87f464 - # via pydantic diff --git a/docker/benchmark/compose.yml b/docker/benchmark/compose.yml index b57a9e06..b979f065 100644 --- a/docker/benchmark/compose.yml +++ b/docker/benchmark/compose.yml @@ -56,22 +56,6 @@ services: volumes: - postgres-data:/var/lib/postgresql - sag-postgres: - image: pgvector/pgvector:pg18 - environment: - POSTGRES_DB: sag_benchmark - POSTGRES_PASSWORD: ${BENCHMARK_POSTGRES_PASSWORD} - POSTGRES_USER: sag_benchmark - healthcheck: - test: - - CMD-SHELL - - pg_isready -U sag_benchmark -d sag_benchmark - interval: 1s - timeout: 5s - retries: 60 - volumes: - - sag-postgres-data:/var/lib/postgresql - qdrant: image: qdrant/qdrant:v1.18.0 volumes: @@ -255,10 +239,6 @@ services: - --target - graphrag - letta-unit: - <<: *unit - image: ${BENCHMARK_LETTA_IMAGE:-elf-benchmark-letta:v4} - openkb-unit: <<: *unit image: ${BENCHMARK_OPENKB_IMAGE:-elf-benchmark-openkb:v4} @@ -271,22 +251,6 @@ services: - --target - openkb - sag-unit: - <<: *unit - image: ${BENCHMARK_SAG_IMAGE:-elf-benchmark-sag:v4} - depends_on: - sag-postgres: - condition: service_healthy - environment: - <<: *unit-environment - DATABASE_URL: postgres://sag_benchmark:${BENCHMARK_POSTGRES_PASSWORD}@sag-postgres:5432/sag_benchmark - EMBEDDING_DIMENSIONS: "1024" - command: - - python3 - - /opt/benchmark/benchmark-unit.py - - --target - - sag - honcho-unit: <<: *unit image: ${BENCHMARK_HONCHO_IMAGE:-elf-benchmark-honcho:v4} @@ -313,6 +277,5 @@ volumes: lightrag-data: lightrag-inputs: falkordb-data: - sag-postgres-data: honcho-postgres-data: honcho-redis-data: diff --git a/docker/benchmark/letta.Dockerfile b/docker/benchmark/letta.Dockerfile deleted file mode 100644 index 39275795..00000000 --- a/docker/benchmark/letta.Dockerfile +++ /dev/null @@ -1,15 +0,0 @@ -FROM letta/letta@sha256:aa66c3eeee13d2dfc40c650d709b550237ee31bfc91942a52fa488a13fa8c102 - -COPY config/benchmark/locks/letta.txt /opt/benchmark/locks/letta.txt -RUN python3 -m venv /opt/benchmark-venv \ - && /opt/benchmark-venv/bin/pip install --no-cache-dir \ - --require-hashes --requirement /opt/benchmark/locks/letta.txt - -COPY scripts/benchmark-unit.py /opt/benchmark/benchmark-unit.py -COPY scripts/benchmark_targets /opt/benchmark/benchmark_targets - -ENV PATH="/opt/benchmark-venv/bin:/app/.venv/bin:${PATH}" -ENV LETTA_BASE_URL="http://127.0.0.1:8283" -ENV PYTHONUNBUFFERED=1 - -CMD ["sh", "-c", "mkdir -p /benchmark/artifacts/raw; export OPENAI_API_KEY=\"${EMBEDDING_API_KEY}\"; /app/letta/server/startup.sh > /benchmark/artifacts/raw/letta-server.log 2>&1 & server_pid=$!; trap 'kill -INT ${server_pid} 2>/dev/null || true' EXIT; python3 /opt/benchmark/benchmark-unit.py --target letta; status=$?; kill -INT ${server_pid} 2>/dev/null || true; wait ${server_pid} 2>/dev/null || true; exit ${status}"] diff --git a/docker/benchmark/sag.Dockerfile b/docker/benchmark/sag.Dockerfile deleted file mode 100644 index f298719b..00000000 --- a/docker/benchmark/sag.Dockerfile +++ /dev/null @@ -1,26 +0,0 @@ -FROM node:22-bookworm-slim - -ARG SAG_REVISION=84a5b8c9bd45944b8a3cd76e0ba4762ce68ccc72 - -RUN apt-get update \ - && apt-get install -y --no-install-recommends ca-certificates git python3 \ - && rm -rf /var/lib/apt/lists/* - -RUN git init /opt/sag \ - && git -C /opt/sag remote add origin https://github.com/Zleap-AI/SAG.git \ - && git -C /opt/sag fetch --depth=1 origin "${SAG_REVISION}" \ - && git -C /opt/sag checkout --detach FETCH_HEAD \ - && test "$(git -C /opt/sag rev-parse HEAD)" = "${SAG_REVISION}" - -WORKDIR /opt/sag -RUN npm ci \ - && npx tsc -p tsconfig.build.json --noEmit - -COPY scripts/benchmark-unit.py /opt/benchmark/benchmark-unit.py -COPY scripts/benchmark_targets /opt/benchmark/benchmark_targets - -ENV SAG_REPO_DIR=/opt/sag -ENV SAG_TSX=/opt/sag/node_modules/.bin/tsx -ENV PYTHONUNBUFFERED=1 - -CMD ["python3", "/opt/benchmark/benchmark-unit.py", "--target", "sag"] diff --git a/docs/log.md b/docs/log.md index 8c0886c4..a5b1be40 100644 --- a/docs/log.md +++ b/docs/log.md @@ -8,6 +8,10 @@ logs. ## 2026-09-21 +- Removed the retired SAG v1 and Letta Python server implementations from the + default competitor matrix, dispatcher, container setup, and dependency locks. + Preserved dated evidence and documented the exact historical reproduction ref. + - Recorded legacy workspace disposition, historical benchmark claim corrections, and the bounded repair queue. Fixed the radar generator's obsolete entrypoint reference and its checked-in report. Removed the misplaced application-level diff --git a/docs/reference/benchmark-target-lifecycle.md b/docs/reference/benchmark-target-lifecycle.md new file mode 100644 index 00000000..05d54056 --- /dev/null +++ b/docs/reference/benchmark-target-lifecycle.md @@ -0,0 +1,63 @@ +--- +type: Reference +title: "Benchmark Target Lifecycle" +description: "Separate current benchmark targets from retired product implementations." +resource: docs/reference/benchmark-target-lifecycle.md +status: active +authority: current_state +owner: benchmarking +last_verified: 2026-09-21 +source_refs: + - https://github.com/Zleap-AI/SAG + - https://github.com/letta-ai/letta/blob/main/SECURITY.md +code_refs: + - config/benchmark/benchmark-v3.json + - scripts/benchmark_contract/fixtures.py + - scripts/benchmark_runner/cli.py +related: + - docs/research/2026-09-21-legacy-workspace-disposition.md +--- +# Benchmark Target Lifecycle + +Purpose: Define the current default comparison set and its historical boundary. +Read this when: Running `cargo make benchmark-competitors` or interpreting old results. +Not this document: A fresh measurement or a claim that retained target pins are latest. + +## Current set + +The default manifest retains ten targets: ELF, mem0, qmd, LightRAG, OpenViking, +Graphiti, GraphRAG, PageIndex, OpenKB, and Honcho. The four suite definitions, +scoring rules, and PageIndex eligibility boundary are unchanged. A reduction in +the matrix is not an improvement in measured performance. Do not compare an +aggregate from the new set directly with a twelve-target historical aggregate. + +## Retired implementations + +| Target | Removed implementation | Reason | Candidate replacement | +| --- | --- | --- | --- | +| SAG | June 26 revision `84a5b8c9bd45944b8a3cd76e0ba4762ce68ccc72`, TypeScript internal-service adapter and its database/container setup | Upstream replaced the architecture on July 14 and states that its old v1 branch is no longer maintained. | Current `zleap-sag` application/API, after an independent adapter review and measured run. | +| Letta | Legacy `letta/letta` Python server, archival-passage adapter and client lock | Upstream explicitly retires the V1 Python server and associated Docker images, including security updates. | `letta-ai/letta-code`, after checking whether its current product boundary matches the suite. | + +Neither project is declared inactive. OpenKB and GraphRAG remain available: +slower development alone does not prove retirement. Other retained pins still +need separate freshness reviews; this change does not silently upgrade them. + +The default manifest, target dispatcher, build list, Compose services, Dockerfiles, +and retired adapter dependencies no longer expose these two old implementations. +Explicit selection of a retired target fails before provider access or artifact +creation. Retired implementation-specific tests are removed; shared provenance, +mutation, failure classification, and scoring tests remain. + +## Historical reproduction + +Use repository revision `a996918ffba45d4055aba0508477b82b4246152b` or the exact +source revision recorded in an old result to inspect the previous manifest, +adapters, containers, and locks. Dependency availability is not guaranteed. +Historical reports and older fixture/smoke lanes remain dated evidence, not the +current competitor entrypoint or a statement that a retired server is supported. +Never run a retired server as a production service. + +Preserve original bundle contents and matrix identities when rendering old +reports. Do not remove rows from a historical result or relabel an old SAG/Letta +measurement as a measurement of its replacement. This cleanup performs no paid +provider calls and produces no new competitor-quality results. diff --git a/docs/reference/index.md b/docs/reference/index.md index c92a9876..c61c76b4 100644 --- a/docs/reference/index.md +++ b/docs/reference/index.md @@ -1,5 +1,8 @@ # Reference Index +- [Benchmark Target Lifecycle](benchmark-target-lifecycle.md): current targets, + retired implementations, and historical reproduction boundaries. + Purpose: Route agents to current structure references and non-procedural orientation. Read this when: You need a stable overview that is not a normative spec or runbook. Not this document: Correctness contracts, execution steps, or latent research. diff --git a/scripts/benchmark-unit.py b/scripts/benchmark-unit.py index 7ba35bee..1529ab03 100644 --- a/scripts/benchmark-unit.py +++ b/scripts/benchmark-unit.py @@ -34,8 +34,6 @@ def parse_args() -> argparse.Namespace: "openviking", "graphiti", "graphrag", - "letta", - "sag", "pageindex", "openkb", "honcho", @@ -57,8 +55,6 @@ def failure_result( "openviking": "native_resource_find", "graphiti": "native_temporal_graph_search", "graphrag": "native_local_search", - "letta": "native_archival_search", - "sag": "native_multi_search_source_id_trace", "openkb": "agent_query_with_native_source_trace", "honcho": "native_hybrid_message_search", }.get(target, "external_embedding") @@ -113,8 +109,6 @@ def main() -> int: "openviking", "graphiti", "graphrag", - "letta", - "sag", "honcho", }: required += ( @@ -127,8 +121,6 @@ def main() -> int: "mem0", "lightrag", "graphiti", - "letta", - "sag", "openkb", "honcho", }: @@ -189,25 +181,6 @@ def main() -> int: except GraphRAGAdapterFailure as error: result = failure_result(args.target, "adapter_failed", error, job_ids) exit_code = 1 - elif args.target == "letta": - from benchmark_targets.letta import ( - LettaAdapterFailure, - LettaProductFailure, - run_letta, - ) - - try: - result = run_letta(INPUT, ARTIFACTS, STATE / "letta") - except LettaProductFailure as error: - result = failure_result(args.target, "product_failed", error, job_ids) - exit_code = 1 - except LettaAdapterFailure as error: - result = failure_result(args.target, "adapter_failed", error, job_ids) - exit_code = 1 - elif args.target == "sag": - from benchmark_targets.sag import run_sag - - result = run_sag(INPUT, ARTIFACTS, STATE / "sag") elif args.target == "pageindex": result = run_pageindex(INPUT, ARTIFACTS, STATE / "pageindex") elif args.target == "openkb": diff --git a/scripts/benchmark_contract/fixtures.py b/scripts/benchmark_contract/fixtures.py index 253b8a81..5f901b29 100644 --- a/scripts/benchmark_contract/fixtures.py +++ b/scripts/benchmark_contract/fixtures.py @@ -16,8 +16,6 @@ "openviking", "graphrag", "graphiti", - "letta", - "sag", "pageindex", "openkb", "honcho", diff --git a/scripts/benchmark_runner/cli.py b/scripts/benchmark_runner/cli.py index 8a64afba..d2c62d04 100644 --- a/scripts/benchmark_runner/cli.py +++ b/scripts/benchmark_runner/cli.py @@ -38,6 +38,10 @@ def main() -> int: args = parse_args() manifest = load_json(args.manifest) validate_manifest(manifest) + if args.only_target and args.only_target not in { + target["id"] for target in manifest["targets"] + }: + raise ValueError(f"unknown or retired target: {args.only_target}") selected_suite_ids = args.suites or [entry["id"] for entry in manifest["suites"]] unknown = set(selected_suite_ids) - {entry["id"] for entry in manifest["suites"]} if unknown: diff --git a/scripts/benchmark_runner/docker.py b/scripts/benchmark_runner/docker.py index 522cb611..7c6b902a 100644 --- a/scripts/benchmark_runner/docker.py +++ b/scripts/benchmark_runner/docker.py @@ -18,8 +18,6 @@ "openviking", "graphiti", "graphrag", - "letta", - "sag", "openkb", "honcho", ) diff --git a/scripts/benchmark_targets/letta.py b/scripts/benchmark_targets/letta.py deleted file mode 100644 index 3ae7fc76..00000000 --- a/scripts/benchmark_targets/letta.py +++ /dev/null @@ -1,527 +0,0 @@ -"""Letta archival-memory benchmark adapter.""" - -from __future__ import annotations - -import hashlib -import json -import os -import time -import urllib.error -import urllib.parse -import urllib.request -import uuid -from pathlib import Path -from typing import Any, Callable, TypeVar - - -LETTA_IMAGE = ( - "letta/letta@sha256:" - "aa66c3eeee13d2dfc40c650d709b550237ee31bfc91942a52fa488a13fa8c102" -) -FORBIDDEN_FIXTURE_KEYS = { - "expected_answer", - "negative_traps", - "qrels", - "required_evidence", - "scoring_rubric", -} -T = TypeVar("T") - - -class LettaProductFailure(RuntimeError): - """The native Letta client or API failed.""" - - -class LettaAdapterFailure(RuntimeError): - """The adapter input, state, or native identity mapping is invalid.""" - - -def _write_json(path: Path, value: Any) -> None: - path.parent.mkdir(parents=True, exist_ok=True) - path.write_text( - json.dumps(value, ensure_ascii=True, indent=2, sort_keys=True) + "\n", - encoding="utf-8", - ) - - -def _as_json(value: Any) -> Any: - if hasattr(value, "model_dump"): - return value.model_dump(mode="json") - if isinstance(value, list): - return [_as_json(item) for item in value] - if isinstance(value, dict): - return {key: _as_json(item) for key, item in value.items()} - return value - - -def _assert_blind(value: Any, path: str = "$") -> None: - if isinstance(value, dict): - leaked = FORBIDDEN_FIXTURE_KEYS.intersection(value) - if leaked: - raise LettaAdapterFailure( - f"Letta fixture leaks evaluator keys at {path}: " - + ", ".join(sorted(leaked)) - ) - for key, child in value.items(): - _assert_blind(child, f"{path}.{key}") - elif isinstance(value, list): - for index, child in enumerate(value): - _assert_blind(child, f"{path}[{index}]") - - -def _load_jobs(input_dir: Path) -> list[dict[str, Any]]: - jobs = [ - json.loads(path.read_text(encoding="utf-8")) - for path in sorted(input_dir.glob("*.json")) - ] - if not jobs: - raise LettaAdapterFailure("Letta received no product fixtures") - for job in jobs: - _assert_blind(job) - return jobs - - -def _fixture_digest(jobs: list[dict[str, Any]]) -> str: - encoded = json.dumps( - jobs, ensure_ascii=True, separators=(",", ":"), sort_keys=True - ).encode("utf-8") - return hashlib.sha256(encoded).hexdigest() - - -def _product_call(label: str, operation: Callable[[], T]) -> T: - try: - return operation() - except LettaAdapterFailure: - raise - except Exception as error: - raise LettaProductFailure(f"Letta {label} failed: {error}") from error - - -def _wait_for_server(base_url: str, timeout_seconds: float) -> None: - deadline = time.monotonic() + timeout_seconds - last_error: Exception | None = None - while time.monotonic() < deadline: - try: - with urllib.request.urlopen( - f"{base_url.rstrip('/')}/v1/health/", timeout=2.0 - ) as response: - if 200 <= response.status < 300: - return - last_error = RuntimeError( - f"health endpoint returned {response.status}" - ) - except (OSError, urllib.error.URLError) as error: - last_error = error - time.sleep(1.0) - raise LettaProductFailure(f"Letta server did not become ready: {last_error}") - - -def _agent_configs() -> tuple[Any, Any]: - from letta_client.types.embedding_config import EmbeddingConfig - from letta_client.types.llm_config import LlmConfig - - required = ( - "CHAT_API_BASE", - "CHAT_MODEL", - "EMBEDDING_API_BASE", - "EMBEDDING_MODEL", - "EMBEDDING_DIMENSIONS", - ) - missing = [name for name in required if not os.environ.get(name)] - if missing: - raise LettaAdapterFailure( - "missing required Letta configuration: " + ", ".join(missing) - ) - reasoning = os.environ.get("CHAT_REASONING_EFFORT") - if reasoning not in {"low", "medium", "high"}: - reasoning = None - return ( - LlmConfig( - model=os.environ["CHAT_MODEL"], - model_endpoint_type="openai", - model_endpoint=os.environ["CHAT_API_BASE"], - context_window=32768, - max_tokens=4096, - reasoning_effort=reasoning, - ), - EmbeddingConfig( - embedding_endpoint_type="openai", - embedding_endpoint=os.environ["EMBEDDING_API_BASE"], - embedding_model=os.environ["EMBEDDING_MODEL"], - embedding_dim=int(os.environ["EMBEDDING_DIMENSIONS"]), - embedding_chunk_size=300, - ), - ) - - -def _search( - base_url: str, agent_id: str, query: str, top_k: int -) -> tuple[dict[str, Any], float]: - started = time.monotonic() - try: - endpoint = ( - f"{base_url.rstrip('/')}/v1/agents/{agent_id}/archival-memory/search?" - + urllib.parse.urlencode({"query": query, "top_k": top_k}) - ) - with urllib.request.urlopen(endpoint, timeout=120.0) as response: - native = json.loads(response.read().decode("utf-8")) - except (OSError, urllib.error.URLError, json.JSONDecodeError) as error: - raise LettaProductFailure(f"Letta archival search failed: {error}") from error - latency_ms = (time.monotonic() - started) * 1000.0 - if not isinstance(native, dict) or not isinstance(native.get("results"), list): - raise LettaAdapterFailure("Letta archival search returned a malformed response") - return native, latency_ms - - -def _mapped_evidence( - native_search: dict[str, Any], passage_identity: dict[str, str] -) -> list[str]: - evidence_ids: list[str] = [] - for result in native_search["results"]: - if not isinstance(result, dict) or not isinstance(result.get("id"), str): - raise LettaAdapterFailure("Letta search result has no native passage id") - evidence_id = passage_identity.get(result["id"]) - if evidence_id is not None and evidence_id not in evidence_ids: - evidence_ids.append(evidence_id) - return evidence_ids - - -def _native_contexts( - native_search: dict[str, Any], passage_identity: dict[str, str] -) -> list[dict[str, Any]]: - """Preserve ranked native passage content for central answer and mutation checks.""" - contexts: list[dict[str, Any]] = [] - for result in native_search["results"]: - if not isinstance(result, dict): - raise LettaAdapterFailure("Letta search returned a malformed passage") - passage_id = result.get("id") - content = result.get("content") - if not isinstance(passage_id, str) or not isinstance(content, str): - raise LettaAdapterFailure( - "Letta search result omitted its native passage id or content" - ) - contexts.append( - { - "evidence_id": passage_identity.get(passage_id), - "text": content, - } - ) - return contexts - - -def _apply_operations( - client: Any, - jobs: list[dict[str, Any]], - receipt_agents: dict[str, Any], -) -> tuple[dict[str, list[dict[str, Any]]], list[dict[str, Any]]]: - """Use Letta's native delete/create APIs for delete and replacement semantics.""" - receipts: dict[str, list[dict[str, Any]]] = {} - native_jobs: list[dict[str, Any]] = [] - for job in jobs: - state = receipt_agents.get(job["job_id"]) - if not isinstance(state, dict): - raise LettaAdapterFailure(f"Letta state is missing job {job['job_id']}") - agent_id = state.get("agent_id") - identities = state.get("passage_identity") - historical_identities = state.get("historical_passage_identity") - if ( - not isinstance(agent_id, str) - or not isinstance(identities, dict) - or not isinstance(historical_identities, dict) - ): - raise LettaAdapterFailure(f"Letta state is malformed for {job['job_id']}") - job_receipts: list[dict[str, Any]] = [] - native_operations: list[dict[str, Any]] = [] - for operation in job.get("operations") or []: - requested_type = operation.get("type") - evidence_id = operation.get("evidence_id") - matching = [ - passage_id - for passage_id, mapped in identities.items() - if mapped == evidence_id - ] - if len(matching) != 1: - raise LettaAdapterFailure( - "Letta native mutation target did not resolve to one passage" - ) - old_passage_id = matching[0] - deleted = _product_call( - "archival passage deletion", - lambda agent_id=agent_id, old_passage_id=old_passage_id: ( - client.agents.passages.delete( - agent_id=agent_id, memory_id=old_passage_id - ) - ), - ) - historical_identities[old_passage_id] = evidence_id - del identities[old_passage_id] - native_type = "delete" - created: list[Any] = [] - if requested_type == "update": - replacement_text = operation.get("text") - if not isinstance(replacement_text, str) or not replacement_text: - raise LettaAdapterFailure( - "Letta update operation has no replacement text" - ) - created = _product_call( - "replacement archival passage creation", - lambda agent_id=agent_id, replacement_text=replacement_text: ( - client.agents.passages.create( - agent_id=agent_id, text=replacement_text - ) - ), - ) - if not isinstance(created, list) or not created: - raise LettaAdapterFailure( - "Letta replacement creation returned no passage identity" - ) - for passage in created: - passage_id = getattr(passage, "id", None) - if not isinstance(passage_id, str) or passage_id in identities: - raise LettaAdapterFailure( - "Letta replacement creation returned an invalid passage identity" - ) - identities[passage_id] = evidence_id - historical_identities[passage_id] = evidence_id - native_type = "replace" - elif requested_type != "delete": - raise LettaAdapterFailure( - f"Letta does not support operation {requested_type!r}" - ) - job_receipts.append( - { - "requested_type": requested_type, - "native_type": native_type, - "classification": "completed", - "native_success": True, - } - ) - native_operations.append( - { - "requested": operation, - "deleted_passage_id": old_passage_id, - "delete_response": _as_json(deleted), - "created_passages": _as_json(created), - } - ) - receipts[job["job_id"]] = job_receipts - native_jobs.append( - {"job_id": job["job_id"], "operations": native_operations} - ) - return receipts, native_jobs - - -def _cold_ingest( - client: Any, jobs: list[dict[str, Any]] -) -> tuple[dict[str, Any], list[dict[str, Any]]]: - llm_config, embedding_config = _agent_configs() - receipt_agents: dict[str, Any] = {} - native_agents: list[dict[str, Any]] = [] - for job in jobs: - agent = _product_call( - "agent creation", - lambda job=job: client.agents.create( - name=f"elf-benchmark-{job['job_id']}-{uuid.uuid4().hex[:12]}", - llm_config=llm_config, - embedding_config=embedding_config, - memory_blocks=[ - { - "label": "benchmark", - "value": "Retrieve source-backed archival memory.", - } - ], - ), - ) - passage_identity: dict[str, str] = {} - created: list[Any] = [] - for item in job["corpus"]["items"]: - passages = _product_call( - "archival passage creation", - lambda item=item: client.agents.passages.create( - agent_id=agent.id, text=item["text"] - ), - ) - if not isinstance(passages, list) or not passages: - raise LettaAdapterFailure( - "Letta passage creation returned no native passage identity" - ) - for passage in passages: - passage_id = getattr(passage, "id", None) - if not isinstance(passage_id, str) or passage_id in passage_identity: - raise LettaAdapterFailure( - "Letta passage creation returned an invalid native identity" - ) - passage_identity[passage_id] = item["evidence_id"] - created.extend(passages) - receipt_agents[job["job_id"]] = { - "agent_id": agent.id, - "passage_identity": passage_identity, - "historical_passage_identity": dict(passage_identity), - } - native_agents.append( - { - "job_id": job["job_id"], - "agent": _as_json(agent), - "created_passages": _as_json(created), - } - ) - return receipt_agents, native_agents - - -def _phase( - client: Any, - base_url: str, - jobs: list[dict[str, Any]], - receipt_agents: dict[str, Any], - *, - phase: str, - native_agents: list[dict[str, Any]] | None = None, - operations: dict[str, list[dict[str, Any]]] | None = None, -) -> tuple[dict[str, Any], dict[str, Any]]: - rows: list[dict[str, Any]] = [] - native_jobs: list[dict[str, Any]] = [] - for job in jobs: - state = receipt_agents.get(job["job_id"]) - if not isinstance(state, dict): - raise LettaAdapterFailure(f"Letta state is missing job {job['job_id']}") - agent_id = state.get("agent_id") - identities = state.get("passage_identity") - historical_identities = state.get("historical_passage_identity") - if ( - not isinstance(agent_id, str) - or not isinstance(identities, dict) - or not isinstance(historical_identities, dict) - ): - raise LettaAdapterFailure(f"Letta state is malformed for {job['job_id']}") - warm_readback: dict[str, Any] | None = None - if phase == "warm": - agent = _product_call( - "warm agent retrieval", - lambda agent_id=agent_id: client.agents.retrieve(agent_id=agent_id), - ) - if getattr(agent, "id", None) != agent_id: - raise LettaAdapterFailure("Letta warm readback changed native agent identity") - passages = _product_call( - "warm passage readback", - lambda agent_id=agent_id: client.agents.passages.list( - agent_id=agent_id, limit=1000 - ), - ) - passage_ids = [getattr(passage, "id", None) for passage in passages] - if ( - any(not isinstance(passage_id, str) for passage_id in passage_ids) - or len(passage_ids) != len(set(passage_ids)) - or set(passage_ids) != set(identities) - ): - raise LettaAdapterFailure( - "Letta warm readback did not preserve post-operation passage identities" - ) - warm_readback = { - "agent": _as_json(agent), - "passages": _as_json(passages), - } - native_search, latency_ms = _search( - base_url, agent_id, job["prompt"]["content"], top_k=5 - ) - result_identities = historical_identities if phase == "warm" else identities - evidence_ids = _mapped_evidence(native_search, result_identities) - contexts = _native_contexts(native_search, result_identities) - rows.append( - { - "job_id": job["job_id"], - "classification": "completed", - "evidence_ids": evidence_ids, - "contexts": contexts, - "operations": (operations or {}).get(job["job_id"], []), - "returned_count": len(native_search["results"]), - "latency_ms": latency_ms, - "native_status": "completed", - "failure": None, - } - ) - native_job = { - "job_id": job["job_id"], - "agent_id": agent_id, - "search": native_search, - } - if warm_readback is not None: - native_job["readback"] = warm_readback - native_jobs.append(native_job) - native_output: dict[str, Any] = {"phase": phase, "jobs": native_jobs} - if native_agents is not None: - native_output["ingest"] = native_agents - return ( - { - "status": "completed", - "jobs": rows, - "adapter_metadata": {"index_reused": phase == "warm"}, - }, - native_output, - ) - - -def run_letta(input_dir: Path, artifacts: Path, state_dir: Path) -> dict[str, Any]: - """Run cold ingest/search and warm search against one self-hosted Letta state.""" - - jobs = _load_jobs(input_dir) - base_url = os.environ.get("LETTA_BASE_URL", "http://127.0.0.1:8283") - _wait_for_server( - base_url, float(os.environ.get("LETTA_STARTUP_TIMEOUT_SECONDS", "120")) - ) - from letta_client import Letta - - client = Letta(base_url=base_url) - state_dir.mkdir(parents=True, exist_ok=True) - receipt_path = state_dir / "cold-ingest.json" - if receipt_path.exists(): - raise LettaAdapterFailure("Letta cold state already exists before ingest") - - ingest_started = time.monotonic() - receipt_agents, native_agents = _cold_ingest(client, jobs) - ingest_duration_ms = (time.monotonic() - ingest_started) * 1000.0 - receipt = { - "image": LETTA_IMAGE, - "fixture_sha256": _fixture_digest(jobs), - "agents": receipt_agents, - } - _write_json(receipt_path, receipt) - cold, cold_native = _phase( - client, - base_url, - jobs, - receipt_agents, - phase="cold", - native_agents=native_agents, - ) - _write_json(artifacts / "raw" / "letta-cold.json", cold_native) - - warm_receipt = json.loads(receipt_path.read_text(encoding="utf-8")) - if warm_receipt != receipt or warm_receipt["fixture_sha256"] != _fixture_digest(jobs): - raise LettaAdapterFailure("Letta warm phase did not reuse the cold receipt") - operation_receipts, native_operations = _apply_operations( - client, jobs, warm_receipt["agents"] - ) - _write_json( - artifacts / "raw" / "letta-operations.json", native_operations - ) - warm, warm_native = _phase( - client, - base_url, - jobs, - warm_receipt["agents"], - phase="warm", - operations=operation_receipts, - ) - _write_json(artifacts / "raw" / "letta-warm.json", warm_native) - - return { - "schema": "elf.benchmark_unit_result/v4", - "target": "letta", - "native_mode": "external_embedding", - "score_eligible": True, - "result_class": "completed", - "warm_reused_state": True, - "ingest_count": 1, - "ingest_duration_ms": round(ingest_duration_ms, 3), - "phases": {"cold": cold, "warm": warm}, - } diff --git a/scripts/benchmark_targets/sag.py b/scripts/benchmark_targets/sag.py deleted file mode 100644 index e99d089d..00000000 --- a/scripts/benchmark_targets/sag.py +++ /dev/null @@ -1,574 +0,0 @@ -"""Docker-contained adapter for Zleap-AI SAG native retrieval.""" - -from __future__ import annotations - -import hashlib -import json -import os -import re -import subprocess -import time -import uuid -from pathlib import Path -from typing import Any - - -SAG_REVISION = "84a5b8c9bd45944b8a3cd76e0ba4762ce68ccc72" -SAG_REPO = Path(os.environ.get("SAG_REPO_DIR", "/opt/sag")) -SAG_TSX = Path(os.environ.get("SAG_TSX", "/opt/sag/node_modules/.bin/tsx")) -FORBIDDEN_PRODUCT_KEYS = { - "expected_answer", - "negative_traps", - "qrels", - "required_evidence", - "scoring_rubric", -} -REQUIRED_PROVIDER_VALUES = { - "CHAT_MODEL": "gpt-5.6-luna", - "CHAT_REASONING_EFFORT": "high", - "EMBEDDING_MODEL": "Qwen3-Embedding-8B", - "EMBEDDING_DIMENSIONS": "1024", -} - - -class SagProductFailure(RuntimeError): - """The pinned SAG runtime failed after the adapter reached it.""" - - -class SagAdapterFailure(RuntimeError): - """The adapter input, configuration, or state contract is invalid.""" - - -class SagConfigurationFailure(SagAdapterFailure): - """The SAG unit does not have its required frozen configuration.""" - - -_NATIVE_RUNNER = r""" -import { createHash } from "node:crypto"; -import { readFile, writeFile } from "node:fs/promises"; -import { ingestionService } from "/opt/sag/src/services/ingestion-service.js"; -import { searchService } from "/opt/sag/src/services/search-service.js"; -import { migrate } from "/opt/sag/src/db/migrate.js"; -import { seed } from "/opt/sag/src/db/seed.js"; -import { pool, closePool } from "/opt/sag/src/db/pool.js"; - -const inputPath = process.argv[2]; -const outputPath = process.argv[3]; - -async function stateDigest(sourceIds) { - const queries = [ - ["sources", "select * from sources where id = any($1::uuid[]) order by id"], - ["documents", "select * from documents where source_id = any($1::uuid[]) order by id"], - ["document_sections", ` - select ds.* from document_sections ds - join documents d on d.id = ds.document_id - where d.source_id = any($1::uuid[]) order by ds.id - `], - ["source_chunks", "select * from source_chunks where source_id = any($1::uuid[]) order by id"], - ["entities", "select * from entities where source_id = any($1::uuid[]) order by id"], - ["events", "select * from events where source_id = any($1::uuid[]) order by id"], - ["event_entities", ` - select ee.* from event_entities ee - join events e on e.id = ee.event_id - where e.source_id = any($1::uuid[]) order by ee.id - `] - ]; - const hash = createHash("sha256"); - const counts = {}; - for (const [name, sql] of queries) { - const result = await pool.query(sql, [sourceIds]); - counts[name] = result.rows.length; - hash.update(name); - hash.update("\0"); - hash.update(JSON.stringify(result.rows)); - hash.update("\0"); - } - return { sha256: hash.digest("hex"), counts }; -} - -async function searchJobs(jobs) { - const results = []; - for (const job of jobs) { - const started = performance.now(); - const search = await searchService.search({ - query: job.query, - sourceIds: job.items.map((item) => item.sourceId), - strategy: "multi", - searchMode: "standard", - topK: 5, - returnTrace: true, - multi: { - entityTopK: 8, - multiTopK: 8, - maxEvents: 8, - rerankTopK: 5, - maxSections: 5, - maxHops: 2 - } - }); - results.push({ - jobId: job.jobId, - latencyMs: performance.now() - started, - search - }); - } - return results; -} - -async function main() { - const input = JSON.parse(await readFile(inputPath, "utf8")); - const sourceIds = input.jobs.flatMap((job) => job.items.map((item) => item.sourceId)); - await migrate(); - await seed(); - - const existing = await pool.query( - "select id from sources where id = any($1::uuid[]) limit 1", - [sourceIds] - ); - if (existing.rowCount) { - throw new Error("SAG cold state already contains a benchmark source"); - } - - const ingest = []; - const ingestStarted = performance.now(); - for (const job of input.jobs) { - for (const item of job.items) { - const result = await ingestionService.ingestDocument({ - sourceId: item.sourceId, - title: item.title, - content: item.content, - metadata: { elfBenchmarkSourceKey: item.sourceKey }, - extract: true, - chunking: { mode: "heading_strict" } - }); - ingest.push({ jobId: job.jobId, sourceKey: item.sourceKey, result }); - } - } - - const ingestLatencyMs = performance.now() - ingestStarted; - const stateBeforeCold = await stateDigest(sourceIds); - const cold = await searchJobs(input.jobs); - const stateAfterCold = await stateDigest(sourceIds); - const warm = await searchJobs(input.jobs); - const stateAfterWarm = await stateDigest(sourceIds); - await writeFile(outputPath, JSON.stringify({ - ingest, - ingestLatencyMs, - stateBeforeCold, - cold, - stateAfterCold, - warm, - stateAfterWarm - }, null, 2) + "\n"); -} - -main() - .then(async () => closePool()) - .catch(async (error) => { - console.error(error instanceof Error ? error.stack || error.message : String(error)); - await closePool(); - process.exit(1); - }); -""" - - -def _write_json(path: Path, value: Any) -> None: - path.parent.mkdir(parents=True, exist_ok=True) - path.write_text( - json.dumps(value, ensure_ascii=True, indent=2, sort_keys=True) + "\n", - encoding="utf-8", - ) - - -def _assert_qrel_blind(value: Any, path: str = "$") -> None: - if isinstance(value, dict): - leaked = FORBIDDEN_PRODUCT_KEYS.intersection(value) - if leaked: - raise SagAdapterFailure( - f"SAG fixture leaks evaluator keys at {path}: " - + ", ".join(sorted(leaked)) - ) - for key, child in value.items(): - _assert_qrel_blind(child, f"{path}.{key}") - elif isinstance(value, list): - for index, child in enumerate(value): - _assert_qrel_blind(child, f"{path}[{index}]") - - -def _load_jobs(input_dir: Path) -> list[dict[str, Any]]: - jobs: list[dict[str, Any]] = [] - for path in sorted(input_dir.glob("*.json")): - try: - job = json.loads(path.read_text(encoding="utf-8")) - _assert_qrel_blind(job) - if not isinstance(job["job_id"], str) or not job["job_id"]: - raise TypeError("job_id must be a non-empty string") - if not isinstance(job["prompt"]["content"], str): - raise TypeError("prompt.content must be a string") - items = job["corpus"]["items"] - if not isinstance(items, list) or not items: - raise TypeError("corpus.items must be a non-empty list") - for item in items: - if not isinstance(item["evidence_id"], str) or not isinstance( - item["text"], str - ): - raise TypeError("corpus evidence_id and text must be strings") - except SagAdapterFailure: - raise - except (KeyError, TypeError, json.JSONDecodeError) as error: - raise SagAdapterFailure( - f"invalid SAG fixture {path.name}: {error}" - ) from error - jobs.append(job) - if not jobs: - raise SagAdapterFailure("SAG received no product fixtures") - return jobs - - -def _verify_revision() -> None: - completed = subprocess.run( - ["git", "-C", str(SAG_REPO), "rev-parse", "HEAD"], - check=False, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - ) - if completed.returncode or completed.stdout.strip() != SAG_REVISION: - raise SagAdapterFailure("SAG checkout does not match the frozen revision") - - -def _required_environment() -> dict[str, str]: - required = ( - "DATABASE_URL", - "CHAT_API_BASE", - "CHAT_API_KEY", - "CHAT_MODEL", - "CHAT_REASONING_EFFORT", - "EMBEDDING_API_BASE", - "EMBEDDING_API_KEY", - "EMBEDDING_MODEL", - "EMBEDDING_DIMENSIONS", - ) - missing = [name for name in required if not os.environ.get(name)] - if missing: - raise SagConfigurationFailure( - "missing SAG configuration: " + ", ".join(missing) - ) - for name, expected in REQUIRED_PROVIDER_VALUES.items(): - if os.environ[name] != expected: - raise SagConfigurationFailure(f"SAG requires {name}={expected}") - return {name: os.environ[name] for name in required} - - -def _source_identity(job_id: str, index: int) -> tuple[str, str]: - identity = f"{job_id}:{index}" - digest = hashlib.sha256(identity.encode("utf-8")).hexdigest()[:16] - source_key = f"item-{index:04d}-{digest}" - source_id = str(uuid.uuid5(uuid.NAMESPACE_URL, f"elf-sag:{identity}")) - return source_id, source_key - - -def _native_input(jobs: list[dict[str, Any]]) -> tuple[dict[str, Any], dict[str, str]]: - source_map: dict[str, str] = {} - native_jobs: list[dict[str, Any]] = [] - for job in jobs: - native_items = [] - for index, item in enumerate(job["corpus"]["items"]): - source_id, source_key = _source_identity(job["job_id"], index) - if source_id in source_map: - raise SagAdapterFailure("SAG generated a duplicate native source id") - source_map[source_id] = item["evidence_id"] - native_items.append( - { - "sourceId": source_id, - "sourceKey": source_key, - "title": f"{job['job_id']} source {index + 1}", - "content": item["text"], - } - ) - native_jobs.append( - { - "jobId": job["job_id"], - "query": job["prompt"]["content"], - "items": native_items, - } - ) - payload = {"jobs": native_jobs} - _assert_qrel_blind(payload) - return payload, source_map - - -def evidence_ids_from_search( - native_search: Any, source_map: dict[str, str] -) -> list[str]: - """Map evidence only through SAG's explicit SearchSection.sourceId field.""" - if not isinstance(native_search, dict): - return [] - sections = native_search.get("sections") - if not isinstance(sections, list): - return [] - output: list[str] = [] - for section in sections: - if not isinstance(section, dict): - continue - source_id = section.get("sourceId") - evidence_id = source_map.get(source_id) if isinstance(source_id, str) else None - if evidence_id is not None and evidence_id not in output: - output.append(evidence_id) - return output - - -def _provider_failure(message: str) -> bool: - lowered = message.lower() - return bool( - re.search(r"(?:request failed|status):\s*[45]\d\d", lowered) - or any( - token in lowered - for token in ("401", "403", "429", "rate limit", "fetch failed") - ) - ) - - -def _sanitized_error(error: BaseException) -> str: - message = str(error) - for name in ("CHAT_API_KEY", "EMBEDDING_API_KEY"): - secret = os.environ.get(name) - if secret: - message = message.replace(secret, "[redacted]") - return message - - -def _terminal_result( - jobs: list[dict[str, Any]], classification: str, reason: str -) -> dict[str, Any]: - def phase() -> dict[str, Any]: - return { - "status": classification, - "jobs": [ - { - "job_id": job.get("job_id", "unknown"), - "classification": classification, - "evidence_ids": [], - "returned_count": 0, - "latency_ms": 0.0, - "native_status": "not_run", - "failure": reason, - } - for job in jobs - ], - "adapter_metadata": {"index_reused": False}, - } - - return { - "schema": "elf.benchmark_unit_result/v4", - "target": "sag", - "native_mode": "native_multi_search_source_id_trace", - "score_eligible": True, - "result_class": classification, - "terminal_reason": reason, - "warm_reused_state": False, - "ingest_count": 0, - "phases": {"cold": phase(), "warm": phase()}, - } - - -def _run_native( - runner_path: Path, - input_path: Path, - output_path: Path, - artifacts: Path, - environment: dict[str, str], -) -> None: - raw_dir = artifacts / "raw" - raw_dir.mkdir(parents=True, exist_ok=True) - env = os.environ.copy() - env.update( - { - "DATABASE_URL": environment["DATABASE_URL"], - "DEFAULT_TENANT_ID": "elf-benchmark", - "EMBEDDING_BASE_URL": environment["EMBEDDING_API_BASE"], - "EMBEDDING_API_KEY": environment["EMBEDDING_API_KEY"], - "EMBEDDING_MODEL": environment["EMBEDDING_MODEL"], - "EMBEDDING_DIMENSIONS": environment["EMBEDDING_DIMENSIONS"], - "LLM_BASE_URL": environment["CHAT_API_BASE"], - "LLM_API_KEY": environment["CHAT_API_KEY"], - "LLM_MODEL": environment["CHAT_MODEL"], - "LLM_REASONING_EFFORT": environment["CHAT_REASONING_EFFORT"], - "LLM_TIMEOUT_MS": os.environ.get("SAG_LLM_TIMEOUT_MS", "120000"), - "LLM_MAX_RETRIES": "1", - "DEFAULT_SEARCH_MODE": "standard", - "INGEST_CONCURRENCY": "1", - "NO_COLOR": "1", - "NODE_ENV": "production", - } - ) - started = time.monotonic() - try: - completed = subprocess.run( - [str(SAG_TSX), str(runner_path), str(input_path), str(output_path)], - cwd=SAG_REPO, - env=env, - check=False, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - text=True, - timeout=int(os.environ.get("SAG_TIMEOUT_SECONDS", "900")), - ) - except subprocess.TimeoutExpired as error: - stdout = error.stdout or "" - stderr = error.stderr or "" - if isinstance(stdout, bytes): - stdout = stdout.decode("utf-8", errors="replace") - if isinstance(stderr, bytes): - stderr = stderr.decode("utf-8", errors="replace") - (raw_dir / "sag-stdout.log").write_text(stdout, encoding="utf-8") - (raw_dir / "sag-stderr.log").write_text(stderr, encoding="utf-8") - raise SagProductFailure("SAG native runner timed out") from error - (raw_dir / "sag-stdout.log").write_text(completed.stdout, encoding="utf-8") - (raw_dir / "sag-stderr.log").write_text(completed.stderr, encoding="utf-8") - if completed.returncode: - detail = (completed.stderr or completed.stdout).strip() - raise SagProductFailure( - f"SAG native runner exited {completed.returncode} after " - f"{time.monotonic() - started:.3f}s: {detail}" - ) - - -def _phase( - jobs: list[dict[str, Any]], - rows: Any, - source_map: dict[str, str], - *, - reused: bool, -) -> dict[str, Any]: - if not isinstance(rows, list): - raise SagAdapterFailure("SAG native search output is malformed") - by_job = { - row.get("jobId"): row for row in rows if isinstance(row, dict) - } - output = [] - for job in jobs: - row = by_job.get(job["job_id"]) - if not isinstance(row, dict): - raise SagAdapterFailure(f"SAG omitted native result for {job['job_id']}") - search = row.get("search") - sections = search.get("sections") if isinstance(search, dict) else None - if not isinstance(sections, list): - raise SagAdapterFailure( - f"SAG returned malformed sections for {job['job_id']}" - ) - output.append( - { - "job_id": job["job_id"], - "classification": "completed", - "evidence_ids": evidence_ids_from_search(search, source_map), - "returned_count": len(sections), - "latency_ms": round(float(row.get("latencyMs") or 0.0), 3), - "native_status": "completed", - "failure": None, - } - ) - return { - "status": "completed", - "jobs": output, - "adapter_metadata": { - "index_reused": reused, - "revision": SAG_REVISION, - "embedding_dimensions": 1024, - "embedding_dimension_deviation": "native_sag_schema_limit", - }, - } - - -def run_sag(input_dir: Path, artifacts: Path, state_dir: Path) -> dict[str, Any]: - """Run one SAG ingest and cold/warm searches against the exact same state.""" - jobs: list[dict[str, Any]] = [] - try: - _verify_revision() - jobs = _load_jobs(input_dir) - environment = _required_environment() - state_dir.mkdir(parents=True, exist_ok=True) - receipt_path = state_dir / "cold-ingest.json" - if receipt_path.exists(): - raise SagAdapterFailure("SAG cold ingest receipt already exists") - native_input, source_map = _native_input(jobs) - runner_path = state_dir / "sag-native-runner.ts" - input_path = state_dir / "sag-native-input.json" - output_path = artifacts / "raw" / "sag-native-output.json" - runner_path.write_text(_NATIVE_RUNNER, encoding="utf-8") - _write_json(input_path, native_input) - _run_native( - runner_path, input_path, output_path, artifacts, environment - ) - native = json.loads(output_path.read_text(encoding="utf-8")) - state_digests = [ - native.get("stateBeforeCold"), - native.get("stateAfterCold"), - native.get("stateAfterWarm"), - ] - if not all(isinstance(item, dict) for item in state_digests): - raise SagAdapterFailure("SAG omitted native state evidence") - hashes = [item.get("sha256") for item in state_digests] - if not hashes[0] or len(set(hashes)) != 1: - raise SagAdapterFailure( - "SAG cold and warm queries did not reuse exact state" - ) - ingest = native.get("ingest") - expected_ingests = sum(len(job["corpus"]["items"]) for job in jobs) - if not isinstance(ingest, list) or len(ingest) != expected_ingests: - raise SagAdapterFailure( - "SAG did not return one native ingest per corpus item" - ) - ingest_latency_ms = native.get("ingestLatencyMs") - if not isinstance(ingest_latency_ms, (int, float)) or ingest_latency_ms < 0: - raise SagAdapterFailure("SAG omitted native ingest latency") - receipt = { - "revision": SAG_REVISION, - "fixture_sha256": hashlib.sha256( - json.dumps( - jobs, ensure_ascii=True, separators=(",", ":"), sort_keys=True - ).encode("utf-8") - ).hexdigest(), - "source_identity": source_map, - "native_state_sha256": hashes[0], - } - _write_json(receipt_path, receipt) - if json.loads(receipt_path.read_text(encoding="utf-8")) != receipt: - raise SagAdapterFailure("SAG cold ingest receipt readback changed") - cold = _phase(jobs, native.get("cold"), source_map, reused=False) - warm = _phase(jobs, native.get("warm"), source_map, reused=True) - except SagConfigurationFailure as error: - return _terminal_result( - jobs, "configuration_failed", _sanitized_error(error) - ) - except SagAdapterFailure as error: - return _terminal_result(jobs, "adapter_failed", _sanitized_error(error)) - except SagProductFailure as error: - reason = _sanitized_error(error) - classification = ( - "provider_failed" if _provider_failure(reason) else "product_failed" - ) - return _terminal_result(jobs, classification, reason) - except ( - KeyError, - OSError, - TypeError, - ValueError, - subprocess.SubprocessError, - json.JSONDecodeError, - ) as error: - return _terminal_result( - jobs, "adapter_failed", _sanitized_error(error) - ) - - return { - "schema": "elf.benchmark_unit_result/v4", - "target": "sag", - "native_mode": "native_multi_search_source_id_trace", - "score_eligible": True, - "result_class": "completed", - "warm_reused_state": True, - "ingest_count": 1, - "ingest_duration_ms": round(float(ingest_latency_ms), 3), - "phases": {"cold": cold, "warm": warm}, - } diff --git a/scripts/tests/test_benchmark_adapters.py b/scripts/tests/test_benchmark_adapters.py index 72033cf3..5ca24558 100644 --- a/scripts/tests/test_benchmark_adapters.py +++ b/scripts/tests/test_benchmark_adapters.py @@ -14,7 +14,6 @@ GRAPHRAG = load_script("benchmark_graphrag", "scripts/benchmark_targets/graphrag.py") -LETTA = load_script("benchmark_letta", "scripts/benchmark_targets/letta.py") OPENVIKING = load_script( "benchmark_openviking", "scripts/benchmark_targets/openviking.py" @@ -89,86 +88,6 @@ async def get_by_uuids(driver: object, uuids: list[str]) -> list[object]: self.assertEqual(len(native), 1) - def test_letta_native_contexts_and_mutation_receipts_use_passage_apis(self) -> None: - native_search = { - "results": [ - {"id": "old-update", "content": "old update text"}, - {"id": "unknown", "content": "unmapped native text"}, - ] - } - self.assertEqual( - LETTA._native_contexts(native_search, {"old-update": "e_update"}), - [ - {"evidence_id": "e_update", "text": "old update text"}, - {"evidence_id": None, "text": "unmapped native text"}, - ], - ) - - class Passages: - def __init__(self) -> None: - self.deleted: list[tuple[str, str]] = [] - - def delete(self, *, agent_id: str, memory_id: str) -> dict[str, bool]: - self.deleted.append((agent_id, memory_id)) - return {"deleted": True} - - def create(self, *, agent_id: str, text: str) -> list[object]: - self.created = (agent_id, text) - return [types.SimpleNamespace(id="replacement")] - - passages = Passages() - client = types.SimpleNamespace( - agents=types.SimpleNamespace(passages=passages) - ) - jobs = [ - { - "job_id": "j_mutation", - "operations": [ - {"type": "update", "evidence_id": "e_update", "text": "new"}, - {"type": "delete", "evidence_id": "e_delete"}, - ], - } - ] - state = { - "j_mutation": { - "agent_id": "agent", - "passage_identity": { - "old-update": "e_update", - "old-delete": "e_delete", - }, - "historical_passage_identity": { - "old-update": "e_update", - "old-delete": "e_delete", - }, - } - } - receipts, native = LETTA._apply_operations(client, jobs, state) - self.assertEqual( - [row["native_type"] for row in receipts["j_mutation"]], - ["replace", "delete"], - ) - self.assertEqual(passages.deleted, [("agent", "old-update"), ("agent", "old-delete")]) - self.assertEqual( - state["j_mutation"]["passage_identity"], {"replacement": "e_update"} - ) - self.assertEqual( - LETTA._mapped_evidence( - {"results": [{"id": "old-delete", "content": "stale"}]}, - state["j_mutation"]["historical_passage_identity"], - ), - ["e_delete"], - ) - self.assertEqual( - state["j_mutation"]["historical_passage_identity"], - { - "old-update": "e_update", - "old-delete": "e_delete", - "replacement": "e_update", - }, - ) - self.assertEqual(len(native[0]["operations"]), 2) - - def test_openviking_uses_native_uri_prefixes_dimensions_and_mutations(self) -> None: source_map = { "viking://resources/job/source": "e_parent", diff --git a/scripts/tests/test_benchmark_contract.py b/scripts/tests/test_benchmark_contract.py index 8d7ce5db..601c2640 100644 --- a/scripts/tests/test_benchmark_contract.py +++ b/scripts/tests/test_benchmark_contract.py @@ -26,7 +26,10 @@ class BenchmarkContractTests(BenchmarkCase): def test_manifest_and_exact_four_suites_validate(self) -> None: validate_manifest(self.manifest) self.assertEqual(set(self.suites), set(SUITE_IDS)) - self.assertEqual(len(self.manifest["targets"]), 12) + self.assertEqual(len(self.manifest["targets"]), 10) + self.assertTrue({"sag", "letta"}.isdisjoint( + target["id"] for target in self.manifest["targets"] + )) self.assertEqual(self.manifest["runner"]["capacity"], 1) for suite_id, expected in { "common-core-v1": 24, diff --git a/scripts/tests/test_benchmark_report.py b/scripts/tests/test_benchmark_report.py index b7e1ef03..37f5a372 100644 --- a/scripts/tests/test_benchmark_report.py +++ b/scripts/tests/test_benchmark_report.py @@ -244,7 +244,7 @@ def test_roadmap_lists_all_failures_and_separates_privacy(self) -> None: def test_decisions_disclose_performance_and_add_thresholded_action(self) -> None: suite = self.subset("common-core-v1", 2) rows = [] - for target, latency, ingest in (("elf", 100.0, 1000.0), ("sag", 5.0, 10.0)): + for target, latency, ingest in (("elf", 100.0, 1000.0), ("qmd", 5.0, 10.0)): evaluation = evaluate_unit( suite, self.completed_unit(target, suite), self.targets[target] ) @@ -262,7 +262,7 @@ def test_decisions_disclose_performance_and_add_thresholded_action(self) -> None } ) bundle = { - "target_pins": {"elf": {}, "sag": {}}, + "target_pins": {"elf": {}, "qmd": {}}, "suite_results": {"common-core-v1": {"results": rows}}, } decisions = "\n".join(report_decisions.five_decisions(bundle)) diff --git a/scripts/tests/test_benchmark_runner.py b/scripts/tests/test_benchmark_runner.py index fbfd0ecd..45d3da31 100644 --- a/scripts/tests/test_benchmark_runner.py +++ b/scripts/tests/test_benchmark_runner.py @@ -12,6 +12,8 @@ from benchmark_runner import answers as runner_answers from benchmark_runner import docker as runner_docker import subprocess +import sys +import tempfile PREFLIGHT = load_script( "benchmark_provider_preflight", "scripts/benchmark-provider-preflight.py" @@ -19,6 +21,19 @@ class BenchmarkRunnerTests(BenchmarkCase): + def test_retired_target_is_rejected_before_artifacts_or_provider_access(self) -> None: + for target in ("letta", "sag"): + with self.subTest(target=target), tempfile.TemporaryDirectory() as directory: + artifacts = Path(directory) / "artifacts" + result = subprocess.run( + [sys.executable, str(REPO / "scripts/benchmark-runner.py"), + "--only-target", target, "--artifact-root", str(artifacts)], + cwd=REPO, capture_output=True, text=True, check=False, + ) + self.assertNotEqual(result.returncode, 0) + self.assertIn(f"unknown or retired target: {target}", result.stderr) + self.assertFalse(artifacts.exists()) + def test_shared_answer_requires_nonempty_text_and_preserves_raw_response(self) -> None: suite = self.subset("common-core-v1") case_id = opaque_job_id(suite["jobs"][0]["job_id"])