Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -79,6 +79,7 @@ pollard-serve-eval = "pollard_serve_eval:main"
pollard-card = "pollard_card:main"
pollard-ggufcheck = "pollard_ggufcompat:main"
pollard-flybrain = "pollard_flybrain:main"
pollard-brainverify = "pollard_brainverify:main"
pollard-connectome = "pollard_connectome:main"
pollard-reclaim = "pollard_reclaim:main"
pollard-exl3-band = "pollard_exl3_band:main"
Expand All @@ -91,7 +92,7 @@ pollard-envmatch = "pollard_envmatch:main"
[tool.setuptools]
package-dir = { "" = "tools" }
py-modules = ["pollard_auto", "pollard_calc", "pollard_fit", "pollard_run", "pollard_fit_dit",
"pollard_flybrain", "pollard_connectome", "pollard_experts", "pollard_route", "pollard_sensitivity", "pollard_smooth", "pollard_rotate", "pollard_precondition", "pollard_eval", "pollard_health", "pollard_pack", "pollard_prune", "pollard_bench", "pollard_export", "pollard_gptq", "pollard_automap", "pollard_abliterate", "pollard_probe", "pollard_probes", "pollard_scorecard", "pollard_kl", "pollard_lowbit", "pollard_verify", "pollard_vllm", "pollard_doctor", "pollard_hf_smooth", "pollard_mx", "pollard_mlx", "pollard_exl3", "pollard_calib", "pollard_palette", "pollard_ls", "pollard_workspace",
"pollard_flybrain", "pollard_brainverify", "pollard_brain_backends", "pollard_connectome", "pollard_experts", "pollard_route", "pollard_sensitivity", "pollard_smooth", "pollard_rotate", "pollard_precondition", "pollard_eval", "pollard_health", "pollard_pack", "pollard_prune", "pollard_bench", "pollard_export", "pollard_gptq", "pollard_automap", "pollard_abliterate", "pollard_probe", "pollard_probes", "pollard_scorecard", "pollard_kl", "pollard_lowbit", "pollard_verify", "pollard_vllm", "pollard_doctor", "pollard_hf_smooth", "pollard_mx", "pollard_mlx", "pollard_exl3", "pollard_calib", "pollard_palette", "pollard_ls", "pollard_workspace",
"pollard_serve_eval", "pollard_card", "pollard_onboard", "pollard_envmatch", "imatrix_fix_gate", "exl3_fix_mtp_ehproj",
"pollard_archfp", "pollard_errtype", "pollard_errsrc", "pollard_recard",
"pollard_ggufcompat", "pollard_reclaim", "pollard_exl3_band", "pollard_refcheck",
Expand Down
2 changes: 1 addition & 1 deletion skills/pollard/SKILL.md
Original file line number Diff line number Diff line change
Expand Up @@ -315,7 +315,7 @@ Holds as documents grow far past what it trained on, with the state size unchang
| 1,536 tokens (12 windows) | 0.0% | 100.0% | 8.8 MB |
| 3,072 tokens (24 windows) | 0.0% | 96.9% | 8.8 MB |

**State is 8,552 slots × 256 = 8.8 MB, constant at any length.** The state update costs ~3% of one
**State is 8,552 slots × 328 = 11.2 MB, constant at any length.** The state update costs ~3% of one
model forward (2.1 ms vs 73.4 ms) and fits on CPU at 3.6 ms, so it can run beside the model instead
of competing with it.

Expand Down
109 changes: 109 additions & 0 deletions tests/test_recipes.py
Original file line number Diff line number Diff line change
Expand Up @@ -1427,6 +1427,115 @@ class Qwen2VLConfig for this kind of AutoModel" -- so a VL model could not be lo



def test_brain_query_default_matches_the_verified_prompt():
"""The query shape is part of the experiment, not a cosmetic default.

A token is filed under the words immediately before it, so retrieval works by reproducing that
context. The verified construction ends with "Answer: The secret word is" -- question AND
continuation. Ship a default that is only the question and a brain measuring 100% measures 46%,
from a memory that is perfectly intact. That default shipped, and a first correction to only the
continuation was wrong in the same way. Both halves, or it is not the measured prompt.
"""
tools = pathlib.Path(__file__).resolve().parent.parent / "tools"
src = (tools / "pollard_flybrain.py").read_text(encoding="utf-8")
verified = "Question: what is the secret word? Answer: The secret word is"
i = src.index('ap.add_argument("--ask"')
assert verified in src[i:i + 400], "--ask default is not the verified prompt"
# and the trainer must teach what the default asks
assert verified in src[:i], "the trainer's ASKS no longer contains the default query"

vsrc = (tools / "pollard_brainverify.py").read_text(encoding="utf-8")
assert verified in vsrc, "the verifier must use the same construction it validates"


def test_brain_payload_codec_is_recorded_not_assumed():
"""A brain must say how its payload is encoded, because reading it the other way is noise.

'tokens' stores this backbone's vocabulary indices; 'bytes' stores UTF-8 text, which any
tokenizer can read back and which is smaller for short words (6x8 against 4x18). A brain written
one way and read the other decodes to garbage with no error, so the codec travels in the
checkpoint and defaults to the original behaviour for every brain written before it existed.
"""
tools = pathlib.Path(__file__).resolve().parent.parent / "tools"
src = (tools / "pollard_flybrain.py").read_text(encoding="utf-8")
assert 'self.meta.get("codec", "tokens")' in src, "codec must default to the original behaviour"
assert '"codec": codec' in src, "the trainer must record the codec it wrote"
assert 'bits = 8 if codec == "bytes"' in src, "a byte payload is 8 bits, not the vocabulary width"
assert "decode_text" in src, "a byte brain needs a text decoder"
# Changing the payload must be ALLOWED, not refused. --continue-from carries trained weights;
# written memory lives in a .flystate file, so a new codebook has nothing stored to corrupt.
# People swap backbones and payloads constantly, and refusing the whole transfer over a
# resizable layer threw away the address path and gate that transfer perfectly well.
i = src.index("if continue_from:")
block = src[i:i + 2000]
assert "raise SystemExit" not in block, "a payload change must not abort the transfer"
assert "payload change" in block, "a payload change must be reported, not silent"
# but a written-memory file from a differently shaped brain IS still refused
assert "state was written by a differently shaped brain" in src, \
"load_state must still refuse a mismatched .flystate -- that file holds real memory"



def test_every_runtime_backend_declares_the_same_four_operations():
"""A brain needs four things from a backbone, and nothing else.

embed(ids), forward(embeds) -> (logits, hidden), forward_ids(ids), out_weight(). Brains ran only
under transformers because those four calls were written inline against one library, not because
of anything in the memory. Any runtime that can be fed EMBEDDINGS can host one -- that is the
hard requirement, since memory is delivered by prepending vectors to the sequence.
"""
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "tools"))
import pollard_brain_backends as B

for cls in (B.Transformers, B.MLX, B.ExLlamaV3, B.LlamaCpp):
for op in ("embed", "forward", "forward_ids", "out_weight"):
assert callable(getattr(cls, op, None)), f"{cls.__name__} is missing {op}()"
assert getattr(cls, "name", "?") != "?", f"{cls.__name__} has no lane name"
assert {c.name for c in (B.Transformers, B.MLX, B.ExLlamaV3, B.LlamaCpp)} == \
{"transformers", "mlx", "exl3", "gguf"}


def test_mlx_output_embedding_is_dequantized():
"""A quantized MLX model reports a PACKED output embedding, and the brain's codes come from it.

A 4-bit Qwen reports (151936, 112) where the real matrix is (151936, 896). Hand the brain packed
bytes and it builds its codebook out of bit-patterns: every stored token decodes to noise and
nothing raises. The backend must dequantize before returning it.
"""
tools = pathlib.Path(__file__).resolve().parent.parent / "tools"
src = (tools / "pollard_brain_backends.py").read_text(encoding="utf-8")
i = src.index("class MLX")
block = src[i:src.index("class ExLlamaV3")]
assert "dequantize" in block, "MLX out_weight must dequantize a packed embedding"
assert 'hasattr(mod, "scales")' in block, "must detect a quantized module before unpacking"



def test_gguf_lane_requires_unpooled_per_token_states():
"""llama.cpp CAN host a brain -- but only unpooled.

The high-level Llama.eval() takes tokens only, which is why this lane looked closed. The C API
has both halves: llama_batch_init(n, embd, seq) carries EMBEDDINGS in `embd`, and
llama_get_embeddings_ith() returns the final hidden state per token. Verified on a Q4_K_M build:
embed (1,6,896) in, logits (1,6,151936) and hidden (1,6,896) out.

Pooling is the trap. With llama.cpp's default the context returns ONE pooled vector for the whole
sequence, so a brain has nothing per-token to address on and every write lands in the same place.
The context must be opened with pooling NONE.
"""
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "tools"))
tools = pathlib.Path(__file__).resolve().parent.parent / "tools"
src = (tools / "pollard_brain_backends.py").read_text(encoding="utf-8")
assert "LLAMA_POOLING_TYPE_NONE" in src, "the GGUF context must disable pooling"
i = src.index("class LlamaCpp")
block = src[i:src.index("def open_backend")]
assert "llama_batch_init" in block and "embd" in block, "must feed embeddings, not ids"
assert "llama_get_embeddings_ith" in block, "must read per-token hidden states"
# llama.cpp does not expose its output embedding, and the brain needs to say so rather than guess
assert "does not expose its output embedding" in block



def main():
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_") and callable(v)]
fails = 0
Expand Down
59 changes: 59 additions & 0 deletions tools/pollard_auto.py
Original file line number Diff line number Diff line change
Expand Up @@ -357,6 +357,56 @@ def _ensure_imatrix(a):
return imat


def attach_brain(brain_path: str, out_dir: str, fmt: str) -> None:
"""Ship a brain alongside a build, with a note saying exactly what it attaches to.

The brain is a separate 11 MB file, not weights to quantize, so it copies into the build
untouched whatever lane this is. What differs per lane is whether it can be USED there yet: the
GPTQ lane loads under transformers, so a brain attaches directly; GGUF, MLX and EXL3 run under
their own engines, which have no hook to inject memory tokens into the residual stream. The file
still travels with the build so the pair never gets separated, and the note says which case this
is instead of leaving someone to find out at run time.
"""
import shutil
if not os.path.isfile(brain_path):
raise SystemExit(f"--brain: no such file {brain_path}")
dest_dir = out_dir if os.path.isdir(out_dir) else os.path.dirname(out_dir) or "."
name = os.path.basename(brain_path)
shutil.copy2(brain_path, os.path.join(dest_dir, name))

meta = {}
try:
import torch
meta = torch.load(brain_path, map_location="cpu", weights_only=False).get("meta", {})
except Exception:
pass
live = fmt in ("gptq",)
note = [f"# Brain: {name}", ""]
note.append(f"- slots x width : {meta.get('neurons','?')} x {meta.get('width','?')}"
f" ({int(meta.get('neurons',0)) * int(meta.get('width',0)) * 4 / 1e6:.1f} MB live state)"
if meta.get("neurons") else "- (could not read brain metadata)")
if meta.get("hidden"):
note.append(f"- trained against a backbone with hidden size {meta['hidden']}")
note.append("")
if live:
note += ["Attach it at run time:", "",
"```python", "from pollard_flybrain import FlyBrain, load_backbone",
"brain = FlyBrain.load(\"" + name + "\").bind(model, tok)",
"brain.feed(open(\"long_document.txt\").read())", "```", ""]
else:
note += [f"The {fmt.upper()} lane runs under its own engine, which has no hook to inject",
"memory tokens into the residual stream, so the brain cannot attach to THIS build",
"yet. It ships here so the pair stays together; use it with the transformers copy of",
"the same backbone, or carry it to another model with:", "",
"```", f"pollard-flybrain --train 900 --continue-from {name} --model <hf-id> ...", "```", ""]
note += ["Verify any brain with:", "",
"```", f"pollard-brainverify --brain {name} --model <hf-id> --filler corpus.txt", "```"]
with open(os.path.join(dest_dir, "BRAIN.md"), "w", encoding="utf-8") as f:
f.write("\n".join(note) + "\n")
print(f" brain: {name} -> {dest_dir}"
+ (" (attaches at run time)" if live else f" (ships with the build; {fmt} cannot host it yet)"))


def main():
ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
ap.add_argument("--gguf", help="f16/bf16 source GGUF (or use --hf to point at HF weights)")
Expand All @@ -367,6 +417,11 @@ def main():
" | exl3 (exllamav3 -- the heavy trellis lane) | mx (Blackwell NVFP4 / any-GPU W4A16, "
"compressed-tensors)")
ap.add_argument("--output", help="output dir/file for the gptq/mlx/mx/exl3 export (else auto-named)")
ap.add_argument("--brain", help="ship a trained brain (pollard-flybrain / human connectome) WITH this\n"
"build, so the pair travels as one artifact. The brain is not\n"
"quantized -- it is 11 MB of memory that attaches to the model at run\n"
"time and can be detached, moved to another backbone and carried on\n"
"with --continue-from.")
ap.add_argument("--sensitivity", help="Pollard sensitivity.json (gptq/mlx/mx allocation; else auto-measured)")
ap.add_argument("--no-measure", dest="measure", action="store_false",
help="skip the auto sensitivity probe on the gptq/mlx/mx lanes (falls back to uniform "
Expand Down Expand Up @@ -459,6 +514,8 @@ def main():
except Exception as e:
print(f" [match-transformers] skipped ({e}); using the current env")
_emit_nongguf(a)
if a.brain and a.run:
attach_brain(a.brain, a.output or ".", a.format)
if not a.run:
print("\n plan only -- re-run with --run to execute.")
return
Expand Down Expand Up @@ -518,6 +575,8 @@ def main():
else:
print(f" 2) (--no-auto-imatrix set and no --imatrix: stock K-quant ladder only. Drop the "
f"flag for the {flagship} flagship -- the winning build, auto-calibrated.)")
if a.brain and a.run:
attach_brain(a.brain, a.out or os.path.dirname(a.gguf) or ".", "gguf")
if not a.run:
print("\n plan only -- re-run with --run to execute.")

Expand Down
Loading
Loading