From 2e60a5d5408b0ef5969b3c3a50e0eb70cb71b3fe Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Sun, 13 Sep 2026 20:45:55 -0500 Subject: [PATCH 01/84] feat: c library form with the per-point infer inline in the header --- python/rosenna/cli.py | 7 ++- python/rosenna/emit_c.py | 33 ++++++++++-- python/rosenna/verify.py | 12 ++++- python/tests/test_emit_c.py | 20 +++++--- python/tests/test_library_form.py | 83 +++++++++++++++++++++++++++++++ 5 files changed, 140 insertions(+), 15 deletions(-) create mode 100644 python/tests/test_library_form.py diff --git a/python/rosenna/cli.py b/python/rosenna/cli.py index 0ae1889..f043419 100644 --- a/python/rosenna/cli.py +++ b/python/rosenna/cli.py @@ -6,7 +6,7 @@ from google.protobuf.message import DecodeError -from .emit_c import emit_c +from .emit_c import emit_c, emit_c_recipe from .emit_fortran import emit_fortran from .frontend import UnsupportedModel, load_graph from .plan import build_plan, validate_model_name @@ -75,11 +75,14 @@ def _cmd_generate(args) -> int: written.append(f90_path) if "c" in langs: source, header = emit_c(plan) + recipe = emit_c_recipe(plan) c_path = outdir / f"{name}.c" h_path = outdir / f"{name}.h" + mk_path = outdir / f"{name}.mk" c_path.write_text(source) h_path.write_text(header) - written += [c_path, h_path] + mk_path.write_text(recipe) + written += [c_path, h_path, mk_path] rwt_path = outdir / f"{name}.rwt" write_weights(plan, graph, rwt_path) diff --git a/python/rosenna/emit_c.py b/python/rosenna/emit_c.py index 740c712..961428a 100644 --- a/python/rosenna/emit_c.py +++ b/python/rosenna/emit_c.py @@ -73,11 +73,28 @@ def emit_c(plan: Plan) -> tuple: lines += _emit_source_head(plan, ctype) lines += _emit_load(plan) lines += _emit_init(plan) - lines += _emit_infer(plan, ctype) source = "\n".join(lines) + "\n" return source, header +def emit_c_recipe(plan: Plan) -> str: + """A Makefile fragment that builds lib.a with the host's offload flags.""" + n = plan.model + return f"""# Generated by rosenna. Build lib{n}.a with the same offload flags as the host. +CC ?= gcc +CFLAGS ?= -O2 -Wall -Wextra -std=c11 +ROSENNA_OFFLOAD_FLAGS ?= + +lib{n}.a: {n}.o +\tar rcs $@ $< +{n}.o: {n}.c {n}.h +\t$(CC) $(CFLAGS) $(ROSENNA_OFFLOAD_FLAGS) -c $< -o $@ +clean: +\trm -f {n}.o lib{n}.a +.PHONY: clean +""" + + def _emit_header(plan: Plan, ctype: str) -> str: m = plan.model guard = f"ROSENNA_{m.upper()}_H" @@ -87,8 +104,16 @@ def _emit_header(plan: Plan, ctype: str) -> str: "", "/* Generated by rosenna. Do not edit. */", "", + "#include ", + "", + ] + for w in plan.weights: + lines.append(f"extern {ctype} {w.symbol}[{_weight_size(w.shape)}];") + if plan.weights: + lines.append("") + lines += _emit_infer(plan, ctype) + lines += [ f"int {m}_init(const char *path);", - f"void {m}_infer(const {ctype} *restrict x, {ctype} *restrict y);", "", "#endif", ] @@ -111,7 +136,7 @@ def _emit_source_head(plan: Plan, ctype: str) -> list: "", ] for w in plan.weights: - lines.append(f"static {ctype} {w.symbol}[{_weight_size(w.shape)}];") + lines.append(f"{ctype} {w.symbol}[{_weight_size(w.shape)}];") lines.append("") return lines @@ -227,7 +252,7 @@ def _emit_infer(plan: Plan, ctype: str) -> list: scratch = sorted((s for s in plan.buffers if s not in ("x", "y")), key=lambda s: int(s[1:])) - lines = [f"void {m}_infer(const {ctype} *restrict x, {ctype} *restrict y) {{"] + lines = [f"static inline void {m}_infer(const {ctype} *restrict x, {ctype} *restrict y) {{"] for sym in scratch: lines.append(f" {ctype} {sym}[{plan.buffers[sym]}];") diff --git a/python/rosenna/verify.py b/python/rosenna/verify.py index 4e99306..a27723f 100644 --- a/python/rosenna/verify.py +++ b/python/rosenna/verify.py @@ -198,9 +198,19 @@ def _run_backend(backend: str, plan, workdir: Path, inputs): (workdir / f"{name}.c").write_text(source) (workdir / f"{name}.h").write_text(header) (workdir / "verify_main.c").write_text(_c_driver(name, n_in, n_out, dtype)) + # Compile the generated source to an object, archive it, and link the + # driver against the archive -- the library form -- rather than + # compiling both sources together, so `verify` exercises the same + # delivery shape a downstream host build uses. + _run("compile", backend, + ["gcc", "-O2", "-Wall", "-Wextra", "-std=c11", "-c", f"{name}.c", "-o", f"{name}.o"], + cwd=workdir) + _run("archive", backend, + ["ar", "rcs", f"lib{name}.a", f"{name}.o"], + cwd=workdir) _run("compile/link", backend, ["gcc", "-O2", "-Wall", "-Wextra", "-std=c11", "-o", "verify_run", - f"{name}.c", "verify_main.c", "-lm"], + "verify_main.c", f"lib{name}.a", "-lm"], cwd=workdir) else: raise ValueError(f"unknown backend {backend!r}") diff --git a/python/tests/test_emit_c.py b/python/tests/test_emit_c.py index 28987c5..06f7026 100644 --- a/python/tests/test_emit_c.py +++ b/python/tests/test_emit_c.py @@ -83,14 +83,18 @@ def test_matches_onnxruntime_f32(tmp_path, golden_model): def test_infer_is_pure_and_has_literal_bounds(golden_model): plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64") source, header = emit_c(plan) - assert "void gemm_small_infer(const double *restrict x, double *restrict y) {" in source + # `infer` is now defined only in the header (a static inline callable + # from inside the host's own offload region); the source never defines + # it. + assert "static inline void gemm_small_infer(const double *restrict x, double *restrict y) {" in header # The scratch buffers come from plan.buffers now (ruling R13), not from a # second allocator private to this emitter: gemm_small's t0 is reused by # both gemms, so the plan sizes it at the larger of the two (3), and the # Fortran backend declares exactly the same set. - assert "double t0[3];" in source - assert "double t1[2];" in source + assert "double t0[3];" in header + assert "double t1[2];" in header assert "malloc" not in source + assert "malloc" not in header assert "restrict" in header @@ -138,11 +142,11 @@ def test_both_backends_agree(tmp_path, golden_model): def test_f32_plan_uses_single_precision_math(golden_model): """An f32 build must call tanhf/expf, not promote every activation to double.""" plan = build_plan(load_graph(golden_model("gemm_big")), dtype="f32") - source, _ = emit_c(plan) - assert "tanhf(" in source - assert "expf(" in source - assert "0.0f" in source - body = "\n".join(l for l in source.splitlines() if "_infer" not in l) + _, header = emit_c(plan) + assert "tanhf(" in header + assert "expf(" in header + assert "0.0f" in header + body = "\n".join(l for l in header.splitlines() if "_infer" not in l) assert " tanh(" not in body and "=tanh(" not in body assert " exp(" not in body and "(exp(" not in body diff --git a/python/tests/test_library_form.py b/python/tests/test_library_form.py new file mode 100644 index 0000000..fd93052 --- /dev/null +++ b/python/tests/test_library_form.py @@ -0,0 +1,83 @@ +import re +import shutil +import subprocess +import numpy as np +import onnxruntime as ort +import pytest +from rosenna.frontend import load_graph +from rosenna.plan import build_plan +from rosenna.weights import write_weights +from rosenna.emit_c import emit_c, emit_c_recipe +from tests.test_emit_fortran import _live_reference + + +def _cc(): + for cand in ("gcc-15", "gcc-14", "gcc-13", "gcc"): + if shutil.which(cand): + return cand + pytest.skip("no C compiler found") + + +def test_header_defines_inline_infer_and_source_does_not(golden_model): + plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64") + source, header = emit_c(plan) + assert "static inline void gemm_small_infer(" in header + # Structural check (controller ruling P1): infer must be defined only in + # the header, never in the source, regardless of how the source happens + # to spell a call to it. + assert "static inline" not in source + assert re.search(r"^void gemm_small_infer\(", source, re.MULTILINE) is None + assert "extern double w0[" in header + assert "\ndouble w0[" in source and "static double w0[" not in source + + +def test_library_and_header_inline_agree(tmp_path, golden_model): + name = "gemm_small" + graph = load_graph(golden_model(name)) + plan = build_plan(graph, dtype="f64") + source, header = emit_c(plan) + (tmp_path / f"{name}.c").write_text(source) + (tmp_path / f"{name}.h").write_text(header) + write_weights(plan, graph, tmp_path / f"{name}.rwt") + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + (tmp_path / "host.c").write_text(f""" +#include +#include "{name}.h" +int main(void) {{ + double x[{n_in}], y[{n_out}]; + int n; if ({name}_init("{name}.rwt")) return 2; + if (scanf("%d", &n) != 1) return 1; + for (int c = 0; c < n; ++c) {{ + for (int i = 0; i < {n_in}; ++i) if (scanf("%lf", &x[i]) != 1) return 1; + {name}_infer(x, y); /* the header inline, called from the host TU */ + for (int i = 0; i < {n_out}; ++i) printf("%.17e ", y[i]); + printf("\\n"); + }} + return 0; +}} +""") + cc = _cc() + subprocess.run([cc, "-O2", "-Wall", "-Wextra", "-std=c11", "-c", f"{name}.c"], cwd=tmp_path, check=True, capture_output=True, text=True) + subprocess.run(["ar", "rcs", f"lib{name}.a", f"{name}.o"], cwd=tmp_path, check=True) + subprocess.run([cc, "-O2", "-Wall", "-Wextra", "-std=c11", "host.c", f"lib{name}.a", "-lm", "-o", "host"], cwd=tmp_path, check=True, capture_output=True, text=True) + session = ort.InferenceSession(golden_model(name)) + inputs, expected = _live_reference(session, session.get_inputs()[0].shape, np.float64, seed=3, batch=8) + if inputs is None: + pytest.skip(f"{name}: onnxruntime reference is all-zero across 10 resampled " + f"batches; its golden-file weights produced a dead model") + stdin = f"{len(inputs)}\n" + "\n".join(" ".join(repr(float(v)) for v in row) for row in inputs) + out = subprocess.run(["./host"], cwd=tmp_path, input=stdin, capture_output=True, text=True, check=True).stdout + got = np.array([[float(v) for v in line.split()] for line in out.strip().splitlines()]) + np.testing.assert_allclose(got, expected, rtol=1e-5, atol=1e-6) + + +def test_recipe_builds_the_library(tmp_path, golden_model): + name = "gemm_small" + graph = load_graph(golden_model(name)) + plan = build_plan(graph, dtype="f64") + source, header = emit_c(plan) + (tmp_path / f"{name}.c").write_text(source) + (tmp_path / f"{name}.h").write_text(header) + (tmp_path / "Makefile").write_text(emit_c_recipe(plan)) + subprocess.run(["make", f"CC={_cc()}"], cwd=tmp_path, check=True, capture_output=True, text=True) + assert (tmp_path / f"lib{name}.a").exists() From 2ca5720822d5370a3e26ff3412065eb9db1d011f Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Sun, 13 Sep 2026 20:55:52 -0500 Subject: [PATCH 02/84] fix: prefix every external weight symbol with the model name --- python/rosenna/emit_c.py | 31 ++++++++--- python/tests/test_library_form.py | 91 ++++++++++++++++++++++++++++++- 2 files changed, 113 insertions(+), 9 deletions(-) diff --git a/python/rosenna/emit_c.py b/python/rosenna/emit_c.py index 961428a..62fe9fd 100644 --- a/python/rosenna/emit_c.py +++ b/python/rosenna/emit_c.py @@ -43,6 +43,21 @@ def _weight_size(shape) -> int: return size +def _c_weight_symbol(model: str, symbol: str) -> str: + """The C external identifier for a plan weight symbol (controller ruling R1). + + plan.py names every model's weights `w0`, `b0`, `w1`, ... uniformly: + those symbols enter the plan hash and Fortran's module scope, where the + `module` keyword already isolates them, so plan.py stays as it is. + Dropping `static` from the C weight definitions (this task) gives them + external linkage, though, and two different models linked into one host + would then collide on `_w0`/`_b0`. Every C site that names a weight + array goes through this one helper, prefixed with the model name, so no + site can drift out of sync with another. + """ + return f"{model}_{symbol}" + + def _weight_index_c(weight_by_symbol: dict, op) -> str: """Decide the accumulation index order from the op's own transB flag. @@ -108,7 +123,7 @@ def _emit_header(plan: Plan, ctype: str) -> str: "", ] for w in plan.weights: - lines.append(f"extern {ctype} {w.symbol}[{_weight_size(w.shape)}];") + lines.append(f"extern {ctype} {_c_weight_symbol(m, w.symbol)}[{_weight_size(w.shape)}];") if plan.weights: lines.append("") lines += _emit_infer(plan, ctype) @@ -130,13 +145,12 @@ def _emit_source_head(plan: Plan, ctype: str) -> list: "#include ", "#include ", "#include ", - "#include ", "", f"static const unsigned char expected_hash[32] = {{ {hash_bytes} }};", "", ] for w in plan.weights: - lines.append(f"{ctype} {w.symbol}[{_weight_size(w.shape)}];") + lines.append(f"{ctype} {_c_weight_symbol(m, w.symbol)}[{_weight_size(w.shape)}];") lines.append("") return lines @@ -144,11 +158,13 @@ def _emit_source_head(plan: Plan, ctype: str) -> list: def _emit_load(plan: Plan) -> list: if not plan.weights: return [] + m = plan.model lines = [ "static int load_tensor(FILE *f, const char *name, int32_t namelen, long pos,", " int64_t length) {", ] for w in plan.weights: + c_sym = _c_weight_symbol(m, w.symbol) # `length` comes from the file's own table of contents; a tensor whose # declared byte count disagrees with the array it is about to fill # means a corrupt file, so reject it rather than short-read into the @@ -156,9 +172,9 @@ def _emit_load(plan: Plan) -> list: lines.append( f" if (namelen == {len(w.name)} && " f"memcmp(name, {_c_string(w.name)}, {len(w.name)}) == 0) {{") - lines.append(f" if (length != (int64_t)sizeof {w.symbol}) return 9;") + lines.append(f" if (length != (int64_t)sizeof {c_sym}) return 9;") lines.append(" if (fseek(f, pos, SEEK_SET) != 0) return 9;") - lines.append(f" if (fread({w.symbol}, sizeof {w.symbol}, 1, f) != 1) return 9;") + lines.append(f" if (fread({c_sym}, sizeof {c_sym}, 1, f) != 1) return 9;") lines.append(" return 0;") lines.append(" }") lines += [" return 7;", "}", ""] @@ -261,12 +277,13 @@ def _emit_infer(plan: Plan, ctype: str) -> list: dst, src = plan.assignment[op.out], plan.assignment[op.inp] if op.kind == "gemm": idx_expr = _weight_index_c(weight_by_symbol, op) - bias_init = f"{op.bias}[i]" if op.bias else _ZERO[plan.dtype] + weight_sym = _c_weight_symbol(m, op.weight) + bias_init = f"{_c_weight_symbol(m, op.bias)}[i]" if op.bias else _ZERO[plan.dtype] lines.append(f" for (int i = 0; i < {op.n_out}; ++i) {{") lines.append(f" {ctype} acc = {bias_init};") lines.append( f" for (int j = 0; j < {op.n_in}; ++j) " - f"acc += {src}[j] * {op.weight}[{idx_expr}];") + f"acc += {src}[j] * {weight_sym}[{idx_expr}];") lines.append(f" {dst}[i] = acc;") lines.append(" }") cur_len = op.n_out diff --git a/python/tests/test_library_form.py b/python/tests/test_library_form.py index fd93052..710281f 100644 --- a/python/tests/test_library_form.py +++ b/python/tests/test_library_form.py @@ -27,8 +27,11 @@ def test_header_defines_inline_infer_and_source_does_not(golden_model): # to spell a call to it. assert "static inline" not in source assert re.search(r"^void gemm_small_infer\(", source, re.MULTILINE) is None - assert "extern double w0[" in header - assert "\ndouble w0[" in source and "static double w0[" not in source + # Weight symbols carry the model-name prefix (controller ruling R1) so + # two different models' weight arrays never collide once `static` is + # dropped and they gain external linkage. + assert "extern double gemm_small_w0[" in header + assert "\ndouble gemm_small_w0[" in source and "static double gemm_small_w0[" not in source def test_library_and_header_inline_agree(tmp_path, golden_model): @@ -71,6 +74,90 @@ def test_library_and_header_inline_agree(tmp_path, golden_model): np.testing.assert_allclose(got, expected, rtol=1e-5, atol=1e-6) +def test_two_models_link_into_one_host(tmp_path, golden_model): + """Two different models' libraries must link into one host binary. + + Reproduces the defect the reviewer found in 2e60a5d: dropping `static` + from the weight definitions, without also prefixing them with the model + name, gives every model's `w0`/`b0`/... external linkage under the same + names (plan.py names weights identically across models), so `gemm_small` + and `gemm_nobias` linked into one binary fail with + `ld: duplicate symbols '_w0'`. Controller ruling R1: every C weight + symbol is prefixed with the model name, so this must link, run, and each + model's output must match its own onnxruntime reference. + """ + names = ["gemm_small", "gemm_nobias"] + plans, sessions = {}, {} + cc = _cc() + obj_args = [] + for name in names: + graph = load_graph(golden_model(name)) + plan = build_plan(graph, dtype="f64") + plans[name] = plan + source, header = emit_c(plan) + (tmp_path / f"{name}.c").write_text(source) + (tmp_path / f"{name}.h").write_text(header) + write_weights(plan, graph, tmp_path / f"{name}.rwt") + subprocess.run([cc, "-O2", "-Wall", "-Wextra", "-std=c11", "-c", f"{name}.c"], + cwd=tmp_path, check=True, capture_output=True, text=True) + subprocess.run(["ar", "rcs", f"lib{name}.a", f"{name}.o"], cwd=tmp_path, check=True) + obj_args.append(f"lib{name}.a") + sessions[name] = ort.InferenceSession(golden_model(name)) + + references = {} + for name in names: + session = sessions[name] + shape = session.get_inputs()[0].shape + inputs, expected = _live_reference(session, shape, np.float64, seed=5, batch=8) + if inputs is None: + pytest.skip(f"{name}: onnxruntime reference is all-zero across 10 resampled " + f"batches; its golden-file weights produced a dead model") + references[name] = (inputs, expected) + + host_lines = ["#include "] + host_lines += [f'#include "{name}.h"' for name in names] + host_lines.append("int main(void) {") + for name in names: + plan = plans[name] + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + host_lines.append(f" double {name}_x[{n_in}], {name}_y[{n_out}];") + host_lines.append(f' if ({name}_init("{name}.rwt")) return 2;') + for name in names: + plan = plans[name] + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + host_lines.append(" { int n; if (scanf(\"%d\", &n) != 1) return 1;") + host_lines.append(" for (int c = 0; c < n; ++c) {") + host_lines.append( + f" for (int i = 0; i < {n_in}; ++i) " + f"if (scanf(\"%lf\", &{name}_x[i]) != 1) return 1;") + host_lines.append(f" {name}_infer({name}_x, {name}_y);") + host_lines.append( + f" for (int i = 0; i < {n_out}; ++i) printf(\"%.17e \", {name}_y[i]);") + host_lines.append(' printf("\\n");') + host_lines.append(" } }") + host_lines.append(" return 0;") + host_lines.append("}") + (tmp_path / "host.c").write_text("\n".join(host_lines) + "\n") + + subprocess.run([cc, "-O2", "-Wall", "-Wextra", "-std=c11", "host.c", *obj_args, "-lm", "-o", "host"], + cwd=tmp_path, check=True, capture_output=True, text=True) + + stdin_parts = [] + for name in names: + inputs, _ = references[name] + stdin_parts.append(str(len(inputs))) + stdin_parts.append("\n".join(" ".join(repr(float(v)) for v in row) for row in inputs)) + stdin = "\n".join(stdin_parts) + "\n" + out = subprocess.run(["./host"], cwd=tmp_path, input=stdin, capture_output=True, text=True, check=True).stdout + all_lines = out.strip().splitlines() + pos = 0 + for name in names: + inputs, expected = references[name] + got = np.array([[float(v) for v in line.split()] for line in all_lines[pos:pos + len(inputs)]]) + pos += len(inputs) + np.testing.assert_allclose(got, expected, rtol=1e-5, atol=1e-6) + + def test_recipe_builds_the_library(tmp_path, golden_model): name = "gemm_small" graph = load_graph(golden_model(name)) From 5ae8d5ffa81aae209148fe5ca6b208dec89e1dda Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Sun, 13 Sep 2026 21:22:23 -0500 Subject: [PATCH 03/84] feat: device decoration for openmp, openacc, cuda and hip hosts; embed weights by default --- .github/workflows/CI.yml | 5 + python/rosenna/cli.py | 28 ++++- python/rosenna/emit_c.py | 157 +++++++++++++++++++++--- python/rosenna/plan.py | 54 +++++++- python/rosenna/verify.py | 23 ++-- python/tests/measure_embed_threshold.py | 88 +++++++++++++ python/tests/test_device_c.py | 116 +++++++++++++++++ python/tests/test_embed.py | 39 ++++++ python/tests/test_emit_c.py | 14 ++- python/tests/test_library_form.py | 10 +- python/tests/test_regressions.py | 18 ++- 11 files changed, 510 insertions(+), 42 deletions(-) create mode 100644 python/tests/measure_embed_threshold.py create mode 100644 python/tests/test_device_c.py create mode 100644 python/tests/test_embed.py diff --git a/.github/workflows/CI.yml b/.github/workflows/CI.yml index b597df6..9823565 100644 --- a/.github/workflows/CI.yml +++ b/.github/workflows/CI.yml @@ -41,6 +41,11 @@ jobs: pip install -e python cd python && python3 -m pytest tests -v + - name: Device-path tests ran on the host (not skipped) + run: | + cd python && python3 -m pytest tests/test_device_c.py -v -rs 2>&1 | tee device.log + ! grep -q "SKIPPED" device.log + - name: Run test cases run: | mkdir -p fLibrary/objFiles diff --git a/python/rosenna/cli.py b/python/rosenna/cli.py index f043419..687f996 100644 --- a/python/rosenna/cli.py +++ b/python/rosenna/cli.py @@ -20,6 +20,15 @@ def _dtype_from_precision(precision: str | None) -> str | None: return _PRECISION_TO_DTYPE.get(precision) if precision else None +def _add_embed_flags(sub: argparse.ArgumentParser) -> None: + group = sub.add_mutually_exclusive_group() + group.add_argument("--embed-weights", dest="embed", action="store_true", default=None, + help="embed weights as constants in the header, regardless of size " + "(default: embed automatically below EMBED_THRESHOLD parameters)") + group.add_argument("--no-embed", dest="embed", action="store_false", + help="always load weights from a .rwt file at runtime") + + def build_parser() -> argparse.ArgumentParser: p = argparse.ArgumentParser(prog="rosenna", description="ONNX to Fortran/C inference code") sub = p.add_subparsers(dest="command", required=True) @@ -31,12 +40,14 @@ def build_parser() -> argparse.ArgumentParser: help="default: the model's own dtype") gen.add_argument("--out", default=".") gen.add_argument("--name", default=None, help="symbol prefix; default: the model file stem") + _add_embed_flags(gen) ver = sub.add_parser("verify", help="compile the generated code and compare against onnxruntime") ver.add_argument("model") ver.add_argument("--lang", choices=["fortran", "c", "both"], default="both") ver.add_argument("--precision", choices=["single", "double"], default=None) ver.add_argument("--cases", type=int, default=16, help="random inputs to compare") + _add_embed_flags(ver) info = sub.add_parser("info", help="report ops, shapes and whether the model is supported") info.add_argument("model") @@ -61,7 +72,7 @@ def shape_of(name: str) -> str: def _cmd_generate(args) -> int: graph = load_graph(args.model, name=args.name) - plan = build_plan(graph, dtype=_dtype_from_precision(args.precision)) + plan = build_plan(graph, dtype=_dtype_from_precision(args.precision), embed=args.embed) validate_model_name(plan.model) outdir = Path(args.out) outdir.mkdir(parents=True, exist_ok=True) @@ -84,9 +95,16 @@ def _cmd_generate(args) -> int: mk_path.write_text(recipe) written += [c_path, h_path, mk_path] - rwt_path = outdir / f"{name}.rwt" - write_weights(plan, graph, rwt_path) - written.append(rwt_path) + # An embedded plan has no weights file to write: every weight is already a + # ROSENNA_CONST array baked into the header. Fortran generation (unchanged + # by this task) still needs a .rwt to load, so it is written whenever + # Fortran is one of the requested languages even if the plan embeds. + if plan.embed and "fortran" not in langs: + print(f"embedded weights ({plan.n_params} parameters)") + else: + rwt_path = outdir / f"{name}.rwt" + write_weights(plan, graph, rwt_path) + written.append(rwt_path) for path in written: print(path) @@ -96,7 +114,7 @@ def _cmd_generate(args) -> int: def _cmd_verify(args) -> int: with tempfile.TemporaryDirectory() as workdir: results = verify_model(args.model, args.lang, _dtype_from_precision(args.precision), - args.cases, workdir) + args.cases, workdir, embed=args.embed) all_ok = True for r in results: status = "ok" if r.ok else "FAIL" diff --git a/python/rosenna/emit_c.py b/python/rosenna/emit_c.py index 62fe9fd..94de7ea 100644 --- a/python/rosenna/emit_c.py +++ b/python/rosenna/emit_c.py @@ -4,6 +4,9 @@ _CTYPE = {"f32": "float", "f64": "double"} _DTYPE_CODE = {"f32": 0, "f64": 1} +# Embedding must be lossless: %.17g round-trips any f64, %.9g any f32 +# (Steele & White / Ryu-style shortest-exact-decimal bounds). +_EMBED_FMT = {"f32": "%.9g", "f64": "%.17g"} # Activation expressions, per plan dtype. An f32 plan must call the float # intrinsics: tanh/exp on a float promote the whole expression to double, so @@ -58,6 +61,44 @@ def _c_weight_symbol(model: str, symbol: str) -> str: return f"{model}_{symbol}" +def _c_weight_ref_macro(model: str, symbol: str) -> str: + """The identifier `infer` uses to read a file-loaded weight. + + A file-loaded plan's header declares two different C names for the same + weight -- the host array `` and, under __CUDACC__/__HIPCC__, the + device pointer `_dev` (controller ruling P4; the pointer is + declared-only until Task 3) -- and `infer` is emitted exactly once, so it + cannot spell either name directly. This macro (itself prefixed with the + model-qualified symbol, so it carries ruling R1's collision safety same + as every other external name here) is `#define`d to whichever of the two + the same __CUDACC__/__HIPCC__ guard selects; `infer`'s body reads only + this name. An embedded plan has no such split -- its weights are one + ROSENNA_CONST array reachable under any guard -- so `infer` reads the + plain symbol directly there and never goes through this macro. + """ + return f"ROSENNA_REF_{_c_weight_symbol(model, symbol)}" + + +def _weight_ref(plan: Plan, model: str, symbol: str) -> str: + return _c_weight_symbol(model, symbol) if plan.embed else _c_weight_ref_macro(model, symbol) + + +def _format_embedded_value(v: float, dtype: str) -> str: + """Format one embedded weight, losslessly, as a C literal of the plan's dtype. + + %g drops the decimal point for an exact integer (`1.0` -> `"1"`), and `1f` + is not a floating-constant in C -- the `f` suffix is only legal directly + after a decimal point or an exponent -- so an f32 value with neither gets + one inserted before the suffix is appended. + """ + s = _EMBED_FMT[dtype] % v + if dtype == "f32": + if not any(c in s for c in ".eE"): + s += ".0" + s += "f" + return s + + def _weight_index_c(weight_by_symbol: dict, op) -> str: """Decide the accumulation index order from the op's own transB flag. @@ -84,11 +125,18 @@ def _weight_index_c(weight_by_symbol: dict, op) -> str: def emit_c(plan: Plan) -> tuple: ctype = _CTYPE[plan.dtype] header = _emit_header(plan, ctype) - lines = [] - lines += _emit_source_head(plan, ctype) - lines += _emit_load(plan) - lines += _emit_init(plan) - source = "\n".join(lines) + "\n" + if plan.embed: + # Every weight is a ROSENNA_CONST array in the header, so the source + # has nothing left to define. (Under __CUDACC__/__HIPCC__ that macro + # is __constant__; compiles and runs on the host, device path + # unvalidated until the GPU gate.) + source = f'/* Generated by rosenna. Do not edit. */\n#include "{plan.model}.h"\n' + else: + lines = [] + lines += _emit_source_head(plan, ctype) + lines += _emit_load(plan) + lines += _emit_init(plan) + source = "\n".join(lines) + "\n" return source, header @@ -121,18 +169,94 @@ def _emit_header(plan: Plan, ctype: str) -> str: "", "#include ", "", + # Every device decoration in this header goes through exactly these + # three macros (controller rulings: brief + P3). Under nvcc/hipcc, + # `infer` is callable from device code and embedded weights live in + # __constant__ memory; under any other compiler both are inert. C has + # no `restrict` keyword once this header is pulled into a C++ (or + # CUDA/HIP, which is always C++) translation unit, so ROSENNA_RESTRICT + # picks the compiler-correct spelling instead of `infer` hardcoding one. + "#if defined(__CUDACC__) || defined(__HIPCC__)", + "#define ROSENNA_DEVICE_FN __host__ __device__", + "#define ROSENNA_CONST __constant__", + "#define ROSENNA_RESTRICT __restrict__", + "#else", + "#define ROSENNA_DEVICE_FN", + "#define ROSENNA_CONST static const", + "#if defined(__cplusplus)", + "#define ROSENNA_RESTRICT __restrict__", + "#else", + "#define ROSENNA_RESTRICT restrict", + "#endif", + "#endif", + "", ] + if not plan.embed: + lines += _emit_weight_declarations(plan, ctype) + lines += _emit_device_region(plan, ctype) + if not plan.embed: + lines += [ + f"int {m}_init(const char *path);", + "", + ] + lines.append("#endif") + return "\n".join(lines) + "\n" + + +def _emit_weight_declarations(plan: Plan, ctype: str) -> list: + """A file-loaded plan's weights: a host array, or (Task 3) a device pointer. + + Controller ruling P4: the `_dev` pointers are declared, not defined -- + nothing fills them in until Task 3's CUDA/HIP init exists. Under a plain + or OpenMP host build the __CUDACC__/__HIPCC__ guard is false, so this + branch is never even compiled; no build in this task can reach it. + Compiles and runs on the host; device path unvalidated. + """ + m = plan.model + if not plan.weights: + return [] + lines = ["#if defined(__CUDACC__) || defined(__HIPCC__)"] + for w in plan.weights: + sym = _c_weight_symbol(m, w.symbol) + lines.append(f"extern {ctype} *{sym}_dev;") + lines.append(f"#define {_c_weight_ref_macro(m, w.symbol)} {sym}_dev") + lines.append("#else") + for w in plan.weights: + sym = _c_weight_symbol(m, w.symbol) + lines.append(f"extern {ctype} {sym}[{_weight_size(w.shape)}];") + lines.append(f"#define {_c_weight_ref_macro(m, w.symbol)} {sym}") + lines += ["#endif", ""] + return lines + + +def _emit_device_region(plan: Plan, ctype: str) -> list: + """The declare-target region: embedded weight consts (if any) plus infer. + + Every OpenMP pragma is guarded by `_OPENMP` and every OpenACC pragma by + `_OPENACC` -- gcc -Wall warns `ignoring '#pragma acc ...' + [-Wunknown-pragmas]` on an unguarded one under a compiler without + OpenACC. Under nvcc/hipcc both guards are false and ROSENNA_DEVICE_FN + supplies the CUDA decoration instead. + """ + lines = ["#ifdef _OPENMP", "#pragma omp declare target", "#endif", ""] + if plan.embed: + lines += _emit_embedded_weights(plan, ctype) + lines += ["#ifdef _OPENACC", "#pragma acc routine seq", "#endif"] + lines += _emit_infer(plan, ctype) + lines += ["#ifdef _OPENMP", "#pragma omp end declare target", "#endif", ""] + return lines + + +def _emit_embedded_weights(plan: Plan, ctype: str) -> list: + m = plan.model + lines = [] for w in plan.weights: - lines.append(f"extern {ctype} {_c_weight_symbol(m, w.symbol)}[{_weight_size(w.shape)}];") + sym = _c_weight_symbol(m, w.symbol) + values = ", ".join(_format_embedded_value(v, plan.dtype) for v in w.values) + lines.append(f"ROSENNA_CONST {ctype} {sym}[{_weight_size(w.shape)}] = {{ {values} }};") if plan.weights: lines.append("") - lines += _emit_infer(plan, ctype) - lines += [ - f"int {m}_init(const char *path);", - "", - "#endif", - ] - return "\n".join(lines) + "\n" + return lines def _emit_source_head(plan: Plan, ctype: str) -> list: @@ -268,7 +392,8 @@ def _emit_infer(plan: Plan, ctype: str) -> list: scratch = sorted((s for s in plan.buffers if s not in ("x", "y")), key=lambda s: int(s[1:])) - lines = [f"static inline void {m}_infer(const {ctype} *restrict x, {ctype} *restrict y) {{"] + lines = [f"ROSENNA_DEVICE_FN static inline void {m}_infer(" + f"const {ctype} *ROSENNA_RESTRICT x, {ctype} *ROSENNA_RESTRICT y) {{"] for sym in scratch: lines.append(f" {ctype} {sym}[{plan.buffers[sym]}];") @@ -277,8 +402,8 @@ def _emit_infer(plan: Plan, ctype: str) -> list: dst, src = plan.assignment[op.out], plan.assignment[op.inp] if op.kind == "gemm": idx_expr = _weight_index_c(weight_by_symbol, op) - weight_sym = _c_weight_symbol(m, op.weight) - bias_init = f"{_c_weight_symbol(m, op.bias)}[i]" if op.bias else _ZERO[plan.dtype] + weight_sym = _weight_ref(plan, m, op.weight) + bias_init = f"{_weight_ref(plan, m, op.bias)}[i]" if op.bias else _ZERO[plan.dtype] lines.append(f" for (int i = 0; i < {op.n_out}; ++i) {{") lines.append(f" {ctype} acc = {bias_init};") lines.append( diff --git a/python/rosenna/plan.py b/python/rosenna/plan.py index da909f4..9fe915d 100644 --- a/python/rosenna/plan.py +++ b/python/rosenna/plan.py @@ -11,8 +11,28 @@ _ACTIVATIONS = {"Relu": "relu", "Tanh": "tanh", "Sigmoid": "sigmoid"} _ITEMSIZE = {"f32": 4, "f64": 8} +_NUMPY_DTYPE = {"f32": np.float32, "f64": np.float64} _IDENTIFIER = re.compile(r"[A-Za-z_][A-Za-z0-9_]*\Z") +# Measured on this machine (gcc-15 -O2 -c on an empty translation unit that +# only #includes the generated header; see tests/measure_embed_threshold.py +# and task-2-report.md for the full table, including an extended sweep past +# 1e6 that locates where the real 5-second crossover falls): +# +# params compile time (s) +# 1056 0.05 +# 10100 0.06 +# 100172 0.12 +# 300852 0.25 +# 1001000 0.80 +# +# Every one of the five measured sizes compiles in under a second, so the +# largest of them -- already a round number -- is the threshold: a model +# under 1,000,000 parameters embeds by default. (The crossover past 5s does +# not occur until several million parameters; see the report for that +# supporting data point.) +EMBED_THRESHOLD = 1_000_000 + def validate_model_name(name: str) -> None: """Reject a model name that cannot be interpolated into a Fortran/C identifier. @@ -37,6 +57,13 @@ class WeightSpec: shape: tuple offset: int nbytes: int + # Populated only when the owning Plan embeds its weights (Plan.embed): + # the flattened values, at the plan's own dtype, that emit_c prints as a + # ROSENNA_CONST array literal. A file-loaded plan leaves this None; the + # values live in the .rwt file instead, and this field enters the hash + # only for an embedded plan (embed is itself part of the hash, so the + # two forms of the same model are already distinct artifacts). + values: tuple | None = None @dataclass(frozen=True) @@ -65,6 +92,8 @@ class Plan: buffers: dict assignment: dict weights: tuple + embed: bool + n_params: int def to_json(self) -> str: return json.dumps(asdict(self), sort_keys=True, separators=(",", ":")) @@ -77,7 +106,14 @@ def _length(t: Tensor) -> int: return int(np.prod(t.shape)) if t.shape else 1 -def build_plan(graph: Graph, dtype: str | None = None) -> Plan: +def _weight_elems(shape: tuple) -> int: + n = 1 + for d in shape: + n *= d + return n + + +def build_plan(graph: Graph, dtype: str | None = None, embed: bool | None = None) -> Plan: validate(graph) if len(graph.inputs) != 1 or len(graph.outputs) != 1: raise UnsupportedModel( @@ -115,7 +151,21 @@ def build_plan(graph: Graph, dtype: str | None = None) -> Plan: flat_in = Tensor(in_t.name, (_length(in_t),), dtype) flat_out = Tensor(out_t.name, (_length(out_t),), dtype) buffers, assignment = _assign_buffers(graph, ops, flat_in, flat_out) - return Plan(graph.name, dtype, flat_in, flat_out, tuple(ops), buffers, assignment, tuple(weights)) + + n_params = sum(_weight_elems(w.shape) for w in weights) + if embed is None: + embed = n_params < EMBED_THRESHOLD + if embed: + np_dtype = _NUMPY_DTYPE[dtype] + weights = [ + WeightSpec(w.name, w.symbol, w.shape, w.offset, w.nbytes, + values=tuple(np.asarray(graph.initializers[w.name], dtype=np_dtype) + .ravel(order="C").tolist())) + for w in weights + ] + + return Plan(graph.name, dtype, flat_in, flat_out, tuple(ops), buffers, assignment, + tuple(weights), embed, n_params) def _assign_buffers(graph: Graph, ops, flat_in: Tensor, flat_out: Tensor): diff --git a/python/rosenna/verify.py b/python/rosenna/verify.py index a27723f..8a5cac3 100644 --- a/python/rosenna/verify.py +++ b/python/rosenna/verify.py @@ -31,7 +31,8 @@ class VerifyResult: ok: bool -def verify_model(model_path, lang: str, dtype: str | None, cases: int, workdir) -> list: +def verify_model(model_path, lang: str, dtype: str | None, cases: int, workdir, + embed: bool | None = None) -> list: """Generate, compile and run `lang` backend(s) for `model_path`, and compare to onnxruntime. Draws `cases` random inputs from a fixed seed and compares every backend's output @@ -56,7 +57,7 @@ def verify_model(model_path, lang: str, dtype: str | None, cases: int, workdir) """ workdir = Path(workdir) graph = load_graph(model_path) - plan = build_plan(graph, dtype=dtype) + plan = build_plan(graph, dtype=dtype, embed=embed) validate_model_name(plan.model) # Model's own dtype, read before --precision is applied: this is what onnxruntime @@ -76,7 +77,11 @@ def verify_model(model_path, lang: str, dtype: str | None, cases: int, workdir) for backend in backends: backend_dir = workdir / backend backend_dir.mkdir(parents=True, exist_ok=True) - write_weights(plan, graph, backend_dir / f"{plan.model}.rwt") + # Fortran (unchanged by this task) always loads weights from a file. + # The C backend only needs one when the plan is not embedding its + # weights as ROSENNA_CONST arrays in the header. + if backend == "fortran" or not plan.embed: + write_weights(plan, graph, backend_dir / f"{plan.model}.rwt") got = _run_backend(backend, plan, backend_dir, inputs) abs_err = np.abs(got - expected) denom = np.maximum(np.abs(expected), np.finfo(np.float64).tiny) @@ -134,16 +139,20 @@ def _fortran_driver(name: str, n_in: int, n_out: int, dtype: str) -> str: """ -def _c_driver(name: str, n_in: int, n_out: int, dtype: str) -> str: +def _c_driver(name: str, n_in: int, n_out: int, dtype: str, embed: bool) -> str: c_type = "double" if dtype == "f64" else "float" fmt = "%lf" if dtype == "f64" else "%f" + # An embedded plan has no `_init`: every weight is already a ROSENNA_CONST + # array in the header, resident from program load. + init = "" if embed else ( + f'int status = {name}_init("{name}.rwt");\n' + f' if (status != 0) {{ printf("init status %d\\n", status); return 1; }}\n ') return f""" #include #include "{name}.h" int main(void) {{ {c_type} x[{n_in}], y[{n_out}]; - int ncases, status = {name}_init("{name}.rwt"); - if (status != 0) {{ printf("init status %d\\n", status); return 1; }} + {init}int ncases; if (scanf("%d", &ncases) != 1) return 1; for (int c = 0; c < ncases; ++c) {{ for (int i = 0; i < {n_in}; ++i) if (scanf("{fmt}", &x[i]) != 1) return 1; @@ -197,7 +206,7 @@ def _run_backend(backend: str, plan, workdir: Path, inputs): source, header = emit_c(plan) (workdir / f"{name}.c").write_text(source) (workdir / f"{name}.h").write_text(header) - (workdir / "verify_main.c").write_text(_c_driver(name, n_in, n_out, dtype)) + (workdir / "verify_main.c").write_text(_c_driver(name, n_in, n_out, dtype, plan.embed)) # Compile the generated source to an object, archive it, and link the # driver against the archive -- the library form -- rather than # compiling both sources together, so `verify` exercises the same diff --git a/python/tests/measure_embed_threshold.py b/python/tests/measure_embed_threshold.py new file mode 100644 index 0000000..e46f166 --- /dev/null +++ b/python/tests/measure_embed_threshold.py @@ -0,0 +1,88 @@ +"""Measure gcc -O2 compile time for a header embedding N weight scalars. + +Not a test: run manually -- + + cd python && python3 tests/measure_embed_threshold.py + +-- to reproduce the table cited in the comment next to EMBED_THRESHOLD in +rosenna/plan.py (and pasted into the Task 2 report). Builds a single-Gemm +ONNX model sized so its embedded weight count is close to each of +1e3/1e4/1e5/3e5/1e6 parameters, emits it with `embed=True`, and times +`gcc -O2 -c` compiling an otherwise-empty translation unit that only +`#include`s the generated header (the header, not the source, holds the +embedded ROSENNA_CONST arrays -- the source is nearly empty for an +embedded plan). +""" +import shutil +import subprocess +import sys +import tempfile +import time +from pathlib import Path + +import numpy as np +import onnx +from onnx import helper, numpy_helper, TensorProto + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from rosenna.emit_c import emit_c +from rosenna.frontend import load_graph +from rosenna.plan import build_plan + +SIZES = [1_000, 10_000, 100_000, 300_000, 1_000_000] + + +def _cc() -> str: + for cand in ("gcc-15", "gcc-14", "gcc-13", "gcc"): + path = shutil.which(cand) + if path: + probe = subprocess.run([cand, "-fopenmp", "-x", "c", "-", "-o", "/dev/null"], + input="int main(void){return 0;}", capture_output=True, text=True) + if probe.returncode == 0: + return cand + raise SystemExit("no C compiler with -fopenmp found") + + +def _make_model(path: Path, target_params: int) -> None: + """A single Gemm layer whose weight+bias element count is close to target_params.""" + n = max(2, round((target_params) ** 0.5)) + rng = np.random.default_rng(0) + w = numpy_helper.from_array(rng.uniform(-1, 1, (n, n)).astype(np.float32), "w") + b = numpy_helper.from_array(rng.uniform(-1, 1, (n,)).astype(np.float32), "b") + node = helper.make_node("Gemm", ["x", "w", "b"], ["y"], name="g0") + x = helper.make_tensor_value_info("x", TensorProto.FLOAT, [1, n]) + y = helper.make_tensor_value_info("y", TensorProto.FLOAT, [1, n]) + g = helper.make_graph([node], "measuretmp", [x], [y], initializer=[w, b]) + m = helper.make_model(g, opset_imports=[helper.make_opsetid("", 13)]) + onnx.save(m, path) + + +def main() -> None: + cc = _cc() + rows = [] + with tempfile.TemporaryDirectory() as td: + td = Path(td) + for target in SIZES: + onnx_path = td / f"m{target}.onnx" + _make_model(onnx_path, target) + graph = load_graph(onnx_path) + plan = build_plan(graph, dtype="f64", embed=True) + _, header = emit_c(plan) + (td / f"m{target}.h").write_text(header) + tu = td / f"m{target}.c" + tu.write_text(f'#include "m{target}.h"\n') + start = time.perf_counter() + subprocess.run([cc, "-O2", "-c", str(tu), "-o", str(td / f"m{target}.o")], + check=True, capture_output=True, text=True) + elapsed = time.perf_counter() - start + rows.append((plan.n_params, elapsed)) + + print(f"compiler: {cc}") + print(f"{'params':>10} {'compile s':>10}") + for n_params, elapsed in rows: + print(f"{n_params:>10} {elapsed:>10.3f}") + + +if __name__ == "__main__": + main() diff --git a/python/tests/test_device_c.py b/python/tests/test_device_c.py new file mode 100644 index 0000000..653dbc3 --- /dev/null +++ b/python/tests/test_device_c.py @@ -0,0 +1,116 @@ +import os +import shutil +import subprocess +import numpy as np +import onnxruntime as ort +import pytest +from rosenna.frontend import load_graph +from rosenna.plan import build_plan +from rosenna.weights import write_weights +from rosenna.emit_c import emit_c +from tests.test_emit_fortran import _live_reference + + +def _omp_cc(): + for cand in ("gcc-15", "gcc-14", "gcc-13", "gcc"): + path = shutil.which(cand) + if not path: + continue + probe = subprocess.run([cand, "-fopenmp", "-x", "c", "-", "-o", os.devnull], + input="int main(void){return 0;}", capture_output=True, text=True) + if probe.returncode == 0: + return cand + pytest.skip("no C compiler with -fopenmp found") + + +def _write(tmp_path, name, plan, graph): + source, header = emit_c(plan) + (tmp_path / f"{name}.c").write_text(source) + (tmp_path / f"{name}.h").write_text(header) + if not plan.embed: + write_weights(plan, graph, tmp_path / f"{name}.rwt") + + +HOST = """ +#include +#include "{name}.h" +int main(void) {{ + int n; {init} + if (scanf("%d", &n) != 1) return 1; + double x[64 * {n_in}], y[64 * {n_out}]; + for (int c = 0; c < n * {n_in}; ++c) if (scanf("%lf", &x[c]) != 1) return 1; +#ifdef _OPENMP + #pragma omp target teams loop map(to: x[0:n*{n_in}]) map(from: y[0:n*{n_out}]) +#endif + for (int p = 0; p < n; ++p) {name}_infer(x + p * {n_in}, y + p * {n_out}); /* the host's own region calls the header inline */ + for (int p = 0; p < n; ++p) {{ for (int i = 0; i < {n_out}; ++i) printf("%.17e ", y[p * {n_out} + i]); printf("\\n"); }} + return 0; +}} +""" + + +def _build_and_run(tmp_path, name, plan, graph, cc, flags, inputs, env=None): + _write(tmp_path, name, plan, graph) + init = "" if plan.embed else f'if ({name}_init("{name}.rwt")) return 2;' + (tmp_path / "host.c").write_text(HOST.format(name=name, n_in=plan.input.shape[0], n_out=plan.output.shape[0], init=init)) + objs = ["host.c"] + if not plan.embed: + r = subprocess.run([cc, *flags, "-c", f"{name}.c"], cwd=tmp_path, capture_output=True, text=True) + assert r.returncode == 0 and r.stderr == "", r.stderr + subprocess.run(["ar", "rcs", f"lib{name}.a", f"{name}.o"], cwd=tmp_path, check=True) + objs.append(f"lib{name}.a") + r = subprocess.run([cc, *flags, *objs, "-lm", "-o", "host"], cwd=tmp_path, capture_output=True, text=True) + assert r.returncode == 0 and r.stderr == "", r.stderr + stdin = f"{len(inputs)}\n" + " ".join(repr(float(v)) for v in inputs.ravel()) + return subprocess.run(["./host"], cwd=tmp_path, input=stdin, capture_output=True, text=True, env=env) + + +@pytest.mark.parametrize("name", ["gemm_small", "gemm_big", "gemm_nobias", "droplet", "batchnet"]) +@pytest.mark.parametrize("embed", [True, False]) +def test_host_region_calls_header_inline_and_matches(tmp_path, golden_model, name, embed): + graph = load_graph(golden_model(name)) + plan = build_plan(graph, dtype="f64", embed=embed) + session = ort.InferenceSession(golden_model(name)) + inputs, expected = _live_reference(session, session.get_inputs()[0].shape, np.float64, seed=5, batch=8) + r = _build_and_run(tmp_path, name, plan, graph, _omp_cc(), ["-O2", "-Wall", "-Wextra", "-std=c11", "-fopenmp"], inputs) + assert r.returncode == 0, r.stderr + got = np.array([[float(v) for v in line.split()] for line in r.stdout.strip().splitlines()]) + np.testing.assert_allclose(got, expected, rtol=1e-5, atol=1e-6) + + +def test_target_regions_are_real(tmp_path, golden_model): + # A host-only libgomp refuses a target region under OMP_TARGET_OFFLOAD=MANDATORY. If the pragmas + # were missing or ignored the program would succeed; this is the cheapest evidence without a GPU. + name = "gemm_small" + graph = load_graph(golden_model(name)); plan = build_plan(graph, dtype="f64") + inputs = np.full((1, plan.input.shape[0]), 0.5) + r = _build_and_run(tmp_path, name, plan, graph, _omp_cc(), ["-O2", "-std=c11", "-fopenmp"], inputs, + env={**os.environ, "OMP_TARGET_OFFLOAD": "MANDATORY"}) + assert r.returncode != 0 and "MANDATORY" in r.stderr + + +def test_plain_compiler_without_openmp_still_matches(tmp_path, golden_model): + # Every decoration is inert under a compiler with no -fopenmp and no CUDA; the numbers must not change. + name = "gemm_small" + graph = load_graph(golden_model(name)); plan = build_plan(graph, dtype="f64") + cc = shutil.which("clang") or shutil.which("cc") or _omp_cc() + session = ort.InferenceSession(golden_model(name)) + inputs, expected = _live_reference(session, session.get_inputs()[0].shape, np.float64, seed=6, batch=4) + r = _build_and_run(tmp_path, name, plan, graph, cc, ["-O2", "-Wall", "-Wextra", "-std=c11"], inputs) + assert r.returncode == 0, r.stderr + got = np.array([[float(v) for v in line.split()] for line in r.stdout.strip().splitlines()]) + np.testing.assert_allclose(got, expected, rtol=1e-5, atol=1e-6) + + +def test_header_carries_exactly_the_two_macros(golden_model): + from rosenna.emit_c import emit_c + from rosenna.frontend import load_graph as lg + # A CUDA/HIP host must see __host__ __device__ and __constant__; nothing else in the header may mention CUDA. + # Controller ruling P2: take the golden path through the golden_model fixture (tests.conftest has no + # standalone golden_path function). Controller ruling P3: a third macro, ROSENNA_RESTRICT, sits next to + # the two above so `restrict` -- not a keyword once this header reaches a C++ (nvcc) translation unit -- + # never appears bare in a signature; assert all three macro names are present. + plan = build_plan(lg(golden_model("gemm_small")), dtype="f64") + _, header = emit_c(plan) + assert header.count("__CUDACC__") == 1 and "__host__ __device__" in header and "__constant__" in header + assert "ROSENNA_DEVICE_FN" in header and "ROSENNA_CONST" in header and "ROSENNA_RESTRICT" in header diff --git a/python/tests/test_embed.py b/python/tests/test_embed.py new file mode 100644 index 0000000..445e3cc --- /dev/null +++ b/python/tests/test_embed.py @@ -0,0 +1,39 @@ +from rosenna.frontend import load_graph +from rosenna.plan import build_plan, EMBED_THRESHOLD +from rosenna.emit_c import emit_c + + +def test_small_model_embeds_by_default(golden_model): + plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64") + assert plan.embed is True and plan.n_params < EMBED_THRESHOLD + source, header = emit_c(plan) + assert "ROSENNA_CONST double w0[4] = {" not in header # symbols carry the model prefix (ruling R1) + assert "ROSENNA_CONST double gemm_small_w0[4] = {" in header + assert "_init(" not in header and "fopen" not in source + + +def test_no_embed_flag_keeps_the_file_path(golden_model): + plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64", embed=False) + source, header = emit_c(plan) + assert "_init(" in header and "fopen" in source + + +def test_embed_changes_the_hash(golden_model): + g = load_graph(golden_model("gemm_small")) + assert build_plan(g, dtype="f64", embed=True).hash() != build_plan(g, dtype="f64", embed=False).hash() + + +def test_embedded_and_file_loaded_agree_exactly(tmp_path, golden_model): + # Embedding prints every weight at full precision; the two builds must produce identical bits. + from tests.test_device_c import _build_and_run, _omp_cc + import numpy as np + name = "gemm_big"; graph = load_graph(golden_model(name)) + inputs = np.random.default_rng(11).uniform(-2, 2, (8, build_plan(graph).input.shape[0])) + outs = [] + for embed in (True, False): + d = tmp_path / ("e" if embed else "f"); d.mkdir() + r = _build_and_run(d, name, build_plan(graph, dtype="f64", embed=embed), graph, _omp_cc(), + ["-O2", "-std=c11", "-fopenmp"], inputs) + assert r.returncode == 0, r.stderr + outs.append(r.stdout) + assert outs[0] == outs[1] diff --git a/python/tests/test_emit_c.py b/python/tests/test_emit_c.py index 06f7026..fd10658 100644 --- a/python/tests/test_emit_c.py +++ b/python/tests/test_emit_c.py @@ -12,8 +12,11 @@ def _build_and_run(tmp_path, onnx_path, name, inputs, dtype="f64"): + # embed=False: this helper's driver always calls `_init` against a + # written .rwt file, the file-loaded contract. The dedicated embed=True/ + # False matrix lives in tests/test_device_c.py and tests/test_embed.py. graph = load_graph(onnx_path) - plan = build_plan(graph, dtype=dtype) + plan = build_plan(graph, dtype=dtype, embed=False) source, header = emit_c(plan) (tmp_path / f"{name}.c").write_text(source) (tmp_path / f"{name}.h").write_text(header) @@ -86,7 +89,8 @@ def test_infer_is_pure_and_has_literal_bounds(golden_model): # `infer` is now defined only in the header (a static inline callable # from inside the host's own offload region); the source never defines # it. - assert "static inline void gemm_small_infer(const double *restrict x, double *restrict y) {" in header + assert ("ROSENNA_DEVICE_FN static inline void gemm_small_infer(" + "const double *ROSENNA_RESTRICT x, double *ROSENNA_RESTRICT y) {") in header # The scratch buffers come from plan.buffers now (ruling R13), not from a # second allocator private to this emitter: gemm_small's t0 is reused by # both gemms, so the plan sizes it at the larger of the two (3), and the @@ -99,10 +103,12 @@ def test_infer_is_pure_and_has_literal_bounds(golden_model): def test_init_rejects_a_foreign_weights_file(tmp_path, golden_model): + # embed=False: this test is specifically about `_init`, which an + # embedded plan's header does not declare. graph = load_graph(golden_model("gemm_small")) - plan = build_plan(graph, dtype="f64") + plan = build_plan(graph, dtype="f64", embed=False) other_graph = load_graph(golden_model("gemm_big")) - other_plan = build_plan(other_graph, dtype="f64") + other_plan = build_plan(other_graph, dtype="f64", embed=False) source, header = emit_c(plan) (tmp_path / "gemm_small.c").write_text(source) (tmp_path / "gemm_small.h").write_text(header) diff --git a/python/tests/test_library_form.py b/python/tests/test_library_form.py index 710281f..11dd6a7 100644 --- a/python/tests/test_library_form.py +++ b/python/tests/test_library_form.py @@ -19,7 +19,9 @@ def _cc(): def test_header_defines_inline_infer_and_source_does_not(golden_model): - plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64") + # embed=False: this test is specifically about the file-loaded contract + # (extern declaration in the header, definition in the source). + plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64", embed=False) source, header = emit_c(plan) assert "static inline void gemm_small_infer(" in header # Structural check (controller ruling P1): infer must be defined only in @@ -37,7 +39,7 @@ def test_header_defines_inline_infer_and_source_does_not(golden_model): def test_library_and_header_inline_agree(tmp_path, golden_model): name = "gemm_small" graph = load_graph(golden_model(name)) - plan = build_plan(graph, dtype="f64") + plan = build_plan(graph, dtype="f64", embed=False) source, header = emit_c(plan) (tmp_path / f"{name}.c").write_text(source) (tmp_path / f"{name}.h").write_text(header) @@ -92,7 +94,7 @@ def test_two_models_link_into_one_host(tmp_path, golden_model): obj_args = [] for name in names: graph = load_graph(golden_model(name)) - plan = build_plan(graph, dtype="f64") + plan = build_plan(graph, dtype="f64", embed=False) plans[name] = plan source, header = emit_c(plan) (tmp_path / f"{name}.c").write_text(source) @@ -161,7 +163,7 @@ def test_two_models_link_into_one_host(tmp_path, golden_model): def test_recipe_builds_the_library(tmp_path, golden_model): name = "gemm_small" graph = load_graph(golden_model(name)) - plan = build_plan(graph, dtype="f64") + plan = build_plan(graph, dtype="f64", embed=False) source, header = emit_c(plan) (tmp_path / f"{name}.c").write_text(source) (tmp_path / f"{name}.h").write_text(header) diff --git a/python/tests/test_regressions.py b/python/tests/test_regressions.py index 53dba4f..5fbb127 100644 --- a/python/tests/test_regressions.py +++ b/python/tests/test_regressions.py @@ -54,7 +54,11 @@ def _init_status(tmp_path, lang, onnx_path, name, rwt_bytes): """Emit `name`'s init for `lang`, hand it `rwt_bytes`, return (exit code, stdout).""" work = tmp_path / lang work.mkdir(exist_ok=True) - plan = build_plan(load_graph(onnx_path), dtype="f64") + # embed=False: this helper hands `_init` crafted/corrupted .rwt + # bytes and checks the status it returns, so it always needs the + # file-loaded contract (and its hash must match the caller's plan, which + # also builds with embed=False -- see the two call sites below). + plan = build_plan(load_graph(onnx_path), dtype="f64", embed=False) (work / f"{name}.rwt").write_bytes(rwt_bytes) if lang == "fortran": (work / f"{name}_model.f90").write_text(emit_fortran(plan)) @@ -169,7 +173,9 @@ def _long_name_model(tmp_path): def test_long_initializer_name_round_trips(tmp_path): assert len(_LONG_NAME) > 128 path = _long_name_model(tmp_path) - plan = build_plan(load_graph(path), dtype="f64") + # embed=False: this test is about the name buffer inside `init`'s table- + # of-contents reader, which an embedded plan's source does not emit. + plan = build_plan(load_graph(path), dtype="f64", embed=False) fsrc = emit_fortran(plan) csrc, _ = emit_c(plan) @@ -192,7 +198,10 @@ def test_oversized_name_length_in_the_file_is_rejected(tmp_path, golden_model, l # byte 60; overwrite the first tensor's name length with 4000. onnx_path = golden_model("gemm_small") graph = load_graph(onnx_path) - plan = build_plan(graph, dtype="f64") + # embed=False to match _init_status's own plan -- both build the same + # model the same way, or their plan hashes (and thus this file's + # embedded expected_hash) would disagree. + plan = build_plan(graph, dtype="f64", embed=False) good = tmp_path / "good.rwt" write_weights(plan, graph, good) blob = bytearray(good.read_bytes()) @@ -245,7 +254,8 @@ def test_relu_propagates_nan_in_both_backends(tmp_path): def test_truncated_weights_file_returns_a_status(tmp_path, golden_model, lang): onnx_path = golden_model("gemm_small") graph = load_graph(onnx_path) - plan = build_plan(graph, dtype="f64") + # embed=False: see the comment in the sibling test above. + plan = build_plan(graph, dtype="f64", embed=False) good = tmp_path / "good.rwt" write_weights(plan, graph, good) rc, out = _init_status(tmp_path, lang, onnx_path, "gemm_small", good.read_bytes()[:-12]) From c26df4f3133c974c102c2fafb2fc8d846a74e724 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Sun, 13 Sep 2026 21:33:17 -0500 Subject: [PATCH 04/84] fix: explain the device macros and _dev pointers in the emitted header --- python/rosenna/emit_c.py | 41 ++++++++++++---- python/tests/test_cli.py | 21 ++++++++ python/tests/test_library_form.py | 82 +++++++++++++++++++++++++++++++ 3 files changed, 134 insertions(+), 10 deletions(-) diff --git a/python/rosenna/emit_c.py b/python/rosenna/emit_c.py index 94de7ea..d8cdd6a 100644 --- a/python/rosenna/emit_c.py +++ b/python/rosenna/emit_c.py @@ -169,13 +169,15 @@ def _emit_header(plan: Plan, ctype: str) -> str: "", "#include ", "", - # Every device decoration in this header goes through exactly these - # three macros (controller rulings: brief + P3). Under nvcc/hipcc, - # `infer` is callable from device code and embedded weights live in - # __constant__ memory; under any other compiler both are inert. C has - # no `restrict` keyword once this header is pulled into a C++ (or - # CUDA/HIP, which is always C++) translation unit, so ROSENNA_RESTRICT - # picks the compiler-correct spelling instead of `infer` hardcoding one. + "/* Under nvcc/hipcc: ROSENNA_DEVICE_FN = __host__ __device__ (infer is", + " callable from device code), ROSENNA_CONST = __constant__ (embedded", + " weights live in device memory), ROSENNA_RESTRICT = __restrict__.", + " Otherwise (plain C, or a host OpenMP/OpenACC build): ROSENNA_DEVICE_FN", + " is empty, ROSENNA_CONST = static const, and ROSENNA_RESTRICT is", + " __restrict__ in C++ or restrict in C. infer is separately wrapped in a", + " guarded OpenMP declare-target region with a guarded OpenACC routine-seq", + " pragma below; both are no-ops unless that compiler defines", + " _OPENMP/_OPENACC. */", "#if defined(__CUDACC__) || defined(__HIPCC__)", "#define ROSENNA_DEVICE_FN __host__ __device__", "#define ROSENNA_CONST __constant__", @@ -210,20 +212,39 @@ def _emit_weight_declarations(plan: Plan, ctype: str) -> list: nothing fills them in until Task 3's CUDA/HIP init exists. Under a plain or OpenMP host build the __CUDACC__/__HIPCC__ guard is false, so this branch is never even compiled; no build in this task can reach it. - Compiles and runs on the host; device path unvalidated. + Compiles and runs on the host; device path unvalidated. The explanation + is emitted into the header itself (not just here), since a solver author + or the Task 3 implementer reads the generated .h, not this module. """ m = plan.model if not plan.weights: return [] - lines = ["#if defined(__CUDACC__) || defined(__HIPCC__)"] + lines = [ + f"/* Device copies of the weights below: defined by the library and", + f" filled in by {m}_init on a CUDA/HIP build (Task 3); not referenced", + f" under a plain or OpenMP host build. */", + "#if defined(__CUDACC__) || defined(__HIPCC__)", + ] for w in plan.weights: sym = _c_weight_symbol(m, w.symbol) lines.append(f"extern {ctype} *{sym}_dev;") - lines.append(f"#define {_c_weight_ref_macro(m, w.symbol)} {sym}_dev") lines.append("#else") for w in plan.weights: sym = _c_weight_symbol(m, w.symbol) lines.append(f"extern {ctype} {sym}[{_weight_size(w.shape)}];") + lines += ["#endif", ""] + + lines += [ + "/* Selects the host array or the device pointer above, so infer's", + " body below is emitted once and reads whichever this build has. */", + "#if defined(__CUDACC__) || defined(__HIPCC__)", + ] + for w in plan.weights: + sym = _c_weight_symbol(m, w.symbol) + lines.append(f"#define {_c_weight_ref_macro(m, w.symbol)} {sym}_dev") + lines.append("#else") + for w in plan.weights: + sym = _c_weight_symbol(m, w.symbol) lines.append(f"#define {_c_weight_ref_macro(m, w.symbol)} {sym}") lines += ["#endif", ""] return lines diff --git a/python/tests/test_cli.py b/python/tests/test_cli.py index 64ba357..f6a92fe 100644 --- a/python/tests/test_cli.py +++ b/python/tests/test_cli.py @@ -13,6 +13,27 @@ def test_generate_writes_all_artifacts(tmp_path, capsys, golden_model): assert "gemm_small.rwt" in out +def test_generate_writes_rwt_with_fortran_but_not_c_only(tmp_path, capsys, golden_model): + # gemm_small auto-embeds (well under EMBED_THRESHOLD). --lang both still + # writes the .rwt because Fortran generation is unchanged by this task + # and always loads weights from a file; --lang c alone has nothing left + # that needs one, since the weights are ROSENNA_CONST arrays baked into + # the header. NOTE: Task 4 (Fortran embedding) is expected to flip the + # first assertion once Fortran also embeds by default -- revisit this + # test then rather than assuming it still holds. + onnx_path = golden_model("gemm_small") + + rc = main(["generate", str(onnx_path), "--lang", "both", "--out", str(tmp_path / "both")]) + assert rc == 0 + assert (tmp_path / "both" / "gemm_small.rwt").exists() + + rc = main(["generate", str(onnx_path), "--lang", "c", "--out", str(tmp_path / "c")]) + assert rc == 0 + assert not (tmp_path / "c" / "gemm_small.rwt").exists() + out = capsys.readouterr().out + assert "embedded weights" in out + + def test_verify_passes_on_a_dense_model(capsys, golden_model): onnx_path = golden_model("gemm_small") rc = main(["verify", str(onnx_path), "--cases", "4"]) diff --git a/python/tests/test_library_form.py b/python/tests/test_library_form.py index 11dd6a7..a94db5d 100644 --- a/python/tests/test_library_form.py +++ b/python/tests/test_library_form.py @@ -160,6 +160,88 @@ def test_two_models_link_into_one_host(tmp_path, golden_model): np.testing.assert_allclose(got, expected, rtol=1e-5, atol=1e-6) +def test_two_embedded_models_link_into_one_host(tmp_path, golden_model): + """Two different models' EMBEDDED weights must link into one host binary. + + Mirrors test_two_models_link_into_one_host above, but for embed=True + (the default for these small models). ROSENNA_CONST resolves to `static + const` on the host, which gives each array internal linkage -- so an + unprefixed `w0` in two headers would not raise a linker collision the + way the file-loaded case's external `w0` did -- but both headers still + land in the same translation unit here (host.c #includes both), and an + unprefixed `w0` would be a duplicate *definition* inside that one TU + regardless of linkage. Ruling R1 already prefixes embedded weight + symbols with the model name (see emit_c._emit_embedded_weights); this + test proves that rather than assuming it. + """ + names = ["gemm_small", "gemm_nobias"] + plans, sessions = {}, {} + cc = _cc() + for name in names: + graph = load_graph(golden_model(name)) + plan = build_plan(graph, dtype="f64") + assert plan.embed is True, f"{name}: expected to auto-embed for this test to be meaningful" + plans[name] = plan + source, header = emit_c(plan) + (tmp_path / f"{name}.c").write_text(source) + (tmp_path / f"{name}.h").write_text(header) + sessions[name] = ort.InferenceSession(golden_model(name)) + + references = {} + for name in names: + session = sessions[name] + shape = session.get_inputs()[0].shape + inputs, expected = _live_reference(session, shape, np.float64, seed=7, batch=8) + if inputs is None: + pytest.skip(f"{name}: onnxruntime reference is all-zero across 10 resampled " + f"batches; its golden-file weights produced a dead model") + references[name] = (inputs, expected) + + host_lines = ["#include "] + host_lines += [f'#include "{name}.h"' for name in names] + host_lines.append("int main(void) {") + for name in names: + plan = plans[name] + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + host_lines.append(f" double {name}_x[{n_in}], {name}_y[{n_out}];") + for name in names: + plan = plans[name] + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + host_lines.append(" { int n; if (scanf(\"%d\", &n) != 1) return 1;") + host_lines.append(" for (int c = 0; c < n; ++c) {") + host_lines.append( + f" for (int i = 0; i < {n_in}; ++i) " + f"if (scanf(\"%lf\", &{name}_x[i]) != 1) return 1;") + host_lines.append(f" {name}_infer({name}_x, {name}_y);") + host_lines.append( + f" for (int i = 0; i < {n_out}; ++i) printf(\"%.17e \", {name}_y[i]);") + host_lines.append(' printf("\\n");') + host_lines.append(" } }") + host_lines.append(" return 0;") + host_lines.append("}") + (tmp_path / "host.c").write_text("\n".join(host_lines) + "\n") + + # No lib{name}.a to link: an embedded plan's .c is nearly empty and + # infer lives entirely in the header, so the host TU alone suffices. + subprocess.run([cc, "-O2", "-Wall", "-Wextra", "-std=c11", "host.c", "-lm", "-o", "host"], + cwd=tmp_path, check=True, capture_output=True, text=True) + + stdin_parts = [] + for name in names: + inputs, _ = references[name] + stdin_parts.append(str(len(inputs))) + stdin_parts.append("\n".join(" ".join(repr(float(v)) for v in row) for row in inputs)) + stdin = "\n".join(stdin_parts) + "\n" + out = subprocess.run(["./host"], cwd=tmp_path, input=stdin, capture_output=True, text=True, check=True).stdout + all_lines = out.strip().splitlines() + pos = 0 + for name in names: + inputs, expected = references[name] + got = np.array([[float(v) for v in line.split()] for line in all_lines[pos:pos + len(inputs)]]) + pos += len(inputs) + np.testing.assert_allclose(got, expected, rtol=1e-5, atol=1e-6) + + def test_recipe_builds_the_library(tmp_path, golden_model): name = "gemm_small" graph = load_graph(golden_model(name)) From 05d2631cffef7c5675a142b252dccf951436652c Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Sun, 13 Sep 2026 22:10:11 -0500 Subject: [PATCH 05/84] feat: native cuda/hip batched kernel with an openmp fallback; init copies weights to the device The batched entry point _infer_batch(n, x, y, stream) now exists in both forms of the C library. Under ROSENNA_BACKEND=cuda|hip it is a one-thread-per- point kernel in _kernel.cu calling the header-inline infer, built by nvcc or hipcc through the runtime macros in rosenna_rt.h; under the default omp backend it is a target teams loop over device pointers (is_device_ptr) in .c. init is the plan step: it loads the file, copies each weight to the device (status 10 on failure) and publishes the addresses to the kernel's translation unit; nothing in the loop path allocates, transfers or synchronizes (controller ruling R5). ROSENNA_CONST is __constant__ under 48 KB of embedded weights and __device__ const above (ruling R4). The generated Makefile selects the backend; generate writes the kernel and the runtime header; a compile-only nvcc job is added to CI. Device path unvalidated until that job and the GPU gate run. --- .github/workflows/CI.yml | 42 ++++ python/rosenna/abi.py | 3 +- python/rosenna/cli.py | 10 +- python/rosenna/emit_c.py | 440 ++++++++++++++++++++++++++++----- python/rosenna/emit_kernel.py | 79 ++++++ python/rosenna/rt_header.py | 45 ++++ python/tests/test_kernel.py | 447 ++++++++++++++++++++++++++++++++++ 7 files changed, 1000 insertions(+), 66 deletions(-) create mode 100644 python/rosenna/emit_kernel.py create mode 100644 python/rosenna/rt_header.py create mode 100644 python/tests/test_kernel.py diff --git a/.github/workflows/CI.yml b/.github/workflows/CI.yml index 9823565..87f7421 100644 --- a/.github/workflows/CI.yml +++ b/.github/workflows/CI.yml @@ -51,3 +51,45 @@ jobs: mkdir -p fLibrary/objFiles chmod +x test/run.sh cd test && ./run.sh + + # Compile-only check of the generated CUDA sources. There is no GPU here and + # no driver is installed: nvcc builds the kernel and the .c-as-C++ library for + # sm_80 and the test asserts the archive exists. Running it is the GPU gate's + # job. This job is the first time the generated CUDA code meets a compiler. + # ubuntu-22.04 on purpose: NVIDIA's ubuntu2404 apt repository starts at CUDA + # 12.5, and 12.4 is the toolkit version this job pins. + nvcc_compile: + runs-on: ubuntu-22.04 + + steps: + - name: Clone roseNNa + uses: actions/checkout@v4 + + - name: Install the CUDA 12.4 compiler and runtime headers (no driver) + run: | + set -euo pipefail + wget -q https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.1-1_all.deb + sudo dpkg -i cuda-keyring_1.1-1_all.deb + sudo apt-get update + sudo apt-get install -y --no-install-recommends cuda-nvcc-12-4 cuda-cudart-dev-12-4 + echo "/usr/local/cuda-12.4/bin" >> "$GITHUB_PATH" + + - name: Check nvcc + run: nvcc --version && gcc --version + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: '3.11' + + - name: Install dependencies + run: | + pip install -r requirements.txt + pip install -e python + + - name: CUDA backend compiles under nvcc (not skipped) + run: | + set -o pipefail + cd python && python3 -m pytest tests/test_kernel.py -v -rs -s -k cuda 2>&1 | tee cuda.log + ! grep -q "SKIPPED" cuda.log + grep -q "PASSED" cuda.log diff --git a/python/rosenna/abi.py b/python/rosenna/abi.py index ea61aac..6ac7b10 100644 --- a/python/rosenna/abi.py +++ b/python/rosenna/abi.py @@ -17,6 +17,7 @@ (7, "weights file holds a tensor this model does not declare"), (8, "a name or rank in the weights file exceeds this model's capacity"), (9, "a read failed: the weights file is truncated or inconsistent"), + (10, "device allocation or copy failed in init"), ] # Floors for the buffers `_init` declares to parse the table of @@ -39,7 +40,7 @@ def status_code_comment(prefix: str, model: str) -> list: next to the routine that returns it, in both languages. """ lines = [f"{prefix} Status codes returned by {model}_init:"] - lines += [f"{prefix} {code} {text}" for code, text in STATUS_CODES] + lines += [f"{prefix} {code:>2} {text}" for code, text in STATUS_CODES] return lines diff --git a/python/rosenna/cli.py b/python/rosenna/cli.py index 687f996..44e4c0e 100644 --- a/python/rosenna/cli.py +++ b/python/rosenna/cli.py @@ -8,6 +8,8 @@ from .emit_c import emit_c, emit_c_recipe from .emit_fortran import emit_fortran +from .emit_kernel import emit_kernel +from .rt_header import rt_header from .frontend import UnsupportedModel, load_graph from .plan import build_plan, validate_model_name from .verify import VerificationError, verify_model @@ -90,10 +92,16 @@ def _cmd_generate(args) -> int: c_path = outdir / f"{name}.c" h_path = outdir / f"{name}.h" mk_path = outdir / f"{name}.mk" + cu_path = outdir / f"{name}_kernel.cu" + rt_path = outdir / "rosenna_rt.h" c_path.write_text(source) h_path.write_text(header) mk_path.write_text(recipe) - written += [c_path, h_path, mk_path] + # The native batched kernel and its runtime map: built only when the + # recipe runs with ROSENNA_BACKEND=cuda|hip, inert otherwise. + cu_path.write_text(emit_kernel(plan)) + rt_path.write_text(rt_header()) + written += [c_path, h_path, mk_path, cu_path, rt_path] # An embedded plan has no weights file to write: every weight is already a # ROSENNA_CONST array baked into the header. Fortran generation (unchanged diff --git a/python/rosenna/emit_c.py b/python/rosenna/emit_c.py index d8cdd6a..f25a756 100644 --- a/python/rosenna/emit_c.py +++ b/python/rosenna/emit_c.py @@ -4,6 +4,24 @@ _CTYPE = {"f32": "float", "f64": "double"} _DTYPE_CODE = {"f32": 0, "f64": 1} +_ITEMSIZE = {"f32": 4, "f64": 8} +_CUDA_GUARD = "#if defined(__CUDACC__) || defined(__HIPCC__)" +_NOT_CUDA_GUARD = "#if !defined(__CUDACC__) && !defined(__HIPCC__)" +# Device-pass guard: nvcc defines __CUDA_ARCH__ and hipcc __HIP_DEVICE_COMPILE__ +# only while compiling for the device, so a header-inline function can read one +# storage in its host instantiation and another in its device instantiation. +_DEVICE_PASS_GUARD = "#if defined(__CUDA_ARCH__) || defined(__HIP_DEVICE_COMPILE__)" + +# Controller ruling R4. CUDA __constant__ memory is 64 KB per module, while +# a model embeds by default below EMBED_THRESHOLD (1M parameters, up to 8 MB +# of f64), so an embedded model whose weights exceed the constant budget must +# be placed in ordinary device memory or nvcc rejects the header. The cut is +# 48 KB of weight bytes, leaving the remaining 16 KB for anything else the +# translation unit puts in constant memory; the decision is per model, made +# once at generation time, and the header records which it took. +CONSTANT_MEMORY_LIMIT = 48 * 1024 +# The one-thread-per-point kernel's block size (emit_kernel). +KERNEL_TILE = 128 # Embedding must be lossless: %.17g round-trips any f64, %.9g any f32 # (Steele & White / Ryu-style shortest-exact-decimal bounds). _EMBED_FMT = {"f32": "%.9g", "f64": "%.17g"} @@ -64,16 +82,16 @@ def _c_weight_symbol(model: str, symbol: str) -> str: def _c_weight_ref_macro(model: str, symbol: str) -> str: """The identifier `infer` uses to read a file-loaded weight. - A file-loaded plan's header declares two different C names for the same - weight -- the host array `` and, under __CUDACC__/__HIPCC__, the - device pointer `_dev` (controller ruling P4; the pointer is - declared-only until Task 3) -- and `infer` is emitted exactly once, so it - cannot spell either name directly. This macro (itself prefixed with the - model-qualified symbol, so it carries ruling R1's collision safety same - as every other external name here) is `#define`d to whichever of the two - the same __CUDACC__/__HIPCC__ guard selects; `infer`'s body reads only - this name. An embedded plan has no such split -- its weights are one - ROSENNA_CONST array reachable under any guard -- so `infer` reads the + A file-loaded plan's header holds two different storages for the same + weight -- the host array `` and, in the device pass of nvcc or + hipcc, an entry of the per-translation-unit __constant__ pointer table + (see _emit_device_weight_table) -- and `infer` is emitted exactly once, + so it cannot spell either name directly. This macro (itself prefixed with + the model-qualified symbol, so it carries ruling R1's collision safety + same as every other external name here) is `#define`d to whichever of + the two the device-pass guard selects; `infer`'s body reads only this + name. An embedded plan has no such split -- its weights are one + ROSENNA_CONST array reachable in either pass -- so `infer` reads the plain symbol directly there and never goes through this macro. """ return f"ROSENNA_REF_{_c_weight_symbol(model, symbol)}" @@ -125,42 +143,83 @@ def _weight_index_c(weight_by_symbol: dict, op) -> str: def emit_c(plan: Plan) -> tuple: ctype = _CTYPE[plan.dtype] header = _emit_header(plan, ctype) + # nvcc and hipcc compile this file as C++ (-x cu / -x hip in the recipe) + # so that init can call the runtime, so every external definition sits + # in an extern "C" block matching the header's declarations; under a C + # compiler the block is not there. + lines = _emit_source_head(plan, ctype) if plan.embed: # Every weight is a ROSENNA_CONST array in the header, so the source - # has nothing left to define. (Under __CUDACC__/__HIPCC__ that macro - # is __constant__; compiles and runs on the host, device path + # has nothing to define or load; it holds only the OpenMP-fallback + # infer_batch. (Under nvcc/hipcc that macro is __constant__ or + # __device__ const; compiles and runs on the host, device path # unvalidated until the GPU gate.) - source = f'/* Generated by rosenna. Do not edit. */\n#include "{plan.model}.h"\n' + pass else: - lines = [] - lines += _emit_source_head(plan, ctype) lines += _emit_load(plan) + lines += _emit_upload(plan) lines += _emit_init(plan) - source = "\n".join(lines) + "\n" + lines += _emit_fallback_infer_batch(plan, ctype) + lines += ["#if defined(__cplusplus)", "}", "#endif"] + source = "\n".join(lines) + "\n" return source, header def emit_c_recipe(plan: Plan) -> str: - """A Makefile fragment that builds lib.a with the host's offload flags.""" + """A Makefile fragment that builds lib.a for one of three backends. + + ROSENNA_BACKEND=omp (the default) compiles .c with the host C + compiler and the host's offload flags; its infer_batch is the OpenMP + target loop over host pointers. ROSENNA_BACKEND=cuda|hip compiles both + .c (as C++, since init then calls the runtime) and _kernel.cu + with nvcc or hipcc; that infer_batch launches the native kernel over + device pointers. The two are never linked together. + """ n = plan.model - return f"""# Generated by rosenna. Build lib{n}.a with the same offload flags as the host. + return f"""# Generated by rosenna. Builds lib{n}.a; ROSENNA_BACKEND selects the batched path. CC ?= gcc CFLAGS ?= -O2 -Wall -Wextra -std=c11 ROSENNA_OFFLOAD_FLAGS ?= - +DEVFLAGS ?= -O2 +# cuda | hip | omp +ROSENNA_BACKEND ?= omp + +ifeq ($(ROSENNA_BACKEND),cuda) +DEVCC ?= nvcc +lib{n}.a: {n}.o {n}_kernel.o +\tar rcs $@ $^ +{n}_kernel.o: {n}_kernel.cu {n}.h rosenna_rt.h +\t$(DEVCC) $(DEVFLAGS) -c $< -o $@ +{n}.o: {n}.c {n}.h rosenna_rt.h +\t$(DEVCC) $(DEVFLAGS) -x cu -c $< -o $@ +else ifeq ($(ROSENNA_BACKEND),hip) +DEVCC ?= hipcc +lib{n}.a: {n}.o {n}_kernel.o +\tar rcs $@ $^ +{n}_kernel.o: {n}_kernel.cu {n}.h rosenna_rt.h +\t$(DEVCC) $(DEVFLAGS) -x hip -c $< -o $@ +{n}.o: {n}.c {n}.h rosenna_rt.h +\t$(DEVCC) $(DEVFLAGS) -x hip -c $< -o $@ +else lib{n}.a: {n}.o -\tar rcs $@ $< +\tar rcs $@ $^ {n}.o: {n}.c {n}.h \t$(CC) $(CFLAGS) $(ROSENNA_OFFLOAD_FLAGS) -c $< -o $@ +endif clean: -\trm -f {n}.o lib{n}.a +\trm -f {n}.o {n}_kernel.o lib{n}.a .PHONY: clean """ +def _embedded_weight_bytes(plan: Plan) -> int: + return plan.n_params * _ITEMSIZE[plan.dtype] if plan.embed else 0 + + def _emit_header(plan: Plan, ctype: str) -> str: m = plan.model guard = f"ROSENNA_{m.upper()}_H" + n_in, n_out = plan.input.shape[0], plan.output.shape[0] lines = [ f"#ifndef {guard}", f"#define {guard}", @@ -169,18 +228,92 @@ def _emit_header(plan: Plan, ctype: str) -> str: "", "#include ", "", + ] + lines += _emit_device_macros(plan) + lines += [ + "#if defined(__cplusplus)", + 'extern "C" {', + "#endif", + "", + ] + if not plan.embed: + lines += _emit_weight_declarations(plan, ctype) + lines += [ + f"int {m}_init(const char *path);", + "", + ] + lines += [ + "/* Batched inference over n points stored contiguously: point p reads", + f" x + p * {n_in} and writes y + p * {n_out}. x and y must already be on", + " the device; init is the only routine that transfers. This call", + " allocates nothing, copies nothing and never synchronizes; the caller", + f" owns the stream. Which implementation the library holds is fixed when", + f" lib{m}.a is built (ROSENNA_BACKEND in {m}.mk); the two are never", + " linked together.", + " cuda/hip backend: x and y are raw device pointers holding n * n_in", + " and n * n_out values; stream is a cudaStream_t / hipStream_t, or", + " NULL for the default stream; the launch is asynchronous on it.", + " omp backend: x and y are device pointers the host obtained from", + " omp_target_alloc or from use_device_ptr on data it mapped (on a", + " host-only build those are the host pointers and the loop runs on", + " the CPU); stream is ignored.", + f" Returns 0, or 10 if a device allocation or copy failed in {m}_init", + " (cuda/hip backend of a file-loaded model; init never having been", + " called counts as that). */", + f"int {m}_infer_batch(int n, const {ctype} *x, {ctype} *y, void *stream);", + "", + "#if defined(__cplusplus)", + "}", + "#endif", + "", + ] + if not plan.embed: + lines += _emit_device_weight_table(plan, ctype) + lines += _emit_device_region(plan, ctype) + lines.append("#endif") + return "\n".join(lines) + "\n" + + +def _emit_device_macros(plan: Plan) -> list: + """The macros at the top of the header, and nothing else defines them. + + ROSENNA_CONST (controller ruling R4): the embedded weights go to + __constant__ only while their total size stays under + CONSTANT_MEMORY_LIMIT, since CUDA constant memory is 64 KB per module; + a larger embedded model reads them from __device__ const global memory + instead. Both are `static` so that each translation unit that includes + the header -- .c compiled as C++, _kernel.cu, and any host + .cu -- gets its own copy with internal linkage: a namespace-scope + __constant__ definition with external linkage in a header is a duplicate + symbol the moment two objects include it. + """ + nbytes = _embedded_weight_bytes(plan) + if plan.embed and nbytes < CONSTANT_MEMORY_LIMIT: + const_qual = "static __constant__" + const_note = (f"static __constant__: the {nbytes} bytes of embedded weights", + f" fit the {CONSTANT_MEMORY_LIMIT}-byte constant-memory budget.") + elif plan.embed: + const_qual = "static __device__ const" + const_note = (f"static __device__ const: the {nbytes} bytes of embedded weights", + f" exceed the {CONSTANT_MEMORY_LIMIT}-byte constant-memory budget.") + else: + const_qual = "static __constant__" + const_note = ("static __constant__ (unused: this model loads its weights", + " from a file).") + return [ "/* Under nvcc/hipcc: ROSENNA_DEVICE_FN = __host__ __device__ (infer is", - " callable from device code), ROSENNA_CONST = __constant__ (embedded", - " weights live in device memory), ROSENNA_RESTRICT = __restrict__.", + " callable from device code), ROSENNA_RESTRICT = __restrict__, and", + f" ROSENNA_CONST = {const_note[0]}", + const_note[1], " Otherwise (plain C, or a host OpenMP/OpenACC build): ROSENNA_DEVICE_FN", " is empty, ROSENNA_CONST = static const, and ROSENNA_RESTRICT is", " __restrict__ in C++ or restrict in C. infer is separately wrapped in a", " guarded OpenMP declare-target region with a guarded OpenACC routine-seq", " pragma below; both are no-ops unless that compiler defines", " _OPENMP/_OPENACC. */", - "#if defined(__CUDACC__) || defined(__HIPCC__)", + _CUDA_GUARD, "#define ROSENNA_DEVICE_FN __host__ __device__", - "#define ROSENNA_CONST __constant__", + f"#define ROSENNA_CONST {const_qual}", "#define ROSENNA_RESTRICT __restrict__", "#else", "#define ROSENNA_DEVICE_FN", @@ -193,59 +326,122 @@ def _emit_header(plan: Plan, ctype: str) -> str: "#endif", "", ] - if not plan.embed: - lines += _emit_weight_declarations(plan, ctype) - lines += _emit_device_region(plan, ctype) - if not plan.embed: - lines += [ - f"int {m}_init(const char *path);", - "", - ] - lines.append("#endif") - return "\n".join(lines) + "\n" def _emit_weight_declarations(plan: Plan, ctype: str) -> list: - """A file-loaded plan's weights: a host array, or (Task 3) a device pointer. - - Controller ruling P4: the `_dev` pointers are declared, not defined -- - nothing fills them in until Task 3's CUDA/HIP init exists. Under a plain - or OpenMP host build the __CUDACC__/__HIPCC__ guard is false, so this - branch is never even compiled; no build in this task can reach it. - Compiles and runs on the host; device path unvalidated. The explanation - is emitted into the header itself (not just here), since a solver author - or the Task 3 implementer reads the generated .h, not this module. + """A file-loaded plan's weights, inside the header's extern "C" block. + + The host arrays are declared under every compiler: _init fills + them from the file, and the host instantiation of infer reads them even + when nvcc or hipcc is the compiler. The `_dev` pointers exist + only under nvcc/hipcc: _init allocates and fills each one (status + 10 on failure) and _infer_batch passes them to the kernel. Under + a plain or OpenMP host build the guard is false and no device pointer + is even declared. """ m = plan.model if not plan.weights: return [] lines = [ - f"/* Device copies of the weights below: defined by the library and", - f" filled in by {m}_init on a CUDA/HIP build (Task 3); not referenced", - f" under a plain or OpenMP host build. */", - "#if defined(__CUDACC__) || defined(__HIPCC__)", + "/* Host arrays filled by init from the weights file. Under an offloading", + " OpenMP or OpenACC build they also have device copies, which init", + " updates once the file is read (the plan step, ruling R5). */", ] - for w in plan.weights: - sym = _c_weight_symbol(m, w.symbol) - lines.append(f"extern {ctype} *{sym}_dev;") - lines.append("#else") + lines += _omp_declare_target_begin() for w in plan.weights: sym = _c_weight_symbol(m, w.symbol) lines.append(f"extern {ctype} {sym}[{_weight_size(w.shape)}];") - lines += ["#endif", ""] - + lines += _omp_declare_target_end() + lines += _acc_declare(plan, "create") lines += [ - "/* Selects the host array or the device pointer above, so infer's", - " body below is emitted once and reads whichever this build has. */", - "#if defined(__CUDACC__) || defined(__HIPCC__)", + "", + "/* Device copies of the arrays above: allocated and filled by", + f" {m}_init on a cuda/hip build, read by the batched kernel; not", + " referenced under a plain or OpenMP host build. */", + _CUDA_GUARD, ] for w in plan.weights: sym = _c_weight_symbol(m, w.symbol) - lines.append(f"#define {_c_weight_ref_macro(m, w.symbol)} {sym}_dev") + lines.append(f"extern {ctype} *{sym}_dev;") + lines += [ + f"/* Called by {m}_init once the copies above exist: publishes them to", + f" the kernel's translation unit ({m}_kernel.cu, where it is defined).", + " Part of the plan step, not of the host API. */", + f"int {_device_bind(m)}(void);", + "#endif", + "", + ] + return lines + + +def _omp_declare_target_begin() -> list: + return ["#ifdef _OPENMP", "#pragma omp declare target", "#endif"] + + +def _omp_declare_target_end() -> list: + return ["#ifdef _OPENMP", "#pragma omp end declare target", "#endif"] + + +def _weight_symbol_list(plan: Plan) -> str: + return ", ".join(_c_weight_symbol(plan.model, w.symbol) for w in plan.weights) + + +def _acc_declare(plan: Plan, clause: str) -> list: + """`#pragma acc declare (every weight)`, guarded by _OPENACC. + + gcc -fopenacc rejects a `routine seq` function that reads a file-scope + array with no `declare` directive, so every weight array infer reads + carries one: `create` for the file-loaded host arrays (init then does + `update device`), `copyin` for the embedded constants. + """ + if not plan.weights: + return [] + return ["#ifdef _OPENACC", f"#pragma acc declare {clause}({_weight_symbol_list(plan)})", "#endif"] + + +def _device_bind(model: str) -> str: + """The plan-step function in _kernel.cu that binds the weight table.""" + return f"{model}_device_bind" + + +def _device_weight_table(model: str) -> str: + """The per-translation-unit __constant__ table of device weight pointers.""" + return f"{model}_devw" + + +def _emit_device_weight_table(plan: Plan, ctype: str) -> list: + """How infer reaches a file-loaded weight from device code. + + Device code cannot read a host global, and without relocatable device + code a __device__ or __constant__ variable cannot be declared extern + across translation units, so the device addresses the host holds in + `_dev` reach the kernel through this `static __constant__` table, + one copy per translation unit, which _device_bind (in the same + translation unit as the kernel) fills once, at the end of init, through + ROSENNA_MEMCPY_TO_SYMBOL. The ROSENNA_REF_ macros then + select the table entry in the device pass and the host array otherwise, + so infer's one body reads the right storage in each of its two + instantiations without ever naming host storage from device code. + """ + m = plan.model + if not plan.weights: + return [] + table = _device_weight_table(m) + lines = [ + "/* Device-side view of the file-loaded weights: a per-translation-unit", + f" __constant__ table of the device pointers, filled once by {m}_init", + f" through {_device_bind(m)}. Device code reads its own translation", + " unit's table; host code, under any compiler, reads the host arrays. */", + _CUDA_GUARD, + f"static __constant__ const {ctype} *{table}[{len(plan.weights)}];", + "#endif", + _DEVICE_PASS_GUARD, + ] + for k, w in enumerate(plan.weights): + lines.append(f"#define {_c_weight_ref_macro(m, w.symbol)} {table}[{k}]") lines.append("#else") for w in plan.weights: - sym = _c_weight_symbol(m, w.symbol) - lines.append(f"#define {_c_weight_ref_macro(m, w.symbol)} {sym}") + lines.append(f"#define {_c_weight_ref_macro(m, w.symbol)} {_c_weight_symbol(m, w.symbol)}") lines += ["#endif", ""] return lines @@ -275,6 +471,7 @@ def _emit_embedded_weights(plan: Plan, ctype: str) -> list: sym = _c_weight_symbol(m, w.symbol) values = ", ".join(_format_embedded_value(v, plan.dtype) for v in w.values) lines.append(f"ROSENNA_CONST {ctype} {sym}[{_weight_size(w.shape)}] = {{ {values} }};") + lines += _acc_declare(plan, "copyin") if plan.weights: lines.append("") return lines @@ -282,24 +479,127 @@ def _emit_embedded_weights(plan: Plan, ctype: str) -> list: def _emit_source_head(plan: Plan, ctype: str) -> list: m = plan.model - hash_bytes = ", ".join(f"0x{b:02x}" for b in bytes.fromhex(plan.hash())) lines = [ "/* Generated by rosenna. Do not edit. */", + "", + "/* On a cuda/hip build the runtime is reached only through the macros in", + " rosenna_rt.h, included ahead of the model header so that the CUDA/HIP", + " declaration keywords the header uses are in scope; a host C compiler", + " never sees it. */", + _CUDA_GUARD, + '#include "rosenna_rt.h"', + "#endif", f'#include "{m}.h"', "", - "#include ", - "#include ", - "#include ", + "#include ", + ] + if not plan.embed: + lines += [ + "#include ", + "#include ", + "#include ", + ] + lines += [ + "", + "/* nvcc and hipcc compile this file as C++ (-x cu / -x hip, see the", + " recipe); the definitions below then carry C linkage to match the", + " header's declarations, so C and Fortran hosts link unchanged. */", + "#if defined(__cplusplus)", + 'extern "C" {', + "#endif", "", + ] + if plan.embed: + return lines + hash_bytes = ", ".join(f"0x{b:02x}" for b in bytes.fromhex(plan.hash())) + lines += [ f"static const unsigned char expected_hash[32] = {{ {hash_bytes} }};", "", ] + # The header's `acc declare create` on the extern declarations already + # covers these definitions (gcc rejects a second declare for the same + # variable); the OpenMP declare-target region is repeated, which is + # allowed and keeps the definition self-describing. + lines += _omp_declare_target_begin() for w in plan.weights: lines.append(f"{ctype} {_c_weight_symbol(m, w.symbol)}[{_weight_size(w.shape)}];") + lines += _omp_declare_target_end() + if plan.weights: + lines.append(_CUDA_GUARD) + for w in plan.weights: + lines.append(f"{ctype} *{_c_weight_symbol(m, w.symbol)}_dev = 0;") + lines.append("#endif") lines.append("") return lines +def _emit_upload(plan: Plan) -> list: + """The cuda/hip half of init: copy the freshly loaded host arrays to the device. + + Only compiled under nvcc/hipcc, where the runtime is reachable through + rosenna_rt.h. A repeated init frees the previous copies first (freeing a + null pointer is a no-op in both runtimes); a failed allocation or copy + leaves that pointer null and returns 10, so a later infer_batch refuses + to launch rather than read an unfilled buffer. The last step publishes + the new addresses to the kernel's translation unit (_device_bind, + in _kernel.cu); after init returns, the loop path transfers + nothing (controller ruling R5). + """ + m = plan.model + if not plan.weights: + return [] + lines = [ + _CUDA_GUARD, + f"static int {m}_upload(void) {{", + ] + for w in plan.weights: + sym = _c_weight_symbol(m, w.symbol) + lines += [ + f" (void)ROSENNA_FREE({sym}_dev);", + f" {sym}_dev = 0;", + f" if (ROSENNA_MALLOC(&{sym}_dev, sizeof {sym}) != ROSENNA_OK) return 10;", + f" if (ROSENNA_MEMCPY_H2D({sym}_dev, {sym}, sizeof {sym}) != ROSENNA_OK) {{", + f" (void)ROSENNA_FREE({sym}_dev);", + f" {sym}_dev = 0;", + f" return 10;", + f" }}", + ] + lines += [f" return {_device_bind(m)}();", "}", "#endif", ""] + return lines + + +def _emit_fallback_infer_batch(plan: Plan, ctype: str) -> list: + """The OpenMP-target infer_batch: the omp backend, over device pointers. + + Controller ruling R5: x and y are already on the device (is_device_ptr + / deviceptr), so the loop path maps, allocates and synchronizes nothing. + Compiled only when the compiler is not nvcc or hipcc; the cuda/hip + backends define the same function in _kernel.cu instead. Under a + host compiler without -fopenmp/-fopenacc the pragmas are inert and this + is a plain loop over host pointers, which is also what use_device_ptr + yields on a host-only build. Compiles and runs on the host; device path + unvalidated. + """ + m = plan.model + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + return [ + _NOT_CUDA_GUARD, + f"int {m}_infer_batch(int n, const {ctype} *x, {ctype} *y, void *stream) {{", + " (void)stream;", + " if (n <= 0) return 0;", + "#if defined(_OPENMP)", + "#pragma omp target teams loop is_device_ptr(x, y)", + "#elif defined(_OPENACC)", + "#pragma acc parallel loop deviceptr(x, y)", + "#endif", + f" for (int p = 0; p < n; ++p) {m}_infer(x + (size_t)p * {n_in}, y + (size_t)p * {n_out});", + " return 0;", + "}", + "#endif", + "", + ] + + def _emit_load(plan: Plan) -> list: if not plan.weights: return [] @@ -390,7 +690,19 @@ def _emit_init(plan: Plan) -> list: " if (fseek(f, tocpos, SEEK_SET) != 0) { fclose(f); return 9; }", " }", " fclose(f);", + " /* The plan step's transfer: the device copies of the arrays, under", + " whichever offload family this build has. */", + "#ifdef _OPENMP", + f"#pragma omp target update to({_weight_symbol_list(plan)})", + "#endif", + "#ifdef _OPENACC", + f"#pragma acc update device({_weight_symbol_list(plan)})", + "#endif", + _CUDA_GUARD, + f" return {m}_upload();", + "#else", " return 0;", + "#endif", "}", "", ] diff --git a/python/rosenna/emit_kernel.py b/python/rosenna/emit_kernel.py new file mode 100644 index 0000000..826186d --- /dev/null +++ b/python/rosenna/emit_kernel.py @@ -0,0 +1,79 @@ +"""Render _kernel.cu: the native batched kernel for nvcc and hipcc. + +One source in the CUDA subset hipcc accepts unchanged; every runtime call +goes through rosenna_rt.h. The kernel is one thread per point calling the +header-inline infer, which is correct by construction because it reuses the +body every host test already checks; a fused tiled GEMM over the batch is a +follow-up once the GPU gate has timed this one. Controller ruling R5: x and +y are device-resident and infer_batch only launches; the one transfer this +file makes (_device_bind, file-loaded plans only) belongs to the plan +step and is called from init. Compiles and runs on the host (through the +omp backend) only; the device path is unvalidated until nvcc has built it +on CI and the GPU gate has run it. +""" +from .emit_c import KERNEL_TILE, _CTYPE, _c_weight_symbol, _device_bind, _device_weight_table +from .plan import Plan + + +def emit_kernel(plan: Plan) -> str: + m = plan.model + ctype = _CTYPE[plan.dtype] + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + lines = [ + "/* Generated by rosenna. Do not edit. Build with nvcc or hipcc. */", + '#include "rosenna_rt.h"', + f'#include "{m}.h"', + "", + "#include ", + "", + f"#define ROSENNA_TILE {KERNEL_TILE}", + "", + ] + if not plan.embed and plan.weights: + table = _device_weight_table(m) + nw = len(plan.weights) + lines += [ + "/* Plan step (controller ruling R5), called by init after it has made", + " the device copies: publishes their addresses to this translation", + " unit's __constant__ table, the one the kernel below reads. This is", + " the only transfer in this file; infer_batch launches and nothing", + " else. */", + f'extern "C" int {_device_bind(m)}(void) {{', + f" const {ctype} *table[{nw}] = {{", + ] + for w in plan.weights: + lines.append(f" {_c_weight_symbol(m, w.symbol)}_dev,") + lines += [ + " };", + f" for (int k = 0; k < {nw}; ++k) if (table[k] == 0) return 10;", + f" if (ROSENNA_MEMCPY_TO_SYMBOL({table}, table, sizeof table) != ROSENNA_OK) return 10;", + " return 0;", + "}", + "", + ] + lines += [ + "/* One thread per point; each calls the same inline body the host uses,", + " now instantiated as a __device__ function. */", + f"static __global__ void {m}_kernel(int n, const {ctype} *__restrict__ x,", + f" {ctype} *__restrict__ y) {{", + " const int p = (int)(blockIdx.x * blockDim.x + threadIdx.x);", + " if (p >= n) return;", + f" {m}_infer(x + (size_t)p * {n_in}, y + (size_t)p * {n_out});", + "}", + "", + f'extern "C" int {m}_infer_batch(int n, const {ctype} *x, {ctype} *y, void *stream) {{', + " if (n <= 0) return 0;", + " const ROSENNA_STREAM_T s = (ROSENNA_STREAM_T)stream;", + ] + if not plan.embed and plan.weights: + # A null device copy means init never ran or failed: status 10 rather + # than a device fault. Reading the host globals is not a transfer. + lines += [f" if ({_c_weight_symbol(m, w.symbol)}_dev == 0) return 10;" for w in plan.weights] + lines += [ + " const int grid = (n + ROSENNA_TILE - 1) / ROSENNA_TILE;", + f" ROSENNA_LAUNCH({m}_kernel, grid, ROSENNA_TILE, s, n, x, y);", + " return 0;", + "}", + "", + ] + return "\n".join(lines) diff --git a/python/rosenna/rt_header.py b/python/rosenna/rt_header.py new file mode 100644 index 0000000..6283c04 --- /dev/null +++ b/python/rosenna/rt_header.py @@ -0,0 +1,45 @@ +"""Render rosenna_rt.h: the one file that names the CUDA or HIP runtime API. + +The generated kernel (`emit_kernel`) and the CUDA/HIP branch of the generated +`.c` reach the runtime only through these macros, so one source builds under +both nvcc and hipcc. The file is identical for every model. +""" + +_RT_HEADER = """\ +/* Generated by rosenna. Do not edit. + Maps the runtime calls the generated sources make onto CUDA or HIP; no + other generated file names a cuda* or hip* symbol. Compiled only by nvcc + (__CUDACC__) or hipcc (__HIPCC__): a host C compiler never sees it. */ +#ifndef ROSENNA_RT_H +#define ROSENNA_RT_H +#if defined(__HIPCC__) +#include +#define ROSENNA_STREAM_T hipStream_t +#define ROSENNA_MALLOC(p, n) hipMalloc((void **)(p), (n)) +#define ROSENNA_MEMCPY_H2D(d, h, n) hipMemcpy((d), (h), (n), hipMemcpyHostToDevice) +#define ROSENNA_MEMCPY_TO_SYMBOL(sym, src, n) \\ + hipMemcpyToSymbol(HIP_SYMBOL(sym), (src), (n), 0, hipMemcpyHostToDevice) +#define ROSENNA_FREE(p) hipFree(p) +#define ROSENNA_OK hipSuccess +#define ROSENNA_SYNC(s) hipStreamSynchronize(s) +#define ROSENNA_LAUNCH(k, g, b, s, ...) k<<<(g), (b), 0, (s)>>>(__VA_ARGS__) +#elif defined(__CUDACC__) +#include +#define ROSENNA_STREAM_T cudaStream_t +#define ROSENNA_MALLOC(p, n) cudaMalloc((void **)(p), (n)) +#define ROSENNA_MEMCPY_H2D(d, h, n) cudaMemcpy((d), (h), (n), cudaMemcpyHostToDevice) +#define ROSENNA_MEMCPY_TO_SYMBOL(sym, src, n) \\ + cudaMemcpyToSymbol((sym), (src), (n), 0, cudaMemcpyHostToDevice) +#define ROSENNA_FREE(p) cudaFree(p) +#define ROSENNA_OK cudaSuccess +#define ROSENNA_SYNC(s) cudaStreamSynchronize(s) +#define ROSENNA_LAUNCH(k, g, b, s, ...) k<<<(g), (b), 0, (s)>>>(__VA_ARGS__) +#else +#error "rosenna_rt.h is for nvcc or hipcc only" +#endif +#endif +""" + + +def rt_header() -> str: + return _RT_HEADER diff --git a/python/tests/test_kernel.py b/python/tests/test_kernel.py new file mode 100644 index 0000000..5a0367a --- /dev/null +++ b/python/tests/test_kernel.py @@ -0,0 +1,447 @@ +import os +import re +import shutil +import subprocess +import numpy as np +import pytest +from onnx import helper, numpy_helper +from rosenna.frontend import load_graph +from rosenna.plan import build_plan +from rosenna.emit_kernel import emit_kernel +from rosenna.rt_header import rt_header +from rosenna.emit_c import emit_c, emit_c_recipe, CONSTANT_MEMORY_LIMIT +from tests.conftest import save_model + + +def test_kernel_source_uses_only_the_rt_macros(golden_model): + plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64") + cu = emit_kernel(plan) + assert "rosenna_rt.h" in cu and "__global__" in cu and 'extern "C" int gemm_small_infer_batch(' in cu + for forbidden in ("cudaMalloc", "hipMalloc", "cudaMemcpy", "hipMemcpy", "<<<"): + assert forbidden not in cu, forbidden # only rosenna_rt.h may name the runtime + + +def test_kernel_source_names_no_runtime_symbol_for_a_file_loaded_plan(golden_model): + # The file-loaded kernel file also holds the plan-step bind of the device + # weight pointers (called by init); that too must go through + # rosenna_rt.h, never a cuda*/hip* name. + plan = build_plan(load_graph(golden_model("gemm_big")), dtype="f64", embed=False) + cu = emit_kernel(plan) + assert re.search(r"\b(cuda|hip)[A-Z]", cu) is None + assert 'extern "C" int gemm_big_device_bind(void) {' in cu + assert "ROSENNA_MEMCPY_TO_SYMBOL(gemm_big_devw, table, sizeof table)" in cu + assert "ROSENNA_LAUNCH(gemm_big_kernel" in cu + # An embedded plan reads its ROSENNA_CONST arrays directly and binds nothing. + cu_e = emit_kernel(build_plan(load_graph(golden_model("gemm_big")), dtype="f64", embed=True)) + assert "ROSENNA_MEMCPY_TO_SYMBOL" not in cu_e and "device_bind" not in cu_e + + +def _function_body(text: str, signature_start: str) -> str: + """The text of one top-level C function, from its signature to its closing brace.""" + start = text.index(signature_start) + end = text.index("\n}\n", start) + 3 + return text[start:end] + + +_LOOP_PATH_FORBIDDEN = ("map(to", "map(from", "map(tofrom", "copyin", "copyout", + "ROSENNA_MALLOC", "ROSENNA_MEMCPY", "ROSENNA_SYNC", + "cudaMemcpy", "hipMemcpy", "cudaMalloc", "hipMalloc", + "cudaDeviceSynchronize", "hipDeviceSynchronize") + + +def test_loop_path_never_transfers(golden_model): + # Controller ruling R5: init is the plan step and the only routine that + # allocates, transfers or synchronizes; x and y are device-resident in + # every backend. No transfer can be observed on a host-only build, so + # this structural check is the CI-able guarantee. Scope: the omp-backend + # infer_batch body in the .c (init is exempt), and in the kernel file the + # kernel and infer_batch bodies. The kernel file's third function, + # _device_bind, is the tail of the plan step: a __constant__ symbol + # can be written only from its own translation unit without relocatable + # device code, so init reaches into the kernel's file to publish the + # weight addresses. It is exempt like init, and checked below to be the + # one place in that file a transfer name appears. + for embed in (True, False): + name = "gemm_big"; plan = build_plan(load_graph(golden_model(name)), dtype="f64", embed=embed) + source, _ = emit_c(plan) + cu = emit_kernel(plan) + loop_path = [ + _function_body(source, f"int {name}_infer_batch("), + _function_body(cu, f"static __global__ void {name}_kernel("), + _function_body(cu, f'extern "C" int {name}_infer_batch('), + ] + for body in loop_path: + for forbidden in _LOOP_PATH_FORBIDDEN: + assert forbidden not in body, (embed, forbidden, body) + rest = cu + for body in loop_path[1:]: + rest = rest.replace(body, "") + if embed: + for forbidden in _LOOP_PATH_FORBIDDEN: + assert forbidden not in rest, (forbidden, rest) + else: + bind = _function_body(cu, f'extern "C" int {name}_device_bind(') + rest = rest.replace(bind, "") + for forbidden in _LOOP_PATH_FORBIDDEN: + assert forbidden not in rest, (forbidden, rest) + assert "ROSENNA_MEMCPY_TO_SYMBOL" in bind + assert f"return {name}_device_bind();" in _function_body(source, f"static int {name}_upload(") + + +def test_rt_header_maps_both_runtimes(): + h = rt_header() + assert "__HIPCC__" in h and "__CUDACC__" in h and "ROSENNA_LAUNCH" in h + # Every macro the generated sources use is defined once per runtime. + for macro in ("ROSENNA_STREAM_T", "ROSENNA_MALLOC", "ROSENNA_MEMCPY_H2D", "ROSENNA_FREE", + "ROSENNA_OK", "ROSENNA_SYNC", "ROSENNA_LAUNCH", "ROSENNA_MEMCPY_TO_SYMBOL"): + assert h.count(f"#define {macro}") == 2, macro + + +def test_header_declares_infer_batch_with_c_linkage_on_both_forms(golden_model): + for embed in (True, False): + plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64", embed=embed) + source, header = emit_c(plan) + assert "int gemm_small_infer_batch(int n, const double *x, double *y, void *stream);" in header + assert 'extern "C" {' in header and "__cplusplus" in header + assert "x and y must already be on\n the device; init is the only routine that transfers." in header + # The OpenMP fallback lives in the .c under the negation of the CUDA/HIP + # guard, over device pointers (ruling R5), each pragma under its own guard. + assert "int gemm_small_infer_batch(int n, const double *x, double *y, void *stream) {" in source + assert "#if !defined(__CUDACC__) && !defined(__HIPCC__)" in source + assert "#if defined(_OPENMP)\n#pragma omp target teams loop is_device_ptr(x, y)\n" in source + assert "#elif defined(_OPENACC)\n#pragma acc parallel loop deviceptr(x, y)\n#endif" in source + + +def test_file_loaded_source_copies_to_the_device_under_the_cuda_guard(golden_model): + plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64", embed=False) + source, header = emit_c(plan) + assert "double *gemm_small_w0_dev = 0;" in source + assert "ROSENNA_MALLOC(&gemm_small_w0_dev, sizeof gemm_small_w0)" in source + assert "ROSENNA_MEMCPY_H2D(gemm_small_w0_dev, gemm_small_w0, sizeof gemm_small_w0)" in source + assert "return 10;" in source and '#include "rosenna_rt.h"' in source + # init ends by publishing the copies to the kernel's translation unit. + assert "return gemm_small_device_bind();" in source + assert "int gemm_small_device_bind(void);" in header + # Under an offloading OpenMP/OpenACC build the host arrays have device + # copies that init updates in the same call (the plan step); each + # directive under its own guard. + weights = "gemm_small_w0, gemm_small_b0, gemm_small_w1, gemm_small_b1" + assert f"#ifdef _OPENMP\n#pragma omp target update to({weights})\n#endif" in source + assert f"#ifdef _OPENACC\n#pragma acc update device({weights})\n#endif" in source + assert f"#pragma acc declare create({weights})" in header + assert "#pragma omp declare target\n#endif\nextern double gemm_small_w0[4];" in header + + +def test_generated_c_is_warning_free_under_openacc(tmp_path, golden_model): + # gcc -fopenacc rejects a `routine seq` function reading a file-scope + # array with no `declare` directive (HEAD before this task failed here); + # both forms must now compile clean, including a host calling the + # `parallel loop deviceptr(x, y)` fallback. + from tests.test_device_c import _omp_cc + cc = _omp_cc() + probe = subprocess.run([cc, "-fopenacc", "-x", "c", "-", "-o", os.devnull], + input="int main(void){return 0;}", capture_output=True, text=True) + if probe.returncode != 0: + pytest.skip(f"{cc} does not accept -fopenacc") + for embed in (True, False): + name = "gemm_small"; plan = build_plan(load_graph(golden_model(name)), dtype="f64", embed=embed) + d = tmp_path / ("e" if embed else "f"); d.mkdir() + source, header = emit_c(plan) + (d / f"{name}.c").write_text(source); (d / f"{name}.h").write_text(header) + (d / "host.c").write_text(f""" +#include "{name}.h" +int main(void) {{ double x[2] = {{0.5, 0.5}}, y[3]; return {name}_infer_batch(1, x, y, 0); }} +""") + r = subprocess.run([cc, "-O2", "-Wall", "-Wextra", "-std=c11", "-fopenacc", f"{name}.c", "host.c", "-lm", "-o", "host"], + cwd=d, capture_output=True, text=True) + assert r.returncode == 0 and r.stderr == "", r.stderr + if embed: + assert subprocess.run(["./host"], cwd=d, capture_output=True).returncode == 0 + # Status 10 is in the emitted legend, next to the routine that returns it. + assert "10 device allocation or copy failed in init" in source + # Device code reads the per-translation-unit __constant__ pointer table; + # host code, under any compiler, reads the host arrays. + assert "static __constant__ const double *gemm_small_devw[4];" in header + assert "#define ROSENNA_REF_gemm_small_w0 gemm_small_devw[0]" in header + assert "#define ROSENNA_REF_gemm_small_w0 gemm_small_w0" in header + assert "__CUDA_ARCH__" in header and "__HIP_DEVICE_COMPILE__" in header + + +def _embedded_plan(tmp_path, name, n_in, n_out): + w = numpy_helper.from_array(np.random.default_rng(1).uniform(-1, 1, (n_in, n_out)).astype(np.float32), "w") + node = helper.make_node("MatMul", ["x", "w"], ["y"], name="m0") + path = save_model(tmp_path, name, [node], [w], (1, n_in), (1, n_out)) + return build_plan(load_graph(path), dtype="f64", embed=True) + + +def test_constant_memory_limit_selects_the_device_qualifier(tmp_path): + # Controller ruling R4: CUDA __constant__ memory is 64 KB per module and + # the embed threshold is 1M parameters, so an embedded model past the + # constant budget must go to __device__ const instead. The cut is 48 KB of + # weight bytes; a header states which it chose. + assert CONSTANT_MEMORY_LIMIT == 48 * 1024 + small = _embedded_plan(tmp_path, "under", 64, 90) # 5760 f64 = 46080 B < 48 KB + big = _embedded_plan(tmp_path, "over", 64, 100) # 6400 f64 = 51200 B > 48 KB + _, h_small = emit_c(small) + _, h_big = emit_c(big) + assert "#define ROSENNA_CONST static __constant__" in h_small + assert "__device__ const" not in h_small + assert "#define ROSENNA_CONST static __device__ const" in h_big + assert "__constant__" not in h_big + assert "46080" in h_small and "51200" in h_big # the comment names the byte count it judged + + +def _omp_host(name, n_in, n_out, npts, init): + """A host that maps its data first and hands infer_batch device pointers (R5). + + Under the gcc host fallback every step is the identity and the batch must + agree bit for bit with the host's own calls of the header inline. + """ + return f""" +#include +#include +#include "{name}.h" +int main(void) {{ + const int nx = {npts} * {n_in}, ny = {npts} * {n_out}; + double *x = malloc(nx * sizeof *x), *y = malloc(ny * sizeof *y), *yb = malloc(ny * sizeof *yb); + int status = 0; + {init} + for (int c = 0; c < nx; ++c) x[c] = 0.01 * c - 0.3; + for (int p = 0; p < {npts}; ++p) {name}_infer(x + p * {n_in}, y + p * {n_out}); +#ifdef _OPENMP + #pragma omp target enter data map(to: x[0:nx]) map(alloc: yb[0:ny]) + #pragma omp target data use_device_ptr(x, yb) +#endif + {{ + status = {name}_infer_batch({npts}, x, yb, 0); + if (status == 0) status = {name}_infer_batch(0, x, yb, 0) ? 4 : 0; + }} +#ifdef _OPENMP + #pragma omp target exit data map(from: yb[0:ny]) map(delete: x[0:nx]) +#endif + if (status) return status; + for (int c = 0; c < ny; ++c) if (y[c] != yb[c]) return 5; + puts("agree"); return 0; }} +""" + + +def _omp_build_and_run(tmp_path, name, plan, cc, init): + source, header = emit_c(plan) + (tmp_path / f"{name}.c").write_text(source); (tmp_path / f"{name}.h").write_text(header) + (tmp_path / "Makefile").write_text(emit_c_recipe(plan)) + r = subprocess.run(["make", f"CC={cc}", "ROSENNA_BACKEND=omp", "ROSENNA_OFFLOAD_FLAGS=-fopenmp"], cwd=tmp_path, capture_output=True, text=True) + assert r.returncode == 0, r.stderr + assert "warning" not in r.stderr, r.stderr + (tmp_path / "host.c").write_text(_omp_host(name, plan.input.shape[0], plan.output.shape[0], 16, init)) + r = subprocess.run([cc, "-O2", "-Wall", "-Wextra", "-std=c11", "-fopenmp", "host.c", f"lib{name}.a", "-lm", "-o", "host"], cwd=tmp_path, capture_output=True, text=True) + assert r.returncode == 0 and r.stderr == "", r.stderr + return subprocess.run(["./host"], cwd=tmp_path, capture_output=True, text=True) + + +def test_omp_backend_infer_batch_matches_inline(tmp_path, golden_model): + # The fallback backend: a library-side target loop over device pointers + # (is_device_ptr) must agree exactly with the host's own loop over the + # header inline. The host maps x and y first and passes what + # use_device_ptr yields, per ruling R5. + from tests.test_device_c import _omp_cc + from rosenna.weights import write_weights + name = "gemm_big"; graph = load_graph(golden_model(name)); plan = build_plan(graph, dtype="f64", embed=False) + write_weights(plan, graph, tmp_path / f"{name}.rwt") + r = _omp_build_and_run(tmp_path, name, plan, _omp_cc(), f'if ({name}_init("{name}.rwt")) return 2;') + assert r.returncode == 0 and r.stdout.strip() == "agree", (r.returncode, r.stdout, r.stderr) + + +def test_omp_backend_builds_an_embedded_plan_too(tmp_path, golden_model): + # An embedded plan's .c holds only the fallback infer_batch; the recipe + # must still produce a library for it, and the batch must agree with the + # inline. + from tests.test_device_c import _omp_cc + name = "gemm_small"; graph = load_graph(golden_model(name)); plan = build_plan(graph, dtype="f64", embed=True) + r = _omp_build_and_run(tmp_path, name, plan, _omp_cc(), "") + assert r.returncode == 0 and r.stdout.strip() == "agree", (r.returncode, r.stdout, r.stderr) + + +def test_omp_target_loop_in_infer_batch_is_real(tmp_path, golden_model): + # As in test_device_c: a host-only libgomp refuses a target region under + # OMP_TARGET_OFFLOAD=MANDATORY, so the library's own loop must fail there + # -- proof the pragma is a real target construct and not ignored. + from tests.test_device_c import _omp_cc + name = "gemm_small"; plan = build_plan(load_graph(golden_model(name)), dtype="f64", embed=True) + source, header = emit_c(plan) + (tmp_path / f"{name}.c").write_text(source); (tmp_path / f"{name}.h").write_text(header) + (tmp_path / "host.c").write_text(f""" +#include "{name}.h" +int main(void) {{ double x[2] = {{0.5, 0.5}}, y[3]; return {name}_infer_batch(1, x, y, 0); }} +""") + cc = _omp_cc() + r = subprocess.run([cc, "-O2", "-std=c11", "-fopenmp", "host.c", f"{name}.c", "-lm", "-o", "host"], cwd=tmp_path, capture_output=True, text=True) + assert r.returncode == 0, r.stderr + r = subprocess.run(["./host"], cwd=tmp_path, capture_output=True, text=True, + env={**os.environ, "OMP_TARGET_OFFLOAD": "MANDATORY"}) + assert r.returncode != 0 and "MANDATORY" in r.stderr + + +def test_generated_c_is_warning_free_under_a_plain_compiler(tmp_path, golden_model): + # clang without -fopenmp and without CUDA: every guard is false and the + # fallback still has to compile clean under -Wall -Wextra. + cc = shutil.which("clang") or shutil.which("cc") + if cc is None: + pytest.skip("no plain C compiler found") + for embed in (True, False): + name = "gemm_small"; plan = build_plan(load_graph(golden_model(name)), dtype="f64", embed=embed) + d = tmp_path / ("e" if embed else "f"); d.mkdir() + source, header = emit_c(plan) + (d / f"{name}.c").write_text(source); (d / f"{name}.h").write_text(header) + r = subprocess.run([cc, "-O2", "-Wall", "-Wextra", "-std=c11", "-c", f"{name}.c"], cwd=d, capture_output=True, text=True) + assert r.returncode == 0 and r.stderr == "", r.stderr + + +def test_generated_c_compiles_as_cpp_with_the_rt_header_stubbed(tmp_path, golden_model): + # nvcc and hipcc compile .c as C++ (-x cu / -x hip). No such compiler + # runs here, so this is the nearest local check: a host C++ compiler with + # __CUDACC__ forced on and the CUDA keywords and runtime replaced by inert + # stand-ins. It catches C-only constructs in the .c, linkage mismatches + # between the header's extern "C" block and the definitions, and any use + # of a runtime name outside rosenna_rt.h. It proves nothing about device + # code generation; that waits for the nvcc CI job. + cxx = shutil.which("clang++") or shutil.which("g++") + if cxx is None: + pytest.skip("no C++ compiler found") + stub = """ +#ifndef ROSENNA_RT_H +#define ROSENNA_RT_H +#include +#include +#include +typedef void *ROSENNA_STREAM_T_; +static inline int rosenna_stub_malloc(void **p, size_t n) { *p = malloc(n); return *p ? 0 : 1; } +static inline int rosenna_stub_h2d(void *d, const void *h, size_t n) { memcpy(d, h, n); return 0; } +static inline int rosenna_stub_free(void *p) { free(p); return 0; } +static inline int rosenna_stub_sync(void *s) { (void)s; return 0; } +#define ROSENNA_STREAM_T ROSENNA_STREAM_T_ +#define ROSENNA_MALLOC(p, n) rosenna_stub_malloc((void **)(p), (n)) +#define ROSENNA_MEMCPY_H2D(d, h, n) rosenna_stub_h2d((d), (h), (n)) +#define ROSENNA_MEMCPY_TO_SYMBOL(sym, src, n) rosenna_stub_h2d((sym), (src), (n)) +#define ROSENNA_FREE(p) rosenna_stub_free(p) +#define ROSENNA_OK 0 +#define ROSENNA_SYNC(s) rosenna_stub_sync(s) +#define ROSENNA_LAUNCH(k, g, b, s, ...) \\ + do { for (blockIdx.x = 0; blockIdx.x < (unsigned)((g) * (b)); ++blockIdx.x) k(__VA_ARGS__); } while (0) +#endif +""" + from rosenna.weights import write_weights + for embed in (True, False): + name = "gemm_small"; graph = load_graph(golden_model(name)); plan = build_plan(graph, dtype="f64", embed=embed) + d = tmp_path / ("e" if embed else "f"); d.mkdir() + source, header = emit_c(plan) + (d / f"{name}.c").write_text(source); (d / f"{name}.h").write_text(header) + (d / "rosenna_rt.h").write_text(stub) + if not embed: + write_weights(plan, graph, d / f"{name}.rwt") + (d / f"{name}_kernel.cu").write_text(emit_kernel(plan)) + # blockIdx/blockDim/threadIdx are CUDA builtins; give the stub the + # three as plain objects so the kernel body parses, with one thread + # per block so the stub launch's loop over blockIdx.x walks the points. + (d / "builtins.h").write_text( + "struct rosenna_dim3 { unsigned int x, y, z; };\n" + "static struct rosenna_dim3 blockIdx = {0, 0, 0};\n" + "static const struct rosenna_dim3 blockDim = {1, 0, 0}, threadIdx = {0, 0, 0};\n") + # -ffp-contract=off on every translation unit: the host may be built + # by a different compiler than the library, and clang contracts + # `acc += a * b` to an FMA by default where g++ in ISO mode does not, + # which breaks the bit-for-bit comparison below for no real reason. + common = [cxx, "-x", "c++", "-std=c++11", "-ffp-contract=off", "-Wall", "-Wextra", "-c", + "-D__CUDACC__=1", "-D__host__=", "-D__device__=", "-D__constant__=", "-D__global__=", + "-include", "builtins.h"] + r = subprocess.run(common + [f"{name}.c", "-o", f"{name}.o"], cwd=d, capture_output=True, text=True) + assert r.returncode == 0, r.stderr + r = subprocess.run(common + [f"{name}_kernel.cu", "-o", f"{name}_kernel.o"], cwd=d, capture_output=True, text=True) + assert r.returncode == 0, r.stderr + # A plain C host TU links against the C++-compiled objects: the API + # has C linkage. With the stubbed runtime, init's upload and bind run + # on host memory and the "launch" is a serial call of the kernel body, + # so the batch must agree with the inline exactly. + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + (d / "host.c").write_text(f""" +#include "{name}.h" +int main(void) {{ double x[4 * {n_in}], y[4 * {n_out}], yb[4 * {n_out}]; + {'' if embed else f'if ({name}_init("none.rwt") != 1) return 2; if ({name}_init("{name}.rwt") != 0) return 3;'} + for (int c = 0; c < 4 * {n_in}; ++c) x[c] = 0.1 * c - 0.2; + for (int p = 0; p < 4; ++p) {name}_infer(x + p * {n_in}, y + p * {n_out}); + if ({name}_infer_batch(4, x, yb, 0)) return 4; + for (int c = 0; c < 4 * {n_out}; ++c) if (y[c] != yb[c]) return 5; + return {name}_infer_batch(0, x, yb, 0); }} +""") + cc = shutil.which("clang") or shutil.which("cc") or shutil.which("gcc") + r = subprocess.run([cc, "-std=c11", "-ffp-contract=off", "-c", "host.c", "-o", "host.o"], cwd=d, capture_output=True, text=True) + assert r.returncode == 0, r.stderr + r = subprocess.run([cxx, "host.o", f"{name}.o", f"{name}_kernel.o", "-lm", "-o", "host"], cwd=d, capture_output=True, text=True) + assert r.returncode == 0, r.stderr + r = subprocess.run(["./host"], cwd=d, capture_output=True, text=True) + assert r.returncode == 0, (r.returncode, r.stderr) + + +def test_recipe_selects_the_backend(golden_model): + plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64", embed=False) + mk = emit_c_recipe(plan) + assert "ROSENNA_BACKEND ?= omp" in mk + assert "ifeq ($(ROSENNA_BACKEND),cuda)" in mk and "else ifeq ($(ROSENNA_BACKEND),hip)" in mk + assert "DEVCC ?= nvcc" in mk and "DEVCC ?= hipcc" in mk + assert "-x cu -c $< -o $@" in mk and "-x hip -c $< -o $@" in mk + assert "gemm_small_kernel.o: gemm_small_kernel.cu gemm_small.h rosenna_rt.h" in mk + assert "$(CC) $(CFLAGS) $(ROSENNA_OFFLOAD_FLAGS) -c $< -o $@" in mk + + +def test_generate_writes_the_kernel_and_rt_header(tmp_path, golden_model): + from rosenna.cli import main + rc = main(["generate", str(golden_model("gemm_small")), "--lang", "c", "--out", str(tmp_path)]) + assert rc == 0 + for f in ["gemm_small.c", "gemm_small.h", "gemm_small.mk", "gemm_small_kernel.cu", "rosenna_rt.h"]: + assert (tmp_path / f).exists(), f + assert (tmp_path / "rosenna_rt.h").read_text() == rt_header() + rc = main(["generate", str(golden_model("gemm_small")), "--lang", "fortran", "--out", str(tmp_path / "f")]) + assert rc == 0 + assert not (tmp_path / "f" / "rosenna_rt.h").exists() + + +def _nvcc_build(d, name, plan): + source, header = emit_c(plan) + (d / f"{name}.c").write_text(source); (d / f"{name}.h").write_text(header) + (d / f"{name}_kernel.cu").write_text(emit_kernel(plan)); (d / "rosenna_rt.h").write_text(rt_header()) + (d / "Makefile").write_text(emit_c_recipe(plan)) + r = subprocess.run(["make", "ROSENNA_BACKEND=cuda", "DEVFLAGS=-O2 -arch=sm_80"], cwd=d, capture_output=True, text=True) + # nvcc's own diagnostics (warnings included) go to the CI log under -s: + # they are the only view of this code a CUDA compiler gives us. + print(f"\n--- nvcc {name} embed={plan.embed} dtype={plan.dtype} ---\n{r.stdout}{r.stderr}") + assert r.returncode == 0, r.stderr + assert (d / f"lib{name}.a").exists() + + +@pytest.mark.skipif(not shutil.which("nvcc"), reason="nvcc not installed") +def test_cuda_backend_compiles(tmp_path, golden_model): + # Compile only: nvcc builds device code with no GPU present. Running needs the GPU gate. + for embed in (True, False): + name = "gemm_small"; graph = load_graph(golden_model(name)); plan = build_plan(graph, dtype="f64", embed=embed) + d = tmp_path / ("e" if embed else "f"); d.mkdir() + _nvcc_build(d, name, plan) + + +@pytest.mark.skipif(not shutil.which("nvcc"), reason="nvcc not installed") +@pytest.mark.parametrize("name", ["gemm_big", "gemm_nobias", "droplet", "batchnet"]) +@pytest.mark.parametrize("dtype", ["f32", "f64"]) +@pytest.mark.parametrize("embed", [True, False]) +def test_cuda_backend_compiles_every_dense_model(tmp_path, golden_model, name, dtype, embed): + # The activations (tanhf/expf and their double forms) and every weight + # layout the plan can produce must also pass nvcc, not just gemm_small. + plan = build_plan(load_graph(golden_model(name)), dtype=dtype, embed=embed) + _nvcc_build(tmp_path, name, plan) + + +@pytest.mark.skipif(not shutil.which("nvcc"), reason="nvcc not installed") +def test_cuda_backend_compiles_past_the_constant_memory_budget(tmp_path): + # An embedded model over CONSTANT_MEMORY_LIMIT takes the __device__ const + # path (ruling R4); nvcc must accept that header too. + plan = _embedded_plan(tmp_path, "over", 64, 100) + assert "__device__ const" in emit_c(plan)[1] + d = tmp_path / "b"; d.mkdir() + _nvcc_build(d, "over", plan) From e2a0c4c463e640f1c33b82ce6496856922ab1ac5 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Sun, 13 Sep 2026 22:39:20 -0500 Subject: [PATCH 06/84] fix: bind per-TU tables from the header, check launches, tighten the no-transfer sweep Review round 1 of the batched kernel. The header now carries _device_bind_here(), the per-translation-unit bind of the __constant__ weight table, which the library's own device_bind forwards to and which any user kernel's translation unit must call after init (ruling R8). The host instantiation of an embedded model's __host__ __device__ infer is an assert stub in a CUDA/HIP build instead of a read of device storage (R9). Launches are checked with a peek, status 11 (R10). infer_batch's pointers are restrict-qualified (R7). The structural no-transfer test sweeps the whole .c outside init and its upload tail, with the OpenMP/OpenACC transfer tokens added (R6), and the stubbed C++ test now compiles the kernel unit with __CUDA_ARCH__ defined so infer reads through the table. Macros are #undef'd before definition; HIP_SYMBOL is explained. --- python/rosenna/abi.py | 3 +- python/rosenna/emit_c.py | 97 +++++++++++++--- python/rosenna/emit_kernel.py | 26 ++--- python/rosenna/rt_header.py | 4 + python/tests/test_kernel.py | 205 +++++++++++++++++++++------------- 5 files changed, 221 insertions(+), 114 deletions(-) diff --git a/python/rosenna/abi.py b/python/rosenna/abi.py index 6ac7b10..068f1ec 100644 --- a/python/rosenna/abi.py +++ b/python/rosenna/abi.py @@ -18,6 +18,7 @@ (8, "a name or rank in the weights file exceeds this model's capacity"), (9, "a read failed: the weights file is truncated or inconsistent"), (10, "device allocation or copy failed in init"), + (11, "kernel launch failed"), ] # Floors for the buffers `_init` declares to parse the table of @@ -39,7 +40,7 @@ def status_code_comment(prefix: str, model: str) -> list: this generator's Python to find out what it means, so the table is emitted next to the routine that returns it, in both languages. """ - lines = [f"{prefix} Status codes returned by {model}_init:"] + lines = [f"{prefix} Status codes ({model}_init; infer_batch returns 0, 10 or 11):"] lines += [f"{prefix} {code:>2} {text}" for code, text in STATUS_CODES] return lines diff --git a/python/rosenna/emit_c.py b/python/rosenna/emit_c.py index f25a756..7604332 100644 --- a/python/rosenna/emit_c.py +++ b/python/rosenna/emit_c.py @@ -227,8 +227,18 @@ def _emit_header(plan: Plan, ctype: str) -> str: "/* Generated by rosenna. Do not edit. */", "", "#include ", - "", ] + if plan.embed: + lines.append("#include ") + else: + lines += [ + "/* The runtime macros the per-translation-unit bind below needs; a", + " host C compiler never sees this include. */", + _CUDA_GUARD, + '#include "rosenna_rt.h"', + "#endif", + ] + lines.append("") lines += _emit_device_macros(plan) lines += [ "#if defined(__cplusplus)", @@ -257,10 +267,10 @@ def _emit_header(plan: Plan, ctype: str) -> str: " omp_target_alloc or from use_device_ptr on data it mapped (on a", " host-only build those are the host pointers and the loop runs on", " the CPU); stream is ignored.", - f" Returns 0, or 10 if a device allocation or copy failed in {m}_init", + f" Returns 0; 10 if a device allocation or copy failed in {m}_init", " (cuda/hip backend of a file-loaded model; init never having been", - " called counts as that). */", - f"int {m}_infer_batch(int n, const {ctype} *x, {ctype} *y, void *stream);", + " called counts as that); 11 if the kernel launch failed (cuda/hip). */", + f"int {m}_infer_batch(int n, const {ctype} *ROSENNA_RESTRICT x, {ctype} *ROSENNA_RESTRICT y, void *stream);", "", "#if defined(__cplusplus)", "}", @@ -300,6 +310,15 @@ def _emit_device_macros(plan: Plan) -> list: const_qual = "static __constant__" const_note = ("static __constant__ (unused: this model loads its weights", " from a file).") + if plan.embed: + stub_note = [ + " ROSENNA_INFER_HOST_STUB is 1 in the host pass of a CUDA/HIP build: the", + " embedded weights are device storage there, so the host instantiation", + " of infer is a stub that asserts (a no-op under NDEBUG); call infer from", + " a kernel, or use infer_batch. Any other compiler computes on the host.", + ] + else: + stub_note = [] return [ "/* Under nvcc/hipcc: ROSENNA_DEVICE_FN = __host__ __device__ (infer is", " callable from device code), ROSENNA_RESTRICT = __restrict__, and", @@ -310,14 +329,27 @@ def _emit_device_macros(plan: Plan) -> list: " __restrict__ in C++ or restrict in C. infer is separately wrapped in a", " guarded OpenMP declare-target region with a guarded OpenACC routine-seq", " pragma below; both are no-ops unless that compiler defines", - " _OPENMP/_OPENACC. */", + " _OPENMP/_OPENACC.", + *stub_note, + " The macros are redefined (#undef first) so that headers of several", + " models can share one translation unit without redefinition warnings. */", + "#undef ROSENNA_DEVICE_FN", + "#undef ROSENNA_CONST", + "#undef ROSENNA_RESTRICT", + "#undef ROSENNA_INFER_HOST_STUB", _CUDA_GUARD, "#define ROSENNA_DEVICE_FN __host__ __device__", f"#define ROSENNA_CONST {const_qual}", "#define ROSENNA_RESTRICT __restrict__", + *(["#if defined(__CUDA_ARCH__) || defined(__HIP_DEVICE_COMPILE__)", + "#define ROSENNA_INFER_HOST_STUB 0", + "#else", + "#define ROSENNA_INFER_HOST_STUB 1", + "#endif"] if plan.embed else ["#define ROSENNA_INFER_HOST_STUB 0"]), "#else", "#define ROSENNA_DEVICE_FN", "#define ROSENNA_CONST static const", + "#define ROSENNA_INFER_HOST_STUB 0", "#if defined(__cplusplus)", "#define ROSENNA_RESTRICT __restrict__", "#else", @@ -429,8 +461,8 @@ def _emit_device_weight_table(plan: Plan, ctype: str) -> list: table = _device_weight_table(m) lines = [ "/* Device-side view of the file-loaded weights: a per-translation-unit", - f" __constant__ table of the device pointers, filled once by {m}_init", - f" through {_device_bind(m)}. Device code reads its own translation", + f" __constant__ table of the device pointers, filled once, after {m}_init,", + f" by {_device_bind(m)}_here below. Device code reads its own translation", " unit's table; host code, under any compiler, reads the host arrays. */", _CUDA_GUARD, f"static __constant__ const {ctype} *{table}[{len(plan.weights)}];", @@ -443,6 +475,30 @@ def _emit_device_weight_table(plan: Plan, ctype: str) -> list: for w in plan.weights: lines.append(f"#define {_c_weight_ref_macro(m, w.symbol)} {_c_weight_symbol(m, w.symbol)}") lines += ["#endif", ""] + nw = len(plan.weights) + lines += [ + f"/* In a CUDA/HIP build, call {m}_device_bind_here() once after {m}_init()", + f" in every translation unit whose kernels call {m}_infer. Embedded", + " models need nothing. Each translation unit holds its own copy of the", + " table above; this fills the including translation unit's copy from", + f" the device addresses init made (lib{m}.a does it for its own kernel).", + " Part of the plan step: it transfers, so never call it from a loop.", + " Returns 0, or 10 if init has not made the device copies. */", + _CUDA_GUARD, + f"static inline int {_device_bind(m)}_here(void) {{", + f" const {ctype} *table[{nw}] = {{", + ] + for w in plan.weights: + lines.append(f" {_c_weight_symbol(m, w.symbol)}_dev,") + lines += [ + " };", + f" for (int k = 0; k < {nw}; ++k) if (table[k] == 0) return 10;", + f" if (ROSENNA_MEMCPY_TO_SYMBOL({table}, table, sizeof table) != ROSENNA_OK) return 10;", + " return 0;", + "}", + "#endif", + "", + ] return lines @@ -481,14 +537,6 @@ def _emit_source_head(plan: Plan, ctype: str) -> list: m = plan.model lines = [ "/* Generated by rosenna. Do not edit. */", - "", - "/* On a cuda/hip build the runtime is reached only through the macros in", - " rosenna_rt.h, included ahead of the model header so that the CUDA/HIP", - " declaration keywords the header uses are in scope; a host C compiler", - " never sees it. */", - _CUDA_GUARD, - '#include "rosenna_rt.h"', - "#endif", f'#include "{m}.h"', "", "#include ", @@ -537,7 +585,7 @@ def _emit_upload(plan: Plan) -> list: """The cuda/hip half of init: copy the freshly loaded host arrays to the device. Only compiled under nvcc/hipcc, where the runtime is reachable through - rosenna_rt.h. A repeated init frees the previous copies first (freeing a + rosenna_rt.h (included by the header under the same guard). A repeated init frees the previous copies first (freeing a null pointer is a no-op in both runtimes); a failed allocation or copy leaves that pointer null and returns 10, so a later infer_batch refuses to launch rather than read an unfilled buffer. The last step publishes @@ -584,7 +632,7 @@ def _emit_fallback_infer_batch(plan: Plan, ctype: str) -> list: n_in, n_out = plan.input.shape[0], plan.output.shape[0] return [ _NOT_CUDA_GUARD, - f"int {m}_infer_batch(int n, const {ctype} *x, {ctype} *y, void *stream) {{", + f"int {m}_infer_batch(int n, const {ctype} *ROSENNA_RESTRICT x, {ctype} *ROSENNA_RESTRICT y, void *stream) {{", " (void)stream;", " if (n <= 0) return 0;", "#if defined(_OPENMP)", @@ -727,6 +775,19 @@ def _emit_infer(plan: Plan, ctype: str) -> list: lines = [f"ROSENNA_DEVICE_FN static inline void {m}_infer(" f"const {ctype} *ROSENNA_RESTRICT x, {ctype} *ROSENNA_RESTRICT y) {{"] + if plan.embed: + # Controller ruling R9: in the host pass of a CUDA/HIP build the + # embedded arrays are device storage (nvcc diagnoses a direct read, + # hip-clang's host shadow is undefined), so that instantiation must + # not touch them. The literals are not duplicated for a host twin. + lines += [ + "#if ROSENNA_INFER_HOST_STUB", + " (void)x;", + " (void)y;", + f' assert(0 && "rosenna: {m}_infer is device-only in a CUDA/HIP build; ' + f'call it from a kernel or use {m}_infer_batch");', + "#else", + ] for sym in scratch: lines.append(f" {ctype} {sym}[{plan.buffers[sym]}];") @@ -751,6 +812,8 @@ def _emit_infer(plan: Plan, ctype: str) -> list: else: raise AssertionError(f"unhandled op kind {op.kind!r}") + if plan.embed: + lines.append("#endif") lines.append("}") lines.append("") return lines diff --git a/python/rosenna/emit_kernel.py b/python/rosenna/emit_kernel.py index 826186d..05e7f75 100644 --- a/python/rosenna/emit_kernel.py +++ b/python/rosenna/emit_kernel.py @@ -11,7 +11,7 @@ omp backend) only; the device path is unvalidated until nvcc has built it on CI and the GPU gate has run it. """ -from .emit_c import KERNEL_TILE, _CTYPE, _c_weight_symbol, _device_bind, _device_weight_table +from .emit_c import KERNEL_TILE, _CTYPE, _c_weight_symbol, _device_bind from .plan import Plan @@ -30,24 +30,13 @@ def emit_kernel(plan: Plan) -> str: "", ] if not plan.embed and plan.weights: - table = _device_weight_table(m) - nw = len(plan.weights) lines += [ "/* Plan step (controller ruling R5), called by init after it has made", - " the device copies: publishes their addresses to this translation", - " unit's __constant__ table, the one the kernel below reads. This is", - " the only transfer in this file; infer_batch launches and nothing", - " else. */", + " the device copies: binds this translation unit's __constant__ table,", + " the one the kernel below reads, through the header's", + f" {_device_bind(m)}_here. Nothing else in this file transfers. */", f'extern "C" int {_device_bind(m)}(void) {{', - f" const {ctype} *table[{nw}] = {{", - ] - for w in plan.weights: - lines.append(f" {_c_weight_symbol(m, w.symbol)}_dev,") - lines += [ - " };", - f" for (int k = 0; k < {nw}; ++k) if (table[k] == 0) return 10;", - f" if (ROSENNA_MEMCPY_TO_SYMBOL({table}, table, sizeof table) != ROSENNA_OK) return 10;", - " return 0;", + f" return {_device_bind(m)}_here();", "}", "", ] @@ -61,7 +50,7 @@ def emit_kernel(plan: Plan) -> str: f" {m}_infer(x + (size_t)p * {n_in}, y + (size_t)p * {n_out});", "}", "", - f'extern "C" int {m}_infer_batch(int n, const {ctype} *x, {ctype} *y, void *stream) {{', + f'extern "C" int {m}_infer_batch(int n, const {ctype} *__restrict__ x, {ctype} *__restrict__ y, void *stream) {{', " if (n <= 0) return 0;", " const ROSENNA_STREAM_T s = (ROSENNA_STREAM_T)stream;", ] @@ -72,6 +61,9 @@ def emit_kernel(plan: Plan) -> str: lines += [ " const int grid = (n + ROSENNA_TILE - 1) / ROSENNA_TILE;", f" ROSENNA_LAUNCH({m}_kernel, grid, ROSENNA_TILE, s, n, x, y);", + " /* A peek, not a sync (ruling R5): a bad configuration or stream is", + " reported now; asynchronous faults surface at the caller's sync. */", + " if (ROSENNA_LAUNCH_STATUS() != ROSENNA_OK) return 11;", " return 0;", "}", "", diff --git a/python/rosenna/rt_header.py b/python/rosenna/rt_header.py index 6283c04..af20dc2 100644 --- a/python/rosenna/rt_header.py +++ b/python/rosenna/rt_header.py @@ -17,12 +17,15 @@ #define ROSENNA_STREAM_T hipStream_t #define ROSENNA_MALLOC(p, n) hipMalloc((void **)(p), (n)) #define ROSENNA_MEMCPY_H2D(d, h, n) hipMemcpy((d), (h), (n), hipMemcpyHostToDevice) +/* HIP_SYMBOL is how HIP spells a symbol argument portably: it expands to X + on hip-clang and to &X on the retired hcc path. */ #define ROSENNA_MEMCPY_TO_SYMBOL(sym, src, n) \\ hipMemcpyToSymbol(HIP_SYMBOL(sym), (src), (n), 0, hipMemcpyHostToDevice) #define ROSENNA_FREE(p) hipFree(p) #define ROSENNA_OK hipSuccess #define ROSENNA_SYNC(s) hipStreamSynchronize(s) #define ROSENNA_LAUNCH(k, g, b, s, ...) k<<<(g), (b), 0, (s)>>>(__VA_ARGS__) +#define ROSENNA_LAUNCH_STATUS() hipPeekAtLastError() #elif defined(__CUDACC__) #include #define ROSENNA_STREAM_T cudaStream_t @@ -34,6 +37,7 @@ #define ROSENNA_OK cudaSuccess #define ROSENNA_SYNC(s) cudaStreamSynchronize(s) #define ROSENNA_LAUNCH(k, g, b, s, ...) k<<<(g), (b), 0, (s)>>>(__VA_ARGS__) +#define ROSENNA_LAUNCH_STATUS() cudaPeekAtLastError() #else #error "rosenna_rt.h is for nvcc or hipcc only" #endif diff --git a/python/tests/test_kernel.py b/python/tests/test_kernel.py index 5a0367a..b13222e 100644 --- a/python/tests/test_kernel.py +++ b/python/tests/test_kernel.py @@ -29,8 +29,10 @@ def test_kernel_source_names_no_runtime_symbol_for_a_file_loaded_plan(golden_mod cu = emit_kernel(plan) assert re.search(r"\b(cuda|hip)[A-Z]", cu) is None assert 'extern "C" int gemm_big_device_bind(void) {' in cu - assert "ROSENNA_MEMCPY_TO_SYMBOL(gemm_big_devw, table, sizeof table)" in cu + assert "return gemm_big_device_bind_here();" in cu assert "ROSENNA_LAUNCH(gemm_big_kernel" in cu + # Ruling R10: the launch is checked with a peek (never a sync), status 11. + assert "if (ROSENNA_LAUNCH_STATUS() != ROSENNA_OK) return 11;" in cu # An embedded plan reads its ROSENNA_CONST arrays directly and binds nothing. cu_e = emit_kernel(build_plan(load_graph(golden_model("gemm_big")), dtype="f64", embed=True)) assert "ROSENNA_MEMCPY_TO_SYMBOL" not in cu_e and "device_bind" not in cu_e @@ -46,46 +48,41 @@ def _function_body(text: str, signature_start: str) -> str: _LOOP_PATH_FORBIDDEN = ("map(to", "map(from", "map(tofrom", "copyin", "copyout", "ROSENNA_MALLOC", "ROSENNA_MEMCPY", "ROSENNA_SYNC", "cudaMemcpy", "hipMemcpy", "cudaMalloc", "hipMalloc", - "cudaDeviceSynchronize", "hipDeviceSynchronize") + "cudaDeviceSynchronize", "hipDeviceSynchronize", + "target update", "update device", "enter data", "exit data", + "omp_target_memcpy", "acc_memcpy") def test_loop_path_never_transfers(golden_model): - # Controller ruling R5: init is the plan step and the only routine that + # Controller rulings R5/R6: init is the plan step and the only routine that # allocates, transfers or synchronizes; x and y are device-resident in # every backend. No transfer can be observed on a host-only build, so - # this structural check is the CI-able guarantee. Scope: the omp-backend - # infer_batch body in the .c (init is exempt), and in the kernel file the - # kernel and infer_batch bodies. The kernel file's third function, - # _device_bind, is the tail of the plan step: a __constant__ symbol - # can be written only from its own translation unit without relocatable - # device code, so init reaches into the kernel's file to publish the - # weight addresses. It is exempt like init, and checked below to be the - # one place in that file a transfer name appears. + # this structural check is the CI-able guarantee. Exempt: the body of + # `_init` and of `_upload`, the static tail of init that + # holds the cuda/hip allocation and copies (init calls it and nothing + # else does). Everything else in the .c, and the whole kernel file, must + # be free of every transfer token. The kernel file's `_device_bind` + # only forwards to the header's `_device_bind_here`, which is where + # the one symbol copy of the plan step lives; that header function is + # checked here to be the sole holder of ROSENNA_MEMCPY_TO_SYMBOL. for embed in (True, False): name = "gemm_big"; plan = build_plan(load_graph(golden_model(name)), dtype="f64", embed=embed) - source, _ = emit_c(plan) + source, header = emit_c(plan) cu = emit_kernel(plan) - loop_path = [ - _function_body(source, f"int {name}_infer_batch("), - _function_body(cu, f"static __global__ void {name}_kernel("), - _function_body(cu, f'extern "C" int {name}_infer_batch('), - ] - for body in loop_path: - for forbidden in _LOOP_PATH_FORBIDDEN: - assert forbidden not in body, (embed, forbidden, body) - rest = cu - for body in loop_path[1:]: - rest = rest.replace(body, "") - if embed: - for forbidden in _LOOP_PATH_FORBIDDEN: - assert forbidden not in rest, (forbidden, rest) - else: - bind = _function_body(cu, f'extern "C" int {name}_device_bind(') - rest = rest.replace(bind, "") - for forbidden in _LOOP_PATH_FORBIDDEN: - assert forbidden not in rest, (forbidden, rest) - assert "ROSENNA_MEMCPY_TO_SYMBOL" in bind - assert f"return {name}_device_bind();" in _function_body(source, f"static int {name}_upload(") + rest_c = source + if not embed: + init = _function_body(source, f"int {name}_init(") + upload = _function_body(source, f"static int {name}_upload(") + assert f"return {name}_upload();" in init and "ROSENNA_MALLOC" in upload + rest_c = rest_c.replace(init, "").replace(upload, "") + for forbidden in _LOOP_PATH_FORBIDDEN: + assert forbidden not in rest_c, (embed, forbidden) + assert forbidden not in cu, (embed, forbidden) + if not embed: + bind_here = _function_body(header, f"static inline int {name}_device_bind_here(") + assert "ROSENNA_MEMCPY_TO_SYMBOL" in bind_here + assert header.count("ROSENNA_MEMCPY_TO_SYMBOL") == 1 + assert f"return {name}_device_bind_here();" in _function_body(cu, f'extern "C" int {name}_device_bind(') def test_rt_header_maps_both_runtimes(): @@ -93,20 +90,26 @@ def test_rt_header_maps_both_runtimes(): assert "__HIPCC__" in h and "__CUDACC__" in h and "ROSENNA_LAUNCH" in h # Every macro the generated sources use is defined once per runtime. for macro in ("ROSENNA_STREAM_T", "ROSENNA_MALLOC", "ROSENNA_MEMCPY_H2D", "ROSENNA_FREE", - "ROSENNA_OK", "ROSENNA_SYNC", "ROSENNA_LAUNCH", "ROSENNA_MEMCPY_TO_SYMBOL"): - assert h.count(f"#define {macro}") == 2, macro + "ROSENNA_OK", "ROSENNA_SYNC", "ROSENNA_LAUNCH", "ROSENNA_MEMCPY_TO_SYMBOL", + "ROSENNA_LAUNCH_STATUS"): + assert h.count(f"#define {macro}(") + h.count(f"#define {macro} ") == 2, macro def test_header_declares_infer_batch_with_c_linkage_on_both_forms(golden_model): for embed in (True, False): plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64", embed=embed) source, header = emit_c(plan) - assert "int gemm_small_infer_batch(int n, const double *x, double *y, void *stream);" in header + # Ruling R7: the batch's pointers are restrict-qualified. + assert ("int gemm_small_infer_batch(int n, const double *ROSENNA_RESTRICT x, " + "double *ROSENNA_RESTRICT y, void *stream);") in header assert 'extern "C" {' in header and "__cplusplus" in header assert "x and y must already be on\n the device; init is the only routine that transfers." in header # The OpenMP fallback lives in the .c under the negation of the CUDA/HIP # guard, over device pointers (ruling R5), each pragma under its own guard. - assert "int gemm_small_infer_batch(int n, const double *x, double *y, void *stream) {" in source + assert ("int gemm_small_infer_batch(int n, const double *ROSENNA_RESTRICT x, " + "double *ROSENNA_RESTRICT y, void *stream) {") in source + assert ('extern "C" int gemm_small_infer_batch(int n, const double *__restrict__ x, ' + "double *__restrict__ y, void *stream) {") in emit_kernel(plan) assert "#if !defined(__CUDACC__) && !defined(__HIPCC__)" in source assert "#if defined(_OPENMP)\n#pragma omp target teams loop is_device_ptr(x, y)\n" in source assert "#elif defined(_OPENACC)\n#pragma acc parallel loop deviceptr(x, y)\n#endif" in source @@ -118,10 +121,17 @@ def test_file_loaded_source_copies_to_the_device_under_the_cuda_guard(golden_mod assert "double *gemm_small_w0_dev = 0;" in source assert "ROSENNA_MALLOC(&gemm_small_w0_dev, sizeof gemm_small_w0)" in source assert "ROSENNA_MEMCPY_H2D(gemm_small_w0_dev, gemm_small_w0, sizeof gemm_small_w0)" in source - assert "return 10;" in source and '#include "rosenna_rt.h"' in source - # init ends by publishing the copies to the kernel's translation unit. + assert "return 10;" in source and '#include "rosenna_rt.h"' not in source # the header includes it + # init ends by publishing the copies to the kernel's translation unit, + # through the header's per-translation-unit bind (ruling R8), which any + # user kernel's translation unit must call as well. assert "return gemm_small_device_bind();" in source assert "int gemm_small_device_bind(void);" in header + assert "static inline int gemm_small_device_bind_here(void) {" in header + assert ("call gemm_small_device_bind_here() once after gemm_small_init()\n" + " in every translation unit whose kernels call gemm_small_infer. Embedded\n" + " models need nothing.") in header + assert '#include "rosenna_rt.h"' in header # Under an offloading OpenMP/OpenACC build the host arrays have device # copies that init updates in the same call (the plan step); each # directive under its own guard. @@ -167,6 +177,27 @@ def test_generated_c_is_warning_free_under_openacc(tmp_path, golden_model): assert "__CUDA_ARCH__" in header and "__HIP_DEVICE_COMPILE__" in header +def test_embedded_infer_is_a_stub_in_the_host_pass_of_a_device_build(golden_model): + # Ruling R9: the host instantiation of __host__ __device__ infer must not + # read the __constant__/__device__ arrays (nvcc diagnoses it; hip-clang's + # host shadow is undefined). It asserts instead, a no-op under NDEBUG. + # The literals are emitted once; there is no host twin. + plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64", embed=True) + _, header = emit_c(plan) + assert "#include " in header + assert "#if defined(__CUDA_ARCH__) || defined(__HIP_DEVICE_COMPILE__)\n#define ROSENNA_INFER_HOST_STUB 0\n#else\n#define ROSENNA_INFER_HOST_STUB 1\n#endif" in header + assert ('#if ROSENNA_INFER_HOST_STUB\n (void)x;\n (void)y;\n assert(0 && "rosenna: gemm_small_infer ' + 'is device-only in a CUDA/HIP build; call it from a kernel or use gemm_small_infer_batch");\n#else') in header + assert header.count("gemm_small_w0[4] = {") == 1 + assert "host instantiation\n of infer is a stub" in header + # A file-loaded plan computes on the host in every pass (it reads the host arrays). + _, header_f = emit_c(build_plan(load_graph(golden_model("gemm_small")), dtype="f64", embed=False)) + assert "ROSENNA_INFER_HOST_STUB 1" not in header_f and "assert(" not in header_f + # Both headers can share one translation unit: every macro is #undef'd first. + for macro in ("ROSENNA_DEVICE_FN", "ROSENNA_CONST", "ROSENNA_RESTRICT", "ROSENNA_INFER_HOST_STUB"): + assert f"#undef {macro}" in header and f"#undef {macro}" in header_f + + def _embedded_plan(tmp_path, name, n_in, n_out): w = numpy_helper.from_array(np.random.default_rng(1).uniform(-1, 1, (n_in, n_out)).astype(np.float32), "w") node = helper.make_node("MatMul", ["x", "w"], ["y"], name="m0") @@ -304,6 +335,15 @@ def test_generated_c_compiles_as_cpp_with_the_rt_header_stubbed(tmp_path, golden # between the header's extern "C" block and the definitions, and any use # of a runtime name outside rosenna_rt.h. It proves nothing about device # code generation; that waits for the nvcc CI job. + # + # Two variants of the kernel translation unit. With __CUDA_ARCH__ defined + # (as in nvcc's device pass) infer reads the file-loaded weights through + # the per-translation-unit table that device_bind_here fills, and the + # embedded body is the real one: init -> upload -> bind -> launch must + # agree bit for bit with the inline, and an unbound table is status 10. + # Without it (nvcc's host pass) the file-loaded kernel reads the host + # arrays and the embedded kernel is the R9 stub, so that variant is + # compiled and linked, and run only for the file-loaded plan. cxx = shutil.which("clang++") or shutil.which("g++") if cxx is None: pytest.skip("no C++ compiler found") @@ -327,58 +367,65 @@ def test_generated_c_compiles_as_cpp_with_the_rt_header_stubbed(tmp_path, golden #define ROSENNA_SYNC(s) rosenna_stub_sync(s) #define ROSENNA_LAUNCH(k, g, b, s, ...) \\ do { for (blockIdx.x = 0; blockIdx.x < (unsigned)((g) * (b)); ++blockIdx.x) k(__VA_ARGS__); } while (0) +#define ROSENNA_LAUNCH_STATUS() 0 #endif """ from rosenna.weights import write_weights for embed in (True, False): - name = "gemm_small"; graph = load_graph(golden_model(name)); plan = build_plan(graph, dtype="f64", embed=embed) - d = tmp_path / ("e" if embed else "f"); d.mkdir() - source, header = emit_c(plan) - (d / f"{name}.c").write_text(source); (d / f"{name}.h").write_text(header) - (d / "rosenna_rt.h").write_text(stub) - if not embed: - write_weights(plan, graph, d / f"{name}.rwt") - (d / f"{name}_kernel.cu").write_text(emit_kernel(plan)) - # blockIdx/blockDim/threadIdx are CUDA builtins; give the stub the - # three as plain objects so the kernel body parses, with one thread - # per block so the stub launch's loop over blockIdx.x walks the points. - (d / "builtins.h").write_text( - "struct rosenna_dim3 { unsigned int x, y, z; };\n" - "static struct rosenna_dim3 blockIdx = {0, 0, 0};\n" - "static const struct rosenna_dim3 blockDim = {1, 0, 0}, threadIdx = {0, 0, 0};\n") - # -ffp-contract=off on every translation unit: the host may be built - # by a different compiler than the library, and clang contracts - # `acc += a * b` to an FMA by default where g++ in ISO mode does not, - # which breaks the bit-for-bit comparison below for no real reason. - common = [cxx, "-x", "c++", "-std=c++11", "-ffp-contract=off", "-Wall", "-Wextra", "-c", - "-D__CUDACC__=1", "-D__host__=", "-D__device__=", "-D__constant__=", "-D__global__=", - "-include", "builtins.h"] - r = subprocess.run(common + [f"{name}.c", "-o", f"{name}.o"], cwd=d, capture_output=True, text=True) - assert r.returncode == 0, r.stderr - r = subprocess.run(common + [f"{name}_kernel.cu", "-o", f"{name}_kernel.o"], cwd=d, capture_output=True, text=True) - assert r.returncode == 0, r.stderr - # A plain C host TU links against the C++-compiled objects: the API - # has C linkage. With the stubbed runtime, init's upload and bind run - # on host memory and the "launch" is a serial call of the kernel body, - # so the batch must agree with the inline exactly. - n_in, n_out = plan.input.shape[0], plan.output.shape[0] - (d / "host.c").write_text(f""" + for arch in (None, "800"): + name = "gemm_small"; graph = load_graph(golden_model(name)); plan = build_plan(graph, dtype="f64", embed=embed) + d = tmp_path / f"{'e' if embed else 'f'}{arch or ''}"; d.mkdir() + source, header = emit_c(plan) + (d / f"{name}.c").write_text(source); (d / f"{name}.h").write_text(header) + (d / "rosenna_rt.h").write_text(stub) + if not embed: + write_weights(plan, graph, d / f"{name}.rwt") + (d / f"{name}_kernel.cu").write_text(emit_kernel(plan)) + # blockIdx/blockDim/threadIdx are CUDA builtins; give the stub the + # three as plain objects so the kernel body parses, with one thread + # per block so the stub launch's loop over blockIdx.x walks the points. + (d / "builtins.h").write_text( + "struct rosenna_dim3 { unsigned int x, y, z; };\n" + "static struct rosenna_dim3 blockIdx = {0, 0, 0};\n" + "static const struct rosenna_dim3 blockDim = {1, 0, 0}, threadIdx = {0, 0, 0};\n") + # -ffp-contract=off on every translation unit: the host may be built + # by a different compiler than the library, and clang contracts + # `acc += a * b` to an FMA by default where g++ in ISO mode does not, + # which breaks the bit-for-bit comparison below for no real reason. + common = [cxx, "-x", "c++", "-std=c++11", "-ffp-contract=off", "-Wall", "-Wextra", "-c", + "-D__CUDACC__=1", "-D__host__=", "-D__device__=", "-D__constant__=", "-D__global__=", + "-include", "builtins.h"] + r = subprocess.run(common + [f"{name}.c", "-o", f"{name}.o"], cwd=d, capture_output=True, text=True) + assert r.returncode == 0, r.stderr + arch_flag = [f"-D__CUDA_ARCH__={arch}"] if arch else [] + r = subprocess.run(common + arch_flag + [f"{name}_kernel.cu", "-o", f"{name}_kernel.o"], cwd=d, capture_output=True, text=True) + assert r.returncode == 0, r.stderr + # A plain C host TU links against the C++-compiled objects: the API + # has C linkage. With the stubbed runtime, init's upload and bind run + # on host memory and the "launch" is a serial call of the kernel body. + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + unbound = "" if embed else f"if ({name}_infer_batch(4, x, yb, 0) != 10) return 6;" + init = "" if embed else f'if ({name}_init("none.rwt") != 1) return 2; if ({name}_init("{name}.rwt") != 0) return 3;' + (d / "host.c").write_text(f""" #include "{name}.h" int main(void) {{ double x[4 * {n_in}], y[4 * {n_out}], yb[4 * {n_out}]; - {'' if embed else f'if ({name}_init("none.rwt") != 1) return 2; if ({name}_init("{name}.rwt") != 0) return 3;'} for (int c = 0; c < 4 * {n_in}; ++c) x[c] = 0.1 * c - 0.2; + {unbound} + {init} for (int p = 0; p < 4; ++p) {name}_infer(x + p * {n_in}, y + p * {n_out}); if ({name}_infer_batch(4, x, yb, 0)) return 4; for (int c = 0; c < 4 * {n_out}; ++c) if (y[c] != yb[c]) return 5; return {name}_infer_batch(0, x, yb, 0); }} """) - cc = shutil.which("clang") or shutil.which("cc") or shutil.which("gcc") - r = subprocess.run([cc, "-std=c11", "-ffp-contract=off", "-c", "host.c", "-o", "host.o"], cwd=d, capture_output=True, text=True) - assert r.returncode == 0, r.stderr - r = subprocess.run([cxx, "host.o", f"{name}.o", f"{name}_kernel.o", "-lm", "-o", "host"], cwd=d, capture_output=True, text=True) - assert r.returncode == 0, r.stderr - r = subprocess.run(["./host"], cwd=d, capture_output=True, text=True) - assert r.returncode == 0, (r.returncode, r.stderr) + cc = shutil.which("clang") or shutil.which("cc") or shutil.which("gcc") + r = subprocess.run([cc, "-std=c11", "-ffp-contract=off", "-c", "host.c", "-o", "host.o"], cwd=d, capture_output=True, text=True) + assert r.returncode == 0, r.stderr + r = subprocess.run([cxx, "host.o", f"{name}.o", f"{name}_kernel.o", "-lm", "-o", "host"], cwd=d, capture_output=True, text=True) + assert r.returncode == 0, r.stderr + if embed and arch is None: + continue # the kernel would hit the R9 host stub's assert + r = subprocess.run(["./host"], cwd=d, capture_output=True, text=True) + assert r.returncode == 0, (embed, arch, r.returncode, r.stderr) def test_recipe_selects_the_backend(golden_model): From fece60cfb9ea9be6107cba278631446dfbfce38f Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Sun, 13 Sep 2026 22:56:23 -0500 Subject: [PATCH 07/84] fix: use non-sticky GetLastError and clarify device_bind_here's per-init contract ROSENNA_LAUNCH_STATUS() now maps to cudaGetLastError()/hipGetLastError() instead of the Peek variants: Peek leaves a stale error in the thread slot, so every later call after one real failure would report status 11 forever, while Get reports once and clears (both stay non-synchronizing). Also fixes the device_bind_here header comment, which said "once after init" -- a repeated init reallocates the device buffers, so every call to init needs its own device_bind_here to follow it. --- python/rosenna/emit_c.py | 2 +- python/rosenna/rt_header.py | 4 ++-- python/tests/test_kernel.py | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/python/rosenna/emit_c.py b/python/rosenna/emit_c.py index 7604332..eb8acbd 100644 --- a/python/rosenna/emit_c.py +++ b/python/rosenna/emit_c.py @@ -477,7 +477,7 @@ def _emit_device_weight_table(plan: Plan, ctype: str) -> list: lines += ["#endif", ""] nw = len(plan.weights) lines += [ - f"/* In a CUDA/HIP build, call {m}_device_bind_here() once after {m}_init()", + f"/* In a CUDA/HIP build, call {m}_device_bind_here() after EVERY call to {m}_init()", f" in every translation unit whose kernels call {m}_infer. Embedded", " models need nothing. Each translation unit holds its own copy of the", " table above; this fills the including translation unit's copy from", diff --git a/python/rosenna/rt_header.py b/python/rosenna/rt_header.py index af20dc2..a6b1064 100644 --- a/python/rosenna/rt_header.py +++ b/python/rosenna/rt_header.py @@ -25,7 +25,7 @@ #define ROSENNA_OK hipSuccess #define ROSENNA_SYNC(s) hipStreamSynchronize(s) #define ROSENNA_LAUNCH(k, g, b, s, ...) k<<<(g), (b), 0, (s)>>>(__VA_ARGS__) -#define ROSENNA_LAUNCH_STATUS() hipPeekAtLastError() +#define ROSENNA_LAUNCH_STATUS() hipGetLastError() #elif defined(__CUDACC__) #include #define ROSENNA_STREAM_T cudaStream_t @@ -37,7 +37,7 @@ #define ROSENNA_OK cudaSuccess #define ROSENNA_SYNC(s) cudaStreamSynchronize(s) #define ROSENNA_LAUNCH(k, g, b, s, ...) k<<<(g), (b), 0, (s)>>>(__VA_ARGS__) -#define ROSENNA_LAUNCH_STATUS() cudaPeekAtLastError() +#define ROSENNA_LAUNCH_STATUS() cudaGetLastError() #else #error "rosenna_rt.h is for nvcc or hipcc only" #endif diff --git a/python/tests/test_kernel.py b/python/tests/test_kernel.py index b13222e..bfb3ae1 100644 --- a/python/tests/test_kernel.py +++ b/python/tests/test_kernel.py @@ -128,7 +128,7 @@ def test_file_loaded_source_copies_to_the_device_under_the_cuda_guard(golden_mod assert "return gemm_small_device_bind();" in source assert "int gemm_small_device_bind(void);" in header assert "static inline int gemm_small_device_bind_here(void) {" in header - assert ("call gemm_small_device_bind_here() once after gemm_small_init()\n" + assert ("call gemm_small_device_bind_here() after EVERY call to gemm_small_init()\n" " in every translation unit whose kernels call gemm_small_infer. Embedded\n" " models need nothing.") in header assert '#include "rosenna_rt.h"' in header From 352c14a4bd85e14e9c66c603fcb6d21921c11a0d Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Sun, 13 Sep 2026 23:08:20 -0500 Subject: [PATCH 08/84] feat: fortran device directives, embedded parameter weights, device copy in init, infer_batch Gives the Fortran module the same device contract Tasks 1-3 gave the C header: infer carries !$omp declare target / !$acc routine seq; an embedded plan's weights are parameter arrays (reshape'd from a flat, C-order literal list so the memory layout matches the file-loaded `protected` form) with no init at all; a file-loaded plan's init makes its weights device-resident with !$omp target update to / !$acc update device right after the load, in the same host call (Fortran has no runtime-API path without CUDA Fortran). A new infer_batch(n, x, y, status) is the OpenMP-target fallback over device-resident arrays (has_device_addr(x, y), no map/copyin/copyout/update -- ruling R5), verified accepted on an explicit-shape dummy under gfortran 15 with no fallback needed. A declared-only bind(C) interface exposes Task 3's native _infer_batch to Fortran hosts holding device pointers. emit_fortran_recipe/_fortran.mk mirrors the C recipe, building lib.a from _model.o. A weight tensor large enough to need more than 255 continuation lines in one array constructor (gemm_big's 1200-element layer) is split into several <=500-element parameter chunks and reassembled with a second, short reshape statement, since F2008's per-statement continuation cap is independent of the 132-column limit already handled. Also flips ruling R3 (Task 2): now that Fortran embeds by default too, `generate` writes a .rwt only for embed=False plans, in every language; verify.py's Fortran build now compiles the module to an object, archives it, and links the driver against the archive, matching the C backend's library-form path. Existing Fortran tests that are specifically about the file-loaded contract (`_init`, `protected` declarations, the load_tensor case-label select) now build with embed=False explicitly, mirroring the same fix already applied on the C side when embedding was introduced there. --- python/rosenna/cli.py | 17 ++- python/rosenna/emit_fortran.py | 197 +++++++++++++++++++++++++--- python/rosenna/verify.py | 38 ++++-- python/tests/test_cli.py | 37 ++++-- python/tests/test_cli_smoke.py | 4 +- python/tests/test_device_fortran.py | 159 ++++++++++++++++++++++ python/tests/test_emit_fortran.py | 20 ++- python/tests/test_regressions.py | 4 +- 8 files changed, 423 insertions(+), 53 deletions(-) create mode 100644 python/tests/test_device_fortran.py diff --git a/python/rosenna/cli.py b/python/rosenna/cli.py index 44e4c0e..586ffa1 100644 --- a/python/rosenna/cli.py +++ b/python/rosenna/cli.py @@ -7,7 +7,7 @@ from google.protobuf.message import DecodeError from .emit_c import emit_c, emit_c_recipe -from .emit_fortran import emit_fortran +from .emit_fortran import emit_fortran, emit_fortran_recipe from .emit_kernel import emit_kernel from .rt_header import rt_header from .frontend import UnsupportedModel, load_graph @@ -84,8 +84,10 @@ def _cmd_generate(args) -> int: written = [] if "fortran" in langs: f90_path = outdir / f"{name}_model.f90" + fmk_path = outdir / f"{name}_fortran.mk" f90_path.write_text(emit_fortran(plan)) - written.append(f90_path) + fmk_path.write_text(emit_fortran_recipe(plan)) + written += [f90_path, fmk_path] if "c" in langs: source, header = emit_c(plan) recipe = emit_c_recipe(plan) @@ -103,11 +105,12 @@ def _cmd_generate(args) -> int: rt_path.write_text(rt_header()) written += [c_path, h_path, mk_path, cu_path, rt_path] - # An embedded plan has no weights file to write: every weight is already a - # ROSENNA_CONST array baked into the header. Fortran generation (unchanged - # by this task) still needs a .rwt to load, so it is written whenever - # Fortran is one of the requested languages even if the plan embeds. - if plan.embed and "fortran" not in langs: + # An embedded plan has no weights file to write in either language: every + # weight is already a `parameter`/ROSENNA_CONST array baked into the + # generated source (controller ruling R3, flipped by Task 4: Fortran now + # embeds by default too, so this no longer depends on which languages + # were requested). + if plan.embed: print(f"embedded weights ({plan.n_params} parameters)") else: rwt_path = outdir / f"{name}.rwt" diff --git a/python/rosenna/emit_fortran.py b/python/rosenna/emit_fortran.py index 0278c6f..1a71c83 100644 --- a/python/rosenna/emit_fortran.py +++ b/python/rosenna/emit_fortran.py @@ -10,6 +10,13 @@ "sigmoid": "1.0_wp / (1.0_wp + exp(-({v})))"} _DTYPE_CODE = {"f32": 0, "f64": 1} +# Embedding must be lossless: %.16e (17 significant digits: one before the +# point, sixteen after) round-trips any f64, %.8e (9 significant digits) any +# f32 -- the same Steele & White / Ryu-style bounds emit_c's %.17g/%.9g rely +# on, spelled as a fixed-width Fortran-legal exponential literal instead of +# C's shortest-form %g (which drops trailing digits %e always keeps). +_EMBED_FMT = {"f32": "%.8e", "f64": "%.16e"} + # Free-form Fortran source is limited to 132 columns (F2008 3.3.2.1). gfortran # 15 accepts a longer line silently, gfortran 13 rejects it with # -Werror=line-truncation, and every gfortran rejects it under -std=f2008, so @@ -70,34 +77,126 @@ def _weight_dims(plan: Plan, symbol: str) -> str: return "(" + ",".join(str(d) for d in reversed(spec.shape)) + ")" +def _weight_dims_list(plan: Plan, symbol: str) -> str: + spec = next(w for w in plan.weights if w.symbol == symbol) + return "[" + ",".join(str(d) for d in reversed(spec.shape)) + "]" + + +def _format_embedded_value(v: float, dtype: str) -> str: + return (_EMBED_FMT[dtype] % v) + "_wp" + + +def _weight_symbol_list(plan: Plan) -> str: + return ", ".join(w.symbol for w in plan.weights) + + +# F2008 free-form source caps a single statement at 255 continuation lines +# (3.3.2.4), independently of the 132-column limit each of those lines +# obeys: gfortran diagnoses an over-long array constructor with "Warning: +# Limit of 255 continuations exceeded", not a column complaint, so +# _wrap_items alone cannot keep a several-hundred-element weight (e.g. +# gemm_big's 40x30 = 1200-element layer) legal. Above _EMBED_CHUNK elements, +# the flat literal list is split into several small `parameter` arrays +# (each well under the continuation limit) and reassembled with one more +# `parameter` statement over their names -- a statement with a handful of +# short identifiers, never close to either limit itself. +_EMBED_CHUNK = 500 + + +def _emit_embedded_weights(plan: Plan) -> list: + """`plan.embed`'s weights, as `parameter` arrays holding the literal values. + + A rank-1 array (every bias) is a plain bracketed list. A rank-2 array + (every Gemm/MatMul weight) is `reshape([flat values], [dims])`: `values` + is already the C-order (row-major) flattening of the ONNX-shaped tensor + (plan.py's `np.ravel(order="C")`), and Fortran's default array + constructor fills its target in column-major order, so reshaping that + same flat sequence into the *reversed* shape reproduces exactly the + memory layout `_weight_dims` already declares for the file-loaded form + -- the same raw-bytes-in trick `load_tensor` relies on, done here at + compile time instead of at file-read time. + """ + lines = [] + for w in plan.weights: + dims = _weight_dims(plan, w.symbol) + items = [_format_embedded_value(v, plan.dtype) for v in w.values] + if len(items) <= _EMBED_CHUNK: + flat = items + else: + chunks = [items[i:i + _EMBED_CHUNK] for i in range(0, len(items), _EMBED_CHUNK)] + chunk_names = [f"{w.symbol}_c{k}" for k in range(len(chunks))] + for cname, chunk in zip(chunk_names, chunks): + lines += _wrap_items(f" real(wp), parameter :: {cname}({len(chunk)}) = [ ", + chunk, " ]", " " * 8) + flat = chunk_names + if len(w.shape) <= 1: + head, tail = f" real(wp), parameter :: {w.symbol}{dims} = [ ", " ]" + else: + head = f" real(wp), parameter :: {w.symbol}{dims} = reshape([ " + tail = f" ], {_weight_dims_list(plan, w.symbol)})" + lines += _wrap_items(head, flat, tail, " " * 8) + return lines + + def emit_fortran(plan: Plan) -> str: m, wp = plan.model, _KIND[plan.dtype] - # Wrap each literal in an explicit int(..., int8) conversion: an - # unsuffixed literal is default INTEGER(4), and initializing an - # INTEGER(1) parameter array from those trips gfortran's -Wconversion - # under -Wextra. A plain "_int8" kind suffix does not work either -- - # gfortran checks a literal's unsigned magnitude against the kind's - # range before applying unary minus, so "-128_int8" is rejected as - # "Integer too big for its kind" even though -128 is in range. - hash_terms = [f"int({b if b < 128 else b - 256}, int8)" for b in bytes.fromhex(plan.hash())] + public = [f"{m}_infer", f"{m}_infer_batch", f"{m}_infer_batch_dev"] + if not plan.embed: + public.insert(0, f"{m}_init") lines = [ f"module {m}_model", " ! Generated by rosenna. Do not edit.", " use iso_fortran_env, only: int8, int32, int64, " + wp, + " use iso_c_binding, only: c_int, c_ptr", " implicit none", " private", - f" public :: {m}_init, {m}_infer", + " public :: " + ", ".join(public), "", f" integer, parameter :: wp = {wp}", ] - lines += _wrap_items(" integer(int8), parameter :: expected_hash(32) = [ ", hash_terms, " ]", - " " * 8) - for w in plan.weights: - lines.append(f" real(wp), protected :: {w.symbol}{_weight_dims(plan, w.symbol)}") - lines += ["", "contains", ""] - lines += _emit_init(plan) - lines += _emit_load(plan) + if not plan.embed: + # Wrap each literal in an explicit int(..., int8) conversion: an + # unsuffixed literal is default INTEGER(4), and initializing an + # INTEGER(1) parameter array from those trips gfortran's -Wconversion + # under -Wextra. A plain "_int8" kind suffix does not work either -- + # gfortran checks a literal's unsigned magnitude against the kind's + # range before applying unary minus, so "-128_int8" is rejected as + # "Integer too big for its kind" even though -128 is in range. + hash_terms = [f"int({b if b < 128 else b - 256}, int8)" for b in bytes.fromhex(plan.hash())] + lines += _wrap_items(" integer(int8), parameter :: expected_hash(32) = [ ", hash_terms, " ]", + " " * 8) + for w in plan.weights: + lines.append(f" real(wp), protected :: {w.symbol}{_weight_dims(plan, w.symbol)}") + else: + lines += _emit_embedded_weights(plan) + if plan.weights: + lines.append(f" !$omp declare target({_weight_symbol_list(plan)})") + if not plan.embed: + # gfortran rejects a `routine seq`/declare-target function that + # reads a file-scope array with no OpenACC `declare` directive of + # its own; an embedded plan's arrays are compile-time constants + # instead and need none. + lines.append(f" !$acc declare create({_weight_symbol_list(plan)})") + lines += [ + "", + " interface", + f" function {m}_infer_batch_dev(n, x, y, stream) &", + f' bind(C, name="{m}_infer_batch") result(status)', + " import :: c_int, c_ptr", + " integer(c_int), value :: n", + " type(c_ptr), value :: x, y, stream", + " integer(c_int) :: status", + " end function", + " end interface", + "", + "contains", + "", + ] + if not plan.embed: + lines += _emit_init(plan) + lines += _emit_load(plan) lines += _emit_infer(plan) + lines += _emit_infer_batch(plan) lines.append(f"end module {m}_model") return "\n".join(lines) + "\n" @@ -179,6 +278,12 @@ def _emit_init(plan: Plan) -> list: " if (ios /= 0) then; status = 9; close(u); return; end if", " end do", " close(u)", + " ! The plan step's transfer (ruling R5): make the freshly loaded", + " ! weights device-resident. Fortran has no runtime-API path without", + " ! CUDA Fortran, so this directive form is the whole of init's device", + " ! copy; one host call both loads and uploads.", + f" !$omp target update to({_weight_symbol_list(plan)})", + f" !$acc update device({_weight_symbol_list(plan)})", " end subroutine", "", ] @@ -247,6 +352,8 @@ def _emit_infer(plan: Plan) -> list: lines = [ f" pure subroutine {m}_infer(x, y)", + " !$omp declare target", + " !$acc routine seq", f" real(wp), intent(in) :: x({n_in})", f" real(wp), intent(out) :: y({n_out})", ] @@ -298,3 +405,61 @@ def _emit_infer(plan: Plan) -> list: lines.append(" end subroutine") lines.append("") return lines + + +def _emit_infer_batch(plan: Plan) -> list: + """The OpenMP-target fallback infer_batch: over device-resident arrays. + + Ruling R5: x and y are already on the device (has_device_addr / the + OpenACC deviceptr twin), so the loop path allocates, maps and transfers + nothing -- the host maps them itself (`!$omp target enter data`) once, + outside this call. `has_device_addr` on this explicit-shape dummy + (rather than the assumed-shape form the spec flags as a compiler risk) + is accepted, without warning, under gfortran 15 -fopenmp -std=f2008; see + task-4-report.md for which gfortran this was verified against. + """ + m = plan.model + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + return [ + f" subroutine {m}_infer_batch(n, x, y, status)", + " integer, intent(in) :: n", + f" real(wp), intent(in) :: x({n_in}, n)", + f" real(wp), intent(out) :: y({n_out}, n)", + " integer, intent(out) :: status", + " integer :: p", + " ! Ruling R5: x and y are already device-resident. No data clause and", + " ! nothing else here transfers, allocates or synchronizes.", + " !$omp target teams loop has_device_addr(x, y)", + " !$acc parallel loop deviceptr(x, y)", + " do p = 1, n", + f" call {m}_infer(x(:, p), y(:, p))", + " end do", + " status = 0", + " end subroutine", + "", + ] + + +def emit_fortran_recipe(plan: Plan) -> str: + """A Makefile fragment that builds lib.a from the generated module. + + Mirrors emit_c_recipe's shape: FC/FFLAGS/ROSENNA_OFFLOAD_FLAGS are + override points for the host's own compiler and offload flags. gfortran + drops _model.mod alongside the object as a side effect of + compiling; nothing in the recipe needs to name it, but a host module + that `use`s this one needs that .mod on its include path. + """ + n = plan.model + return f"""# Generated by rosenna. Builds lib{n}.a from {n}_model.f90 (and {n}_model.mod). +FC ?= gfortran +FFLAGS ?= -O2 -Wall -Wextra -std=f2008 +ROSENNA_OFFLOAD_FLAGS ?= + +lib{n}.a: {n}_model.o +\tar rcs $@ $^ +{n}_model.o: {n}_model.f90 +\t$(FC) $(FFLAGS) $(ROSENNA_OFFLOAD_FLAGS) -c $< -o $@ +clean: +\trm -f {n}_model.o {n}_model.mod lib{n}.a +.PHONY: clean +""" diff --git a/python/rosenna/verify.py b/python/rosenna/verify.py index 8a5cac3..6f6644a 100644 --- a/python/rosenna/verify.py +++ b/python/rosenna/verify.py @@ -77,10 +77,10 @@ def verify_model(model_path, lang: str, dtype: str | None, cases: int, workdir, for backend in backends: backend_dir = workdir / backend backend_dir.mkdir(parents=True, exist_ok=True) - # Fortran (unchanged by this task) always loads weights from a file. - # The C backend only needs one when the plan is not embedding its - # weights as ROSENNA_CONST arrays in the header. - if backend == "fortran" or not plan.embed: + # Both backends now embed by default: a plan that embeds has no + # weights file to load in either language (every weight is a + # `parameter`/ROSENNA_CONST array baked into the generated source). + if not plan.embed: write_weights(plan, graph, backend_dir / f"{plan.model}.rwt") got = _run_backend(backend, plan, backend_dir, inputs) abs_err = np.abs(got - expected) @@ -115,8 +115,16 @@ def _live_inputs(session, shape, cases: int, model_path, np_dtype): f"passing comparison here would not demonstrate anything") -def _fortran_driver(name: str, n_in: int, n_out: int, dtype: str) -> str: +def _fortran_driver(name: str, n_in: int, n_out: int, dtype: str, embed: bool) -> str: real_kind = "real64" if dtype == "f64" else "real32" + # An embedded plan has no `_init`: every weight is already a `parameter` + # array in the generated module, resident from program load. + init = "" if embed else f""" + call {name}_init('{name}.rwt', status) + if (status /= 0) then + print *, 'init status', status + stop 1 + end if""" return f""" program verify_main use {name}_model @@ -124,12 +132,7 @@ def _fortran_driver(name: str, n_in: int, n_out: int, dtype: str) -> str: implicit none real({real_kind}) :: x({n_in}), y({n_out}) integer :: status, i, ncases - read(*,*) ncases - call {name}_init('{name}.rwt', status) - if (status /= 0) then - print *, 'init status', status - stop 1 - end if + read(*,*) ncases{init} do i = 1, ncases read(*,*) x call {name}_infer(x, y) @@ -197,10 +200,19 @@ def _run_backend(backend: str, plan, workdir: Path, inputs): dtype = plan.dtype if backend == "fortran": (workdir / f"{name}_model.f90").write_text(emit_fortran(plan)) - (workdir / "verify_main.f90").write_text(_fortran_driver(name, n_in, n_out, dtype)) + (workdir / "verify_main.f90").write_text(_fortran_driver(name, n_in, n_out, dtype, plan.embed)) + # Compile the module to an object, archive it, and link the driver + # against the archive -- the library form -- rather than compiling + # both sources together, mirroring the C backend below. + _run("compile", backend, + ["gfortran", "-O2", "-Wall", "-Wextra", "-c", f"{name}_model.f90"], + cwd=workdir) + _run("archive", backend, + ["ar", "rcs", f"lib{name}.a", f"{name}_model.o"], + cwd=workdir) _run("compile/link", backend, ["gfortran", "-O2", "-Wall", "-Wextra", "-o", "verify_run", - f"{name}_model.f90", "verify_main.f90"], + "verify_main.f90", f"lib{name}.a"], cwd=workdir) elif backend == "c": source, header = emit_c(plan) diff --git a/python/tests/test_cli.py b/python/tests/test_cli.py index f6a92fe..b2a54d9 100644 --- a/python/tests/test_cli.py +++ b/python/tests/test_cli.py @@ -3,9 +3,12 @@ def test_generate_writes_all_artifacts(tmp_path, capsys, golden_model): + # --no-embed: gemm_small auto-embeds (well under EMBED_THRESHOLD) in both + # languages now (Task 4), so this test forces the file-loaded contract to + # exercise the .rwt-writing path it asserts on. onnx_path = golden_model("gemm_small") rc = main(["generate", str(onnx_path), - "--lang", "both", "--precision", "double", "--out", str(tmp_path)]) + "--lang", "both", "--precision", "double", "--out", str(tmp_path), "--no-embed"]) assert rc == 0 for f in ["gemm_small_model.f90", "gemm_small.c", "gemm_small.h", "gemm_small.rwt"]: assert (tmp_path / f).exists(), f @@ -13,19 +16,15 @@ def test_generate_writes_all_artifacts(tmp_path, capsys, golden_model): assert "gemm_small.rwt" in out -def test_generate_writes_rwt_with_fortran_but_not_c_only(tmp_path, capsys, golden_model): - # gemm_small auto-embeds (well under EMBED_THRESHOLD). --lang both still - # writes the .rwt because Fortran generation is unchanged by this task - # and always loads weights from a file; --lang c alone has nothing left - # that needs one, since the weights are ROSENNA_CONST arrays baked into - # the header. NOTE: Task 4 (Fortran embedding) is expected to flip the - # first assertion once Fortran also embeds by default -- revisit this - # test then rather than assuming it still holds. +def test_generate_writes_rwt_only_when_not_embedding(tmp_path, capsys, golden_model): + # gemm_small auto-embeds (well under EMBED_THRESHOLD) as of Task 4 in + # both languages, so neither --lang both nor --lang c writes a .rwt by + # default; --no-embed is what brings it back, regardless of --lang. onnx_path = golden_model("gemm_small") rc = main(["generate", str(onnx_path), "--lang", "both", "--out", str(tmp_path / "both")]) assert rc == 0 - assert (tmp_path / "both" / "gemm_small.rwt").exists() + assert not (tmp_path / "both" / "gemm_small.rwt").exists() rc = main(["generate", str(onnx_path), "--lang", "c", "--out", str(tmp_path / "c")]) assert rc == 0 @@ -33,6 +32,24 @@ def test_generate_writes_rwt_with_fortran_but_not_c_only(tmp_path, capsys, golde out = capsys.readouterr().out assert "embedded weights" in out + rc = main(["generate", str(onnx_path), "--lang", "both", "--no-embed", + "--out", str(tmp_path / "noembed")]) + assert rc == 0 + assert (tmp_path / "noembed" / "gemm_small.rwt").exists() + + +def test_generate_writes_the_fortran_recipe(tmp_path, golden_model): + onnx_path = golden_model("gemm_small") + rc = main(["generate", str(onnx_path), "--lang", "fortran", "--out", str(tmp_path)]) + assert rc == 0 + assert (tmp_path / "gemm_small_model.f90").exists() + mk = tmp_path / "gemm_small_fortran.mk" + assert mk.exists() + assert "libgemm_small.a: gemm_small_model.o" in mk.read_text() + # The C recipe (a separate file, a separate object) is untouched by a + # Fortran-only generate. + assert not (tmp_path / "gemm_small.mk").exists() + def test_verify_passes_on_a_dense_model(capsys, golden_model): onnx_path = golden_model("gemm_small") diff --git a/python/tests/test_cli_smoke.py b/python/tests/test_cli_smoke.py index ea212f6..ef50a99 100644 --- a/python/tests/test_cli_smoke.py +++ b/python/tests/test_cli_smoke.py @@ -8,9 +8,11 @@ def test_help_exits_zero(capsys): assert "generate" in capsys.readouterr().out def test_generate_writes_model_file(tmp_path, capsys, golden_model): + # --no-embed: gemm_small auto-embeds by default in both languages (Task 4), + # which would leave no .rwt to assert on below. onnx_path = golden_model("gemm_small") rc = main(["generate", str(onnx_path), "--lang", "both", "--precision", "double", - "--out", str(tmp_path), "--name", "mymodel"]) + "--out", str(tmp_path), "--name", "mymodel", "--no-embed"]) assert rc == 0 assert (tmp_path / "mymodel_model.f90").exists() assert (tmp_path / "mymodel.c").exists() diff --git a/python/tests/test_device_fortran.py b/python/tests/test_device_fortran.py new file mode 100644 index 0000000..81d0db0 --- /dev/null +++ b/python/tests/test_device_fortran.py @@ -0,0 +1,159 @@ +import os +import shutil +import subprocess +import numpy as np +import onnxruntime as ort +import pytest +from rosenna.frontend import load_graph +from rosenna.plan import build_plan +from rosenna.weights import write_weights +from rosenna.emit_fortran import emit_fortran, emit_fortran_recipe +from tests.test_emit_fortran import _live_reference + + +def _omp_fc(): + fc = shutil.which("gfortran") + if not fc: + pytest.skip("no gfortran") + probe = subprocess.run([fc, "-fopenmp", "-x", "f95", "-", "-o", os.devnull], + input="end\n", capture_output=True, text=True) + if probe.returncode != 0: + pytest.skip("gfortran without -fopenmp") + return fc + + +HOST = """ +program host + use {name}_model + use iso_fortran_env, only: real64 + implicit none + real(real64) :: x({n_in}, 64), y({n_out}, 64), yb({n_out}, 64) + integer :: n, p, status + {init_lines} + read(*,*) n + read(*,*) x(:, 1:n) + !$omp target enter data map(to: x) map(alloc: y, yb) + !$omp target teams loop + do p = 1, n + call {name}_infer(x(:, p), y(:, p)) + end do + call {name}_infer_batch(n, x, yb, status) ! x, yb already on the device (R5) + !$omp target exit data map(from: y, yb) map(delete: x) + if (status /= 0) stop 4 + ! abs(...) > 0, not /=: an exact-bits comparison without tripping + ! gfortran's -Wcompare-reals on a bare real (in)equality. + if (any(abs(y(:, 1:n) - yb(:, 1:n)) > 0.0_real64)) stop 5 + do p = 1, n + print '({n_out}(es24.16,1x))', y(:, p) + end do +end program +""" + + +def _build_and_run(tmp_path, name, embed, inputs, golden_model): + graph = load_graph(golden_model(name)) + plan = build_plan(graph, dtype="f64", embed=embed) + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + (tmp_path / f"{name}_model.f90").write_text(emit_fortran(plan)) + init_lines = "" if embed else f'call {name}_init("{name}.rwt", status); if (status /= 0) stop 2' + if not embed: + write_weights(plan, graph, tmp_path / f"{name}.rwt") + (tmp_path / "host.f90").write_text(HOST.format(name=name, n_in=n_in, n_out=n_out, init_lines=init_lines)) + fc = _omp_fc() + flags = ["-O2", "-Wall", "-Wextra", "-std=f2008", "-fopenmp"] + r = subprocess.run([fc, *flags, "-c", f"{name}_model.f90"], cwd=tmp_path, capture_output=True, text=True) + assert r.returncode == 0 and r.stderr == "", r.stderr + subprocess.run(["ar", "rcs", f"lib{name}.a", f"{name}_model.o"], cwd=tmp_path, check=True) + r = subprocess.run([fc, *flags, "host.f90", f"lib{name}.a", "-o", "host"], cwd=tmp_path, capture_output=True, text=True) + assert r.returncode == 0 and r.stderr == "", r.stderr + stdin = f"{len(inputs)}\n" + "\n".join(" ".join(repr(float(v)) for v in row) for row in inputs) + out = subprocess.run(["./host"], cwd=tmp_path, input=stdin, capture_output=True, text=True, check=True).stdout + return np.array([[float(v) for v in line.split()] for line in out.strip().splitlines()]) + + +@pytest.mark.parametrize("name", ["gemm_small", "gemm_big", "gemm_nobias", "droplet", "batchnet"]) +@pytest.mark.parametrize("embed", [True, False]) +def test_host_region_calls_module_infer_and_matches(tmp_path, golden_model, name, embed): + session = ort.InferenceSession(golden_model(name)) + inputs, expected = _live_reference(session, session.get_inputs()[0].shape, np.float64, seed=8, batch=8) + if inputs is None: + pytest.skip(f"{name}: onnxruntime reference is all-zero across 10 resampled " + f"batches; its golden-file weights produced a dead model") + got = _build_and_run(tmp_path, name, embed, inputs, golden_model) + np.testing.assert_allclose(got, expected, rtol=1e-5, atol=1e-6) + + +def test_fortran_target_regions_are_real(tmp_path, golden_model): + # A host-only libgomp refuses a target region under OMP_TARGET_OFFLOAD=MANDATORY. + # If the pragmas were missing or ignored the program would succeed; this is the + # cheapest evidence without a GPU (mirrors tests/test_device_c.py). + name = "gemm_small" + session = ort.InferenceSession(golden_model(name)) + inputs, _ = _live_reference(session, session.get_inputs()[0].shape, np.float64, seed=9, batch=1) + assert inputs is not None + graph = load_graph(golden_model(name)); plan = build_plan(graph, dtype="f64", embed=True) + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + (tmp_path / f"{name}_model.f90").write_text(emit_fortran(plan)) + (tmp_path / "host.f90").write_text(HOST.format(name=name, n_in=n_in, n_out=n_out, init_lines="")) + fc = _omp_fc() + subprocess.run([fc, "-O2", "-std=f2008", "-fopenmp", f"{name}_model.f90", "host.f90", "-o", "host"], + cwd=tmp_path, check=True, capture_output=True) + stdin = "1\n" + " ".join(repr(float(v)) for v in inputs[0]) + r = subprocess.run(["./host"], cwd=tmp_path, input=stdin, capture_output=True, text=True, + env={**os.environ, "OMP_TARGET_OFFLOAD": "MANDATORY"}) + assert r.returncode != 0 and "MANDATORY" in r.stderr + + +def test_generated_fortran_still_fits_in_132_columns(golden_model): + for name in ["gemm_small", "gemm_big", "gemm_nobias", "droplet", "batchnet"]: + for embed in (True, False): + src = emit_fortran(build_plan(load_graph(golden_model(name)), dtype="f64", embed=embed)) + longest = max(len(line) for line in src.splitlines()) + assert longest <= 132, (name, embed, longest) + + +def test_infer_batch_body_never_transfers(golden_model): + # Ruling R5, structural check: infer_batch's own body holds no map/copyin/ + # copyout/update/enter-exit-data token -- init is exempt (it is the plan + # step and the only routine that transfers). + forbidden = ("map(", "copyin", "copyout", "target update", "update device", "enter data", "exit data") + for name in ["gemm_small", "gemm_big", "gemm_nobias", "droplet", "batchnet"]: + for embed in (True, False): + plan = build_plan(load_graph(golden_model(name)), dtype="f64", embed=embed) + src = emit_fortran(plan) + start = src.index(f"subroutine {name}_infer_batch(") + end = src.index("end subroutine", start) + body = src[start:end] + for token in forbidden: + assert token not in body, (name, embed, token) + + +def test_embedded_module_has_no_init_and_file_loaded_does(golden_model): + for embed, expect_init in ((True, False), (False, True)): + plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64", embed=embed) + src = emit_fortran(plan) + assert ("subroutine gemm_small_init(" in src) == expect_init + assert ("real(wp), parameter :: w0" in src) == embed + assert ("real(wp), protected :: w0" in src) == (not embed) + + +def test_bind_c_interface_targets_the_c_infer_batch_symbol(golden_model): + plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64") + src = emit_fortran(plan) + assert 'bind(C, name="gemm_small_infer_batch") result(status)' in src + assert "gemm_small_infer_batch_dev" in src + assert "type(c_ptr), value :: x, y, stream" in src + + +def test_fortran_recipe_builds_the_library(tmp_path, golden_model): + name = "gemm_small" + plan = build_plan(load_graph(golden_model(name)), dtype="f64") + (tmp_path / f"{name}_model.f90").write_text(emit_fortran(plan)) + (tmp_path / "Makefile").write_text(emit_fortran_recipe(plan)) + fc = _omp_fc() + subprocess.run(["make", f"FC={fc}"], cwd=tmp_path, check=True, capture_output=True, text=True) + assert (tmp_path / f"lib{name}.a").exists() + assert (tmp_path / f"{name}_model.o").exists() + subprocess.run(["make", "clean"], cwd=tmp_path, check=True, capture_output=True, text=True) + assert not (tmp_path / f"lib{name}.a").exists() + assert not (tmp_path / f"{name}_model.o").exists() diff --git a/python/tests/test_emit_fortran.py b/python/tests/test_emit_fortran.py index 509e200..96a75b9 100644 --- a/python/tests/test_emit_fortran.py +++ b/python/tests/test_emit_fortran.py @@ -11,8 +11,11 @@ def _build_and_run(tmp_path, onnx_path, name, inputs, dtype="f64"): + # embed=False: this helper's driver always calls `_init` against a + # written .rwt file, the file-loaded contract. The dedicated embed=True/ + # False matrix lives in tests/test_device_fortran.py and tests/test_embed.py. graph = load_graph(onnx_path) - plan = build_plan(graph, dtype=dtype) + plan = build_plan(graph, dtype=dtype, embed=False) (tmp_path / f"{name}_model.f90").write_text(emit_fortran(plan)) write_weights(plan, graph, tmp_path / f"{name}.rwt") n_in, n_out = plan.input.shape[0], plan.output.shape[0] @@ -120,7 +123,10 @@ def test_matches_onnxruntime_f32(tmp_path, golden_model): def test_infer_is_pure_and_has_literal_bounds(golden_model): - plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64") + # embed=False: this test is specifically about the file-loaded contract + # (`protected` module variables filled by `init`), which an embedded + # plan's module does not declare. + plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64", embed=False) src = emit_fortran(plan) assert "pure subroutine gemm_small_infer" in src assert "real(wp), protected :: w0(2,2)" in src @@ -128,10 +134,12 @@ def test_infer_is_pure_and_has_literal_bounds(golden_model): def test_init_rejects_a_foreign_weights_file(tmp_path, golden_model): + # embed=False: this test is specifically about `_init`, which an + # embedded plan's module does not declare. graph = load_graph(golden_model("gemm_small")) - plan = build_plan(graph, dtype="f64") + plan = build_plan(graph, dtype="f64", embed=False) other_graph = load_graph(golden_model("gemm_big")) - other_plan = build_plan(other_graph, dtype="f64") + other_plan = build_plan(other_graph, dtype="f64", embed=False) (tmp_path / "gemm_small_model.f90").write_text(emit_fortran(plan)) write_weights(other_plan, other_graph, tmp_path / "gemm_small.rwt") (tmp_path / "main.f90").write_text(""" @@ -162,7 +170,9 @@ def test_case_labels_escape_quotes(tmp_path): m = helper.make_model(g, opset_imports=[helper.make_opsetid("", 13)]) path = tmp_path / "quoted.onnx" onnx.save(m, path) - src = emit_fortran(build_plan(load_graph(path), dtype="f64")) + # embed=False: the case-label select lives in `load_tensor`, which an + # embedded plan's module does not emit. + src = emit_fortran(build_plan(load_graph(path), dtype="f64", embed=False)) assert "case ('layer.0''weight')" in src (tmp_path / "quoted_model.f90").write_text(src) subprocess.run(["gfortran", "-O2", "-Wall", "-Wextra", "-c", "quoted_model.f90"], diff --git a/python/tests/test_regressions.py b/python/tests/test_regressions.py index 5fbb127..3fe7fd9 100644 --- a/python/tests/test_regressions.py +++ b/python/tests/test_regressions.py @@ -372,7 +372,9 @@ def test_generated_fortran_fits_in_132_columns(golden_model, name, dtype): def test_generated_fortran_fits_in_132_columns_with_a_long_tensor_name(tmp_path): # The name buffer is plan-derived now, so a 176-character initializer name # becomes a 176-character case label unless the literal is continued. - src = emit_fortran(build_plan(load_graph(_long_name_model(tmp_path)), dtype="f64")) + # embed=False: the case label this test asserts on lives in `load_tensor`, + # which an embedded plan's module does not emit. + src = emit_fortran(build_plan(load_graph(_long_name_model(tmp_path)), dtype="f64", embed=False)) assert _over_long_fortran_lines(src) == [] # Wrapped, not dropped: the literal is continued across lines, so the # head of the name is still there and the tail follows a leading `&`. From e078931f9056d81c2aa7df9e8a9fc2500cc312a1 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Sun, 13 Sep 2026 23:23:01 -0500 Subject: [PATCH 09/84] fix: give the fortran archive its own name so both recipes can build in one directory Ruling R13. generate --lang both writes both .mk (C) and _fortran.mk (Fortran) into one output directory, and both recipes archived into lib.a: ar rcs appends, so building both there in sequence silently merged gemm_small_model.o into the C archive, and either recipe's `clean` then deleted the shared file. Reproduced and confirmed with a forced rebuild (ar t showed both gemm_small.o and gemm_small_model.o in one lib.a). The Fortran archive is now lib_f.a; a Fortran host that also links the native CUDA/HIP kernel links both archives: -l_f -l. verify.py's Fortran build follows the same name even though its fortran/c backends already build in separate directories, so the two names are never the same anywhere a caller might build both recipes in one place. Adds the two-archive test that would have caught this (build both recipes into one directory via `generate --lang both`, assert two distinct archives each containing only its own object per `ar t`, and that cleaning one leaves the other's archive and object alone), plus test_generated_fortran_is_warning_free_under_openacc mirroring the existing C test -fsyntax-only under -fopenacc for all five dense models, both embed states. --- python/rosenna/emit_fortran.py | 16 ++++++-- python/rosenna/verify.py | 9 ++++- python/tests/test_cli.py | 60 ++++++++++++++++++++++++++++- python/tests/test_device_fortran.py | 33 +++++++++++++++- 4 files changed, 109 insertions(+), 9 deletions(-) diff --git a/python/rosenna/emit_fortran.py b/python/rosenna/emit_fortran.py index 1a71c83..db9ee1d 100644 --- a/python/rosenna/emit_fortran.py +++ b/python/rosenna/emit_fortran.py @@ -441,25 +441,33 @@ def _emit_infer_batch(plan: Plan) -> list: def emit_fortran_recipe(plan: Plan) -> str: - """A Makefile fragment that builds lib.a from the generated module. + """A Makefile fragment that builds lib_f.a from the generated module. Mirrors emit_c_recipe's shape: FC/FFLAGS/ROSENNA_OFFLOAD_FLAGS are override points for the host's own compiler and offload flags. gfortran drops _model.mod alongside the object as a side effect of compiling; nothing in the recipe needs to name it, but a host module that `use`s this one needs that .mod on its include path. + + Controller ruling R13: the archive is lib_f.a, not lib.a -- + `generate --lang both` writes this recipe and emit_c_recipe's into ONE + output directory, and `ar rcs` APPENDS to an existing archive, so two + recipes sharing one archive name silently merge their objects into it + the moment both are built there (and either recipe's `clean` then + deletes the other's artifacts too). A Fortran host that also links the + native CUDA/HIP kernel links both archives: `-l_f -l`. """ n = plan.model - return f"""# Generated by rosenna. Builds lib{n}.a from {n}_model.f90 (and {n}_model.mod). + return f"""# Generated by rosenna. Builds lib{n}_f.a from {n}_model.f90 (and {n}_model.mod). FC ?= gfortran FFLAGS ?= -O2 -Wall -Wextra -std=f2008 ROSENNA_OFFLOAD_FLAGS ?= -lib{n}.a: {n}_model.o +lib{n}_f.a: {n}_model.o \tar rcs $@ $^ {n}_model.o: {n}_model.f90 \t$(FC) $(FFLAGS) $(ROSENNA_OFFLOAD_FLAGS) -c $< -o $@ clean: -\trm -f {n}_model.o {n}_model.mod lib{n}.a +\trm -f {n}_model.o {n}_model.mod lib{n}_f.a .PHONY: clean """ diff --git a/python/rosenna/verify.py b/python/rosenna/verify.py index 6f6644a..592b0ab 100644 --- a/python/rosenna/verify.py +++ b/python/rosenna/verify.py @@ -207,12 +207,17 @@ def _run_backend(backend: str, plan, workdir: Path, inputs): _run("compile", backend, ["gfortran", "-O2", "-Wall", "-Wextra", "-c", f"{name}_model.f90"], cwd=workdir) + # lib_f.a, not lib.a (ruling R13): the C backend's own + # archive is lib.a, and although verify's fortran/c backends + # build in separate directories (no collision here), the two names + # must never be the same anywhere a caller might build both recipes + # in one place (cli.py's `generate --lang both`, in particular). _run("archive", backend, - ["ar", "rcs", f"lib{name}.a", f"{name}_model.o"], + ["ar", "rcs", f"lib{name}_f.a", f"{name}_model.o"], cwd=workdir) _run("compile/link", backend, ["gfortran", "-O2", "-Wall", "-Wextra", "-o", "verify_run", - "verify_main.f90", f"lib{name}.a"], + "verify_main.f90", f"lib{name}_f.a"], cwd=workdir) elif backend == "c": source, header = emit_c(plan) diff --git a/python/tests/test_cli.py b/python/tests/test_cli.py index b2a54d9..fa10a3a 100644 --- a/python/tests/test_cli.py +++ b/python/tests/test_cli.py @@ -1,5 +1,8 @@ +import subprocess import rosenna.verify as verify_mod from rosenna.cli import main +from tests.test_library_form import _cc +from tests.test_device_fortran import _omp_fc def test_generate_writes_all_artifacts(tmp_path, capsys, golden_model): @@ -45,12 +48,67 @@ def test_generate_writes_the_fortran_recipe(tmp_path, golden_model): assert (tmp_path / "gemm_small_model.f90").exists() mk = tmp_path / "gemm_small_fortran.mk" assert mk.exists() - assert "libgemm_small.a: gemm_small_model.o" in mk.read_text() + # Ruling R13: the Fortran archive is lib_f.a, not lib.a -- + # the latter is the C recipe's archive, and the two must never collide + # when both recipes build in the same directory (see the two-archive + # test below, which is what would have caught that defect). + assert "libgemm_small_f.a: gemm_small_model.o" in mk.read_text() # The C recipe (a separate file, a separate object) is untouched by a # Fortran-only generate. assert not (tmp_path / "gemm_small.mk").exists() +def test_both_recipes_build_distinct_archives_in_one_directory(tmp_path, golden_model): + # Controller ruling R13, reproducing the reviewer's finding on 352c14a: + # `generate --lang both` writes both .mk (C) and _fortran.mk + # (Fortran) into ONE output directory, and both recipes used to archive + # into the same lib.a -- `ar rcs` APPENDS, so building both there + # in sequence silently merged gemm_small_model.o into gemm_small.a's own + # archive, and either recipe's `clean` then deleted the shared file. + # This is exactly the scenario the bind(C) interface serves: a Fortran + # host that `use`s the module AND links the native CUDA/HIP kernel needs + # both archives to coexist, distinctly, in one place. + name = "gemm_small" + onnx_path = golden_model(name) + rc = main(["generate", str(onnx_path), "--lang", "both", "--out", str(tmp_path), "--no-embed"]) + assert rc == 0 + + cc, fc = _cc(), _omp_fc() + subprocess.run(["make", "-f", f"{name}.mk", f"CC={cc}"], cwd=tmp_path, + check=True, capture_output=True, text=True) + subprocess.run(["make", "-f", f"{name}_fortran.mk", f"FC={fc}"], cwd=tmp_path, + check=True, capture_output=True, text=True) + + c_archive, f_archive = tmp_path / f"lib{name}.a", tmp_path / f"lib{name}_f.a" + assert c_archive.exists() and f_archive.exists() + assert c_archive != f_archive + + def _members(archive): + # One member per line; BSD ar (macOS) also lists a "__.SYMDEF SORTED" + # pseudo-member (its own symbol table) that GNU ar's `ar t` omits -- + # filter it out rather than split() on whitespace, since its name + # itself contains a space. + out = subprocess.run(["ar", "t", str(archive)], cwd=tmp_path, + check=True, capture_output=True, text=True).stdout + return [line for line in out.splitlines() if not line.startswith("__.SYMDEF")] + + # Each archive holds only its own object -- not the other's, and not both + # (the merged-archive defect: one .a holding gemm_small_model.o AND + # gemm_small.o together, silently, because `ar rcs` appends). + assert _members(c_archive) == [f"{name}.o"] + assert _members(f_archive) == [f"{name}_model.o"] + + # `clean` on one recipe never touches the other's archive or object. + subprocess.run(["make", "-f", f"{name}_fortran.mk", "clean"], cwd=tmp_path, + check=True, capture_output=True, text=True) + assert not f_archive.exists() and not (tmp_path / f"{name}_model.o").exists() + assert c_archive.exists() and (tmp_path / f"{name}.o").exists() + + subprocess.run(["make", "-f", f"{name}.mk", "clean"], cwd=tmp_path, + check=True, capture_output=True, text=True) + assert not c_archive.exists() and not (tmp_path / f"{name}.o").exists() + + def test_verify_passes_on_a_dense_model(capsys, golden_model): onnx_path = golden_model("gemm_small") rc = main(["verify", str(onnx_path), "--cases", "4"]) diff --git a/python/tests/test_device_fortran.py b/python/tests/test_device_fortran.py index 81d0db0..8c76913 100644 --- a/python/tests/test_device_fortran.py +++ b/python/tests/test_device_fortran.py @@ -137,6 +137,32 @@ def test_embedded_module_has_no_init_and_file_loaded_does(golden_model): assert ("real(wp), protected :: w0" in src) == (not embed) +def test_generated_fortran_is_warning_free_under_openacc(tmp_path, golden_model): + # Mirrors tests/test_kernel.py::test_generated_c_is_warning_free_under_openacc. + # gfortran -fopenacc rejects a `routine seq` function reading a file-scope + # array with no `declare` directive of its own (why _emit_embedded_weights + # skips `!$acc declare create` -- an embedded plan's arrays are compile- + # time constants and need none, but a file-loaded plan's `protected` + # arrays do); this is the test that guards that comment. + fc = shutil.which("gfortran") + if not fc: + pytest.skip("no gfortran") + probe = subprocess.run([fc, "-fopenacc", "-x", "f95", "-", "-o", os.devnull], + input="end\n", capture_output=True, text=True) + if probe.returncode != 0: + pytest.skip(f"{fc} does not accept -fopenacc") + for name in ["gemm_small", "gemm_big", "gemm_nobias", "droplet", "batchnet"]: + for embed in (True, False): + plan = build_plan(load_graph(golden_model(name)), dtype="f64", embed=embed) + src_path = tmp_path / f"{name}_{embed}_model.f90" + src_path.write_text(emit_fortran(plan)) + r = subprocess.run( + [fc, "-O2", "-Wall", "-Wextra", "-std=f2008", "-fopenacc", "-fsyntax-only", + src_path.name], + cwd=tmp_path, capture_output=True, text=True) + assert r.returncode == 0 and r.stderr == "", (name, embed, r.stderr) + + def test_bind_c_interface_targets_the_c_infer_batch_symbol(golden_model): plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64") src = emit_fortran(plan) @@ -152,8 +178,11 @@ def test_fortran_recipe_builds_the_library(tmp_path, golden_model): (tmp_path / "Makefile").write_text(emit_fortran_recipe(plan)) fc = _omp_fc() subprocess.run(["make", f"FC={fc}"], cwd=tmp_path, check=True, capture_output=True, text=True) - assert (tmp_path / f"lib{name}.a").exists() + # lib_f.a, not lib.a (ruling R13): see + # tests/test_cli.py::test_both_recipes_build_distinct_archives_in_one_directory + # for why the two names must never collide. + assert (tmp_path / f"lib{name}_f.a").exists() assert (tmp_path / f"{name}_model.o").exists() subprocess.run(["make", "clean"], cwd=tmp_path, check=True, capture_output=True, text=True) - assert not (tmp_path / f"lib{name}.a").exists() + assert not (tmp_path / f"lib{name}_f.a").exists() assert not (tmp_path / f"{name}_model.o").exists() From 0fcd3bdc6192f3798eea68745060f328dad594db Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Sun, 13 Sep 2026 23:51:16 -0500 Subject: [PATCH 10/84] feat: gpu-gate command and the microfd closure worked example Adds `rosenna gpu-gate`, the script a user runs on a real GPU machine to validate the device library: it generates gemm_big embedded and file-loaded in both languages, builds the C library with the chosen batched backend and the Fortran library with the host compiler, then runs three harnesses (C per-point, Fortran per-point, and infer_batch over device-resident data), each checked against onnxruntime and timed per point, and writes every command and its output to gate-report.md. --host-fallback drops the OMP_TARGET_OFFLOAD=MANDATORY requirement so the omp backend, and the script itself, can be exercised end to end on a machine with no accelerator; that is how tests/test_gate.py runs it here. Moves _live_reference from tests/test_emit_fortran.py into rosenna/verify.py so gate.py can reuse it without importing test code; every existing import of it keeps working via a re-export. Adds the microfd_closure worked example: closure.py exports a 9-16-1 Tanh MLP closure model, and patch.md documents (as an unapplied diff against a reference copy of microfd.c) the per-cell closure() kernel, the muf blend in face(), the g.nut allocation and device mapping, and the batched alternative via closure_infer_batch. README.md states the device path is host-validated only until gpu-gate has actually run on a GPU machine. .gitignore gains an allowlist entry for python/examples/**/*.md. --- .gitignore | 1 + python/examples/microfd_closure/README.md | 72 +++ python/examples/microfd_closure/closure.py | 56 ++ python/examples/microfd_closure/patch.md | 208 +++++++ python/rosenna/cli.py | 50 ++ python/rosenna/gate.py | 647 +++++++++++++++++++++ python/rosenna/verify.py | 38 ++ python/tests/test_emit_fortran.py | 34 +- python/tests/test_gate.py | 29 + 9 files changed, 1106 insertions(+), 29 deletions(-) create mode 100644 python/examples/microfd_closure/README.md create mode 100644 python/examples/microfd_closure/closure.py create mode 100644 python/examples/microfd_closure/patch.md create mode 100644 python/rosenna/gate.py create mode 100644 python/tests/test_gate.py diff --git a/.gitignore b/.gitignore index 6726778..4469476 100644 --- a/.gitignore +++ b/.gitignore @@ -14,6 +14,7 @@ !*.toml !goldenFiles/mnist/mnist.onnx !instructions/* +!python/examples/**/*.md reading.f90 userTesting.f90 linearV3copy.f90 diff --git a/python/examples/microfd_closure/README.md b/python/examples/microfd_closure/README.md new file mode 100644 index 0000000..a74187f --- /dev/null +++ b/python/examples/microfd_closure/README.md @@ -0,0 +1,72 @@ +# microfd closure: a worked example + +The device path this repository generates (OpenMP-target, OpenACC, CUDA and +HIP hosts; a native batched kernel; the same contract from C and Fortran) is +validated by running `rosenna gpu-gate` on a machine with a real accelerator +and recording the report it writes. Until that run has happened and its +`gate-report.md` is on record, everything below -- including this example -- +is host-validated only: it compiles and runs on the host; the device path is +unvalidated. + +## What this is + +microfd is a compact, single-file 3D compressible Navier-Stokes solver +(finite volume, WENO5-Z + HLLC, SSP-RK3, MPI + OpenMP target offload). This +directory shows how a rosenna model plugs into it as a per-cell turbulence +closure: a small MLP that maps the nine components of the local +velocity-gradient tensor to a turbulent-viscosity correction, evaluated once +per cell inside microfd's own offloaded loop. `patch.md` is written against +the reference copy of `microfd.c` used to design it; nothing here compiles +microfd itself, and no microfd source is vendored into this repository. + +- `closure.py` -- builds and exports `closure.onnx`: a 9-input, 16-hidden + (Tanh), 1-output MLP, deterministically initialized. +- `patch.md` -- the exact edits to `microfd.c` (as a documented diff, not + applied to any repository -- nothing here compiles microfd) that add the + closure to the solver's viscous flux, plus the alternative call convention + for a larger network. + +## Generating the code + +``` +cd python +python3 examples/microfd_closure/closure.py # writes closure.onnx +python3 -c " +from rosenna.cli import main +main(['generate', 'examples/microfd_closure/closure.onnx', + '--lang', 'c', '--precision', 'double', '--out', 'examples/microfd_closure/gen']) +" +``` + +`--precision double` matches microfd's own arithmetic, which is `double` +throughout; the default (the model's own dtype, float32) would otherwise +force a cast at every call site. The model is tiny (177 parameters, well +under the 1,000,000-parameter embed threshold), so it embeds by default: +`generate` writes `closure.h` with the weights baked in as `ROSENNA_CONST` +arrays, `closure.c` (the OpenMP-fallback `infer_batch`, needed only for the +batched alternative in `patch.md`), `closure_kernel.cu` (the native CUDA/HIP +`infer_batch`, likewise), `closure.mk`, and no `.rwt` file -- an embedded +plan has nothing to load at runtime. + +Because the plan embeds, the per-point path in `patch.md` needs only +`#include "closure.h"`: `closure_infer` is `static inline` and fully +resident in the header, so there is no `closure_init` to call and nothing to +link into microfd's own build. The batched alternative (`closure_infer_batch`) +is not header-inline -- it is always defined in `closure.c` / the kernel's +`.cu` file, whichever `ROSENNA_BACKEND` `closure.mk` was built with -- so +that path does link `libclosure.a`. + +## Validating on a GPU machine + +``` +rosenna gpu-gate --cc gcc-15 --fc gfortran --flags -fopenmp \ + --backend cuda --devcc nvcc --out /tmp/rosenna-gate +``` + +records `gate-report.md`: every command it ran, every line of output, the +compiler versions, and nanoseconds per point for each of the three +harnesses. Run it with `--backend hip --devcc hipcc` on an AMD GPU, or +`--backend omp` to check the OpenMP-target fallback on either. Only after +that report exists for the backend and hardware you actually run microfd on +should the closure above be described as device-validated rather than +host-validated. diff --git a/python/examples/microfd_closure/closure.py b/python/examples/microfd_closure/closure.py new file mode 100644 index 0000000..7bb609d --- /dev/null +++ b/python/examples/microfd_closure/closure.py @@ -0,0 +1,56 @@ +"""Export closure.onnx: a per-cell turbulence-closure MLP for the microfd worked example. + +9 inputs -- the velocity-gradient tensor du_i/dx_j at one cell, flattened +row-major (du/dx, du/dy, du/dz, dv/dx, dv/dy, dv/dz, dw/dx, dw/dy, dw/dz) -- +one hidden layer of 16 units with Tanh, and one output (a turbulent-viscosity +correction, added to the molecular mu at a face in patch.md). No output +activation: the closure is a signed correction, not a probability or a +strictly positive quantity, and the network is free to clip or scale it +downstream (patch.md's `muf` line does exactly that: mu + max(...)). + +Deterministic: torch.manual_seed pins the initialization so re-running this +script reproduces the same closure.onnx byte for byte (module weight order +and Kaiming/uniform default init are themselves deterministic given the +seed). + +This is documentation, not a validated closure model: see README.md in this +directory for what "validated" means here, and patch.md for the exact edits +to microfd.c. Nothing in this repository compiles microfd.c. +""" +from pathlib import Path + +import torch +import torch.nn as nn + +N_IN, N_HIDDEN, N_OUT = 9, 16, 1 + + +class Closure(nn.Module): + def __init__(self): + super().__init__() + self.hidden = nn.Linear(N_IN, N_HIDDEN) + self.act = nn.Tanh() + self.output = nn.Linear(N_HIDDEN, N_OUT) + + def forward(self, x): + return self.output(self.act(self.hidden(x))) + + +def main(): + torch.manual_seed(0) + model = Closure().eval() + example = torch.zeros(1, N_IN) + out_path = Path(__file__).parent / "closure.onnx" + torch.onnx.export( + model, example, str(out_path), + export_params=True, dynamo=False, + opset_version=10, + do_constant_folding=True, + input_names=["input"], + output_names=["output"], + ) + print(out_path) + + +if __name__ == "__main__": + main() diff --git a/python/examples/microfd_closure/patch.md b/python/examples/microfd_closure/patch.md new file mode 100644 index 0000000..0737007 --- /dev/null +++ b/python/examples/microfd_closure/patch.md @@ -0,0 +1,208 @@ +# patch.md: wiring the closure into microfd.c + +This documents the edits to `microfd.c` that add the per-cell closure from +`closure.py` to the solver's viscous flux. It is documentation: nothing in +this repository compiles microfd, the diff is not applied anywhere, and no +microfd source is vendored here. Line numbers and surrounding context are +taken from the 231-line reference copy of `microfd.c` used to design this +patch. Device residency for the generated code is unvalidated until +`rosenna gpu-gate` has run on a GPU machine (see README.md); this patch is +correspondingly "compiles and runs on the host; device path unvalidated". + +The closure model is `closure.onnx` (9 inputs: the velocity-gradient tensor +du_i/dx_j, flattened as du/dx, du/dy, du/dz, dv/dx, dv/dy, dv/dz, dw/dx, +dw/dy, dw/dz; 16 hidden units, Tanh; 1 output, no activation), generated +with: + +``` +rosenna generate closure.onnx --lang c --precision double --out . +``` + +into the same directory as `microfd.c`. The plan embeds (177 parameters), +so `closure_infer` is `static inline` in `closure.h` with the weights baked +in: no `closure_init`, and nothing to link for the per-point path below. +`closure_infer_batch` (used only by the batched alternative at the end of +this file) is not header-inline; it always lives in `closure.c` / the +kernel's `.cu` file, so that path links `libclosure.a`. + +## 1. Include the header + +```diff + #include + #include ++#include "closure.h" // closure_infer: 9 velocity gradients -> nut, header-inline (embedded plan) + + #define NV 5 // rho, rho u, rho v, rho w, E +``` + +## 2. A place to keep the turbulent viscosity: `g.nut` + +`g` already holds one padded-block array per quantity (`q`, `q1`, `w`, `F`, +the halo buffers); `nut` is one more, one value per cell (not `NV*nc` like +`q`/`w`, since it is a single scalar field). + +```diff + double L[3], o[3], h[3], gamma, mu, pr, cfl, tend, t; +- double *q, *q1, *w, *F, *sbuf[2], *rbuf[2]; // F holds all three directions: [d][NV][nc] ++ double *q, *q1, *w, *F, *sbuf[2], *rbuf[2]; // F holds all three directions: [d][NV][nc] ++ double *nut; // turbulent/SGS viscosity from the closure, one value per cell + MPI_Comm comm; +``` + +## 3. The `closure()` kernel + +Placed after `prim()` (which fills `w`, the primitives array `closure()` +reads) and before `face()` (which reads `g.nut`), so it slots directly into +`rhs_eval`'s existing sequence. It uses microfd's own `LOCALS`/`FOR3`/`IDX` +macros and its own naming convention for the primitives array (`w[nc+c]` = +u, `w[2*nc+c]` = v, `w[3*nc+c]` = the third velocity component, named `s` +throughout microfd.c to avoid colliding with the `w` array itself). The +gradients are ordinary second-order central differences at the cell center, +distinct from `face()`'s one-sided/averaged stencil at a face: + +```diff + static void prim(const double*q){ // conserved -> primitive over the whole padded block + LOCALS; double*w=g.w; const double gm=g.gamma-1; + #pragma omp target teams loop + for(size_t c=0;c0)` viscous block, blend the molecular viscosity +with the face-averaged turbulent viscosity from the two cells straddling +the face, and use that blend (`muf`, not `mu`) in the stress: + +```diff + if(mu>0){ // viscous stress and heat flux at the face, 2nd-order central + const long st[3]={1,sx,sy}; const double h[3]={h0,h1,h2}; double du[3][3], div=0; ++ const double muf=mu+.5*(g.nut[c]+g.nut[c+s]); // molecular + face-averaged closure viscosity + for(int a=0;a<3;a++) for(int b=0;b<3;b++) if(a==b||a==d||b==d){ const double*u=w+(1+a)*nc+c; const long t=st[b]; // off-normal off-diagonal terms are dead + du[a][b]= b==d ? (u[s]-u[0])/h[d] : (u[t]-u[-t]+u[s+t]-u[s-t])/(4*h[b]); } // normal: two cells; tangential: averaged central + for(int a=0;a<3;a++) div+=du[a][a]; + f[4]-=kap*(w[4*nc+c+s]/w[c+s]-w[4*nc+c]/w[c])/h[d]; // heat flux with T = p/rho +- for(int m=0;m<3;m++){ const int a=(d+m)%3; const double tau=mu*(du[a][d]+du[d][a]-(a==d)*2./3*div); ++ for(int m=0;m<3;m++){ const int a=(d+m)%3; const double tau=muf*(du[a][d]+du[d][a]-(a==d)*2./3*div); + f[1+m]-=tau; f[4]-=tau*.5*(w[(1+a)*nc+c]+w[(1+a)*nc+c+s]); } + } +``` + +`kap` (the heat-flux conductivity) is left on the molecular `mu`, computed +earlier in `face()` from `g.mu`; blending it too would need a turbulent +Prandtl number, which is outside the scope of this example. + +## 6. Allocate and map `g.nut` + +In `main`, alongside the other padded-block arrays: + +```diff + const size_t m=NV*g.nc; double**arr[]={&g.q,&g.q1,&g.w,&g.F,&g.sbuf[0],&g.sbuf[1],&g.rbuf[0],&g.rbuf[1]}; + for(int i=0;i<8;i++) if(!(*arr[i]=calloc(i<3?m:i==3?3*m:g.nbuf,sizeof(double)))) die("out of memory"); ++ if(!(g.nut=calloc(g.nc,sizeof(double)))) die("out of memory"); // one turbulent-viscosity value per cell + { LOCALS; for(int k=0;k argparse.ArgumentParser: info = sub.add_parser("info", help="report ops, shapes and whether the model is supported") info.add_argument("model") + + gate = sub.add_parser( + "gpu-gate", + help="build and run the device-library validation harnesses on a GPU machine " + "(gemm_big, embedded and file-loaded, both languages, three harnesses); " + "writes gate-report.md") + gate.add_argument("--cc", required=True, help="host C compiler") + gate.add_argument("--fc", required=True, help="host Fortran compiler") + gate.add_argument("--flags", default="", help="host offload flags, e.g. -fopenmp") + gate.add_argument("--backend", choices=["cuda", "hip", "omp"], required=True, + help="which infer_batch implementation to build and exercise") + gate.add_argument("--devcc", default=None, help="nvcc or hipcc; required for --backend cuda|hip") + gate.add_argument("--devflags", default="", help="device compiler flags") + gate.add_argument("--out", default=".", help="directory for generated sources and gate-report.md") + gate.add_argument("--host-fallback", action="store_true", + help="drop the OMP_TARGET_OFFLOAD=MANDATORY requirement so the omp " + "backend can be exercised on a machine with no accelerator") return p @@ -134,6 +152,12 @@ def _cmd_verify(args) -> int: return 0 if all_ok else 1 +def _cmd_gate(args) -> int: + return run_gate(cc=args.cc, fc=args.fc, flags=args.flags, backend=args.backend, + devcc=args.devcc, devflags=args.devflags, out=args.out, + host_fallback=args.host_fallback) + + def _cmd_info(args) -> int: graph = load_graph(args.model) for line in _describe_ops(graph): @@ -147,7 +171,31 @@ def _cmd_info(args) -> int: return 0 +_DASH_VALUED_OPTIONS = ("--flags", "--devflags") + + +def _join_dash_valued_options(argv: list[str]) -> list[str]: + """Let --flags/--devflags take a value that itself starts with '-' (e.g. -fopenmp). + + argparse treats any token starting with a prefix character as a + candidate option string, even one no parser here defines, so `--flags + -fopenmp` (two argv entries) fails with "expected one argument" -- + exactly the invocation shape `rosenna gpu-gate` needs for real compiler + flags. Folding it into one `--flags=-fopenmp` entry first sidesteps + argparse's option-likely-string heuristic entirely; the `=` form always + works because argparse never re-examines what follows `=`. + """ + out = list(argv) + i = 0 + while i < len(out) - 1: + if out[i] in _DASH_VALUED_OPTIONS and out[i + 1].startswith("-"): + out[i:i + 2] = [f"{out[i]}={out[i + 1]}"] + i += 1 + return out + + def main(argv: list[str] | None = None) -> int: + argv = _join_dash_valued_options(sys.argv[1:] if argv is None else argv) args = build_parser().parse_args(argv) try: if args.command == "generate": @@ -156,6 +204,8 @@ def main(argv: list[str] | None = None) -> int: return _cmd_verify(args) if args.command == "info": return _cmd_info(args) + if args.command == "gpu-gate": + return _cmd_gate(args) raise AssertionError(f"unhandled command {args.command!r}") except UnsupportedModel as e: print(f"rosenna: {e}", file=sys.stderr) diff --git a/python/rosenna/gate.py b/python/rosenna/gate.py new file mode 100644 index 0000000..3797993 --- /dev/null +++ b/python/rosenna/gate.py @@ -0,0 +1,647 @@ +"""rosenna gpu-gate: the script a user runs on a real GPU machine. + +None of the CUDA/HIP path has ever been compiled or run on the machine that +wrote it (no nvcc, hipcc, or GPU). This script is the evidence that fact +cannot produce: it generates the gemm_big plan embedded and file-loaded, in +both languages, builds the C library with the chosen batched backend and the +Fortran library with the host compiler, then runs three harnesses -- a +microfd-shaped per-point host in C, the same in Fortran, and a host that +hands device-resident data to infer_batch -- each compared against +onnxruntime and timed per point. Every command, every line of its output, +the compiler versions and the timings go into gate-report.md; a failure at +any step still writes the report and the process exits 1. + +`--host-fallback` drops the OMP_TARGET_OFFLOAD=MANDATORY requirement so the +omp backend can be exercised end to end on a machine with no accelerator +(this is how the test suite runs this script); without it, a machine with no +working offload device fails loudly instead of silently falling back to the +host, which is the whole point of running under MANDATORY in the first +place. + +Device residency is not claimed anywhere until this script has actually run +on a GPU machine and its report recorded. Until then: compiles and runs on +the host; device path unvalidated. +""" +import os +import platform +import shutil +import subprocess +import sys +from pathlib import Path + +import numpy as np +import onnxruntime as ort + +from .emit_c import emit_c, emit_c_recipe +from .emit_fortran import emit_fortran, emit_fortran_recipe +from .emit_kernel import emit_kernel +from .frontend import load_graph +from .plan import build_plan, validate_model_name +from .rt_header import rt_header +from .verify import _live_reference +from .weights import write_weights + +_REPO_ROOT = Path(__file__).resolve().parents[2] +_MODEL = "gemm_big" +_TIMED_ITERS = 1_000_000 +_RTOL, _ATOL = 1e-5, 1e-6 +_RUN_TIMEOUT = 300 + + +class _Report: + """Accumulates gate-report.md; every command and every output line goes in.""" + + def __init__(self): + self.lines = [] + + def h(self, text: str, level: int = 2) -> None: + self.lines.append(f"\n{'#' * level} {text}\n") + + def p(self, text: str) -> None: + self.lines.append(text) + + def block(self, label: str, text: str) -> None: + self.lines.append(f"{label}:\n```\n{text}\n```") + + def command(self, label: str, args: list, cwd=None) -> None: + where = f" (in {cwd})" if cwd is not None else "" + self.lines.append(f"\n**{label}**{where}\n\n```\n$ {' '.join(str(a) for a in args)}\n```") + + def outcome(self, proc) -> None: + self.lines.append(f"exit status: {proc.returncode}") + if proc.stdout: + self.block("stdout", proc.stdout) + if proc.stderr: + self.block("stderr", proc.stderr) + + def text(self) -> str: + return "\n".join(self.lines) + "\n" + + +class _FakeProc: + """Stands in for subprocess.CompletedProcess when the executable itself is missing.""" + + def __init__(self, returncode, stderr): + self.returncode = returncode + self.stdout = "" + self.stderr = stderr + + +def _sh(report: _Report, label: str, args: list, cwd=None, env=None, input_text=None, + timeout=_RUN_TIMEOUT): + """Run one subprocess step, logging the command and its full output either way.""" + report.command(label, args, cwd) + try: + proc = subprocess.run(args, cwd=cwd, env=env, input=input_text, + capture_output=True, text=True, timeout=timeout) + except FileNotFoundError as e: + proc = _FakeProc(127, f"{args[0]}: not found ({e.strerror})") + except subprocess.TimeoutExpired as e: + proc = _FakeProc(124, f"timed out after {timeout}s\nstdout so far:\n{e.stdout}\nstderr so far:\n{e.stderr}") + report.outcome(proc) + return proc + + +def _record_versions(report: _Report, cc: str, fc: str, devcc) -> None: + report.h("toolchain", 3) + report.p(f"platform: {platform.platform()}") + for label, exe in (("cc", cc), ("fc", fc), ("devcc", devcc)): + if not exe: + continue + proc = subprocess.run([exe, "--version"], capture_output=True, text=True) + report.block(f"{label} ({exe}) --version", proc.stdout or proc.stderr or "(no output)") + + +def _ensure_model(report: _Report) -> Path: + onnx_path = _REPO_ROOT / "goldenFiles" / _MODEL / f"{_MODEL}.onnx" + if not onnx_path.exists(): + gen = _REPO_ROOT / "goldenFiles" / _MODEL / f"{_MODEL}.py" + _sh(report, "generate the golden gemm_big model", [sys.executable, str(gen)], + cwd=_REPO_ROOT / "test") + return onnx_path + + +def _generate(outdir: Path, onnx_path: Path, embed: bool): + outdir.mkdir(parents=True, exist_ok=True) + graph = load_graph(str(onnx_path)) + plan = build_plan(graph, dtype="f64", embed=embed) + validate_model_name(plan.model) + name = plan.model + (outdir / f"{name}_model.f90").write_text(emit_fortran(plan)) + (outdir / f"{name}_fortran.mk").write_text(emit_fortran_recipe(plan)) + source, header = emit_c(plan) + (outdir / f"{name}.c").write_text(source) + (outdir / f"{name}.h").write_text(header) + (outdir / f"{name}.mk").write_text(emit_c_recipe(plan)) + (outdir / f"{name}_kernel.cu").write_text(emit_kernel(plan)) + (outdir / "rosenna_rt.h").write_text(rt_header()) + if not plan.embed: + write_weights(plan, graph, outdir / f"{name}.rwt") + return plan + + +def _build_c_lib(report: _Report, outdir: Path, plan, cc, flags, backend, devcc, devflags) -> bool: + name = plan.model + label = "embedded" if plan.embed else "file-loaded" + args = ["make", "-f", f"{name}.mk", f"ROSENNA_BACKEND={backend}"] + if backend == "omp": + args += [f"CC={cc}", f"ROSENNA_OFFLOAD_FLAGS={flags}"] + else: + args += [f"DEVCC={devcc or ('nvcc' if backend == 'cuda' else 'hipcc')}"] + if devflags: + args.append(f"DEVFLAGS={devflags}") + proc = _sh(report, f"build c library ({label}, backend={backend})", args, cwd=outdir) + return proc.returncode == 0 and (outdir / f"lib{name}.a").exists() + + +def _build_fortran_lib(report: _Report, outdir: Path, plan, fc, flags) -> bool: + name = plan.model + label = "embedded" if plan.embed else "file-loaded" + args = ["make", "-f", f"{name}_fortran.mk", f"FC={fc}", f"ROSENNA_OFFLOAD_FLAGS={flags}"] + proc = _sh(report, f"build fortran library ({label})", args, cwd=outdir) + return proc.returncode == 0 and (outdir / f"lib{name}_f.a").exists() + + +def _stdin_for(inputs) -> str: + return f"{len(inputs)}\n" + "\n".join(" ".join(repr(float(v)) for v in row) for row in inputs) + + +def _check_output(report: _Report, stdout: str, expected) -> tuple: + lines = [l for l in stdout.strip().splitlines() if l.strip()] + data_lines = [l for l in lines if not l.startswith("TIMING")] + timing_lines = [l for l in lines if l.startswith("TIMING")] + if len(data_lines) != len(expected): + report.p(f"FAIL: expected {len(expected)} output rows from onnxruntime's own batch, " + f"got {len(data_lines)}") + return False, None + got = np.array([[float(v) for v in l.split()] for l in data_lines]) + try: + np.testing.assert_allclose(got, expected, rtol=_RTOL, atol=_ATOL) + except AssertionError as e: + report.block("FAIL: output does not match the onnxruntime reference", str(e)) + return False, None + report.p(f"matches the onnxruntime reference (rtol={_RTOL}, atol={_ATOL})") + ns = None + if timing_lines: + ns = float(timing_lines[-1].split()[-1]) + report.p(f"{ns:.3f} ns per point") + else: + report.p("FAIL: no TIMING line in the harness output") + return False, None + return True, ns + + +_C_HARNESS1 = """/* rosenna gpu-gate: microfd-shaped per-point host, C. */ +#include +#include +#include +#include "{name}.h" +int main(void) {{ + {init} + int n; + if (scanf("%d", &n) != 1) return 1; + double *x = malloc(sizeof(double) * (size_t)n * {n_in}); + double *y = malloc(sizeof(double) * (size_t)n * {n_out}); + for (int c = 0; c < n * {n_in}; ++c) if (scanf("%lf", &x[c]) != 1) return 1; +#ifdef _OPENMP + #pragma omp target teams loop map(to: x[0:n*{n_in}]) map(from: y[0:n*{n_out}]) +#endif + for (int p = 0; p < n; ++p) {name}_infer(x + p * {n_in}, y + p * {n_out}); /* microfd's own target loop calling the header inline */ + for (int p = 0; p < n; ++p) {{ + for (int i = 0; i < {n_out}; ++i) printf("%.17e ", y[p * {n_out} + i]); + printf("\\n"); + }} + long ntime = {ntime}L; + double t0 = omp_get_wtime(); +#ifdef _OPENMP + #pragma omp target teams loop map(to: x[0:n*{n_in}]) map(from: y[0:n*{n_out}]) +#endif + for (long p = 0; p < ntime; ++p) {{ + int b = (int)(p % n); + {name}_infer(x + b * {n_in}, y + b * {n_out}); + }} + double t1 = omp_get_wtime(); + printf("TIMING %.6f\\n", (t1 - t0) * 1.0e9 / (double)ntime); + free(x); free(y); + return 0; +}} +""" + +_F_HARNESS2 = """ +program host + use {name}_model + use iso_fortran_env, only: real64 + implicit none + real(real64), allocatable :: x(:,:), y(:,:) + integer :: n, p, status, b, ntime, t + integer(8) :: c0, c1, crate + real(real64) :: ns_per_point + status = 0 + {init_lines} + read(*,*) n + allocate(x({n_in}, n), y({n_out}, n)) + read(*,*) x + !$omp target teams loop map(to: x) map(from: y) + do p = 1, n + call {name}_infer(x(:, p), y(:, p)) + end do + do p = 1, n + print '({n_out}(es24.16,1x))', y(:, p) + end do + ntime = {ntime} + call system_clock(count=c0, count_rate=crate) + !$omp target teams loop map(to: x) map(from: y) + do t = 1, ntime + b = mod(t - 1, n) + 1 + call {name}_infer(x(:, b), y(:, b)) + end do + call system_clock(count=c1) + ns_per_point = real(c1 - c0, real64) / real(crate, real64) * 1.0e9_real64 / real(ntime, real64) + print '(A, ES24.16)', 'TIMING ', ns_per_point +end program +""" + +_C_HARNESS3_OMP = """/* rosenna gpu-gate: infer_batch over device-resident data (omp backend), C. */ +#include +#include +#include +#include "{name}.h" +int main(void) {{ + {init} + int n; + if (scanf("%d", &n) != 1) return 1; + double *x = malloc(sizeof(double) * (size_t)n * {n_in}); + double *y = malloc(sizeof(double) * (size_t)n * {n_out}); + for (int c = 0; c < n * {n_in}; ++c) if (scanf("%lf", &x[c]) != 1) return 1; + int status; +#ifdef _OPENMP + #pragma omp target enter data map(to: x[0:n*{n_in}]) map(alloc: y[0:n*{n_out}]) + #pragma omp target data use_device_ptr(x, y) +#endif + {{ + status = {name}_infer_batch(n, x, y, 0); + }} +#ifdef _OPENMP + #pragma omp target exit data map(from: y[0:n*{n_out}]) map(delete: x[0:n*{n_in}]) +#endif + if (status != 0) return 20 + status; + for (int p = 0; p < n; ++p) {{ + for (int i = 0; i < {n_out}; ++i) printf("%.17e ", y[p * {n_out} + i]); + printf("\\n"); + }} + /* Timing: ONE call over ntime points (a batch is meant to be called once + over many points, not called many times over a small batch -- the + latter pays a construct-entry cost per call and times that instead of + the kernel). Values are the correctness batch tiled; only infer_batch's + own cost is timed, not the tiling. */ + long ntime = {ntime}L; + double *xt = malloc(sizeof(double) * (size_t)ntime * {n_in}); + double *yt = malloc(sizeof(double) * (size_t)ntime * {n_out}); + for (long t = 0; t < ntime; ++t) {{ + int b = (int)(t % n); + for (int i = 0; i < {n_in}; ++i) xt[t * {n_in} + i] = x[b * {n_in} + i]; + }} + double t0 = omp_get_wtime(); +#ifdef _OPENMP + #pragma omp target enter data map(to: xt[0:ntime*{n_in}]) map(alloc: yt[0:ntime*{n_out}]) + #pragma omp target data use_device_ptr(xt, yt) +#endif + {{ + status = {name}_infer_batch((int)ntime, xt, yt, 0); + }} +#ifdef _OPENMP + #pragma omp target exit data map(from: yt[0:ntime*{n_out}]) map(delete: xt[0:ntime*{n_in}]) +#endif + double t1 = omp_get_wtime(); + if (status != 0) return 30 + status; + printf("TIMING %.6f\\n", (t1 - t0) * 1.0e9 / (double)ntime); + free(x); free(y); free(xt); free(yt); + return 0; +}} +""" + +_F_HARNESS3_OMP = """ +program host + use {name}_model + use iso_fortran_env, only: real64 + implicit none + real(real64), allocatable :: x(:,:), y(:,:), xt(:,:), yt(:,:) + integer :: n, p, status, ntime, t, b + integer(8) :: c0, c1, crate + real(real64) :: ns_per_point + status = 0 + {init_lines} + read(*,*) n + allocate(x({n_in}, n), y({n_out}, n)) + read(*,*) x + !$omp target enter data map(to: x) map(alloc: y) + call {name}_infer_batch(n, x, y, status) ! x, y already on the device (R5) + !$omp target exit data map(from: y) map(delete: x) + if (status /= 0) stop 20 + do p = 1, n + print '({n_out}(es24.16,1x))', y(:, p) + end do + ! Timing: ONE call over ntime points (a batch is meant to be called once + ! over many points, not called many times over a small batch), values + ! tiled from the correctness batch above; only infer_batch itself is timed. + ntime = {ntime} + allocate(xt({n_in}, ntime), yt({n_out}, ntime)) + do t = 1, ntime + b = mod(t - 1, n) + 1 + xt(:, t) = x(:, b) + end do + call system_clock(count=c0, count_rate=crate) + !$omp target enter data map(to: xt) map(alloc: yt) + call {name}_infer_batch(ntime, xt, yt, status) + !$omp target exit data map(from: yt) map(delete: xt) + call system_clock(count=c1) + if (status /= 0) stop 21 + ns_per_point = real(c1 - c0, real64) / real(crate, real64) * 1.0e9_real64 / real(ntime, real64) + print '(A, ES24.16)', 'TIMING ', ns_per_point +end program +""" + +_DEV_HARNESS3 = """/* rosenna gpu-gate: infer_batch over raw device pointers ({backend}), written by the gate. */ +#include +#include +#include +#include "{name}.h" +int main(void) {{ + {init} + int n; + if (scanf("%d", &n) != 1) return 1; + double *hx = (double*)malloc(sizeof(double) * (size_t)n * {n_in}); + double *hy = (double*)malloc(sizeof(double) * (size_t)n * {n_out}); + for (int c = 0; c < n * {n_in}; ++c) if (scanf("%lf", &hx[c]) != 1) return 1; + double *dx = 0, *dy = 0; + if ({p}Malloc((void**)&dx, sizeof(double) * (size_t)n * {n_in}) != {p}Success) return 2; + if ({p}Malloc((void**)&dy, sizeof(double) * (size_t)n * {n_out}) != {p}Success) return 2; + if ({p}Memcpy(dx, hx, sizeof(double) * (size_t)n * {n_in}, {p}MemcpyHostToDevice) != {p}Success) return 2; + int status = {name}_infer_batch(n, dx, dy, 0); + if (status != 0) return 20 + status; + if ({p}Memcpy(hy, dy, sizeof(double) * (size_t)n * {n_out}, {p}MemcpyDeviceToHost) != {p}Success) return 3; + for (int p2 = 0; p2 < n; ++p2) {{ + for (int i = 0; i < {n_out}; ++i) printf("%.17e ", hy[p2 * {n_out} + i]); + printf("\\n"); + }} + /* Timing: ONE call over ntime points (a batch is meant to be called once + over many points), values tiled from the correctness batch above on + the host, copied to the device once, before the clock starts. */ + long ntime = {ntime}L; + double *hxt = (double*)malloc(sizeof(double) * (size_t)ntime * {n_in}); + for (long t = 0; t < ntime; ++t) {{ + int b = (int)(t % n); + for (int i = 0; i < {n_in}; ++i) hxt[t * {n_in} + i] = hx[b * {n_in} + i]; + }} + double *dxt = 0, *dyt = 0; + if ({p}Malloc((void**)&dxt, sizeof(double) * (size_t)ntime * {n_in}) != {p}Success) return 4; + if ({p}Malloc((void**)&dyt, sizeof(double) * (size_t)ntime * {n_out}) != {p}Success) return 4; + if ({p}Memcpy(dxt, hxt, sizeof(double) * (size_t)ntime * {n_in}, {p}MemcpyHostToDevice) != {p}Success) return 4; + struct timespec t0, t1; + clock_gettime(CLOCK_MONOTONIC, &t0); + /* No cudaMemcpy/hipMemcpy in this call (ruling R5): it only launches. The + nsys check (cuda backend, when nsys is on PATH) asserts that + structurally from the profile, not just by inspection of this source. */ + status = {name}_infer_batch((int)ntime, dxt, dyt, 0); + if (status != 0) return 30 + status; + {p}DeviceSynchronize(); + clock_gettime(CLOCK_MONOTONIC, &t1); + double secs = (double)(t1.tv_sec - t0.tv_sec) + (double)(t1.tv_nsec - t0.tv_nsec) * 1e-9; + printf("TIMING %.6f\\n", secs * 1.0e9 / (double)ntime); + {p}Free(dx); {p}Free(dy); {p}Free(dxt); {p}Free(dyt); + free(hx); free(hy); free(hxt); + return 0; +}} +""" + + +def _run_c_harness1(report, cfg_dir, plan, cc, flags, inputs, expected, env) -> bool: + name = plan.model + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + init = "" if plan.embed else f'if ({name}_init("{name}.rwt")) return 2;' + (cfg_dir / "gate_harness1.c").write_text(_C_HARNESS1.format( + name=name, n_in=n_in, n_out=n_out, init=init, ntime=_TIMED_ITERS)) + objs = ["gate_harness1.c"] + if not plan.embed: + objs.append(f"lib{name}.a") + cc_proc = _sh(report, "compile c per-point harness", + [cc, "-O2", "-Wall", "-Wextra", "-std=c11", *flags.split(), *objs, "-lm", + "-o", "gate_harness1"], cwd=cfg_dir) + if cc_proc.returncode != 0: + return False + run_proc = _sh(report, "run c per-point harness", ["./gate_harness1"], cwd=cfg_dir, + env=env, input_text=_stdin_for(inputs)) + if run_proc.returncode != 0: + return False + ok, _ = _check_output(report, run_proc.stdout, expected) + return ok + + +def _run_fortran_harness2(report, cfg_dir, plan, fc, flags, inputs, expected, env) -> bool: + name = plan.model + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + init_lines = "" if plan.embed else ( + f'call {name}_init("{name}.rwt", status); if (status /= 0) stop 2') + (cfg_dir / "gate_harness2.f90").write_text(_F_HARNESS2.format( + name=name, n_in=n_in, n_out=n_out, init_lines=init_lines, ntime=_TIMED_ITERS)) + fc_proc = _sh(report, "compile fortran per-point harness", + [fc, "-O2", "-Wall", "-Wextra", "-std=f2008", *flags.split(), + "gate_harness2.f90", f"lib{name}_f.a", "-o", "gate_harness2"], cwd=cfg_dir) + if fc_proc.returncode != 0: + return False + run_proc = _sh(report, "run fortran per-point harness", ["./gate_harness2"], cwd=cfg_dir, + env=env, input_text=_stdin_for(inputs)) + if run_proc.returncode != 0: + return False + ok, _ = _check_output(report, run_proc.stdout, expected) + return ok + + +def _run_c_harness3_omp(report, cfg_dir, plan, cc, flags, inputs, expected, env) -> bool: + name = plan.model + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + init = "" if plan.embed else f'if ({name}_init("{name}.rwt")) return 2;' + (cfg_dir / "gate_harness3.c").write_text(_C_HARNESS3_OMP.format( + name=name, n_in=n_in, n_out=n_out, init=init, ntime=_TIMED_ITERS)) + cc_proc = _sh(report, "compile c infer_batch harness (omp)", + [cc, "-O2", "-Wall", "-Wextra", "-std=c11", *flags.split(), + "gate_harness3.c", f"lib{name}.a", "-lm", "-o", "gate_harness3"], cwd=cfg_dir) + if cc_proc.returncode != 0: + return False + run_proc = _sh(report, "run c infer_batch harness (omp)", ["./gate_harness3"], cwd=cfg_dir, + env=env, input_text=_stdin_for(inputs)) + if run_proc.returncode != 0: + return False + ok, _ = _check_output(report, run_proc.stdout, expected) + return ok + + +def _run_fortran_harness3_omp(report, cfg_dir, plan, fc, flags, inputs, expected, env) -> bool: + name = plan.model + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + init_lines = "" if plan.embed else ( + f'call {name}_init("{name}.rwt", status); if (status /= 0) stop 2') + (cfg_dir / "gate_harness3.f90").write_text(_F_HARNESS3_OMP.format( + name=name, n_in=n_in, n_out=n_out, init_lines=init_lines, ntime=_TIMED_ITERS)) + fc_proc = _sh(report, "compile fortran infer_batch harness (omp)", + [fc, "-O2", "-Wall", "-Wextra", "-std=f2008", *flags.split(), + "gate_harness3.f90", f"lib{name}_f.a", "-o", "gate_harness3_f"], cwd=cfg_dir) + if fc_proc.returncode != 0: + return False + run_proc = _sh(report, "run fortran infer_batch harness (omp)", ["./gate_harness3_f"], + cwd=cfg_dir, env=env, input_text=_stdin_for(inputs)) + if run_proc.returncode != 0: + return False + ok, _ = _check_output(report, run_proc.stdout, expected) + return ok + + +def _run_dev_harness3(report, cfg_dir, plan, devcc, devflags, backend, inputs, expected) -> bool: + name = plan.model + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + prefix = "hip" if backend == "hip" else "cuda" + init = "" if plan.embed else f'if ({name}_init("{name}.rwt")) return 2;' + (cfg_dir / "gate_harness3.cu").write_text(_DEV_HARNESS3.format( + name=name, n_in=n_in, n_out=n_out, init=init, ntime=_TIMED_ITERS, p=prefix, + backend=backend)) + x_flag = "hip" if backend == "hip" else "cu" + proc = _sh(report, f"compile {backend} infer_batch harness", + [devcc, *devflags.split(), "-x", x_flag, + "gate_harness3.cu", f"lib{name}.a", "-o", "gate_harness3_dev"], cwd=cfg_dir) + if proc.returncode != 0: + return False + run_proc = _sh(report, f"run {backend} infer_batch harness", ["./gate_harness3_dev"], + cwd=cfg_dir, input_text=_stdin_for(inputs)) + if run_proc.returncode != 0: + return False + ok, _ = _check_output(report, run_proc.stdout, expected) + return ok + + +def _run_nsys_check(report: _Report, cfg_dir: Path) -> bool: + report.h("nsys check: cudaMemcpy count inside the timed infer_batch loop (ruling R5)", 4) + nsys = shutil.which("nsys") + if not nsys: + report.p("nsys not found on PATH; the cudaMemcpy-count check was NOT run " + "(recorded here rather than silently skipped).") + return True + stats_base = cfg_dir / "gate_nsys_profile" + proc = _sh(report, "nsys profile --stats=true", + [nsys, "profile", "--stats=true", "--force-overwrite=true", + "-o", str(stats_base), "./gate_harness3_dev"], cwd=cfg_dir) + combined = (proc.stdout or "") + (proc.stderr or "") + count = combined.count("cudaMemcpy") + report.p(f"cudaMemcpy occurrences reported by nsys: {count}") + if count != 0: + report.p("FAIL: infer_batch's timed loop must never call cudaMemcpy (ruling R5)") + return False + return proc.returncode == 0 + + +def run_gate(*, cc: str, fc: str, flags: str, backend: str, devcc=None, devflags: str = "", + out: str = ".", host_fallback: bool = False) -> int: + out_dir = Path(out) + out_dir.mkdir(parents=True, exist_ok=True) + report_path = out_dir / "gate-report.md" + report = _Report() + ok = True + try: + report.h("rosenna gpu-gate report", 1) + report.p(f"- model: {_MODEL}") + report.p(f"- backend: {backend}") + report.p(f"- host-fallback: {host_fallback}") + report.p(f"- cc: {cc}") + report.p(f"- fc: {fc}") + report.p(f"- flags: {flags!r}") + if backend != "omp": + report.p(f"- devcc: {devcc}") + report.p(f"- devflags: {devflags!r}") + if host_fallback: + report.p("- OMP_TARGET_OFFLOAD=MANDATORY is NOT set (host-fallback mode): " + "this run exercises the omp-backend contract end to end on a machine " + "with no accelerator, the same host-fallback contract " + "tests/test_device_c.py and tests/test_device_fortran.py already cover.") + else: + report.p("- OMP_TARGET_OFFLOAD=MANDATORY is set for every omp-backend harness: " + "a machine with no working offload device must fail here, loudly, " + "rather than silently pass by falling back to the host.") + + _record_versions(report, cc, fc, devcc if backend != "omp" else None) + + onnx_path = _ensure_model(report) + session = ort.InferenceSession(str(onnx_path)) + shape = session.get_inputs()[0].shape + inputs, expected = _live_reference(session, shape, np.float64, seed=42, batch=8) + if inputs is None: + report.p("FATAL: the onnxruntime reference for gemm_big is dead across every " + "resampled batch; nothing here would demonstrate anything.") + report_path.write_text(report.text()) + return 1 + + env = dict(os.environ) + if not host_fallback: + env["OMP_TARGET_OFFLOAD"] = "MANDATORY" + + for embed in (True, False): + label = "embedded" if embed else "file-loaded" + report.h(f"{_MODEL}: {label}", 2) + cfg_dir = out_dir / ("embedded" if embed else "file_loaded") + plan = _generate(cfg_dir, onnx_path, embed) + + c_built = _build_c_lib(report, cfg_dir, plan, cc, flags, backend, devcc, devflags) + f_built = _build_fortran_lib(report, cfg_dir, plan, fc, flags) + + report.h("c harness: per-point infer via target teams loop", 3) + if c_built: + if not _run_c_harness1(report, cfg_dir, plan, cc, flags, inputs, expected, env): + ok = False + else: + report.p("skipped: c library build failed") + ok = False + + report.h("fortran harness: per-point infer via target teams loop", 3) + if f_built: + if not _run_fortran_harness2(report, cfg_dir, plan, fc, flags, inputs, expected, env): + ok = False + else: + report.p("skipped: fortran library build failed") + ok = False + + report.h("infer_batch harness: device-resident data", 3) + if backend == "omp": + if c_built: + if not _run_c_harness3_omp(report, cfg_dir, plan, cc, flags, inputs, expected, env): + ok = False + else: + report.p("skipped: c library build failed") + ok = False + if f_built: + if not _run_fortran_harness3_omp(report, cfg_dir, plan, fc, flags, inputs, expected, env): + ok = False + else: + report.p("skipped: fortran library build failed") + ok = False + else: + if c_built: + dev_ok = _run_dev_harness3(report, cfg_dir, plan, devcc or + ("nvcc" if backend == "cuda" else "hipcc"), + devflags, backend, inputs, expected) + if not dev_ok: + ok = False + if backend == "cuda" and dev_ok: + if not _run_nsys_check(report, cfg_dir): + ok = False + else: + report.p(f"skipped: c library build failed (backend={backend})") + ok = False + + report.h("result", 2) + report.p("PASS: every configuration matched." if ok else + "FAIL: at least one configuration above did not match or did not run.") + report_path.write_text(report.text()) + return 0 if ok else 1 + except Exception as e: # noqa: BLE001 -- the report must still be written + report.h("FATAL", 2) + report.p(f"gate raised an unexpected exception: {type(e).__name__}: {e}") + report_path.write_text(report.text()) + return 1 diff --git a/python/rosenna/verify.py b/python/rosenna/verify.py index 592b0ab..9684601 100644 --- a/python/rosenna/verify.py +++ b/python/rosenna/verify.py @@ -22,6 +22,44 @@ class VerificationError(RuntimeError): """A comparison would not mean anything (e.g. the reference is dead).""" +def _live_reference(session, shape, dtype, seed=0, batch=8, max_attempts=10): + """Resample input batches until the onnxruntime reference itself is alive. + + Non-degeneracy is a property of the randomly generated fixture, not of + the code under test: several golden models (e.g. gemm_small) have no + manual_seed, so their weights differ on every regeneration, and an + all-zero reference (a dead model, e.g. every pre-activation negative + into a final ReLU) is a property of that draw of weights -- correct + generated code reproducing a dead model must *also* be all zero, so no + assertion on our own output can tell the two cases apart. The fix + belongs here, on the reference, before we ever build or run anything. + + `dtype` is a numpy dtype (e.g. np.float64), not a plan dtype string + ("f32"/"f64") -- this helper draws and feeds inputs at that numpy dtype + directly. + + Returns (inputs, expected) for the first batch whose reference has at + least two non-zero values across the whole batch, or (None, None) if + max_attempts batches all came back dead. + + Moved here (from tests/test_emit_fortran.py) so that `rosenna/gate.py` + can reuse it without importing test code; tests/test_emit_fortran.py + re-exports the same name so every existing `from tests.test_emit_fortran + import _live_reference` keeps working unchanged. + """ + rng = np.random.default_rng(seed) + for _ in range(max_attempts): + inputs = rng.uniform(-2, 2, (batch, int(np.prod(shape)))).astype(dtype) + expected = np.array([ + session.run(None, {session.get_inputs()[0].name: + row.reshape(shape).astype(np.float32)})[0].ravel() + for row in inputs + ]) + if np.count_nonzero(expected) >= 2: + return inputs, expected + return None, None + + @dataclass(frozen=True) class VerifyResult: lang: str # "fortran" | "c" diff --git a/python/tests/test_emit_fortran.py b/python/tests/test_emit_fortran.py index 96a75b9..5ed1188 100644 --- a/python/tests/test_emit_fortran.py +++ b/python/tests/test_emit_fortran.py @@ -6,6 +6,11 @@ from rosenna.plan import build_plan from rosenna.weights import write_weights from rosenna.emit_fortran import emit_fortran +# _live_reference now lives in rosenna/verify.py (rosenna/gate.py needs it too, +# and cannot import test code); re-exported here under its original name so +# every existing `from tests.test_emit_fortran import _live_reference` keeps +# working unchanged. +from rosenna.verify import _live_reference DENSE = ["gemm_small", "gemm_big", "gemm_nobias", "droplet", "batchnet"] @@ -49,35 +54,6 @@ def _build_and_run(tmp_path, onnx_path, name, inputs, dtype="f64"): return np.array([[float(v) for v in line.split()] for line in out.strip().splitlines()]) -def _live_reference(session, shape, dtype, seed=0, batch=8, max_attempts=10): - """Resample input batches until the onnxruntime reference itself is alive. - - Non-degeneracy is a property of the randomly generated fixture, not of - the code under test: several golden models (e.g. gemm_small) have no - manual_seed, so their weights differ on every regeneration, and an - all-zero reference (a dead model, e.g. every pre-activation negative - into a final ReLU) is a property of that draw of weights -- correct - generated code reproducing a dead model must *also* be all zero, so no - assertion on our own output can tell the two cases apart. The fix - belongs here, on the reference, before we ever build or run anything. - - Returns (inputs, expected) for the first batch whose reference has at - least two non-zero values across the whole batch, or (None, None) if - max_attempts batches all came back dead. - """ - rng = np.random.default_rng(seed) - for _ in range(max_attempts): - inputs = rng.uniform(-2, 2, (batch, int(np.prod(shape)))).astype(dtype) - expected = np.array([ - session.run(None, {session.get_inputs()[0].name: - row.reshape(shape).astype(np.float32)})[0].ravel() - for row in inputs - ]) - if np.count_nonzero(expected) >= 2: - return inputs, expected - return None, None - - @pytest.mark.parametrize("name", DENSE) def test_matches_onnxruntime(tmp_path, golden_model, name): onnx_path = golden_model(name) diff --git a/python/tests/test_gate.py b/python/tests/test_gate.py new file mode 100644 index 0000000..11912f1 --- /dev/null +++ b/python/tests/test_gate.py @@ -0,0 +1,29 @@ +import shutil +import pytest +from rosenna.cli import main +from tests.test_device_c import _omp_cc + + +def test_gate_runs_in_host_fallback_mode_and_writes_a_report(tmp_path, golden_model): + # On a machine without a GPU the gate runs with --host-fallback, which drops the + # MANDATORY requirement but exercises every other step, so the script itself is tested. + golden_model("gemm_big") + cc = _omp_cc() + fc = shutil.which("gfortran") or pytest.skip("no gfortran") + rc = main(["gpu-gate", "--cc", cc, "--fc", fc, "--flags", "-fopenmp", "--backend", "omp", + "--host-fallback", "--out", str(tmp_path)]) + assert rc == 0 + report = (tmp_path / "gate-report.md").read_text() + for key in ("gemm_big", "embedded", "file-loaded", "fortran", "c", "infer_batch", + "backend: omp", "ns per point", "host-fallback"): + assert key in report + + +def test_gate_fails_loudly_when_offload_is_mandatory_and_absent(tmp_path, golden_model): + golden_model("gemm_big") + cc = _omp_cc() + fc = shutil.which("gfortran") or pytest.skip("no gfortran") + rc = main(["gpu-gate", "--cc", cc, "--fc", fc, "--flags", "-fopenmp", "--backend", "omp", + "--out", str(tmp_path)]) + assert rc == 1 + assert "MANDATORY" in (tmp_path / "gate-report.md").read_text() From 8c7f0d346732d842ef5487bac0ecc590226987f2 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Mon, 14 Sep 2026 00:15:28 -0500 Subject: [PATCH 11/84] fix: link cuda/hip harnesses with the device compiler, scope the nsys check, time only the call Fixes from review round 1 of gate.py (two Critical, three Important; nothing CUDA-related has ever run on this machine, so these are traced-by-reading fixes, not verified-by-running): - Ruling R14 (Critical): _run_c_harness1 now compiles gate_harness1.c with the host compiler and its offload flags as before, but links it with the device compiler (--devcc) for --backend cuda|hip instead of the host compiler and -lm. A file-loaded plan's lib.a was built by nvcc/hipcc for those backends (init's upload/bind calls cudaMalloc/cudaMemcpy/cudaMemcpyToSymbol or the hip equivalents), so the plain host compiler left those undefined on every real run. --backend omp keeps the host compiler as the link driver. - Ruling R15 (Critical): _run_nsys_check no longer counts "cudaMemcpy" over the whole profiled harness (which legitimately memcpys in its untimed setup and would fail an R5-compliant infer_batch). It now brackets only the timed infer_batch call with an nvtx range (-DROSENNA_GATE_NVTX=1 in the .cu driver), profiles with --capture-range=nvtx --nvtx-capture=rosenna_timed, and reads the count from `nsys stats --report cuda_api_sum --format csv`. Skips (without failing the gate) when nsys is absent, when the nvtx header does not compile, or for --backend hip (rocprof scoping is a follow-up). - Ruling R16 (Important): the omp infer_batch harnesses (C and Fortran) now time only the infer_batch call -- target enter/exit data move outside the t0/c0..t1/c1 window, matching what the cuda/hip driver already did. - Important: --help gains an epilog with the three concrete host-compiler pairings (nvc/nvfortran+cuda, amdclang/amdflang+hip, gcc/gfortran+omp+host-fallback); --devcc's help text no longer says "required" (it has a default). README's example swaps the gcc-15/--backend cuda pairing (gcc-15 cannot offload) for the nvc one. - Ruling R17 (Important): closure()'s ghost-layer gap in patch.md is fixed, not just documented -- it now writes nut over the padded block minus one layer on each side (the widest range where its central difference has both neighbours in bounds, which the reference microfd.c's halo() has already filled), not FOR3's interior-only range, so face()'s read one ghost cell into the low boundary is no longer silently zero. The batched alternative's gather loop is fixed the same way; its own remaining edge case (closure_infer_batch is called over all nc cells, one layer wider than the gather loop fills) is called out explicitly instead of reusing the old blanket caveat. Minor: patch.md's no-op -/+ diff pair in section 2 is now a context line; --flags/--devflags carry a comment pointing at _join_dash_valued_options. --- python/examples/microfd_closure/README.md | 18 +- python/examples/microfd_closure/patch.md | 60 +++--- python/rosenna/cli.py | 22 ++- python/rosenna/gate.py | 216 +++++++++++++++++++--- 4 files changed, 256 insertions(+), 60 deletions(-) diff --git a/python/examples/microfd_closure/README.md b/python/examples/microfd_closure/README.md index a74187f..7031d8e 100644 --- a/python/examples/microfd_closure/README.md +++ b/python/examples/microfd_closure/README.md @@ -59,14 +59,20 @@ that path does link `libclosure.a`. ## Validating on a GPU machine ``` -rosenna gpu-gate --cc gcc-15 --fc gfortran --flags -fopenmp \ +rosenna gpu-gate --cc nvc --fc nvfortran --flags "-mp=gpu -gpu=cc80" \ --backend cuda --devcc nvcc --out /tmp/rosenna-gate ``` records `gate-report.md`: every command it ran, every line of output, the compiler versions, and nanoseconds per point for each of the three -harnesses. Run it with `--backend hip --devcc hipcc` on an AMD GPU, or -`--backend omp` to check the OpenMP-target fallback on either. Only after -that report exists for the backend and hardware you actually run microfd on -should the closure above be described as device-validated rather than -host-validated. +harnesses. `--cc`/`--fc` must be a HOST compiler capable of OpenMP target +offload (the pairing above, NVIDIA HPC SDK's `nvc`/`nvfortran`, not a plain +`gcc` -- `gcc-15` from Homebrew, for instance, has no offload device to +target and would silently run every per-point harness on the host even +though `--backend cuda` asks for the native kernel); `rosenna gpu-gate +--help` lists the AMD (`amdclang`/`amdflang`/`hip`) and no-GPU +(`gcc`/`gfortran`/`omp --host-fallback`) pairings too. Run it with `--backend +hip --devcc hipcc` on an AMD GPU, or `--backend omp` to check the +OpenMP-target fallback on either. Only after that report exists for the +backend and hardware you actually run microfd on should the closure above be +described as device-validated rather than host-validated. diff --git a/python/examples/microfd_closure/patch.md b/python/examples/microfd_closure/patch.md index 0737007..f59a03c 100644 --- a/python/examples/microfd_closure/patch.md +++ b/python/examples/microfd_closure/patch.md @@ -43,8 +43,7 @@ the halo buffers); `nut` is one more, one value per cell (not `NV*nc` like ```diff double L[3], o[3], h[3], gamma, mu, pr, cfl, tend, t; -- double *q, *q1, *w, *F, *sbuf[2], *rbuf[2]; // F holds all three directions: [d][NV][nc] -+ double *q, *q1, *w, *F, *sbuf[2], *rbuf[2]; // F holds all three directions: [d][NV][nc] + double *q, *q1, *w, *F, *sbuf[2], *rbuf[2]; // F holds all three directions: [d][NV][nc] + double *nut; // turbulent/SGS viscosity from the closure, one value per cell MPI_Comm comm; ``` @@ -53,9 +52,9 @@ the halo buffers); `nut` is one more, one value per cell (not `NV*nc` like Placed after `prim()` (which fills `w`, the primitives array `closure()` reads) and before `face()` (which reads `g.nut`), so it slots directly into -`rhs_eval`'s existing sequence. It uses microfd's own `LOCALS`/`FOR3`/`IDX` -macros and its own naming convention for the primitives array (`w[nc+c]` = -u, `w[2*nc+c]` = v, `w[3*nc+c]` = the third velocity component, named `s` +`rhs_eval`'s existing sequence. It uses microfd's own `LOCALS`/`IDX` macros +and its own naming convention for the primitives array (`w[nc+c]` = u, +`w[2*nc+c]` = v, `w[3*nc+c]` = the third velocity component, named `s` throughout microfd.c to avoid colliding with the `w` array itself). The gradients are ordinary second-order central differences at the cell center, distinct from `face()`'s one-sided/averaged stencil at a face: @@ -70,7 +69,18 @@ distinct from `face()`'s one-sided/averaged stencil at a face: +static void closure(void){ // per-cell turbulent viscosity from the velocity-gradient closure model + LOCALS; const double*w=g.w; double*nut=g.nut; const double h0=g.h[0],h1=g.h[1],h2=g.h[2]; -+ FOR3(NG,NG,NG,){ ++ // Range is the padded block minus one layer on each side (index 1 to ++ // nx+2*NG-2 along x, and likewise y, z) -- NOT FOR3's interior-only range ++ // (NG to n[d]+NG-1). face(d) reads g.nut one ghost cell into the low ++ // boundary of each direction (its own loop starts at i0=NG-(d==0) etc., ++ // to reach the boundary face using the adjacent ghost cell), and halo() ++ // has already filled every ghost layer by the time closure() runs here ++ // (right after prim(), itself right after halo(), in rhs_eval below), so ++ // the central difference is valid at every index with both neighbours in ++ // bounds -- exactly this range, symmetric on both sides, expressed ++ // directly rather than through FOR3 since the bounds differ from it. ++ #pragma omp target teams loop collapse(3) ++ for(int k=1;k argparse.ArgumentParser: gate = sub.add_parser( "gpu-gate", + formatter_class=argparse.RawDescriptionHelpFormatter, help="build and run the device-library validation harnesses on a GPU machine " "(gemm_big, embedded and file-loaded, both languages, three harnesses); " - "writes gate-report.md") + "writes gate-report.md", + epilog="""\ +Harnesses 1 and 2 (the per-point C and Fortran hosts) always need a HOST +compiler capable of OpenMP target offload, regardless of --backend: the +per-point infer() call always goes through the host compiler's own offload +region, never through --devcc. Concrete pairings: + + --cc nvc --fc nvfortran --flags "-mp=gpu -gpu=cc80" --backend cuda --devcc nvcc + --cc amdclang --fc amdflang --flags "-fopenmp --offload-arch=gfx90a" --backend hip --devcc hipcc + --cc gcc --fc gfortran --flags -fopenmp --backend omp --host-fallback (no GPU) +""") gate.add_argument("--cc", required=True, help="host C compiler") gate.add_argument("--fc", required=True, help="host Fortran compiler") + # A value here that itself starts with '-' (e.g. -fopenmp, or a + # multi-flag string like "-mp=gpu -gpu=cc80") is handled by + # _join_dash_valued_options below, not by argparse's own parsing. gate.add_argument("--flags", default="", help="host offload flags, e.g. -fopenmp") gate.add_argument("--backend", choices=["cuda", "hip", "omp"], required=True, help="which infer_batch implementation to build and exercise") - gate.add_argument("--devcc", default=None, help="nvcc or hipcc; required for --backend cuda|hip") + gate.add_argument("--devcc", default=None, + help="device compiler for --backend cuda|hip, and the link driver " + "for the file-loaded C harnesses there (ruling R14); NOT " + "required -- default: nvcc for cuda, hipcc for hip") + # Same dash-valued handling as --flags; see the comment above. gate.add_argument("--devflags", default="", help="device compiler flags") gate.add_argument("--out", default=".", help="directory for generated sources and gate-report.md") gate.add_argument("--host-fallback", action="store_true", diff --git a/python/rosenna/gate.py b/python/rosenna/gate.py index 3797993..45f9993 100644 --- a/python/rosenna/gate.py +++ b/python/rosenna/gate.py @@ -22,6 +22,8 @@ on a GPU machine and its report recorded. Until then: compiles and runs on the host; device path unvalidated. """ +import csv +import io import os import platform import shutil @@ -301,18 +303,24 @@ def _check_output(report: _Report, stdout: str, expected) -> tuple: int b = (int)(t % n); for (int i = 0; i < {n_in}; ++i) xt[t * {n_in} + i] = x[b * {n_in} + i]; }} - double t0 = omp_get_wtime(); + /* Ruling R16: the mapping (a real transfer, exempt from R5 since it is + the harness's own setup, not inside infer_batch) happens before t0 + and is undone after t1, so the timed window holds only the call -- + matching what the cuda/hip .cu driver already does. */ #ifdef _OPENMP #pragma omp target enter data map(to: xt[0:ntime*{n_in}]) map(alloc: yt[0:ntime*{n_out}]) +#endif + double t0 = omp_get_wtime(); +#ifdef _OPENMP #pragma omp target data use_device_ptr(xt, yt) #endif {{ status = {name}_infer_batch((int)ntime, xt, yt, 0); }} + double t1 = omp_get_wtime(); #ifdef _OPENMP #pragma omp target exit data map(from: yt[0:ntime*{n_out}]) map(delete: xt[0:ntime*{n_in}]) #endif - double t1 = omp_get_wtime(); if (status != 0) return 30 + status; printf("TIMING %.6f\\n", (t1 - t0) * 1.0e9 / (double)ntime); free(x); free(y); free(xt); free(yt); @@ -350,11 +358,13 @@ def _check_output(report: _Report, stdout: str, expected) -> tuple: b = mod(t - 1, n) + 1 xt(:, t) = x(:, b) end do - call system_clock(count=c0, count_rate=crate) + ! Ruling R16: map before c0 and unmap after c1, so the timed window + ! holds only the infer_batch call, matching the cuda/hip .cu driver. !$omp target enter data map(to: xt) map(alloc: yt) + call system_clock(count=c0, count_rate=crate) call {name}_infer_batch(ntime, xt, yt, status) - !$omp target exit data map(from: yt) map(delete: xt) call system_clock(count=c1) + !$omp target exit data map(from: yt) map(delete: xt) if (status /= 0) stop 21 ns_per_point = real(c1 - c0, real64) / real(crate, real64) * 1.0e9_real64 / real(ntime, real64) print '(A, ES24.16)', 'TIMING ', ns_per_point @@ -366,6 +376,15 @@ def _check_output(report: _Report, stdout: str, expected) -> tuple: #include #include #include "{name}.h" +#ifdef ROSENNA_GATE_NVTX +/* Ruling R15: only defined (via -DROSENNA_GATE_NVTX=1) for the separate + build the nsys check compiles, so the ordinary timed run above never + needs this header. nvtx3 is documented as header-only (it loads + libnvToolsExt itself at runtime); _run_nsys_check retries the link with + -lnvToolsExt if the no-link form fails, since that has not been verified + against every toolkit version here. */ +#include +#endif int main(void) {{ {init} int n; @@ -399,10 +418,19 @@ def _check_output(report: _Report, stdout: str, expected) -> tuple: if ({p}Memcpy(dxt, hxt, sizeof(double) * (size_t)ntime * {n_in}, {p}MemcpyHostToDevice) != {p}Success) return 4; struct timespec t0, t1; clock_gettime(CLOCK_MONOTONIC, &t0); - /* No cudaMemcpy/hipMemcpy in this call (ruling R5): it only launches. The - nsys check (cuda backend, when nsys is on PATH) asserts that - structurally from the profile, not just by inspection of this source. */ + /* No cudaMemcpy/hipMemcpy in this call (ruling R5). Ruling R15: the nsys + check brackets ONLY this call with an nvtx range and profiles with + --capture-range=nvtx, so its cudaMemcpy count is scoped to the call + itself, not to this driver's untimed setup above (which legitimately + memcpys) -- counting across the whole profile would fail an + R5-compliant infer_batch. */ +#ifdef ROSENNA_GATE_NVTX + nvtxRangePushA("rosenna_timed"); +#endif status = {name}_infer_batch((int)ntime, dxt, dyt, 0); +#ifdef ROSENNA_GATE_NVTX + nvtxRangePop(); +#endif if (status != 0) return 30 + status; {p}DeviceSynchronize(); clock_gettime(CLOCK_MONOTONIC, &t1); @@ -415,19 +443,41 @@ def _check_output(report: _Report, stdout: str, expected) -> tuple: """ -def _run_c_harness1(report, cfg_dir, plan, cc, flags, inputs, expected, env) -> bool: +def _run_c_harness1(report, cfg_dir, plan, cc, flags, backend, devcc, devflags, + inputs, expected, env) -> bool: name = plan.model n_in, n_out = plan.input.shape[0], plan.output.shape[0] init = "" if plan.embed else f'if ({name}_init("{name}.rwt")) return 2;' (cfg_dir / "gate_harness1.c").write_text(_C_HARNESS1.format( name=name, n_in=n_in, n_out=n_out, init=init, ntime=_TIMED_ITERS)) - objs = ["gate_harness1.c"] + # Ruling R14: compilation always goes through the HOST compiler with its + # own offload flags (nvc's -mp=gpu, amdclang's -fopenmp + # --offload-arch=..., or plain -fopenmp) -- this step does not change + # with --backend. Only the final LINK does: for cuda/hip, a file-loaded + # plan's lib.a was built by nvcc/hipcc (init's upload/bind calls + # cudaMalloc/cudaMemcpy/cudaMemcpyToSymbol or the hip equivalents), so + # linking it with the plain host compiler and -lm leaves those + # undefined on every real run. Using the device compiler as the LINK + # driver instead pulls in the runtime library and its -L path from the + # toolkit itself, with nothing hardcoded here; --backend omp keeps the + # host compiler as the link driver, since its infer_batch is pure + # OpenMP with no runtime-API calls to resolve. + cc_proc = _sh(report, "compile c per-point harness (host compiler, host offload flags)", + [cc, "-O2", "-Wall", "-Wextra", "-std=c11", *flags.split(), + "-c", "gate_harness1.c", "-o", "gate_harness1.o"], cwd=cfg_dir) + if cc_proc.returncode != 0: + return False + objs = ["gate_harness1.o"] if not plan.embed: objs.append(f"lib{name}.a") - cc_proc = _sh(report, "compile c per-point harness", - [cc, "-O2", "-Wall", "-Wextra", "-std=c11", *flags.split(), *objs, "-lm", - "-o", "gate_harness1"], cwd=cfg_dir) - if cc_proc.returncode != 0: + if backend == "omp": + link_cmd = [cc, *flags.split(), *objs, "-lm", "-o", "gate_harness1"] + link_label = "link c per-point harness (host compiler, --backend omp)" + else: + link_cmd = [devcc, *devflags.split(), *objs, "-lm", "-o", "gate_harness1"] + link_label = f"link c per-point harness (device compiler {devcc}, --backend {backend})" + link_proc = _sh(report, link_label, link_cmd, cwd=cfg_dir) + if link_proc.returncode != 0: return False run_proc = _sh(report, "run c per-point harness", ["./gate_harness1"], cwd=cfg_dir, env=env, input_text=_stdin_for(inputs)) @@ -518,24 +568,122 @@ def _run_dev_harness3(report, cfg_dir, plan, devcc, devflags, backend, inputs, e return ok -def _run_nsys_check(report: _Report, cfg_dir: Path) -> bool: - report.h("nsys check: cudaMemcpy count inside the timed infer_batch loop (ruling R5)", 4) +def _probe_nvtx_header(report: _Report, cfg_dir: Path, devcc: str, devflags: str) -> bool: + """Compile-only probe for with the device compiler. + + Ruling R15: if the header is not found, the nsys check is skipped with + a named reason rather than failing the gate or attempting to build the + nvtx-instrumented variant anyway. + """ + (cfg_dir / "gate_nvtx_probe.cu").write_text( + "#include \nint main(void){return 0;}\n") + proc = _sh(report, "probe for ", + [devcc, *devflags.split(), "-x", "cu", "-c", "gate_nvtx_probe.cu", + "-o", "gate_nvtx_probe.o"], cwd=cfg_dir) + return proc.returncode == 0 + + +def _sum_cudamemcpy_calls(csv_text: str) -> int: + """Sum the Num Calls column of every cuda_api_sum row whose Name starts with cudaMemcpy. + + Column names/casing can drift slightly across Nsight Systems versions, + so this matches case-insensitively by substring ("name", "num calls") + rather than an exact header string. + """ + reader = csv.DictReader(io.StringIO(csv_text)) + if not reader.fieldnames: + return 0 + name_col = next((f for f in reader.fieldnames if "name" in f.lower()), None) + calls_col = next((f for f in reader.fieldnames + if "num calls" in f.lower() or "numcalls" in f.lower().replace(" ", "")), + None) + if not name_col or not calls_col: + return 0 + total = 0 + for row in reader: + name = (row.get(name_col) or "").strip() + if name.startswith("cudaMemcpy"): + total += int(float(row.get(calls_col) or 0)) + return total + + +def _run_nsys_check(report: _Report, cfg_dir: Path, plan, devcc: str, devflags: str, + backend: str, inputs) -> bool: + """Ruling R15: assert zero cudaMemcpy calls inside the timed infer_batch call only. + + Profiling the whole harness and counting "cudaMemcpy" across nsys's + free-text output (the previous approach) would fail an R5-compliant + infer_batch: the driver's own untimed setup (H2D copies before the + clock starts) legitimately calls cudaMemcpy. Scoped instead with an + nvtx range around only the timed call, `nsys profile + --capture-range=nvtx --nvtx-capture=rosenna_timed`, and the count read + from `nsys stats --report cuda_api_sum --format csv` on the resulting + report. None of this has ever run (no nvcc/nsys here); see the task + report for what remains unexercised. + """ + report.h("nsys check: cudaMemcpy count inside the nvtx-scoped infer_batch call (ruling R15)", 4) + if backend == "hip": + report.p("nsys check skipped: --backend hip (rocprof scoping of the call is a " + "follow-up; nsys/nvtx are CUDA-only).") + return True nsys = shutil.which("nsys") if not nsys: report.p("nsys not found on PATH; the cudaMemcpy-count check was NOT run " "(recorded here rather than silently skipped).") return True + if not _probe_nvtx_header(report, cfg_dir, devcc, devflags): + report.p("nsys check skipped: nvtx header not found " + "( did not compile with this device compiler).") + return True + + name = plan.model + n_in, n_out = plan.input.shape[0], plan.output.shape[0] + init = "" if plan.embed else f'if ({name}_init("{name}.rwt")) return 2;' + (cfg_dir / "gate_harness3.cu").write_text(_DEV_HARNESS3.format( + name=name, n_in=n_in, n_out=n_out, init=init, ntime=_TIMED_ITERS, p="cuda", + backend=backend)) + # nvtx3 () is documented as header-only: it loads + # libnvToolsExt itself at runtime rather than needing it at link time. + # Some toolkit versions still expect an explicit link; try without + # -lnvToolsExt first (the documented form) and retry once with it if + # linking fails, noting which form was needed. Neither path has run here. + base_cmd = [devcc, *devflags.split(), "-DROSENNA_GATE_NVTX=1", "-x", "cu", + "gate_harness3.cu", f"lib{name}.a"] + proc = _sh(report, "compile nvtx-bracketed infer_batch harness (no explicit -lnvToolsExt)", + [*base_cmd, "-o", "gate_harness3_nvtx"], cwd=cfg_dir) + if proc.returncode != 0: + proc = _sh(report, "compile nvtx-bracketed infer_batch harness (retry: -lnvToolsExt)", + [*base_cmd, "-lnvToolsExt", "-o", "gate_harness3_nvtx"], cwd=cfg_dir) + if proc.returncode != 0: + report.p("nsys check skipped: the nvtx-bracketed driver did not link, " + "with or without -lnvToolsExt.") + return True + stats_base = cfg_dir / "gate_nsys_profile" - proc = _sh(report, "nsys profile --stats=true", - [nsys, "profile", "--stats=true", "--force-overwrite=true", - "-o", str(stats_base), "./gate_harness3_dev"], cwd=cfg_dir) - combined = (proc.stdout or "") + (proc.stderr or "") - count = combined.count("cudaMemcpy") - report.p(f"cudaMemcpy occurrences reported by nsys: {count}") + profile_proc = _sh( + report, "nsys profile --capture-range=nvtx --nvtx-capture=rosenna_timed --stats=true", + [nsys, "profile", "--capture-range=nvtx", "--nvtx-capture=rosenna_timed", + "--stats=true", "--force-overwrite=true", "-o", str(stats_base), + "./gate_harness3_nvtx"], cwd=cfg_dir, input_text=_stdin_for(inputs)) + if profile_proc.returncode != 0: + report.p("FAIL: nsys profile did not complete successfully") + return False + + report_file = stats_base.with_suffix(".nsys-rep") + stats_proc = _sh(report, "nsys stats --report cuda_api_sum --format csv", + [nsys, "stats", "--report", "cuda_api_sum", "--format", "csv", + str(report_file)], cwd=cfg_dir) + if stats_proc.returncode != 0: + report.p("FAIL: nsys stats did not complete successfully") + return False + + count = _sum_cudamemcpy_calls(stats_proc.stdout) + report.p(f"cudaMemcpy* Num Calls inside the nvtx-scoped infer_batch call, from " + f"cuda_api_sum: {count}") if count != 0: - report.p("FAIL: infer_batch's timed loop must never call cudaMemcpy (ruling R5)") + report.p("FAIL: infer_batch's timed call must never call cudaMemcpy (ruling R5)") return False - return proc.returncode == 0 + return True def run_gate(*, cc: str, fc: str, flags: str, backend: str, devcc=None, devflags: str = "", @@ -566,7 +714,12 @@ def run_gate(*, cc: str, fc: str, flags: str, backend: str, devcc=None, devflags "a machine with no working offload device must fail here, loudly, " "rather than silently pass by falling back to the host.") - _record_versions(report, cc, fc, devcc if backend != "omp" else None) + # Resolved once, used everywhere a device compiler command is needed: + # --devcc has a default (it is not required, see --help), so every + # call site uses this instead of repeating the fallback logic. + resolved_devcc = devcc or ("nvcc" if backend == "cuda" else "hipcc") + + _record_versions(report, cc, fc, resolved_devcc if backend != "omp" else None) onnx_path = _ensure_model(report) session = ort.InferenceSession(str(onnx_path)) @@ -593,7 +746,8 @@ def run_gate(*, cc: str, fc: str, flags: str, backend: str, devcc=None, devflags report.h("c harness: per-point infer via target teams loop", 3) if c_built: - if not _run_c_harness1(report, cfg_dir, plan, cc, flags, inputs, expected, env): + if not _run_c_harness1(report, cfg_dir, plan, cc, flags, backend, resolved_devcc, + devflags, inputs, expected, env): ok = False else: report.p("skipped: c library build failed") @@ -623,13 +777,17 @@ def run_gate(*, cc: str, fc: str, flags: str, backend: str, devcc=None, devflags ok = False else: if c_built: - dev_ok = _run_dev_harness3(report, cfg_dir, plan, devcc or - ("nvcc" if backend == "cuda" else "hipcc"), + dev_ok = _run_dev_harness3(report, cfg_dir, plan, resolved_devcc, devflags, backend, inputs, expected) if not dev_ok: ok = False - if backend == "cuda" and dev_ok: - if not _run_nsys_check(report, cfg_dir): + else: + # Ruling R15: invoked for both backends; it skips + # itself (with a named reason) for hip, and for cuda + # when nsys or the nvtx header is unavailable, none + # of which fails the gate on its own. + if not _run_nsys_check(report, cfg_dir, plan, resolved_devcc, devflags, + backend, inputs): ok = False else: report.p(f"skipped: c library build failed (backend={backend})") From 82aca4463c88cb62a48f2fe248d2a01dffda7991 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Mon, 14 Sep 2026 00:26:47 -0500 Subject: [PATCH 12/84] fix: nsys parse reports inconclusive, forward host offload flags at link, drop the live-reference assert Fix round 2 (two new rulings added after round 1 went out, plus one pre-existing test defect the reviewer found; nothing CUDA-related has run on this machine, so R18/R19 are traced-by-reading fixes): - Ruling R18: _sum_cudamemcpy_calls now returns a structured NsysParseResult(parsed, count) instead of a bare int. `parsed` is True only when the Name/Num Calls columns were recognised AND at least one data row was actually read; an unparseable CSV (no fieldnames, no recognised columns, or a header with zero data rows) comes back as parsed=False, never as a bare 0. _run_nsys_check now treats parsed=False as a gate FAILURE ("nsys check inconclusive: could not parse cuda_api_sum"), with the first lines of what nsys actually produced in the report, rather than the previous silent pass. New unit tests in tests/test_gate.py exercise all three cases from the ruling: two cudaMemcpy* rows summing to 5 plus a launch row (parsed, count=5); zero memcpy rows plus a launch row (parsed, count=0); and empty/unrelated CSV text (not parsed, in both cases). - Ruling R19: _run_c_harness1's cuda/hip link line now forwards the HOST offload flags (e.g. nvc's -mp=gpu -gpu=cc80) into the device-compiler link, not just --devflags -- new _host_flags_for_devcc_link wraps each host flag as -Xcompiler for cuda (nvcc otherwise treats an unrecognised flag as compiler-only) and passes them through directly for hip (hipcc is clang-based, like the expected amdclang host pairing). Without this, gate_harness1.o's own OpenMP-target runtime (from the host compiler) could be unresolved at a link driven purely by devflags. README.md's example now also documents the alternative -- host compiler as the link driver with the runtime named explicitly (-L$CUDA_HOME/lib64 -lcudart / -L$ROCM_PATH/lib -lamdhip64) -- for a toolchain that rejects the device-compiler-as-driver form. - Ruling R20 (pre-existing defect): test_device_fortran.py:: test_fortran_target_regions_are_real called _live_reference and asserted `inputs is not None` instead of skipping on a dead model, unlike every other _live_reference call site in the test suite -- it failed deterministically once gemm_small's current (unseeded) weights happened to produce a dead model. The MANDATORY check only needs one point and a libgomp refusal, not a live onnxruntime reference, so it now uses a constant input (np.full(..., 0.5)), mirroring test_device_c.py::test_target_regions_are_real, which already did this correctly. The file's one other _live_reference call site already used the pytest.skip dead-model guard; no other fix was needed there. Full suite: 165 passed, 18 skipped, 0 failed (162 + 3 new _sum_cudamemcpy_calls unit tests; the previous round's 161/18/1 is now clean). --- python/examples/microfd_closure/README.md | 17 +++++ python/rosenna/gate.py | 90 ++++++++++++++++++++--- python/tests/test_device_fortran.py | 10 ++- python/tests/test_gate.py | 47 ++++++++++++ 4 files changed, 148 insertions(+), 16 deletions(-) diff --git a/python/examples/microfd_closure/README.md b/python/examples/microfd_closure/README.md index 7031d8e..a0a0f23 100644 --- a/python/examples/microfd_closure/README.md +++ b/python/examples/microfd_closure/README.md @@ -76,3 +76,20 @@ hip --devcc hipcc` on an AMD GPU, or `--backend omp` to check the OpenMP-target fallback on either. Only after that report exists for the backend and hardware you actually run microfd on should the closure above be described as device-validated rather than host-validated. + +For the file-loaded configuration's per-point harness under `--backend +cuda|hip`, the gate links the generated library with the device compiler as +the link driver (`nvcc`/`hipcc`, forwarding the host offload flags through +as `-Xcompiler ` for cuda, directly for hip -- ruling R14/R19); this +is the form `gate.py` builds and neither form has been verified against a +real toolchain here. If that link fails on a toolchain that rejects a +device compiler as the driver for a host-compiled object, the documented +way out is linking with the HOST compiler instead and naming the CUDA/HIP +runtime explicitly: + +``` +# cuda +nvc -mp=gpu -gpu=cc80 gate_harness1.o libgemm_big.a -L$CUDA_HOME/lib64 -lcudart -lm -o gate_harness1 +# hip +amdclang -fopenmp --offload-arch=gfx90a gate_harness1.o libgemm_big.a -L$ROCM_PATH/lib -lamdhip64 -lm -o gate_harness1 +``` diff --git a/python/rosenna/gate.py b/python/rosenna/gate.py index 45f9993..882e7a0 100644 --- a/python/rosenna/gate.py +++ b/python/rosenna/gate.py @@ -29,6 +29,7 @@ import shutil import subprocess import sys +from dataclasses import dataclass from pathlib import Path import numpy as np @@ -443,6 +444,30 @@ def _check_output(report: _Report, stdout: str, expected) -> tuple: """ +def _host_flags_for_devcc_link(flags: str, backend: str) -> list: + """Ruling R19: forward the HOST offload flags into the device-compiler link line. + + Ruling R14 uses the device compiler as the link driver for a + file-loaded plan under --backend cuda|hip, which resolves the CUDA/HIP + runtime symbols init needs -- but on its own it never sees the host + compiler's own offload-runtime flags (nvc's -mp=gpu -gpu=cc80), so that + runtime can be unresolved at link too, since gate_harness1.o was + compiled by the host compiler with those flags. nvcc treats an + unrecognised flag as compiler-only unless wrapped -Xcompiler , so + each host flag is forwarded that way for cuda; hipcc is clang-based + (like amdclang, the expected host compiler pairing) and accepts the + host flags directly. Neither form has been verified against a real + toolchain here (see the task report). + """ + host_flags = flags.split() + if backend == "hip": + return host_flags + out = [] + for f in host_flags: + out += ["-Xcompiler", f] + return out + + def _run_c_harness1(report, cfg_dir, plan, cc, flags, backend, devcc, devflags, inputs, expected, env) -> bool: name = plan.model @@ -474,8 +499,15 @@ def _run_c_harness1(report, cfg_dir, plan, cc, flags, backend, devcc, devflags, link_cmd = [cc, *flags.split(), *objs, "-lm", "-o", "gate_harness1"] link_label = "link c per-point harness (host compiler, --backend omp)" else: - link_cmd = [devcc, *devflags.split(), *objs, "-lm", "-o", "gate_harness1"] - link_label = f"link c per-point harness (device compiler {devcc}, --backend {backend})" + # Ruling R19: the host offload flags (nvc's -mp=gpu -gpu=cc80, or + # amdclang's -fopenmp --offload-arch=...) are forwarded into this + # link too -- gate_harness1.o's own OpenMP-target runtime needs + # them, and devflags alone (nvcc's/hipcc's own flags) does not + # supply them. + link_cmd = [devcc, *devflags.split(), *_host_flags_for_devcc_link(flags, backend), + *objs, "-lm", "-o", "gate_harness1"] + link_label = (f"link c per-point harness (device compiler {devcc}, --backend {backend}, " + f"host offload flags forwarded)") link_proc = _sh(report, link_label, link_cmd, cwd=cfg_dir) if link_proc.returncode != 0: return False @@ -583,28 +615,51 @@ def _probe_nvtx_header(report: _Report, cfg_dir: Path, devcc: str, devflags: str return proc.returncode == 0 -def _sum_cudamemcpy_calls(csv_text: str) -> int: - """Sum the Num Calls column of every cuda_api_sum row whose Name starts with cudaMemcpy. +@dataclass(frozen=True) +class NsysParseResult: + """Ruling R18: a structured parse result, not a bare int. + + `parsed` is True only when the Name/Num Calls columns were both + recognised AND at least one row was actually read -- proving the CSV + was genuinely parsed, not just that an empty or unrelated header + happened to match nothing. `count` (the sum of Num Calls over every row + whose Name starts with "cudaMemcpy") is meaningful only when `parsed` + is True: an unparsed 0 must never be read as a passing zero, since the + whole point of this check is R5 evidence. + """ + parsed: bool + count: int + + +def _sum_cudamemcpy_calls(csv_text: str) -> NsysParseResult: + """Parse `nsys stats --report cuda_api_sum --format csv` output. Column names/casing can drift slightly across Nsight Systems versions, - so this matches case-insensitively by substring ("name", "num calls") - rather than an exact header string. + so columns are matched case-insensitively by substring ("name", "num + calls") rather than an exact header string. """ reader = csv.DictReader(io.StringIO(csv_text)) if not reader.fieldnames: - return 0 + return NsysParseResult(False, 0) name_col = next((f for f in reader.fieldnames if "name" in f.lower()), None) calls_col = next((f for f in reader.fieldnames if "num calls" in f.lower() or "numcalls" in f.lower().replace(" ", "")), None) if not name_col or not calls_col: - return 0 + return NsysParseResult(False, 0) + rows_read = 0 total = 0 for row in reader: + rows_read += 1 name = (row.get(name_col) or "").strip() if name.startswith("cudaMemcpy"): total += int(float(row.get(calls_col) or 0)) - return total + if rows_read == 0: + # Header recognised but no data rows: still inconclusive, not a + # genuine (parsed) zero -- an empty cuda_api_sum table is at least + # as likely to mean "nsys produced nothing useful" as "zero calls". + return NsysParseResult(False, 0) + return NsysParseResult(True, total) def _run_nsys_check(report: _Report, cfg_dir: Path, plan, devcc: str, devflags: str, @@ -677,10 +732,21 @@ def _run_nsys_check(report: _Report, cfg_dir: Path, plan, devcc: str, devflags: report.p("FAIL: nsys stats did not complete successfully") return False - count = _sum_cudamemcpy_calls(stats_proc.stdout) + result = _sum_cudamemcpy_calls(stats_proc.stdout) + if not result.parsed: + # Ruling R18: an unparseable export must never read as a passing + # zero -- it is treated as a gate failure, since the check's whole + # purpose is R5 evidence and an inconclusive parse provides none. + report.p("nsys check inconclusive: could not parse cuda_api_sum " + "(no recognised Name/Num Calls columns, or no data rows read).") + raw_lines = stats_proc.stdout.splitlines()[:8] + report.block("first lines of `nsys stats --report cuda_api_sum --format csv`", + "\n".join(raw_lines) if raw_lines else "(empty output)") + report.p("FAIL: an inconclusive parse counts as a gate failure, not a pass.") + return False report.p(f"cudaMemcpy* Num Calls inside the nvtx-scoped infer_batch call, from " - f"cuda_api_sum: {count}") - if count != 0: + f"cuda_api_sum: {result.count}") + if result.count != 0: report.p("FAIL: infer_batch's timed call must never call cudaMemcpy (ruling R5)") return False return True diff --git a/python/tests/test_device_fortran.py b/python/tests/test_device_fortran.py index 8c76913..e0556c7 100644 --- a/python/tests/test_device_fortran.py +++ b/python/tests/test_device_fortran.py @@ -86,13 +86,15 @@ def test_host_region_calls_module_infer_and_matches(tmp_path, golden_model, name def test_fortran_target_regions_are_real(tmp_path, golden_model): # A host-only libgomp refuses a target region under OMP_TARGET_OFFLOAD=MANDATORY. # If the pragmas were missing or ignored the program would succeed; this is the - # cheapest evidence without a GPU (mirrors tests/test_device_c.py). + # cheapest evidence without a GPU (mirrors tests/test_device_c.py). This check only + # needs the program to run one point and be refused by libgomp -- it does not compare + # against onnxruntime -- so a constant input (not _live_reference) is enough, and + # cannot itself be a dead-model false pass/fail like test_host_region_calls_module_ + # infer_and_matches above needs to guard against. name = "gemm_small" - session = ort.InferenceSession(golden_model(name)) - inputs, _ = _live_reference(session, session.get_inputs()[0].shape, np.float64, seed=9, batch=1) - assert inputs is not None graph = load_graph(golden_model(name)); plan = build_plan(graph, dtype="f64", embed=True) n_in, n_out = plan.input.shape[0], plan.output.shape[0] + inputs = np.full((1, n_in), 0.5) (tmp_path / f"{name}_model.f90").write_text(emit_fortran(plan)) (tmp_path / "host.f90").write_text(HOST.format(name=name, n_in=n_in, n_out=n_out, init_lines="")) fc = _omp_fc() diff --git a/python/tests/test_gate.py b/python/tests/test_gate.py index 11912f1..60bf86a 100644 --- a/python/tests/test_gate.py +++ b/python/tests/test_gate.py @@ -1,8 +1,55 @@ import shutil import pytest from rosenna.cli import main +from rosenna.gate import _sum_cudamemcpy_calls from tests.test_device_c import _omp_cc +# One header shape `nsys stats --report cuda_api_sum --format csv` actually +# produces, close enough to exercise the parser's real column-matching path +# rather than a hand-simplified stand-in. +_NSYS_CSV_HEADER = ( + '"Time (%)","Total Time (ns)","Num Calls","Avg (ns)","Med (ns)",' + '"Min (ns)","Max (ns)","StdDev (ns)","Name"\n' +) + + +def test_sum_cudamemcpy_calls_sums_the_matching_rows(): + # Ruling R18 (a): two cudaMemcpy* rows (3 + 2 = 5 calls) plus one + # unrelated cudaLaunchKernel row; only the memcpy rows count. + csv_text = _NSYS_CSV_HEADER + ( + '45.0,12345,3,4115.0,4000.0,3900.0,4500.0,120.5,"cudaMemcpyAsync"\n' + '30.0,8000,2,4000.0,4000.0,3900.0,4100.0,50.0,"cudaMemcpyHtoD"\n' + '25.0,6000,10,600.0,600.0,500.0,700.0,20.0,"cudaLaunchKernel"\n' + ) + result = _sum_cudamemcpy_calls(csv_text) + assert result.parsed is True + assert result.count == 5 + + +def test_sum_cudamemcpy_calls_is_a_parsed_zero_with_no_memcpy_rows(): + # Ruling R18 (b): a genuinely parsed export with zero cudaMemcpy* rows + # (only a launch row) is a real pass, not a fallback/unparsed zero -- + # `parsed` distinguishes the two. + csv_text = _NSYS_CSV_HEADER + ( + '100.0,6000,10,600.0,600.0,500.0,700.0,20.0,"cudaLaunchKernel"\n' + ) + result = _sum_cudamemcpy_calls(csv_text) + assert result.parsed is True + assert result.count == 0 + + +def test_sum_cudamemcpy_calls_reports_not_parsed_rather_than_a_false_zero(): + # Ruling R18 (c): neither an empty string nor an unrelated-columns CSV + # may come back as `parsed=True, count=0` -- that would be a silent + # pass on a check whose whole purpose is R5 evidence. + empty = _sum_cudamemcpy_calls("") + assert empty.parsed is False + assert empty.count == 0 + + unrelated = _sum_cudamemcpy_calls("foo,bar\n1,2\n3,4\n") + assert unrelated.parsed is False + assert unrelated.count == 0 + def test_gate_runs_in_host_fallback_mode_and_writes_a_report(tmp_path, golden_model): # On a machine without a GPU the gate runs with --host-fallback, which drops the From a4fd89ebbeb695bd5b4b0139e72bd5576d4cb145 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Mon, 14 Sep 2026 00:43:46 -0500 Subject: [PATCH 13/84] docs: how to generate, build and call a model from C and Fortran on the host or a GPU Every command and code block in python/README.md is checked by tests/test_docs.py: the C and Fortran examples are generated from gemm_small and compiled/run as printed, the status table is rendered from abi.STATUS_CODES, and every `rosenna ...` line parses. Also fixes two test_device_c.py tests that used _live_reference's result without the dead-model pytest.skip guard the sibling tests already have. --- .gitignore | 1 + README.md | 8 + python/README.md | 365 ++++++++++++++++++++++++++++++++++ python/tests/test_device_c.py | 6 + python/tests/test_docs.py | 74 +++++++ 5 files changed, 454 insertions(+) create mode 100644 python/README.md create mode 100644 python/tests/test_docs.py diff --git a/.gitignore b/.gitignore index 4469476..bf8d376 100644 --- a/.gitignore +++ b/.gitignore @@ -15,6 +15,7 @@ !goldenFiles/mnist/mnist.onnx !instructions/* !python/examples/**/*.md +!python/README.md reading.f90 userTesting.f90 linearV3copy.f90 diff --git a/README.md b/README.md index 0844ffc..9758905 100644 --- a/README.md +++ b/README.md @@ -203,6 +203,14 @@ stops with an error. The `onnxWeights.txt` fallback described under Hello RoseNN Please see [this document](https://github.com/comp-physics/roseNNa/blob/master/doc/opensource.md) on how to extend roseNNa to new network models and [this document](https://github.com/comp-physics/roseNNa/blob/master/doc/methodology.md) on the details of the roseNNa pipeline. +## Python code generator + +`python/` holds a second, newer way to use roseNNa: a generator that reads an ONNX model and emits a small, self-contained Fortran module and/or C library, callable per point from inside your own OpenMP-target, OpenACC, CUDA or HIP loop, with its weights device-resident. + +Two paths currently coexist in this repository. The `fLibrary/` runtime library described in the rest of this README supports every op roseNNa implements (RNNs, CNNs, MLPs). The generator in `python/` supports dense (Gemm/MatMul + Relu/Tanh/Sigmoid) models only, but its output is GPU-callable. The generator is meant to replace the library once it covers everything the library does; until then, use `fLibrary/` for anything the generator does not yet support. + +See [python/README.md](python/README.md) for how to install, generate, build and call generated code from C or Fortran. + ## Citation You can cite this work as diff --git a/python/README.md b/python/README.md new file mode 100644 index 0000000..9e7528f --- /dev/null +++ b/python/README.md @@ -0,0 +1,365 @@ +# rosenna: ONNX to a GPU-callable Fortran/C library + +This is the roseNNa code generator: it reads a dense ONNX model and emits a +small, self-contained Fortran module and/or C library that a solver written +in C or Fortran links directly, and calls per point inside its own compute +loop -- on the host, or on a GPU under OpenMP target offload, OpenACC, CUDA +or HIP. + +## What you get + +`rosenna generate model.onnx` turns an ONNX model into `lib.a` (C) and +`lib_f.a` (Fortran, a module in the archive): `_infer` is a +plain per-point function you call inside your own GPU loop, exactly like any +other device-callable routine in your solver, and its weights are +device-resident -- baked into the generated source as constants for a small +model, or loaded once at startup and copied to the device for a large one. + +## Install + +```sh +pip install -e python +``` + +The generator itself only needs Python (`onnx`, `numpy`, `onnxruntime` for +`verify`). Building generated code needs a compiler: + +- host path (no accelerator, or the OpenMP-target host fallback): `gcc`/`gfortran` + with `-fopenmp` (on macOS, Homebrew's `gcc-15`/`gfortran-15` -- Apple's + `clang`-based `gcc` has no `-fopenmp`). +- GPU path: `nvc`/`nvfortran` (NVIDIA HPC SDK) for OpenMP-target or OpenACC on + an NVIDIA GPU; `amdclang`/`amdflang` for OpenMP-target on an AMD GPU; `icx`/`ifx` + for OpenMP-target on an Intel GPU. `nvcc` or `hipcc` if you also want the + native batched kernel (`--backend cuda|hip`, see [Call it from C](#call-it-from-c)). + +## Generate + +```sh +rosenna generate model.onnx --lang both --precision single --out build/ +``` + +`--lang` selects `fortran`, `c`, or `both` (default); `--precision` selects +`single` or `double` and defaults to the model's own dtype (see +[Precision](#precision)); `--name` sets the symbol prefix and defaults to the +model file's stem. `generate` prints every file it wrote: + +| File | Written when | What it is | +|---|---|---| +| `.h` | `--lang c\|both` | the header: `_infer` (per-point, device-decorated), `_infer_batch` | +| `.c` | `--lang c\|both` | weight loading and `_init` (file-loaded models only), the OpenMP-fallback `_infer_batch` | +| `_kernel.cu` | `--lang c\|both` | the native CUDA/HIP batched kernel; inert unless built with `ROSENNA_BACKEND=cuda\|hip` | +| `rosenna_rt.h` | `--lang c\|both` | the CUDA/HIP runtime macro mapping; identical for every model | +| `.mk` | `--lang c\|both` | the C build recipe: builds `lib.a` for `ROSENNA_BACKEND=cuda\|hip\|omp` | +| `_model.f90` | `--lang fortran\|both` | the Fortran module | +| `_fortran.mk` | `--lang fortran\|both` | the Fortran build recipe: builds `lib_f.a` | +| `.rwt` | file-loaded weights only | the weights file `_init` reads | + +Both recipes write into the same output directory and build there: a +`--lang both` run gives you one directory holding both archives. + +A model embeds its weights as constants (`ROSENNA_CONST` in C, a Fortran +`parameter` array) automatically when it has fewer than `EMBED_THRESHOLD` +(1,000,000) parameters; above that it is file-loaded by default. `--embed-weights` +forces embedding regardless of size; `--no-embed` forces a `.rwt` file +regardless of size. An embedded model has no `_init` at all -- there is +nothing to load -- and no `.rwt` file is written for it. + +## Call it from C + +Two paths call the same generated code. This example is generated from +`gemm_small` with `--name model`; it embeds by default, so it has no +`model_init` to call (the commented-out line below shows the file-loaded +form). It reads its inputs from a fixed array, calls `model_infer` in its +own offload loop, calls `model_infer_batch` once, and exits non-zero if the +two disagree: + +```c +#include +#include +#include "model.h" + +#define NPTS 4 + +int main(void) { + /* Fixed inputs: NPTS points of n_in=2 values each. */ + double x[NPTS * 2] = { + 0.10, 0.20, + 0.30, -0.10, + -0.20, 0.50, + 1.00, -1.00, + }; + double y_loop[NPTS * 3]; + double y_batch[NPTS * 3]; + int status = 0; + + /* File-loaded models only: gemm_small embeds by default, so this + generated header has no model_init to call. + if (model_init("model.rwt") != 0) return 1; */ + + /* (a) The per-point path: model_infer inside your own offload loop. */ +#if defined(_OPENMP) + #pragma omp target teams loop map(to: x[0:NPTS * 2]) map(from: y_loop[0:NPTS * 3]) +#endif + for (int p = 0; p < NPTS; ++p) + model_infer(x + p * 2, y_loop + p * 3); + + /* (b) The batched path: model_infer_batch takes device-resident data in + every backend and never allocates, transfers or synchronizes itself. + Under the omp backend the host maps its own arrays and hands + infer_batch the mapped device pointers (use_device_ptr needs a + pointer variable, not an array, hence xp/yp); a cuda/hip caller + passes raw device pointers here instead and skips this mapping. */ +#if defined(_OPENMP) + { + double *xp = x, *yp = y_batch; + #pragma omp target data map(to: x[0:NPTS * 2]) map(from: y_batch[0:NPTS * 3]) \ + use_device_ptr(xp, yp) + { + status = model_infer_batch(NPTS, xp, yp, NULL); + } + } +#else + status = model_infer_batch(NPTS, x, y_batch, NULL); +#endif + if (status != 0) return 1; + + for (int i = 0; i < NPTS * 3; ++i) { + if (fabs(y_loop[i] - y_batch[i]) > 1e-9) { + fprintf(stderr, "mismatch at %d: %.17g vs %.17g\n", i, y_loop[i], y_batch[i]); + return 1; + } + } + return 0; +} +``` + +Generate and build it (`--precision double` here only to keep the example's +own arithmetic in `double` throughout; see [Precision](#precision)): + +```sh +rosenna generate model.onnx --lang c --precision double --out build/ --name model +``` +```sh +make -f model.mk ROSENNA_BACKEND=omp CC=gcc-15 ROSENNA_OFFLOAD_FLAGS=-fopenmp +``` + +`ROSENNA_BACKEND` selects which `model_infer_batch` the archive holds -- +`cuda`/`hip` build `model_kernel.cu` with `DEVCC` (default `nvcc`/`hipcc`) +and launch the native kernel over raw device pointers; `omp` (the default) +builds only `model.c` with the host compiler and runs the OpenMP-target +fallback shown above. The two are never linked together. A cuda/hip build +of a *file-loaded* model needs one more call: after every `model_init`, call +`model_device_bind_here()` in every translation unit whose kernels call +`model_infer` (an embedded model needs neither). + +### No transfers in the loop + +`_init` is the plan step and the only routine that allocates or +transfers. Nothing in the loop path -- `_infer` or +`_infer_batch` -- allocates, transfers or synchronizes; the caller +owns the stream (`model_infer_batch`'s last argument), and `infer_batch` +never even looks at it beyond passing it to the launch. An embedded model's +`_infer` is also device-only under `nvcc`/`hipcc` -- its host +instantiation asserts -- so on those compilers call it from a kernel, or use +`infer_batch`. + +Host offload flags, for the per-point path and the `omp` backend: + +| Host compiler | Host flags (the per-point path and the `omp` backend) | +|---|---| +| nvc / nvfortran (NVIDIA, OpenMP) | `-mp=gpu -gpu=cc80` (or your `-gpu=` target) | +| nvc / nvfortran (NVIDIA, OpenACC) | `-acc -gpu=cc80` | +| amdclang / amdflang (AMD) | `-fopenmp --offload-arch=gfx90a` (or your arch) | +| icx / ifx (Intel) | `-fopenmp -fopenmp-targets=spir64` | +| gcc / gfortran, host fallback | `-fopenmp` | + +`DEVFLAGS`, for the batched backend: + +| Batched backend | `ROSENNA_BACKEND` | `DEVFLAGS` | +|---|---|---| +| CUDA | `cuda` | `-O2 -arch=sm_80` (or your arch) | +| HIP | `hip` | `-O2 --offload-arch=gfx90a` (or your arch) | +| OpenMP fallback | `omp` | none; uses the host flags | + +## Call it from Fortran + +The same two paths, through `use _model`. This is the same +`gemm_small` model as above (`--name model`), built with `--lang fortran`, +so it also embeds and has no `model_init`: + +```fortran +program host + use model_model + use iso_fortran_env, only: real64 + implicit none + integer, parameter :: npts = 4 + real(real64) :: x(2, npts), y_loop(3, npts), y_batch(3, npts) + integer :: p, status + + x(:, 1) = [ 0.10_real64, 0.20_real64] + x(:, 2) = [ 0.30_real64, -0.10_real64] + x(:, 3) = [-0.20_real64, 0.50_real64] + x(:, 4) = [ 1.00_real64, -1.00_real64] + + ! File-loaded models only: gemm_small embeds by default, so this + ! generated module has no model_init to call. + ! call model_init('model.rwt', status) + ! if (status /= 0) stop 1 + + ! (a) The per-point path: model_infer inside your own offload loop. + !$omp target teams loop map(to: x) map(from: y_loop) + do p = 1, npts + call model_infer(x(:, p), y_loop(:, p)) + end do + + ! (b) The batched path: model_infer_batch takes device-resident arrays. + ! The host maps its own arrays and hands infer_batch the mapped device + ! addresses (use_device_addr); a cuda/hip caller reaches the same + ! contract through model_infer_batch_dev and c_loc of device memory. + !$omp target data map(to: x) map(from: y_batch) use_device_addr(x, y_batch) + call model_infer_batch(npts, x, y_batch, status) + !$omp end target data + if (status /= 0) stop 1 + + if (maxval(abs(y_loop - y_batch)) > 1.0e-9_real64) stop 1 +end program +``` + +`model_infer` is `pure`; `model_infer_batch(n, x, y, status)` returns its +status (0, 10 or 11 -- see [Status codes](#status-codes)) as an `intent(out)` +argument rather than a function result, so it can be called from inside a +plain (non-`pure`) host subroutine. Build and run it: + +```sh +rosenna generate model.onnx --lang fortran --precision double --out build/ --name model +``` +```sh +make -f model_fortran.mk FC=gfortran ROSENNA_OFFLOAD_FLAGS=-fopenmp +``` +```sh +gfortran -O2 -std=f2008 -fopenmp -I. host.f90 -L. -lmodel_f -o host +``` + +`gfortran` drops `model_model.mod` next to the object it compiles; `-J DIR` +during the library build sends it to `DIR` instead of the current +directory, and a host that `use`s the module then needs `-I DIR` on its own +compile line to find it (`-I.` above, since the example builds both in the +same directory). + +A Fortran host reaches the batched path two ways. `model_infer_batch` as +shown above is always Fortran's own OpenMP-target fallback, compiled +straight into `lib_f.a`, so it links nothing else. The module also +declares a second route straight to the native kernel: the `bind(C)` +interface `model_infer_batch_dev`, bound to the plain C symbol +`model_infer_batch` that `lib.a` provides -- whichever kernel its +`ROSENNA_BACKEND` was built with (see [Call it from C](#call-it-from-c)). +That route needs `c_ptr`s to device-resident memory, which OpenACC's +`host_data use_device` produces from a mapped Fortran array: + +```fortran +use iso_c_binding, only: c_loc, c_null_ptr +integer :: status +!$acc host_data use_device(x, y_batch) +status = model_infer_batch_dev(npts, c_loc(x), c_loc(y_batch), c_null_ptr) +!$acc end host_data +``` + +and links both archives: `-lmodel_f -lmodel`. + +`model_infer` is not itself inlined across the `use model_model` boundary by +every compiler, so a Fortran host's own offload loop generally gets a real +call per point, not an inlined one, unless the build enables cross-module +inlining (`gfortran -flto`, nvfortran `-Minline`). + +## Precision + +`--precision` defaults to the model's own dtype -- `float32` for a PyTorch +export via `torch.onnx.export`, since that is what PyTorch trains and +exports in. A double-precision host can still call single-precision +generated code: `model_infer`'s `x`/`y` are the plan's own C `float` / +Fortran `real(real32)`, so the host converts at the call site -- an +implicit narrowing conversion for a C `double` array passed element by +element, or an explicit `real(x, real32)` going in and `real(y_f32, real64)` +coming back out in Fortran. `--precision single` is the usual GPU choice +regardless of the host's own precision: consumer and even most datacenter +GPUs run FP64 at a small fraction of their FP32 throughput, so a solver +whose accuracy budget tolerates it gets a substantial speedup from +generating (and calling) the single-precision code even from a +double-precision caller. + +## Status codes + +`_init` and `_infer_batch` return one of these (rendered here +from `rosenna.abi.STATUS_CODES`, the one place the table is defined): + +| Code | Meaning | +|---|---| +| 0 | success | +| 1 | cannot open the weights file | +| 2 | not a roseNNa weights file (bad magic) | +| 3 | weights file version is not supported | +| 4 | weights file dtype does not match this generated code | +| 5 | weights file endianness does not match this machine | +| 6 | weights file plan hash does not match this generated code | +| 7 | weights file holds a tensor this model does not declare | +| 8 | a name or rank in the weights file exceeds this model's capacity | +| 9 | a read failed: the weights file is truncated or inconsistent | +| 10 | device allocation or copy failed in init | +| 11 | kernel launch failed | + +Codes 0-9 are `_init`'s; `_infer_batch` only ever returns 0, 10 +or 11 (10 and 11 are cuda/hip only -- the `omp` backend's fallback loop +cannot itself fail once its arguments are device-resident, so it always +returns 0). + +## Verify + +```sh +rosenna verify model.onnx --lang both --cases 16 +``` + +`verify` generates, compiles and runs the per-point `_infer` path on +the host, for one or both languages, and compares its output against +onnxruntime running the same model over the same random inputs. It proves +the generated arithmetic is correct on the host; it never builds or runs the +batched device path (`_infer_batch`, the native kernel, or the +`omp`/`acc` fallbacks under a real offload device), because that needs a GPU +this machine may not have. + +```sh +rosenna gpu-gate --help +``` +```sh +rosenna gpu-gate --cc gcc --fc gfortran --flags=-fopenmp --backend omp --host-fallback --out gate-report/ +``` + +`gpu-gate` is the check that does exercise the device path: on a machine +with a real accelerator (and the matching compilers -- `--help` lists the +NVIDIA, AMD and no-GPU pairings), it generates a model, builds it for the +chosen `--backend`, and runs three harnesses -- a per-point C host, a +per-point Fortran host, and a host that hands device-resident data to +`infer_batch` -- each compared against onnxruntime and timed, writing every +command and its output to `gate-report.md`. + +Until `gpu-gate` has been run on a GPU machine, the device path is +unvalidated: everything above compiles and runs on the host, and the CUDA +and HIP decoration compiles under `nvcc`/`hipcc` in CI, but none of it has +executed on a device. See `python/examples/microfd_closure/` for a worked +example of wiring a generated model into a solver, with the same caveat. + +## Limits + +- Supported ops: `Gemm`, `MatMul`, `Relu`, `Tanh`, `Sigmoid` -- dense MLPs + only (no convolution, pooling, batch norm, or recurrent ops), one point + per call (every value is rank 1 or a rank-2 tensor with leading dimension + 1), one input and one output tensor. +- A file-loaded model's `_infer` reads unset (zero-initialized static) + weights if `_init` was never called, or failed, before it. Nothing + in the loop path checks this -- checking it there would be the transfer + and synchronization ruled out under [No transfers in the loop](#no-transfers-in-the-loop). +- The native batched kernel (`ROSENNA_BACKEND=cuda|hip`) launches one thread + per point in this release; a fused, tiled batched GEMM is planned once the + GPU gate has timed this one. +- There is no SYCL backend. An Intel GPU is reached through the `omp` + fallback (`icx`/`ifx` with `-fopenmp -fopenmp-targets=spir64`), not a + native kernel. diff --git a/python/tests/test_device_c.py b/python/tests/test_device_c.py index 653dbc3..043e558 100644 --- a/python/tests/test_device_c.py +++ b/python/tests/test_device_c.py @@ -72,6 +72,9 @@ def test_host_region_calls_header_inline_and_matches(tmp_path, golden_model, nam plan = build_plan(graph, dtype="f64", embed=embed) session = ort.InferenceSession(golden_model(name)) inputs, expected = _live_reference(session, session.get_inputs()[0].shape, np.float64, seed=5, batch=8) + if inputs is None: + pytest.skip(f"{name}: onnxruntime reference is all-zero across 10 resampled " + f"batches; its golden-file weights produced a dead model") r = _build_and_run(tmp_path, name, plan, graph, _omp_cc(), ["-O2", "-Wall", "-Wextra", "-std=c11", "-fopenmp"], inputs) assert r.returncode == 0, r.stderr got = np.array([[float(v) for v in line.split()] for line in r.stdout.strip().splitlines()]) @@ -96,6 +99,9 @@ def test_plain_compiler_without_openmp_still_matches(tmp_path, golden_model): cc = shutil.which("clang") or shutil.which("cc") or _omp_cc() session = ort.InferenceSession(golden_model(name)) inputs, expected = _live_reference(session, session.get_inputs()[0].shape, np.float64, seed=6, batch=4) + if inputs is None: + pytest.skip(f"{name}: onnxruntime reference is all-zero across 10 resampled " + f"batches; its golden-file weights produced a dead model") r = _build_and_run(tmp_path, name, plan, graph, cc, ["-O2", "-Wall", "-Wextra", "-std=c11"], inputs) assert r.returncode == 0, r.stderr got = np.array([[float(v) for v in line.split()] for line in r.stdout.strip().splitlines()]) diff --git a/python/tests/test_docs.py b/python/tests/test_docs.py new file mode 100644 index 0000000..398fe40 --- /dev/null +++ b/python/tests/test_docs.py @@ -0,0 +1,74 @@ +import re +import shutil +import subprocess +from pathlib import Path +import pytest +from rosenna.abi import STATUS_CODES +from tests.test_device_c import _omp_cc + +README = Path(__file__).resolve().parents[1] / "README.md" + + +def _blocks(lang): + text = README.read_text() + return re.findall(rf"```{lang}\n(.*?)```", text, re.S) + + +def test_readme_has_every_required_section(): + text = README.read_text() + for heading in ["What you get", "Install", "Generate", "Call it from C", "Call it from Fortran", + "Precision", "Status codes", "Verify", "Limits"]: + assert f"## {heading}" in text, heading + + +def test_status_table_matches_abi(): + text = README.read_text() + for code, meaning in STATUS_CODES: + assert f"| {code} |" in text and meaning in text, (code, meaning) + + +def test_every_shell_command_in_the_readme_parses(): + # Each ```sh block is a sequence of commands the user is told to run; every one must at least + # name a real subcommand or make target, so a renamed flag cannot leave the docs stale. + from rosenna.cli import build_parser + parser = build_parser() + for block in _blocks("sh"): + for line in block.strip().splitlines(): + if line.startswith("rosenna "): + args = line.split()[1:] + try: + parser.parse_args([a for a in args if not a.startswith("<")] or ["--help"]) + except SystemExit as e: + assert e.code == 0, line + + +def test_c_example_in_the_readme_compiles_and_runs(tmp_path, golden_model): + # The README's C example is generated against gemm_small and must build and run as printed. + from rosenna.cli import main + assert main(["generate", str(golden_model("gemm_small")), "--lang", "c", "--precision", "double", + "--out", str(tmp_path), "--name", "model"]) == 0 + (blocks,) = [b for b in _blocks("c") if "model_infer(" in b][:1] or [None] + assert blocks, "README has no C example calling model_infer" + (tmp_path / "host.c").write_text(blocks) + cc = _omp_cc() + subprocess.run(["make", "-f", "model.mk", f"CC={cc}", "ROSENNA_OFFLOAD_FLAGS=-fopenmp"], cwd=tmp_path, check=True, capture_output=True, text=True) + r = subprocess.run([cc, "-O2", "-std=c11", "-fopenmp", "host.c", "-L.", "-lmodel", "-lm", "-o", "host"], cwd=tmp_path, capture_output=True, text=True) + assert r.returncode == 0, r.stderr + assert subprocess.run(["./host"], cwd=tmp_path, capture_output=True, text=True).returncode == 0 + + +def test_fortran_example_in_the_readme_compiles_and_runs(tmp_path, golden_model): + from rosenna.cli import main + assert main(["generate", str(golden_model("gemm_small")), "--lang", "fortran", "--precision", "double", + "--out", str(tmp_path), "--name", "model"]) == 0 + (blocks,) = [b for b in _blocks("fortran") if "model_infer(" in b][:1] or [None] + assert blocks, "README has no Fortran example calling model_infer" + (tmp_path / "host.f90").write_text(blocks) + fc = shutil.which("gfortran") or pytest.skip("no gfortran") + subprocess.run(["make", "-f", "model_fortran.mk", f"FC={fc}", "ROSENNA_OFFLOAD_FLAGS=-fopenmp"], cwd=tmp_path, check=True, capture_output=True, text=True) + # --lang fortran writes libmodel_f.a (model_fortran.mk), not libmodel.a: --lang + # c/both's C recipe (model.mk -> libmodel.a) is not generated by this call, so the + # link line below names the archive the Fortran recipe actually built. + r = subprocess.run([fc, "-O2", "-std=f2008", "-fopenmp", "-I.", "host.f90", "-L.", "-lmodel_f", "-o", "host"], cwd=tmp_path, capture_output=True, text=True) + assert r.returncode == 0, r.stderr + assert subprocess.run(["./host"], cwd=tmp_path, capture_output=True, text=True).returncode == 0 From df7741b89d1c29ff188ccdc1d3193e470999dee9 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Mon, 14 Sep 2026 00:49:28 -0500 Subject: [PATCH 14/84] docs: state the constant-memory cutoff and verify's model-dtype tolerance Review round 1: documents the silent __constant__ vs __device__ const cutoff at 48 KB (ruling R4) in "Call it from C", states that `verify` compares at the ONNX model's own dtype tolerance rather than --precision's in "Verify", and extends the Limits bullet on unset file-loaded weights to cover the omp infer_batch fallback (no status code there; only cuda/hip returns 10). --- python/README.md | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/python/README.md b/python/README.md index 9e7528f..f60ff68 100644 --- a/python/README.md +++ b/python/README.md @@ -152,6 +152,18 @@ of a *file-loaded* model needs one more call: after every `model_init`, call `model_device_bind_here()` in every translation unit whose kernels call `model_infer` (an embedded model needs neither). +Embedded weights on a CUDA/HIP build go to one of two storage classes, +decided per model at generate time, not at build time: under 48 KB (12,288 +float32 or 6,144 float64 parameters) `ROSENNA_CONST` is `__constant__` +(cached, broadcast to every thread reading the same address in a warp); at +or over 48 KB it is `__device__ const` (ordinary global memory), because +CUDA constant memory is 64 KB per module and the cut leaves 16 KB of that +for anything else the translation unit puts there. Both storage classes +compute the same result -- a model over the threshold still runs correctly, +just without the constant-cache broadcast -- and the header's own comment +on `ROSENNA_CONST` states which one a given model got, so check it there if +a per-grid-point call's throughput is on the critical path. + ### No transfers in the loop `_init` is the plan step and the only routine that allocates or @@ -326,6 +338,14 @@ batched device path (`_infer_batch`, the native kernel, or the `omp`/`acc` fallbacks under a real offload device), because that needs a GPU this machine may not have. +The comparison tolerance is keyed on the ONNX model's own dtype, not on +`--precision`: onnxruntime always computes a float32 model's reference in +float32, so `rosenna verify --precision double` on a float32 PyTorch export +is still compared at float32 tolerance (`rtol=1e-5`, `atol=1e-6`), not +float64, however precisely the generated code itself computes. A genuinely +float64 ONNX model is compared at the tight tolerance (`rtol=1e-9`, +`atol=1e-12`) regardless of `--precision`. + ```sh rosenna gpu-gate --help ``` @@ -357,6 +377,9 @@ example of wiring a generated model into a solver, with the same caveat. weights if `_init` was never called, or failed, before it. Nothing in the loop path checks this -- checking it there would be the transfer and synchronization ruled out under [No transfers in the loop](#no-transfers-in-the-loop). + The `omp` backend's `_infer_batch` fallback calls `_infer` per + point and has the same silent behavior; only the cuda/hip path's + `_infer_batch` catches this, returning status 10. - The native batched kernel (`ROSENNA_BACKEND=cuda|hip`) launches one thread per point in this release; a fused, tiled batched GEMM is planned once the GPU gate has timed this one. From 0be7a1a5ceddbbea878d687a9de8db619833e577 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Mon, 14 Sep 2026 01:16:45 -0500 Subject: [PATCH 15/84] fix: device-pass guard needs the compiler macro, accept __HIP__, release device copies on a failed bind - The device-pass guard is now (__CUDACC__ && __CUDA_ARCH__) || ((__HIPCC__ || __HIP__) && __HIP_DEVICE_COMPILE__) (ruling R23): clang's OpenMP nvptx device pass defines __CUDA_ARCH__ without __CUDACC__ and selected the __constant__ table that only the CUDA/HIP guard declares. - Every CUDA/HIP guard in the header, source and rosenna_rt.h accepts __HIP__ next to __HIPCC__, so the header does not depend on the hipcc wrapper's flag order. - infer is emitted as 'static inline ROSENNA_DEVICE_FN', the order CUDA's own headers use. - A failed device_bind (or allocation, or copy) in init releases and nulls every device copy, so infer_batch returns 10 rather than launching over a stale table. - The kernel's launch-check comment names GetLastError (ruling R11); _emit_upload's docstring is reflowed. - Tests: the device-pass guard under a plain compiler with __CUDA_ARCH__ forced on; the three-macro header test also covers the file-loaded header; the release helper is asserted. --- python/rosenna/emit_c.py | 62 +++++++++++++++++++++++++---------- python/rosenna/emit_kernel.py | 4 +-- python/rosenna/rt_header.py | 5 +-- python/tests/test_device_c.py | 48 ++++++++++++++++++++++++--- python/tests/test_emit_c.py | 2 +- python/tests/test_kernel.py | 15 +++++++-- 6 files changed, 108 insertions(+), 28 deletions(-) diff --git a/python/rosenna/emit_c.py b/python/rosenna/emit_c.py index eb8acbd..1305c93 100644 --- a/python/rosenna/emit_c.py +++ b/python/rosenna/emit_c.py @@ -5,12 +5,22 @@ _CTYPE = {"f32": "float", "f64": "double"} _DTYPE_CODE = {"f32": 0, "f64": 1} _ITEMSIZE = {"f32": 4, "f64": 8} -_CUDA_GUARD = "#if defined(__CUDACC__) || defined(__HIPCC__)" -_NOT_CUDA_GUARD = "#if !defined(__CUDACC__) && !defined(__HIPCC__)" +# The CUDA/HIP compiler guard. hipcc's wrapper adds -D__HIPCC__ itself, and +# hip-clang defines __HIP__ for any HIP compilation, so accepting either keeps +# the header independent of the wrapper's flag order. +_IS_CUDA = "defined(__CUDACC__)" +_IS_HIP = "(defined(__HIPCC__) || defined(__HIP__))" +_CUDA_GUARD = f"#if {_IS_CUDA} || {_IS_HIP}" +_NOT_CUDA_GUARD = "#if !defined(__CUDACC__) && !defined(__HIPCC__) && !defined(__HIP__)" # Device-pass guard: nvcc defines __CUDA_ARCH__ and hipcc __HIP_DEVICE_COMPILE__ # only while compiling for the device, so a header-inline function can read one # storage in its host instantiation and another in its device instantiation. -_DEVICE_PASS_GUARD = "#if defined(__CUDA_ARCH__) || defined(__HIP_DEVICE_COMPILE__)" +# Each arch macro is tested together with its compiler macro (ruling R23): +# clang's OpenMP nvptx device pass defines __CUDA_ARCH__ without __CUDACC__, +# and there the host arrays, not the __constant__ table (which only the +# CUDA/HIP guard declares), are the storage that exists. +_DEVICE_PASS_GUARD = (f"#if ({_IS_CUDA} && defined(__CUDA_ARCH__)) || " + f"({_IS_HIP} && defined(__HIP_DEVICE_COMPILE__))") # Controller ruling R4. CUDA __constant__ memory is 64 KB per module, while # a model embeds by default below EMBED_THRESHOLD (1M parameters, up to 8 MB @@ -585,12 +595,15 @@ def _emit_upload(plan: Plan) -> list: """The cuda/hip half of init: copy the freshly loaded host arrays to the device. Only compiled under nvcc/hipcc, where the runtime is reachable through - rosenna_rt.h (included by the header under the same guard). A repeated init frees the previous copies first (freeing a - null pointer is a no-op in both runtimes); a failed allocation or copy - leaves that pointer null and returns 10, so a later infer_batch refuses - to launch rather than read an unfilled buffer. The last step publishes + rosenna_rt.h (included by the header under the same guard). A repeated + init frees the previous copies first (_release; freeing a null + pointer is a no-op in both runtimes). A failed allocation or copy, or a + failed bind, releases every copy again and returns 10, so a later + infer_batch refuses to launch (its null check) rather than read an + unfilled buffer or, after a failed bind, launch over a table that still + holds the previous addresses. The bind is the last step: it publishes the new addresses to the kernel's translation unit (_device_bind, - in _kernel.cu); after init returns, the loop path transfers + in _kernel.cu). After init returns, the loop path transfers nothing (controller ruling R5). """ m = plan.model @@ -598,21 +611,34 @@ def _emit_upload(plan: Plan) -> list: return [] lines = [ _CUDA_GUARD, - f"static int {m}_upload(void) {{", + f"static void {m}_release(void) {{", ] for w in plan.weights: sym = _c_weight_symbol(m, w.symbol) lines += [ f" (void)ROSENNA_FREE({sym}_dev);", f" {sym}_dev = 0;", - f" if (ROSENNA_MALLOC(&{sym}_dev, sizeof {sym}) != ROSENNA_OK) return 10;", - f" if (ROSENNA_MEMCPY_H2D({sym}_dev, {sym}, sizeof {sym}) != ROSENNA_OK) {{", - f" (void)ROSENNA_FREE({sym}_dev);", - f" {sym}_dev = 0;", - f" return 10;", - f" }}", ] - lines += [f" return {_device_bind(m)}();", "}", "#endif", ""] + lines += [ + "}", + "", + f"static int {m}_upload(void) {{", + f" {m}_release();", + ] + fail = f"{{ {m}_release(); return 10; }}" + for w in plan.weights: + sym = _c_weight_symbol(m, w.symbol) + lines += [ + f" if (ROSENNA_MALLOC(&{sym}_dev, sizeof {sym}) != ROSENNA_OK) {fail}", + f" if (ROSENNA_MEMCPY_H2D({sym}_dev, {sym}, sizeof {sym}) != ROSENNA_OK) {fail}", + ] + lines += [ + f" if ({_device_bind(m)}() != 0) {fail}", + " return 0;", + "}", + "#endif", + "", + ] return lines @@ -773,7 +799,9 @@ def _emit_infer(plan: Plan, ctype: str) -> list: scratch = sorted((s for s in plan.buffers if s not in ("x", "y")), key=lambda s: int(s[1:])) - lines = [f"ROSENNA_DEVICE_FN static inline void {m}_infer(" + # Storage class first, then the attribute macro: the order CUDA's own + # headers use for `static inline __host__ __device__`. + lines = [f"static inline ROSENNA_DEVICE_FN void {m}_infer(" f"const {ctype} *ROSENNA_RESTRICT x, {ctype} *ROSENNA_RESTRICT y) {{"] if plan.embed: # Controller ruling R9: in the host pass of a CUDA/HIP build the diff --git a/python/rosenna/emit_kernel.py b/python/rosenna/emit_kernel.py index 05e7f75..16cba52 100644 --- a/python/rosenna/emit_kernel.py +++ b/python/rosenna/emit_kernel.py @@ -61,8 +61,8 @@ def emit_kernel(plan: Plan) -> str: lines += [ " const int grid = (n + ROSENNA_TILE - 1) / ROSENNA_TILE;", f" ROSENNA_LAUNCH({m}_kernel, grid, ROSENNA_TILE, s, n, x, y);", - " /* A peek, not a sync (ruling R5): a bad configuration or stream is", - " reported now; asynchronous faults surface at the caller's sync. */", + " /* GetLastError, not a sync (ruling R5): a bad configuration or stream", + " is reported now; asynchronous faults surface at the caller's sync. */", " if (ROSENNA_LAUNCH_STATUS() != ROSENNA_OK) return 11;", " return 0;", "}", diff --git a/python/rosenna/rt_header.py b/python/rosenna/rt_header.py index a6b1064..0023e6d 100644 --- a/python/rosenna/rt_header.py +++ b/python/rosenna/rt_header.py @@ -9,10 +9,11 @@ /* Generated by rosenna. Do not edit. Maps the runtime calls the generated sources make onto CUDA or HIP; no other generated file names a cuda* or hip* symbol. Compiled only by nvcc - (__CUDACC__) or hipcc (__HIPCC__): a host C compiler never sees it. */ + (__CUDACC__) or hipcc (__HIPCC__, or hip-clang's own __HIP__): a host C + compiler never sees it. */ #ifndef ROSENNA_RT_H #define ROSENNA_RT_H -#if defined(__HIPCC__) +#if defined(__HIPCC__) || defined(__HIP__) #include #define ROSENNA_STREAM_T hipStream_t #define ROSENNA_MALLOC(p, n) hipMalloc((void **)(p), (n)) diff --git a/python/tests/test_device_c.py b/python/tests/test_device_c.py index 043e558..9c7b76d 100644 --- a/python/tests/test_device_c.py +++ b/python/tests/test_device_c.py @@ -108,15 +108,55 @@ def test_plain_compiler_without_openmp_still_matches(tmp_path, golden_model): np.testing.assert_allclose(got, expected, rtol=1e-5, atol=1e-6) -def test_header_carries_exactly_the_two_macros(golden_model): - from rosenna.emit_c import emit_c - from rosenna.frontend import load_graph as lg +def test_header_carries_exactly_the_three_macros(golden_model): + from rosenna.emit_c import _CUDA_GUARD, _DEVICE_PASS_GUARD # A CUDA/HIP host must see __host__ __device__ and __constant__; nothing else in the header may mention CUDA. # Controller ruling P2: take the golden path through the golden_model fixture (tests.conftest has no # standalone golden_path function). Controller ruling P3: a third macro, ROSENNA_RESTRICT, sits next to # the two above so `restrict` -- not a keyword once this header reaches a C++ (nvcc) translation unit -- # never appears bare in a signature; assert all three macro names are present. - plan = build_plan(lg(golden_model("gemm_small")), dtype="f64") + plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64") _, header = emit_c(plan) assert header.count("__CUDACC__") == 1 and "__host__ __device__" in header and "__constant__" in header assert "ROSENNA_DEVICE_FN" in header and "ROSENNA_CONST" in header and "ROSENNA_RESTRICT" in header + # The file-loaded header mentions __CUDACC__ more often (the rosenna_rt.h + # include, the _dev declarations, the __constant__ table and its bind, and + # the device-pass guard), but only ever on one of the two guard lines: + # every occurrence in the emitted text is accounted for by those two. + _, header_f = emit_c(build_plan(load_graph(golden_model("gemm_small")), dtype="f64", embed=False)) + guard_lines = [l for l in header_f.splitlines() if "__CUDACC__" in l] + assert guard_lines and all(l in (_CUDA_GUARD, _DEVICE_PASS_GUARD) for l in guard_lines), guard_lines + assert header_f.count("__CUDACC__") == len(guard_lines) + assert header_f.count(_CUDA_GUARD) == 5 and header_f.count(_DEVICE_PASS_GUARD) == 1 + + +def test_device_pass_guard_needs_the_cuda_compiler_not_just_the_arch(tmp_path, golden_model): + # Ruling R23: clang's OpenMP nvptx device pass defines __CUDA_ARCH__ without + # __CUDACC__ (reproduced with `clang -cc1 -triple nvptx64-nvidia-cuda + # -fopenmp -fopenmp-is-target-device -E -dM`). The device-pass guard must + # test the compiler macro together with the arch macro, or ROSENNA_REF_* + # selects the __constant__ table, which only the CUDA/HIP guard declares: + # an undeclared identifier. Under a plain compiler with __CUDA_ARCH__ + # forced on, the file-loaded header has to compile and read the host arrays. + name = "gemm_big" + graph = load_graph(golden_model(name)) + plan = build_plan(graph, dtype="f64", embed=False) + _write(tmp_path, name, plan, graph) + (tmp_path / "host.c").write_text(HOST.format(name=name, n_in=plan.input.shape[0], n_out=plan.output.shape[0], + init=f'if ({name}_init("{name}.rwt")) return 2;')) + cc = shutil.which("clang") or shutil.which("cc") or _omp_cc() + flags = ["-O2", "-Wall", "-Wextra", "-std=c11", "-D__CUDA_ARCH__=800"] + for src in ("host.c", f"{name}.c"): + r = subprocess.run([cc, *flags, "-c", src], cwd=tmp_path, capture_output=True, text=True) + assert r.returncode == 0 and r.stderr == "", (src, r.stderr) + pre = subprocess.run([cc, *flags, "-E", "host.c"], cwd=tmp_path, capture_output=True, text=True) + assert pre.returncode == 0 + assert f"{name}_devw" not in pre.stdout and f"{name}_w0[" in pre.stdout + # The guard the emitter writes is the compound one, and the arch macro + # never stands alone on a guard line. + header = (tmp_path / f"{name}.h").read_text() + assert ("#if (defined(__CUDACC__) && defined(__CUDA_ARCH__)) || " + "((defined(__HIPCC__) || defined(__HIP__)) && defined(__HIP_DEVICE_COMPILE__))") in header + for line in header.splitlines(): + if "__CUDA_ARCH__" in line and line.startswith("#if"): + assert "__CUDACC__" in line, line diff --git a/python/tests/test_emit_c.py b/python/tests/test_emit_c.py index fd10658..4abe0ef 100644 --- a/python/tests/test_emit_c.py +++ b/python/tests/test_emit_c.py @@ -89,7 +89,7 @@ def test_infer_is_pure_and_has_literal_bounds(golden_model): # `infer` is now defined only in the header (a static inline callable # from inside the host's own offload region); the source never defines # it. - assert ("ROSENNA_DEVICE_FN static inline void gemm_small_infer(" + assert ("static inline ROSENNA_DEVICE_FN void gemm_small_infer(" "const double *ROSENNA_RESTRICT x, double *ROSENNA_RESTRICT y) {") in header # The scratch buffers come from plan.buffers now (ruling R13), not from a # second allocator private to this emitter: gemm_small's t0 is reused by diff --git a/python/tests/test_kernel.py b/python/tests/test_kernel.py index bfb3ae1..e5a2533 100644 --- a/python/tests/test_kernel.py +++ b/python/tests/test_kernel.py @@ -31,7 +31,7 @@ def test_kernel_source_names_no_runtime_symbol_for_a_file_loaded_plan(golden_mod assert 'extern "C" int gemm_big_device_bind(void) {' in cu assert "return gemm_big_device_bind_here();" in cu assert "ROSENNA_LAUNCH(gemm_big_kernel" in cu - # Ruling R10: the launch is checked with a peek (never a sync), status 11. + # Ruling R10/R11: the launch is checked with GetLastError (never a sync), status 11. assert "if (ROSENNA_LAUNCH_STATUS() != ROSENNA_OK) return 11;" in cu # An embedded plan reads its ROSENNA_CONST arrays directly and binds nothing. cu_e = emit_kernel(build_plan(load_graph(golden_model("gemm_big")), dtype="f64", embed=True)) @@ -75,6 +75,7 @@ def test_loop_path_never_transfers(golden_model): upload = _function_body(source, f"static int {name}_upload(") assert f"return {name}_upload();" in init and "ROSENNA_MALLOC" in upload rest_c = rest_c.replace(init, "").replace(upload, "") + # _release (called by upload) only frees: no transfer token. for forbidden in _LOOP_PATH_FORBIDDEN: assert forbidden not in rest_c, (embed, forbidden) assert forbidden not in cu, (embed, forbidden) @@ -125,7 +126,17 @@ def test_file_loaded_source_copies_to_the_device_under_the_cuda_guard(golden_mod # init ends by publishing the copies to the kernel's translation unit, # through the header's per-translation-unit bind (ruling R8), which any # user kernel's translation unit must call as well. - assert "return gemm_small_device_bind();" in source + assert "if (gemm_small_device_bind() != 0) { gemm_small_release(); return 10; }" in source + # A failed bind (like a failed allocation or copy) frees and nulls every + # copy -- the same release a repeated init starts with -- so infer_batch + # then returns 10 instead of launching over a table that still holds the + # previous addresses. + release = _function_body(source, "static void gemm_small_release(void) {") + for sym in ("w0", "b0", "w1", "b1"): + assert f"(void)ROSENNA_FREE(gemm_small_{sym}_dev);\n gemm_small_{sym}_dev = 0;" in release + upload = _function_body(source, "static int gemm_small_upload(void) {") + assert upload.count("{ gemm_small_release(); return 10; }") == 2 * 4 + 1 + assert " gemm_small_release();\n" in upload assert "int gemm_small_device_bind(void);" in header assert "static inline int gemm_small_device_bind_here(void) {" in header assert ("call gemm_small_device_bind_here() after EVERY call to gemm_small_init()\n" From 55d7cf4844350fb2bb5101360a3af13ac99fd314 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Mon, 14 Sep 2026 01:17:28 -0500 Subject: [PATCH 16/84] test: library-form test asserts the storage-class-first infer signature --- python/tests/test_library_form.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/python/tests/test_library_form.py b/python/tests/test_library_form.py index a94db5d..42c2f8a 100644 --- a/python/tests/test_library_form.py +++ b/python/tests/test_library_form.py @@ -23,7 +23,7 @@ def test_header_defines_inline_infer_and_source_does_not(golden_model): # (extern declaration in the header, definition in the source). plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64", embed=False) source, header = emit_c(plan) - assert "static inline void gemm_small_infer(" in header + assert "static inline ROSENNA_DEVICE_FN void gemm_small_infer(" in header # Structural check (controller ruling P1): infer must be defined only in # the header, never in the source, regardless of how the source happens # to spell a call to it. From 3cba88de995afac6e2047fd0a25082e513ea2c74 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Mon, 14 Sep 2026 01:23:02 -0500 Subject: [PATCH 17/84] fix: gpu-gate builds the omp archive with the host compiler for the per-point harness, host compiler drives its link - Ruling R21 (C1): under --backend cuda|hip the gate builds two C archives per configuration: the omp-backend lib.a with --cc and --flags in the configuration directory, linked by the per-point C harness, and the cuda|hip lib.a with --devcc in _lib/ (a copy of the sources), linked by the infer_batch .cu driver. Both build lines are recorded. --help states the underlying library limitation. - Ruling R22 (C2): the host compiler compiles and links harness 1 in every backend, against the omp archive; _host_flags_for_devcc_link and the -Xcompiler forwarding are gone. The device compiler compiles and links only the .cu driver. - I2: the Fortran infer_batch harness (and the test host) wrap the call in target data use_device_addr, so has_device_addr sees device addresses. - I3: no -x cu|hip on the .cu driver lines (a -x before the archive makes clang-based hipcc compile the archive as source). - I4: nsys stats runs with -q and the parser starts at the first line naming Num Calls and Name, skipping the stdout preamble; test added. - I6 (ruling R24): -Wall -Wextra -std=c11|f2008 only when the compiler's basename starts with gcc, gfortran, cc or clang; CFLAGS=/FFLAGS= are passed to both recipes explicitly; test added. - I9: nsys profile gets --capture-range-end=stop. - --devcc is shlex-split wherever it becomes argv; a missing compiler in the version probe is recorded, not raised; report command lines are shell-quoted; the per-point timing loops state the benign output race. --- python/rosenna/cli.py | 23 ++- python/rosenna/gate.py | 278 +++++++++++++++++----------- python/tests/test_device_fortran.py | 7 +- python/tests/test_gate.py | 42 +++++ 4 files changed, 240 insertions(+), 110 deletions(-) diff --git a/python/rosenna/cli.py b/python/rosenna/cli.py index e92118a..8c03e9b 100644 --- a/python/rosenna/cli.py +++ b/python/rosenna/cli.py @@ -70,6 +70,21 @@ def build_parser() -> argparse.ArgumentParser: --cc nvc --fc nvfortran --flags "-mp=gpu -gpu=cc80" --backend cuda --devcc nvcc --cc amdclang --fc amdflang --flags "-fopenmp --offload-arch=gfx90a" --backend hip --devcc hipcc --cc gcc --fc gfortran --flags -fopenmp --backend omp --host-fallback (no GPU) + +Under --backend cuda|hip the gate builds TWO C archives per configuration: +an omp-backend lib.a with --cc and --flags (in the configuration +directory) that the per-point C harness links, and the cuda|hip lib.a +with --devcc (in _lib/) that the infer_batch .cu driver links. That +is a limitation of the generated library, not only of the gate: a per-point +OpenMP or OpenACC host calling _infer on a FILE-LOADED model must link +the omp-backend archive built by that same host compiler +(make -f .mk ROSENNA_BACKEND=omp CC= ROSENNA_OFFLOAD_FLAGS=""), +because only that build gives the weight arrays the declare-target device +copies the host's offload loop reads; the cuda/hip archive serves +_infer_batch and CUDA/HIP kernels that call _device_bind_here(). +Embedded models work in every backend. -Wall -Wextra -std=c11|f2008 are added +only when --cc/--fc is gcc, gfortran, cc or clang (by basename); any other +compiler gets -O2 and --flags. """) gate.add_argument("--cc", required=True, help="host C compiler") gate.add_argument("--fc", required=True, help="host Fortran compiler") @@ -80,9 +95,11 @@ def build_parser() -> argparse.ArgumentParser: gate.add_argument("--backend", choices=["cuda", "hip", "omp"], required=True, help="which infer_batch implementation to build and exercise") gate.add_argument("--devcc", default=None, - help="device compiler for --backend cuda|hip, and the link driver " - "for the file-loaded C harnesses there (ruling R14); NOT " - "required -- default: nvcc for cuda, hipcc for hip") + help="device compiler for --backend cuda|hip: builds the cuda|hip " + "archive and compiles and links the infer_batch .cu driver " + "(ruling R22; the host compiler links every other harness); " + "may carry arguments (\"nvcc -ccbin nvc++\"); NOT required -- " + "default: nvcc for cuda, hipcc for hip") # Same dash-valued handling as --flags; see the comment above. gate.add_argument("--devflags", default="", help="device compiler flags") gate.add_argument("--out", default=".", help="directory for generated sources and gate-report.md") diff --git a/python/rosenna/gate.py b/python/rosenna/gate.py index 882e7a0..6e9fdb9 100644 --- a/python/rosenna/gate.py +++ b/python/rosenna/gate.py @@ -3,10 +3,13 @@ None of the CUDA/HIP path has ever been compiled or run on the machine that wrote it (no nvcc, hipcc, or GPU). This script is the evidence that fact cannot produce: it generates the gemm_big plan embedded and file-loaded, in -both languages, builds the C library with the chosen batched backend and the -Fortran library with the host compiler, then runs three harnesses -- a -microfd-shaped per-point host in C, the same in Fortran, and a host that -hands device-resident data to infer_batch -- each compared against +both languages, builds the omp-backend C archive and the Fortran library +with the host compiler and, under --backend cuda|hip, the native-kernel +archive with the device compiler as well (ruling R21: the per-point host +harness links the host compiler's own archive in every backend, the +infer_batch driver links the device compiler's), then runs three harnesses +-- a microfd-shaped per-point host in C, the same in Fortran, and a host +that hands device-resident data to infer_batch -- each compared against onnxruntime and timed per point. Every command, every line of its output, the compiler versions and the timings go into gate-report.md; a failure at any step still writes the report and the process exits 1. @@ -26,6 +29,7 @@ import io import os import platform +import shlex import shutil import subprocess import sys @@ -49,6 +53,13 @@ _TIMED_ITERS = 1_000_000 _RTOL, _ATOL = 1e-5, 1e-6 _RUN_TIMEOUT = 300 +# Ruling R24: -Wall -Wextra -std=c11|f2008 are added only for a compiler whose +# basename says it takes them; any other --cc/--fc (nvc, nvfortran, amdclang, +# amdflang, flang, icx, ifx) gets -O2 and the user's --flags, nothing else. +_GNU_STYLE_PREFIXES = ("gcc", "gfortran", "cc", "clang") +# The generated C sources one configuration directory holds; the device +# compiler's archive is built from a copy of them in its own subdirectory. +_C_SOURCE_FILES = ("{name}.c", "{name}.h", "rosenna_rt.h", "{name}_kernel.cu", "{name}.mk") class _Report: @@ -67,8 +78,9 @@ def block(self, label: str, text: str) -> None: self.lines.append(f"{label}:\n```\n{text}\n```") def command(self, label: str, args: list, cwd=None) -> None: + """Log one command, shell-quoted so a multi-word argument reads back as one.""" where = f" (in {cwd})" if cwd is not None else "" - self.lines.append(f"\n**{label}**{where}\n\n```\n$ {' '.join(str(a) for a in args)}\n```") + self.lines.append(f"\n**{label}**{where}\n\n```\n$ {shlex.join(str(a) for a in args)}\n```") def outcome(self, proc) -> None: self.lines.append(f"exit status: {proc.returncode}") @@ -105,14 +117,34 @@ def _sh(report: _Report, label: str, args: list, cwd=None, env=None, input_text= return proc +def _gnu_style(compiler: str) -> bool: + return Path(shlex.split(compiler)[0]).name.startswith(_GNU_STYLE_PREFIXES) + + +def _c_flags(cc: str) -> list: + """The C flags the gate adds before the user's --flags (ruling R24).""" + return ["-O2", "-Wall", "-Wextra", "-std=c11"] if _gnu_style(cc) else ["-O2"] + + +def _f_flags(fc: str) -> list: + """The Fortran flags the gate adds before the user's --flags (ruling R24).""" + return ["-O2", "-Wall", "-Wextra", "-std=f2008"] if _gnu_style(fc) else ["-O2"] + + def _record_versions(report: _Report, cc: str, fc: str, devcc) -> None: report.h("toolchain", 3) report.p(f"platform: {platform.platform()}") for label, exe in (("cc", cc), ("fc", fc), ("devcc", devcc)): if not exe: continue - proc = subprocess.run([exe, "--version"], capture_output=True, text=True) - report.block(f"{label} ({exe}) --version", proc.stdout or proc.stderr or "(no output)") + # --devcc may carry its own arguments ("nvcc -ccbin nvc++"), so it is + # split like a shell word list wherever it becomes argv. + try: + proc = subprocess.run([*shlex.split(exe), "--version"], capture_output=True, text=True) + text = proc.stdout or proc.stderr or "(no output)" + except FileNotFoundError as e: + text = f"{exe}: not found ({e.strerror})" + report.block(f"{label} ({exe}) --version", text) def _ensure_model(report: _Report) -> Path: @@ -143,24 +175,63 @@ def _generate(outdir: Path, onnx_path: Path, embed: bool): return plan -def _build_c_lib(report: _Report, outdir: Path, plan, cc, flags, backend, devcc, devflags) -> bool: +@dataclass(frozen=True) +class _CLibs: + """The C archives one configuration builds (ruling R21). + + `host`: lib.a built by the HOST compiler (ROSENNA_BACKEND=omp with + the host offload flags) in the configuration directory, for the per-point + C harness in every backend. A file-loaded model's weight arrays reach the + host compiler's offload region only through the `declare target` device + copies that this build makes and that init's `target update` fills; an + nvcc/hipcc build of .c has neither (_OPENMP is not defined there), + so linking that archive into a per-point OpenMP/OpenACC host leaves the + offload loop reading weights that do not exist on the device. + `dev`: lib.a built by the device compiler (ROSENNA_BACKEND=cuda|hip) + from a copy of the sources in _lib/, for the .cu driver of the + infer_batch harness. None under --backend omp, where that harness links + `host` instead. Either is None when its build failed. + """ + host: Path | None + dev: Path | None + + +def _build_c_libs(report: _Report, outdir: Path, plan, cc, flags, backend, devcc, devflags) -> _CLibs: name = plan.model label = "embedded" if plan.embed else "file-loaded" - args = ["make", "-f", f"{name}.mk", f"ROSENNA_BACKEND={backend}"] + # CFLAGS is passed explicitly (ruling R24): the recipe's own default is the + # gcc-style set, which must never reach a vendor host compiler. + args = ["make", "-f", f"{name}.mk", "ROSENNA_BACKEND=omp", f"CC={cc}", + f"CFLAGS={' '.join(_c_flags(cc))}", f"ROSENNA_OFFLOAD_FLAGS={flags}"] + proc = _sh(report, f"build c library ({label}, backend=omp, host compiler: " + "serves the per-point harness)", args, cwd=outdir) + host = outdir / f"lib{name}.a" + if not (proc.returncode == 0 and host.exists()): + host = None if backend == "omp": - args += [f"CC={cc}", f"ROSENNA_OFFLOAD_FLAGS={flags}"] - else: - args += [f"DEVCC={devcc or ('nvcc' if backend == 'cuda' else 'hipcc')}"] - if devflags: - args.append(f"DEVFLAGS={devflags}") - proc = _sh(report, f"build c library ({label}, backend={backend})", args, cwd=outdir) - return proc.returncode == 0 and (outdir / f"lib{name}.a").exists() + return _CLibs(host, None) + dev_dir = outdir / f"{backend}_lib" + dev_dir.mkdir(exist_ok=True) + for f in _C_SOURCE_FILES: + shutil.copyfile(outdir / f.format(name=name), dev_dir / f.format(name=name)) + args = ["make", "-f", f"{name}.mk", f"ROSENNA_BACKEND={backend}", f"DEVCC={devcc}"] + if devflags: + args.append(f"DEVFLAGS={devflags}") + proc = _sh(report, f"build c library ({label}, backend={backend}, device compiler, in " + f"{dev_dir.name}/: serves the infer_batch harness)", args, cwd=dev_dir) + dev = dev_dir / f"lib{name}.a" + if not (proc.returncode == 0 and dev.exists()): + dev = None + return _CLibs(host, dev) def _build_fortran_lib(report: _Report, outdir: Path, plan, fc, flags) -> bool: name = plan.model label = "embedded" if plan.embed else "file-loaded" - args = ["make", "-f", f"{name}_fortran.mk", f"FC={fc}", f"ROSENNA_OFFLOAD_FLAGS={flags}"] + # FFLAGS explicitly, for the same reason as CFLAGS above (ruling R24): + # flang rejects the recipe's default -std=f2008. + args = ["make", "-f", f"{name}_fortran.mk", f"FC={fc}", f"FFLAGS={' '.join(_f_flags(fc))}", + f"ROSENNA_OFFLOAD_FLAGS={flags}"] proc = _sh(report, f"build fortran library ({label})", args, cwd=outdir) return proc.returncode == 0 and (outdir / f"lib{name}_f.a").exists() @@ -214,6 +285,10 @@ def _check_output(report: _Report, stdout: str, expected) -> tuple: for (int i = 0; i < {n_out}; ++i) printf("%.17e ", y[p * {n_out} + i]); printf("\\n"); }} + /* Timing loop: b cycles through the n (= 8) correctness points, so the + iterations sharing a b all write the same eight output slots. That + race is benign and intentional: every writer of a slot stores the + same value, and y is not read after the loop. */ long ntime = {ntime}L; double t0 = omp_get_wtime(); #ifdef _OPENMP @@ -251,6 +326,10 @@ def _check_output(report: _Report, stdout: str, expected) -> tuple: do p = 1, n print '({n_out}(es24.16,1x))', y(:, p) end do + ! Timing loop: b cycles through the n (= 8) correctness points, so the + ! iterations sharing a b all write the same eight output columns. That + ! race is benign and intentional: every writer of a column stores the + ! same value, and y is not read after the loop. ntime = {ntime} call system_clock(count=c0, count_rate=crate) !$omp target teams loop map(to: x) map(from: y) @@ -343,8 +422,13 @@ def _check_output(report: _Report, stdout: str, expected) -> tuple: read(*,*) n allocate(x({n_in}, n), y({n_out}, n)) read(*,*) x + ! x and y are mapped first, and the call sees their device addresses + ! (use_device_addr): infer_batch's has_device_addr clause needs those, + ! not the host addresses (ruling R5). !$omp target enter data map(to: x) map(alloc: y) - call {name}_infer_batch(n, x, y, status) ! x, y already on the device (R5) + !$omp target data use_device_addr(x, y) + call {name}_infer_batch(n, x, y, status) + !$omp end target data !$omp target exit data map(from: y) map(delete: x) if (status /= 0) stop 20 do p = 1, n @@ -362,9 +446,11 @@ def _check_output(report: _Report, stdout: str, expected) -> tuple: ! Ruling R16: map before c0 and unmap after c1, so the timed window ! holds only the infer_batch call, matching the cuda/hip .cu driver. !$omp target enter data map(to: xt) map(alloc: yt) + !$omp target data use_device_addr(xt, yt) call system_clock(count=c0, count_rate=crate) call {name}_infer_batch(ntime, xt, yt, status) call system_clock(count=c1) + !$omp end target data !$omp target exit data map(from: yt) map(delete: xt) if (status /= 0) stop 21 ns_per_point = real(c1 - c0, real64) / real(crate, real64) * 1.0e9_real64 / real(ntime, real64) @@ -444,71 +530,32 @@ def _check_output(report: _Report, stdout: str, expected) -> tuple: """ -def _host_flags_for_devcc_link(flags: str, backend: str) -> list: - """Ruling R19: forward the HOST offload flags into the device-compiler link line. - - Ruling R14 uses the device compiler as the link driver for a - file-loaded plan under --backend cuda|hip, which resolves the CUDA/HIP - runtime symbols init needs -- but on its own it never sees the host - compiler's own offload-runtime flags (nvc's -mp=gpu -gpu=cc80), so that - runtime can be unresolved at link too, since gate_harness1.o was - compiled by the host compiler with those flags. nvcc treats an - unrecognised flag as compiler-only unless wrapped -Xcompiler , so - each host flag is forwarded that way for cuda; hipcc is clang-based - (like amdclang, the expected host compiler pairing) and accepts the - host flags directly. Neither form has been verified against a real - toolchain here (see the task report). - """ - host_flags = flags.split() - if backend == "hip": - return host_flags - out = [] - for f in host_flags: - out += ["-Xcompiler", f] - return out - - -def _run_c_harness1(report, cfg_dir, plan, cc, flags, backend, devcc, devflags, - inputs, expected, env) -> bool: +def _run_c_harness1(report, cfg_dir, plan, cc, flags, host_lib, inputs, expected, env) -> bool: name = plan.model n_in, n_out = plan.input.shape[0], plan.output.shape[0] init = "" if plan.embed else f'if ({name}_init("{name}.rwt")) return 2;' (cfg_dir / "gate_harness1.c").write_text(_C_HARNESS1.format( name=name, n_in=n_in, n_out=n_out, init=init, ntime=_TIMED_ITERS)) - # Ruling R14: compilation always goes through the HOST compiler with its - # own offload flags (nvc's -mp=gpu, amdclang's -fopenmp - # --offload-arch=..., or plain -fopenmp) -- this step does not change - # with --backend. Only the final LINK does: for cuda/hip, a file-loaded - # plan's lib.a was built by nvcc/hipcc (init's upload/bind calls - # cudaMalloc/cudaMemcpy/cudaMemcpyToSymbol or the hip equivalents), so - # linking it with the plain host compiler and -lm leaves those - # undefined on every real run. Using the device compiler as the LINK - # driver instead pulls in the runtime library and its -L path from the - # toolkit itself, with nothing hardcoded here; --backend omp keeps the - # host compiler as the link driver, since its infer_batch is pure - # OpenMP with no runtime-API calls to resolve. + # Rulings R21/R22: the HOST compiler, with its own offload flags (nvc's + # -mp=gpu -gpu=cc80, amdclang's -fopenmp --offload-arch=..., or plain + # -fopenmp), both compiles and links this harness in every backend. The + # archive it links (file-loaded plans only) is the omp-backend one the + # same host compiler built (_CLibs.host), so no CUDA/HIP runtime is + # involved and nothing is forwarded to a device compiler: nvcc's default + # host compiler is g++, which rejects -mp=gpu, so a device-compiler link + # of this object cannot work, and the omp archive is the only one whose + # weight arrays have the declare-target copies this offload loop reads. cc_proc = _sh(report, "compile c per-point harness (host compiler, host offload flags)", - [cc, "-O2", "-Wall", "-Wextra", "-std=c11", *flags.split(), + [cc, *_c_flags(cc), *flags.split(), "-c", "gate_harness1.c", "-o", "gate_harness1.o"], cwd=cfg_dir) if cc_proc.returncode != 0: return False objs = ["gate_harness1.o"] if not plan.embed: - objs.append(f"lib{name}.a") - if backend == "omp": - link_cmd = [cc, *flags.split(), *objs, "-lm", "-o", "gate_harness1"] - link_label = "link c per-point harness (host compiler, --backend omp)" - else: - # Ruling R19: the host offload flags (nvc's -mp=gpu -gpu=cc80, or - # amdclang's -fopenmp --offload-arch=...) are forwarded into this - # link too -- gate_harness1.o's own OpenMP-target runtime needs - # them, and devflags alone (nvcc's/hipcc's own flags) does not - # supply them. - link_cmd = [devcc, *devflags.split(), *_host_flags_for_devcc_link(flags, backend), - *objs, "-lm", "-o", "gate_harness1"] - link_label = (f"link c per-point harness (device compiler {devcc}, --backend {backend}, " - f"host offload flags forwarded)") - link_proc = _sh(report, link_label, link_cmd, cwd=cfg_dir) + objs.append(str(host_lib.relative_to(cfg_dir))) + link_proc = _sh(report, "link c per-point harness (host compiler, host offload flags, " + "omp-backend archive)", + [cc, *flags.split(), *objs, "-lm", "-o", "gate_harness1"], cwd=cfg_dir) if link_proc.returncode != 0: return False run_proc = _sh(report, "run c per-point harness", ["./gate_harness1"], cwd=cfg_dir, @@ -527,7 +574,7 @@ def _run_fortran_harness2(report, cfg_dir, plan, fc, flags, inputs, expected, en (cfg_dir / "gate_harness2.f90").write_text(_F_HARNESS2.format( name=name, n_in=n_in, n_out=n_out, init_lines=init_lines, ntime=_TIMED_ITERS)) fc_proc = _sh(report, "compile fortran per-point harness", - [fc, "-O2", "-Wall", "-Wextra", "-std=f2008", *flags.split(), + [fc, *_f_flags(fc), *flags.split(), "gate_harness2.f90", f"lib{name}_f.a", "-o", "gate_harness2"], cwd=cfg_dir) if fc_proc.returncode != 0: return False @@ -539,15 +586,15 @@ def _run_fortran_harness2(report, cfg_dir, plan, fc, flags, inputs, expected, en return ok -def _run_c_harness3_omp(report, cfg_dir, plan, cc, flags, inputs, expected, env) -> bool: +def _run_c_harness3_omp(report, cfg_dir, plan, cc, flags, host_lib, inputs, expected, env) -> bool: name = plan.model n_in, n_out = plan.input.shape[0], plan.output.shape[0] init = "" if plan.embed else f'if ({name}_init("{name}.rwt")) return 2;' (cfg_dir / "gate_harness3.c").write_text(_C_HARNESS3_OMP.format( name=name, n_in=n_in, n_out=n_out, init=init, ntime=_TIMED_ITERS)) cc_proc = _sh(report, "compile c infer_batch harness (omp)", - [cc, "-O2", "-Wall", "-Wextra", "-std=c11", *flags.split(), - "gate_harness3.c", f"lib{name}.a", "-lm", "-o", "gate_harness3"], cwd=cfg_dir) + [cc, *_c_flags(cc), *flags.split(), "gate_harness3.c", + str(host_lib.relative_to(cfg_dir)), "-lm", "-o", "gate_harness3"], cwd=cfg_dir) if cc_proc.returncode != 0: return False run_proc = _sh(report, "run c infer_batch harness (omp)", ["./gate_harness3"], cwd=cfg_dir, @@ -566,7 +613,7 @@ def _run_fortran_harness3_omp(report, cfg_dir, plan, fc, flags, inputs, expected (cfg_dir / "gate_harness3.f90").write_text(_F_HARNESS3_OMP.format( name=name, n_in=n_in, n_out=n_out, init_lines=init_lines, ntime=_TIMED_ITERS)) fc_proc = _sh(report, "compile fortran infer_batch harness (omp)", - [fc, "-O2", "-Wall", "-Wextra", "-std=f2008", *flags.split(), + [fc, *_f_flags(fc), *flags.split(), "gate_harness3.f90", f"lib{name}_f.a", "-o", "gate_harness3_f"], cwd=cfg_dir) if fc_proc.returncode != 0: return False @@ -578,7 +625,7 @@ def _run_fortran_harness3_omp(report, cfg_dir, plan, fc, flags, inputs, expected return ok -def _run_dev_harness3(report, cfg_dir, plan, devcc, devflags, backend, inputs, expected) -> bool: +def _run_dev_harness3(report, cfg_dir, plan, devcc, devflags, backend, dev_lib, inputs, expected) -> bool: name = plan.model n_in, n_out = plan.input.shape[0], plan.output.shape[0] prefix = "hip" if backend == "hip" else "cuda" @@ -586,10 +633,15 @@ def _run_dev_harness3(report, cfg_dir, plan, devcc, devflags, backend, inputs, e (cfg_dir / "gate_harness3.cu").write_text(_DEV_HARNESS3.format( name=name, n_in=n_in, n_out=n_out, init=init, ntime=_TIMED_ITERS, p=prefix, backend=backend)) - x_flag = "hip" if backend == "hip" else "cu" - proc = _sh(report, f"compile {backend} infer_batch harness", - [devcc, *devflags.split(), "-x", x_flag, - "gate_harness3.cu", f"lib{name}.a", "-o", "gate_harness3_dev"], cwd=cfg_dir) + # Ruling R22: the device compiler compiles AND links this driver (it + # supplies its own runtime), against the archive it built itself. No + # `-x cu|hip`: the driver is a .cu, which both compilers take as device + # source by extension, and a -x before the archive would make + # clang-based hipcc compile the archive as source too. + proc = _sh(report, f"compile and link {backend} infer_batch harness (device compiler)", + [*shlex.split(devcc), *devflags.split(), + "gate_harness3.cu", str(dev_lib.relative_to(cfg_dir)), "-o", "gate_harness3_dev"], + cwd=cfg_dir) if proc.returncode != 0: return False run_proc = _sh(report, f"run {backend} infer_batch harness", ["./gate_harness3_dev"], @@ -610,7 +662,7 @@ def _probe_nvtx_header(report: _Report, cfg_dir: Path, devcc: str, devflags: str (cfg_dir / "gate_nvtx_probe.cu").write_text( "#include \nint main(void){return 0;}\n") proc = _sh(report, "probe for ", - [devcc, *devflags.split(), "-x", "cu", "-c", "gate_nvtx_probe.cu", + [*shlex.split(devcc), *devflags.split(), "-c", "gate_nvtx_probe.cu", "-o", "gate_nvtx_probe.o"], cwd=cfg_dir) return proc.returncode == 0 @@ -634,11 +686,20 @@ class NsysParseResult: def _sum_cudamemcpy_calls(csv_text: str) -> NsysParseResult: """Parse `nsys stats --report cuda_api_sum --format csv` output. - Column names/casing can drift slightly across Nsight Systems versions, - so columns are matched case-insensitively by substring ("name", "num - calls") rather than an exact header string. + On stdout the CSV comes after a preamble ("Generating SQLite file ...", + "Processing ...", a "** CUDA API Summary" title); the gate asks for -q + to drop it, but does not rely on that: the header is the first line + naming both a "Num Calls" and a "Name" column, case-insensitively, and + parsing starts there. Column names/casing can drift slightly across + Nsight Systems versions, so the columns are then matched by substring + rather than exact string. """ - reader = csv.DictReader(io.StringIO(csv_text)) + lines = csv_text.splitlines() + start = next((i for i, line in enumerate(lines) + if "num calls" in line.lower() and "name" in line.lower()), None) + if start is None: + return NsysParseResult(False, 0) + reader = csv.DictReader(io.StringIO("\n".join(lines[start:]))) if not reader.fieldnames: return NsysParseResult(False, 0) name_col = next((f for f in reader.fieldnames if "name" in f.lower()), None) @@ -663,7 +724,7 @@ def _sum_cudamemcpy_calls(csv_text: str) -> NsysParseResult: def _run_nsys_check(report: _Report, cfg_dir: Path, plan, devcc: str, devflags: str, - backend: str, inputs) -> bool: + backend: str, dev_lib: Path, inputs) -> bool: """Ruling R15: assert zero cudaMemcpy calls inside the timed infer_batch call only. Profiling the whole harness and counting "cudaMemcpy" across nsys's @@ -702,8 +763,8 @@ def _run_nsys_check(report: _Report, cfg_dir: Path, plan, devcc: str, devflags: # Some toolkit versions still expect an explicit link; try without # -lnvToolsExt first (the documented form) and retry once with it if # linking fails, noting which form was needed. Neither path has run here. - base_cmd = [devcc, *devflags.split(), "-DROSENNA_GATE_NVTX=1", "-x", "cu", - "gate_harness3.cu", f"lib{name}.a"] + base_cmd = [*shlex.split(devcc), *devflags.split(), "-DROSENNA_GATE_NVTX=1", + "gate_harness3.cu", str(dev_lib.relative_to(cfg_dir))] proc = _sh(report, "compile nvtx-bracketed infer_batch harness (no explicit -lnvToolsExt)", [*base_cmd, "-o", "gate_harness3_nvtx"], cwd=cfg_dir) if proc.returncode != 0: @@ -715,18 +776,23 @@ def _run_nsys_check(report: _Report, cfg_dir: Path, plan, devcc: str, devflags: return True stats_base = cfg_dir / "gate_nsys_profile" + # --capture-range-end=stop: profiling stops when the nvtx range closes + # and the harness runs on to completion (the default for an nvtx + # capture range shuts the application down instead). profile_proc = _sh( - report, "nsys profile --capture-range=nvtx --nvtx-capture=rosenna_timed --stats=true", + report, "nsys profile --capture-range=nvtx --nvtx-capture=rosenna_timed " + "--capture-range-end=stop --stats=true", [nsys, "profile", "--capture-range=nvtx", "--nvtx-capture=rosenna_timed", - "--stats=true", "--force-overwrite=true", "-o", str(stats_base), - "./gate_harness3_nvtx"], cwd=cfg_dir, input_text=_stdin_for(inputs)) + "--capture-range-end=stop", "--stats=true", "--force-overwrite=true", + "-o", str(stats_base), "./gate_harness3_nvtx"], cwd=cfg_dir, + input_text=_stdin_for(inputs)) if profile_proc.returncode != 0: report.p("FAIL: nsys profile did not complete successfully") return False report_file = stats_base.with_suffix(".nsys-rep") - stats_proc = _sh(report, "nsys stats --report cuda_api_sum --format csv", - [nsys, "stats", "--report", "cuda_api_sum", "--format", "csv", + stats_proc = _sh(report, "nsys stats -q --report cuda_api_sum --format csv", + [nsys, "stats", "-q", "--report", "cuda_api_sum", "--format", "csv", str(report_file)], cwd=cfg_dir) if stats_proc.returncode != 0: report.p("FAIL: nsys stats did not complete successfully") @@ -807,16 +873,15 @@ def run_gate(*, cc: str, fc: str, flags: str, backend: str, devcc=None, devflags cfg_dir = out_dir / ("embedded" if embed else "file_loaded") plan = _generate(cfg_dir, onnx_path, embed) - c_built = _build_c_lib(report, cfg_dir, plan, cc, flags, backend, devcc, devflags) + libs = _build_c_libs(report, cfg_dir, plan, cc, flags, backend, resolved_devcc, devflags) f_built = _build_fortran_lib(report, cfg_dir, plan, fc, flags) report.h("c harness: per-point infer via target teams loop", 3) - if c_built: - if not _run_c_harness1(report, cfg_dir, plan, cc, flags, backend, resolved_devcc, - devflags, inputs, expected, env): + if libs.host is not None: + if not _run_c_harness1(report, cfg_dir, plan, cc, flags, libs.host, inputs, expected, env): ok = False else: - report.p("skipped: c library build failed") + report.p("skipped: c library build (omp backend, host compiler) failed") ok = False report.h("fortran harness: per-point infer via target teams loop", 3) @@ -829,8 +894,9 @@ def run_gate(*, cc: str, fc: str, flags: str, backend: str, devcc=None, devflags report.h("infer_batch harness: device-resident data", 3) if backend == "omp": - if c_built: - if not _run_c_harness3_omp(report, cfg_dir, plan, cc, flags, inputs, expected, env): + if libs.host is not None: + if not _run_c_harness3_omp(report, cfg_dir, plan, cc, flags, libs.host, + inputs, expected, env): ok = False else: report.p("skipped: c library build failed") @@ -842,9 +908,9 @@ def run_gate(*, cc: str, fc: str, flags: str, backend: str, devcc=None, devflags report.p("skipped: fortran library build failed") ok = False else: - if c_built: + if libs.dev is not None: dev_ok = _run_dev_harness3(report, cfg_dir, plan, resolved_devcc, - devflags, backend, inputs, expected) + devflags, backend, libs.dev, inputs, expected) if not dev_ok: ok = False else: @@ -853,10 +919,10 @@ def run_gate(*, cc: str, fc: str, flags: str, backend: str, devcc=None, devflags # when nsys or the nvtx header is unavailable, none # of which fails the gate on its own. if not _run_nsys_check(report, cfg_dir, plan, resolved_devcc, devflags, - backend, inputs): + backend, libs.dev, inputs): ok = False else: - report.p(f"skipped: c library build failed (backend={backend})") + report.p(f"skipped: c library build (backend={backend}, device compiler) failed") ok = False report.h("result", 2) diff --git a/python/tests/test_device_fortran.py b/python/tests/test_device_fortran.py index e0556c7..b0aa618 100644 --- a/python/tests/test_device_fortran.py +++ b/python/tests/test_device_fortran.py @@ -37,7 +37,12 @@ def _omp_fc(): do p = 1, n call {name}_infer(x(:, p), y(:, p)) end do - call {name}_infer_batch(n, x, yb, status) ! x, yb already on the device (R5) + ! infer_batch's has_device_addr wants the mapped arrays' device addresses, + ! which use_device_addr supplies (ruling R5; on this host-only build they + ! are the host addresses, so the omission would have been invisible). + !$omp target data use_device_addr(x, yb) + call {name}_infer_batch(n, x, yb, status) + !$omp end target data !$omp target exit data map(from: y, yb) map(delete: x) if (status /= 0) stop 4 ! abs(...) > 0, not /=: an exact-bits comparison without tripping diff --git a/python/tests/test_gate.py b/python/tests/test_gate.py index 60bf86a..9071ca9 100644 --- a/python/tests/test_gate.py +++ b/python/tests/test_gate.py @@ -51,6 +51,42 @@ def test_sum_cudamemcpy_calls_reports_not_parsed_rather_than_a_false_zero(): assert unrelated.count == 0 +def test_sum_cudamemcpy_calls_skips_the_stdout_preamble(): + # `nsys stats --format csv` on stdout is preceded by progress lines and a + # report title; the header is the first line naming Num Calls and Name, + # not the first line of stdout. + preamble = ( + "Generating SQLite file gate_nsys_profile.sqlite from gate_nsys_profile.nsys-rep\n" + "Processing [gate_nsys_profile.sqlite] with [/opt/nvidia/nsight-systems/reports/cuda_api_sum.py]...\n" + "\n" + " ** CUDA API Summary (cuda_api_sum):\n" + "\n" + ) + csv_text = preamble + _NSYS_CSV_HEADER + ( + '60.0,12000,3,4000.0,4000.0,3900.0,4100.0,50.0,"cudaMemcpy"\n' + '40.0,6000,10,600.0,600.0,500.0,700.0,20.0,"cudaLaunchKernel"\n' + ) + result = _sum_cudamemcpy_calls(csv_text) + assert result.parsed is True + assert result.count == 3 + # A preamble with no CSV after it is still not parsed. + assert _sum_cudamemcpy_calls(preamble) == _sum_cudamemcpy_calls("") + + +def test_gate_flags_follow_the_compiler_basename(): + # Ruling R24: the gcc-style warning and -std flags only for gcc, gfortran, + # cc and clang; a vendor compiler gets -O2 and the user's --flags. + from rosenna.gate import _c_flags, _f_flags + assert _c_flags("gcc-15") == ["-O2", "-Wall", "-Wextra", "-std=c11"] + assert _c_flags("/usr/bin/clang") == ["-O2", "-Wall", "-Wextra", "-std=c11"] + assert _c_flags("cc") == ["-O2", "-Wall", "-Wextra", "-std=c11"] + assert _f_flags("gfortran") == ["-O2", "-Wall", "-Wextra", "-std=f2008"] + for vendor in ("nvc", "nvfortran", "amdclang", "amdflang", "flang", "icx", "ifx", + "/opt/nvidia/hpc_sdk/Linux_x86_64/24.5/compilers/bin/nvc"): + assert _c_flags(vendor) == ["-O2"], vendor + assert _f_flags(vendor) == ["-O2"], vendor + + def test_gate_runs_in_host_fallback_mode_and_writes_a_report(tmp_path, golden_model): # On a machine without a GPU the gate runs with --host-fallback, which drops the # MANDATORY requirement but exercises every other step, so the script itself is tested. @@ -64,6 +100,12 @@ def test_gate_runs_in_host_fallback_mode_and_writes_a_report(tmp_path, golden_mo for key in ("gemm_big", "embedded", "file-loaded", "fortran", "c", "infer_batch", "backend: omp", "ns per point", "host-fallback"): assert key in report + # Rulings R21/R24: the recipes get CFLAGS/FFLAGS explicitly and the host + # compiler links the per-point harness against the omp-backend archive. + assert f"CC={cc} 'CFLAGS=-O2 -Wall -Wextra -std=c11' ROSENNA_OFFLOAD_FLAGS=-fopenmp" in report + assert f"FC={fc} 'FFLAGS=-O2 -Wall -Wextra -std=f2008' ROSENNA_OFFLOAD_FLAGS=-fopenmp" in report + assert "backend=omp, host compiler: serves the per-point harness" in report + assert f"$ {cc} -fopenmp gate_harness1.o libgemm_big.a -lm -o gate_harness1" in report def test_gate_fails_loudly_when_offload_is_mandatory_and_absent(tmp_path, golden_model): From 4121d3f3c66b9edc71680556f23395b7af0f6ef8 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Mon, 14 Sep 2026 01:23:31 -0500 Subject: [PATCH 18/84] fix: hoist g.nut into face()'s locals in the microfd patch; CI grep only fails on compiler-absence skips - I5: patch.md section 5 adds nut=g.nut to face()'s locals line and reads nut[c], nut[c+s] inside the FOR3 region, matching microfd's rule that kernels copy g.* to locals before offloading (closure() in section 3 already did). - I7 (ruling R25): the device-path CI step sets pipefail so pytest's exit status survives the tee, and greps only for the compiler-absence skip reasons; a dead-model skip is not a CI failure. --- .github/workflows/CI.yml | 8 ++++++-- python/examples/microfd_closure/patch.md | 20 ++++++++++++++++---- 2 files changed, 22 insertions(+), 6 deletions(-) diff --git a/.github/workflows/CI.yml b/.github/workflows/CI.yml index 87f7421..27766b7 100644 --- a/.github/workflows/CI.yml +++ b/.github/workflows/CI.yml @@ -41,10 +41,14 @@ jobs: pip install -e python cd python && python3 -m pytest tests -v - - name: Device-path tests ran on the host (not skipped) + # A dead-model skip (the golden models are regenerated unseeded each run + # and can come out all-zero) is legitimate; a skip for a missing compiler + # is not. pipefail keeps pytest's own exit status through the tee. + - name: Device-path tests ran on the host (not skipped for a missing compiler) run: | + set -o pipefail cd python && python3 -m pytest tests/test_device_c.py -v -rs 2>&1 | tee device.log - ! grep -q "SKIPPED" device.log + ! grep -E -q "SKIPPED.*(no C compiler|-fopenmp|no gfortran)" device.log - name: Run test cases run: | diff --git a/python/examples/microfd_closure/patch.md b/python/examples/microfd_closure/patch.md index f59a03c..122ab43 100644 --- a/python/examples/microfd_closure/patch.md +++ b/python/examples/microfd_closure/patch.md @@ -114,14 +114,26 @@ must stop at `nx+2*NG-2` (i.e. `i0)` viscous block, blend the molecular viscosity -with the face-averaged turbulent viscosity from the two cells straddling -the face, and use that blend (`muf`, not `mu`) in the stress: +`face()` hoists every `g.*` it uses into locals before its `FOR3` region +(`microfd.c:18`: kernels copy scalars to locals before offloading), so the +closure field is hoisted the same way -- a `g.nut` dereference inside the +region would map the whole `g` struct with an unattached host pointer +instead of using the `nut` array section mapped in section 6: + +```diff + static void face(int d){ // flux through the face c+1/2 normal to d, stored in F at cell c +- LOCALS; const double gam=g.gamma, mu=g.mu, kap=mu*gam/((gam-1)*g.pr), h0=g.h[0],h1=g.h[1],h2=g.h[2]; const double*w=g.w; double*F=g.F+(size_t)d*NV*nc; ++ LOCALS; const double gam=g.gamma, mu=g.mu, kap=mu*gam/((gam-1)*g.pr), h0=g.h[0],h1=g.h[1],h2=g.h[2]; const double*w=g.w, *nut=g.nut; double*F=g.F+(size_t)d*NV*nc; +``` + +Then, inside the existing `if(mu>0)` viscous block, blend the molecular +viscosity with the face-averaged turbulent viscosity from the two cells +straddling the face, and use that blend (`muf`, not `mu`) in the stress: ```diff if(mu>0){ // viscous stress and heat flux at the face, 2nd-order central const long st[3]={1,sx,sy}; const double h[3]={h0,h1,h2}; double du[3][3], div=0; -+ const double muf=mu+.5*(g.nut[c]+g.nut[c+s]); // molecular + face-averaged closure viscosity ++ const double muf=mu+.5*(nut[c]+nut[c+s]); // molecular + face-averaged closure viscosity for(int a=0;a<3;a++) for(int b=0;b<3;b++) if(a==b||a==d||b==d){ const double*u=w+(1+a)*nc+c; const long t=st[b]; // off-normal off-diagonal terms are dead du[a][b]= b==d ? (u[s]-u[0])/h[d] : (u[t]-u[-t]+u[s+t]-u[s-t])/(4*h[b]); } // normal: two cells; tangential: averaged central for(int a=0;a<3;a++) div+=du[a][a]; From b0f2a823ea5318a2e42bcba459842c53d91d131b Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Mon, 14 Sep 2026 01:24:25 -0500 Subject: [PATCH 19/84] docs: state the cuda/hip archive limitation for per-point file-loaded hosts, the runtime link flags, and what CI has and has not run - python/README.md: the opening says generate writes sources and recipes and make builds the archives, and points at Verify's residency caveat; ROSENNA_BACKEND / --backend are a recipe variable and a gate flag, not a generate flag; the Call it from C section states that a cuda/hip archive does not serve the per-point path of a file-loaded model and gives the omp-backend build line for the host compiler; the bind(C) route names the runtime link flags (-lcudart / nvfortran -cuda, -lamdhip64); the Verify section no longer claims nvcc/hipcc compile in CI; Limits carries the limitation and its planned resolution. - examples/microfd_closure/README.md: the two-archive gate build replaces the retired device-compiler link and -Xcompiler forwarding text. --- python/README.md | 64 ++++++++++++++++++----- python/examples/microfd_closure/README.md | 30 +++++------ 2 files changed, 66 insertions(+), 28 deletions(-) diff --git a/python/README.md b/python/README.md index f60ff68..2e867d0 100644 --- a/python/README.md +++ b/python/README.md @@ -8,12 +8,16 @@ or HIP. ## What you get -`rosenna generate model.onnx` turns an ONNX model into `lib.a` (C) and -`lib_f.a` (Fortran, a module in the archive): `_infer` is a -plain per-point function you call inside your own GPU loop, exactly like any -other device-callable routine in your solver, and its weights are -device-resident -- baked into the generated source as constants for a small -model, or loaded once at startup and copied to the device for a large one. +`rosenna generate model.onnx` writes the sources and build recipes for a C +library and a Fortran module; `make -f .mk` and `make -f +_fortran.mk` then build `lib.a` (C) and `lib_f.a` +(Fortran, a module in the archive). `_infer` is a plain per-point +function you call inside your own GPU loop, exactly like any other +device-callable routine in your solver. Its weights are baked into the +generated source as constants for a small model, or loaded once at startup +by `_init` for a large one, and the generated code is written so that +they live on the device in either case -- with the caveat, stated under +[Verify](#verify), that the device path has not yet been run on a GPU. ## Install @@ -30,7 +34,9 @@ The generator itself only needs Python (`onnx`, `numpy`, `onnxruntime` for - GPU path: `nvc`/`nvfortran` (NVIDIA HPC SDK) for OpenMP-target or OpenACC on an NVIDIA GPU; `amdclang`/`amdflang` for OpenMP-target on an AMD GPU; `icx`/`ifx` for OpenMP-target on an Intel GPU. `nvcc` or `hipcc` if you also want the - native batched kernel (`--backend cuda|hip`, see [Call it from C](#call-it-from-c)). + native batched kernel: `ROSENNA_BACKEND=cuda|hip` when you run the C recipe + (and `--backend cuda|hip` to `rosenna gpu-gate`); `generate` itself has no + backend flag and always writes every file (see [Call it from C](#call-it-from-c)). ## Generate @@ -152,6 +158,27 @@ of a *file-loaded* model needs one more call: after every `model_init`, call `model_device_bind_here()` in every translation unit whose kernels call `model_infer` (an embedded model needs neither). +A cuda/hip archive does not serve the per-point path of a file-loaded +model. `nvcc`/`hipcc` compile `model.c` with `_OPENMP` and `_OPENACC` +undefined, so the weight arrays get no `declare target` device copies and +`model_init`'s `target update` is not compiled; a host translation unit +compiled by `nvc -mp=gpu` (or `amdclang`, or `gcc` with offload) that calls +`model_infer` inside its own offload loop then reads device copies that do +not exist. So a per-point OpenMP or OpenACC host calling `model_infer` on a +file-loaded model must link the `omp`-backend archive built by that same +host compiler with its offload flags: + +```sh +make -f model.mk ROSENNA_BACKEND=omp CC=nvc ROSENNA_OFFLOAD_FLAGS="-mp=gpu -gpu=cc80" +``` + +The cuda/hip archive serves `model_infer_batch` and your own CUDA/HIP +kernels that call `model_infer` after `model_device_bind_here()`. Build +both archives in separate directories if one program needs both. Embedded +models are unaffected: every translation unit holds its own copy of the +constants, so they work with every backend. See [Limits](#limits) for the +planned resolution. + Embedded weights on a CUDA/HIP build go to one of two storage classes, decided per model at generate time, not at build time: under 48 KB (12,288 float32 or 6,144 float64 parameters) `ROSENNA_CONST` is `__constant__` @@ -276,7 +303,12 @@ status = model_infer_batch_dev(npts, c_loc(x), c_loc(y_batch), c_null_ptr) !$acc end host_data ``` -and links both archives: `-lmodel_f -lmodel`. +and links both archives plus the runtime the cuda/hip archive was built +against, which a host that is not itself linked by `nvcc`/`hipcc` has to +name explicitly: `-lmodel_f -lmodel -L$CUDA_HOME/lib64 -lcudart` for CUDA +(or `nvfortran -cuda`, which links it for you), `-lmodel_f -lmodel +-L$ROCM_PATH/lib -lamdhip64` for HIP. With an `omp`-backend `libmodel.a` +nothing extra is needed. `model_infer` is not itself inlined across the `use model_model` boundary by every compiler, so a Fortran host's own offload loop generally gets a real @@ -362,10 +394,12 @@ per-point Fortran host, and a host that hands device-resident data to command and its output to `gate-report.md`. Until `gpu-gate` has been run on a GPU machine, the device path is -unvalidated: everything above compiles and runs on the host, and the CUDA -and HIP decoration compiles under `nvcc`/`hipcc` in CI, but none of it has -executed on a device. See `python/examples/microfd_closure/` for a worked -example of wiring a generated model into a solver, with the same caveat. +unvalidated: everything above compiles and runs on the host, but none of +it has executed on a device. A compile-only `nvcc` job exists in CI +(`.github/workflows/CI.yml`, `nvcc_compile`) and its result will be +reported here after its first run; `hipcc` is only ever exercised by the +gate. See `python/examples/microfd_closure/` for a worked example of +wiring a generated model into a solver, with the same caveat. ## Limits @@ -380,6 +414,12 @@ example of wiring a generated model into a solver, with the same caveat. The `omp` backend's `_infer_batch` fallback calls `_infer` per point and has the same silent behavior; only the cuda/hip path's `_infer_batch` catches this, returning status 10. +- A cuda/hip archive of a file-loaded model serves `_infer_batch` and + CUDA/HIP kernels only; a per-point OpenMP/OpenACC host must link the + `omp`-backend archive built by its own compiler (see [Call it from + C](#call-it-from-c)). The planned resolution is that the host compiler + always compiles `.c` and `_kernel.cu` owns every CUDA/HIP + symbol behind `-DROSENNA_NATIVE_KERNEL`, so one archive serves both paths. - The native batched kernel (`ROSENNA_BACKEND=cuda|hip`) launches one thread per point in this release; a fused, tiled batched GEMM is planned once the GPU gate has timed this one. diff --git a/python/examples/microfd_closure/README.md b/python/examples/microfd_closure/README.md index a0a0f23..0d399f0 100644 --- a/python/examples/microfd_closure/README.md +++ b/python/examples/microfd_closure/README.md @@ -77,19 +77,17 @@ OpenMP-target fallback on either. Only after that report exists for the backend and hardware you actually run microfd on should the closure above be described as device-validated rather than host-validated. -For the file-loaded configuration's per-point harness under `--backend -cuda|hip`, the gate links the generated library with the device compiler as -the link driver (`nvcc`/`hipcc`, forwarding the host offload flags through -as `-Xcompiler ` for cuda, directly for hip -- ruling R14/R19); this -is the form `gate.py` builds and neither form has been verified against a -real toolchain here. If that link fails on a toolchain that rejects a -device compiler as the driver for a host-compiled object, the documented -way out is linking with the HOST compiler instead and naming the CUDA/HIP -runtime explicitly: - -``` -# cuda -nvc -mp=gpu -gpu=cc80 gate_harness1.o libgemm_big.a -L$CUDA_HOME/lib64 -lcudart -lm -o gate_harness1 -# hip -amdclang -fopenmp --offload-arch=gfx90a gate_harness1.o libgemm_big.a -L$ROCM_PATH/lib -lamdhip64 -lm -o gate_harness1 -``` +Under `--backend cuda|hip` the gate builds two C archives per +configuration (rulings R21/R22): the per-point C harness is compiled and +linked by the HOST compiler with its offload flags against an `omp`-backend +`libgemm_big.a` that the same host compiler built, and the `.cu` driver of +the `infer_batch` harness is compiled and linked by the device compiler +against the `cuda|hip` archive in `_lib/`. That is the same rule a +solver has to follow: a per-point OpenMP/OpenACC host calling +`_infer` on a file-loaded model links the `omp`-backend archive its +own compiler built, never the cuda/hip one (see the `Call it from C` +section of `python/README.md`); microfd's closure embeds, so it is not +affected. A host that does link the cuda/hip archive itself (for +`_infer_batch`) names the runtime explicitly, `-L$CUDA_HOME/lib64 +-lcudart` or `-L$ROCM_PATH/lib -lamdhip64`, unless `nvcc`/`hipcc` or +`nvfortran -cuda` drives the link. From 375523f290c21b8fae9f43b887a06b50cbac2875 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Mon, 14 Sep 2026 01:29:16 -0500 Subject: [PATCH 20/84] fix: embedded fortran weights are initialized protected arrays so gfortran -fopenacc can compile the module gfortran 15 -fopenacc materializes a parameter array read inside a routine seq as a static, demands an OpenACC declare for it, and refuses a declare on a named constant, so an embedded Fortran module did not compile at all under -fopenacc (-c; the diagnostic comes after the front end, and the suite's -fsyntax-only check let it through). The embedded weights are now initialized protected module arrays with declare copyin, the same declare-target line as before; the chunk arrays stay parameter since only the initializer names them. The OpenACC test compiles for real, the module test asserts the new declarations, README wording follows. --- python/README.md | 6 +++-- python/rosenna/cli.py | 2 +- python/rosenna/emit_fortran.py | 35 +++++++++++++++++++---------- python/tests/test_device_fortran.py | 23 +++++++++++-------- 4 files changed, 42 insertions(+), 24 deletions(-) diff --git a/python/README.md b/python/README.md index 2e867d0..140f4fe 100644 --- a/python/README.md +++ b/python/README.md @@ -63,8 +63,10 @@ model file's stem. `generate` prints every file it wrote: Both recipes write into the same output directory and build there: a `--lang both` run gives you one directory holding both archives. -A model embeds its weights as constants (`ROSENNA_CONST` in C, a Fortran -`parameter` array) automatically when it has fewer than `EMBED_THRESHOLD` +A model embeds its weights as constants (`ROSENNA_CONST` in C, an +initialized `protected` module array in Fortran, since gfortran's OpenACC +cannot read a `parameter` array from a device routine) automatically when +it has fewer than `EMBED_THRESHOLD` (1,000,000) parameters; above that it is file-loaded by default. `--embed-weights` forces embedding regardless of size; `--no-embed` forces a `.rwt` file regardless of size. An embedded model has no `_init` at all -- there is diff --git a/python/rosenna/cli.py b/python/rosenna/cli.py index 8c03e9b..30073e0 100644 --- a/python/rosenna/cli.py +++ b/python/rosenna/cli.py @@ -159,7 +159,7 @@ def _cmd_generate(args) -> int: written += [c_path, h_path, mk_path, cu_path, rt_path] # An embedded plan has no weights file to write in either language: every - # weight is already a `parameter`/ROSENNA_CONST array baked into the + # weight is already an initialized `protected`/ROSENNA_CONST array baked into the # generated source (controller ruling R3, flipped by Task 4: Fortran now # embeds by default too, so this no longer depends on which languages # were requested). diff --git a/python/rosenna/emit_fortran.py b/python/rosenna/emit_fortran.py index db9ee1d..755da09 100644 --- a/python/rosenna/emit_fortran.py +++ b/python/rosenna/emit_fortran.py @@ -97,14 +97,25 @@ def _weight_symbol_list(plan: Plan) -> str: # _wrap_items alone cannot keep a several-hundred-element weight (e.g. # gemm_big's 40x30 = 1200-element layer) legal. Above _EMBED_CHUNK elements, # the flat literal list is split into several small `parameter` arrays -# (each well under the continuation limit) and reassembled with one more -# `parameter` statement over their names -- a statement with a handful of -# short identifiers, never close to either limit itself. +# (each well under the continuation limit) and reassembled by the weight's +# own initializer over their names -- a statement with a handful of short +# identifiers, never close to either limit itself. _EMBED_CHUNK = 500 def _emit_embedded_weights(plan: Plan) -> list: - """`plan.embed`'s weights, as `parameter` arrays holding the literal values. + """`plan.embed`'s weights, as initialized `protected` module arrays. + + Not `parameter`: gfortran -fopenacc materializes a named-constant array + read inside a `routine seq` as a static and then demands an OpenACC + `declare` for it, which it refuses on a named constant ("not a + variable"), so an embedded module could not be compiled (-c; the error + is raised after the front end, so -fsyntax-only does not see it). An + initialized `protected` module variable takes `declare copyin` and + `declare target`, gets its device copy at program start under either + offload family, and is as read-only outside the module as a constant. + The chunk arrays a long literal list is split into (see _EMBED_CHUNK) + stay `parameter`: only the initializer names them, never the routine. A rank-1 array (every bias) is a plain bracketed list. A rank-2 array (every Gemm/MatMul weight) is `reshape([flat values], [dims])`: `values` @@ -130,9 +141,9 @@ def _emit_embedded_weights(plan: Plan) -> list: chunk, " ]", " " * 8) flat = chunk_names if len(w.shape) <= 1: - head, tail = f" real(wp), parameter :: {w.symbol}{dims} = [ ", " ]" + head, tail = f" real(wp), protected :: {w.symbol}{dims} = [ ", " ]" else: - head = f" real(wp), parameter :: {w.symbol}{dims} = reshape([ " + head = f" real(wp), protected :: {w.symbol}{dims} = reshape([ " tail = f" ], {_weight_dims_list(plan, w.symbol)})" lines += _wrap_items(head, flat, tail, " " * 8) return lines @@ -171,12 +182,12 @@ def emit_fortran(plan: Plan) -> str: lines += _emit_embedded_weights(plan) if plan.weights: lines.append(f" !$omp declare target({_weight_symbol_list(plan)})") - if not plan.embed: - # gfortran rejects a `routine seq`/declare-target function that - # reads a file-scope array with no OpenACC `declare` directive of - # its own; an embedded plan's arrays are compile-time constants - # instead and need none. - lines.append(f" !$acc declare create({_weight_symbol_list(plan)})") + # gfortran rejects a `routine seq` function that reads a module array + # with no OpenACC `declare` directive of its own: `create` for the + # file-loaded arrays (init then does `update device`), `copyin` for + # the embedded ones, whose initializer is the device copy's value. + clause = "copyin" if plan.embed else "create" + lines.append(f" !$acc declare {clause}({_weight_symbol_list(plan)})") lines += [ "", " interface", diff --git a/python/tests/test_device_fortran.py b/python/tests/test_device_fortran.py index b0aa618..c0662e1 100644 --- a/python/tests/test_device_fortran.py +++ b/python/tests/test_device_fortran.py @@ -140,17 +140,22 @@ def test_embedded_module_has_no_init_and_file_loaded_does(golden_model): plan = build_plan(load_graph(golden_model("gemm_small")), dtype="f64", embed=embed) src = emit_fortran(plan) assert ("subroutine gemm_small_init(" in src) == expect_init - assert ("real(wp), parameter :: w0" in src) == embed - assert ("real(wp), protected :: w0" in src) == (not embed) + # Both forms are `protected` module arrays; the embedded one carries + # its initializer (see _emit_embedded_weights for why not `parameter`). + assert ("real(wp), protected :: w0(2,2) = reshape([" in src) == embed + assert ("real(wp), protected :: w0(2,2)\n" in src) == (not embed) + assert ("!$acc declare copyin(w0, b0, w1, b1)" in src) == embed + assert ("!$acc declare create(w0, b0, w1, b1)" in src) == (not embed) def test_generated_fortran_is_warning_free_under_openacc(tmp_path, golden_model): # Mirrors tests/test_kernel.py::test_generated_c_is_warning_free_under_openacc. - # gfortran -fopenacc rejects a `routine seq` function reading a file-scope - # array with no `declare` directive of its own (why _emit_embedded_weights - # skips `!$acc declare create` -- an embedded plan's arrays are compile- - # time constants and need none, but a file-loaded plan's `protected` - # arrays do); this is the test that guards that comment. + # gfortran -fopenacc rejects a `routine seq` function reading a module + # array with no `declare` directive of its own, and refuses a `declare` + # on a `parameter` array (why _emit_embedded_weights emits initialized + # `protected` arrays with `declare copyin`). A real compile (-c), not + # -fsyntax-only: the diagnostic comes after the front end and + # -fsyntax-only let an uncompilable embedded module through. fc = shutil.which("gfortran") if not fc: pytest.skip("no gfortran") @@ -164,8 +169,8 @@ def test_generated_fortran_is_warning_free_under_openacc(tmp_path, golden_model) src_path = tmp_path / f"{name}_{embed}_model.f90" src_path.write_text(emit_fortran(plan)) r = subprocess.run( - [fc, "-O2", "-Wall", "-Wextra", "-std=f2008", "-fopenacc", "-fsyntax-only", - src_path.name], + [fc, "-O2", "-Wall", "-Wextra", "-std=f2008", "-fopenacc", "-c", + src_path.name, "-o", f"{name}_{embed}_model.o"], cwd=tmp_path, capture_output=True, text=True) assert r.returncode == 0 and r.stderr == "", (name, embed, r.stderr) From 0f8e553523fdbe8b8f6dc6b09e21d8059de2f30b Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Mon, 14 Sep 2026 01:31:49 -0500 Subject: [PATCH 21/84] fix: gpu-gate records an unrunnable harness as a failed step instead of aborting the run _sh caught only FileNotFoundError; a produced-but-not-executable output (PermissionError) escaped to the FATAL handler and skipped every later configuration. Any OSError from the exec is now a 127 step in the report. --- python/rosenna/gate.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/python/rosenna/gate.py b/python/rosenna/gate.py index 6e9fdb9..3ac9710 100644 --- a/python/rosenna/gate.py +++ b/python/rosenna/gate.py @@ -109,8 +109,10 @@ def _sh(report: _Report, label: str, args: list, cwd=None, env=None, input_text= try: proc = subprocess.run(args, cwd=cwd, env=env, input=input_text, capture_output=True, text=True, timeout=timeout) - except FileNotFoundError as e: - proc = _FakeProc(127, f"{args[0]}: not found ({e.strerror})") + except OSError as e: + # A missing executable, or one the previous step left unrunnable: + # recorded as a failed step, so the remaining configurations still run. + proc = _FakeProc(127, f"{args[0]}: cannot run ({e.strerror})") except subprocess.TimeoutExpired as e: proc = _FakeProc(124, f"timed out after {timeout}s\nstdout so far:\n{e.stdout}\nstderr so far:\n{e.stderr}") report.outcome(proc) From 933909dca79a851a766ee0d76b2a8e760e3831c5 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Mon, 14 Sep 2026 01:55:29 -0500 Subject: [PATCH 22/84] test: filter driver-level toolchain notices in every device test, as the C tests already do --- python/tests/conftest.py | 26 ++++++++++++++++++++++++ python/tests/test_device_c.py | 10 +++++++--- python/tests/test_device_fortran.py | 10 +++++++--- python/tests/test_kernel.py | 11 ++++++---- python/tests/test_regressions.py | 31 ++++++----------------------- 5 files changed, 53 insertions(+), 35 deletions(-) diff --git a/python/tests/conftest.py b/python/tests/conftest.py index e26c9e6..1a34d31 100644 --- a/python/tests/conftest.py +++ b/python/tests/conftest.py @@ -1,4 +1,5 @@ """Fixtures for golden file models and for inline models built with onnx.helper.""" +import re import subprocess import sys from pathlib import Path @@ -8,6 +9,31 @@ import pytest from onnx import helper, numpy_helper, TensorProto +# A diagnostic about the generated source carries a :: location. +# A driver-level notice instead names the tool as its "location" -- for +# example Apple clang on the macOS CI runner prints, on every invocation and +# whatever the source, +# clang: warning: overriding deployment version from '16.0' to '26.0' [-Woverriding-deployment-version] +# which is about the SDK versus the deployment target and nothing to do with +# our C (ruling R21). gfortran's own multi-line diagnostics keep their +# `:::` header and a bare `Warning: ...` line, neither of +# which this pattern matches, so they survive. +_DRIVER_NOTICE = re.compile(r"^[^\s:]+: (warning|note): ") + + +def _source_diagnostics(stderr: str): + """Split compiler stderr into (about the source, driver-level noise).""" + kept, dropped = [], [] + for line in stderr.splitlines(): + (dropped if _DRIVER_NOTICE.match(line) else kept).append(line) + return "\n".join(kept).strip(), "\n".join(dropped).strip() + + +def _assert_warning_free(lang: str, stderr: str) -> None: + kept, dropped = _source_diagnostics(stderr) + assert kept == "", (f"{lang}: diagnostics about the generated source:\n{kept}\n" + f"(driver-level notices ignored: {dropped or 'none'})") + def save_model(directory, name, nodes, inits, in_shape, out_shape, elem=TensorProto.FLOAT): """Save a one-input, one-output ONNX graph as /.onnx and return the path.""" diff --git a/python/tests/test_device_c.py b/python/tests/test_device_c.py index 9c7b76d..1789f3f 100644 --- a/python/tests/test_device_c.py +++ b/python/tests/test_device_c.py @@ -8,6 +8,7 @@ from rosenna.plan import build_plan from rosenna.weights import write_weights from rosenna.emit_c import emit_c +from tests.conftest import _assert_warning_free from tests.test_emit_fortran import _live_reference @@ -56,11 +57,13 @@ def _build_and_run(tmp_path, name, plan, graph, cc, flags, inputs, env=None): objs = ["host.c"] if not plan.embed: r = subprocess.run([cc, *flags, "-c", f"{name}.c"], cwd=tmp_path, capture_output=True, text=True) - assert r.returncode == 0 and r.stderr == "", r.stderr + assert r.returncode == 0, r.stderr + _assert_warning_free("gcc", r.stderr) subprocess.run(["ar", "rcs", f"lib{name}.a", f"{name}.o"], cwd=tmp_path, check=True) objs.append(f"lib{name}.a") r = subprocess.run([cc, *flags, *objs, "-lm", "-o", "host"], cwd=tmp_path, capture_output=True, text=True) - assert r.returncode == 0 and r.stderr == "", r.stderr + assert r.returncode == 0, r.stderr + _assert_warning_free("gcc", r.stderr) stdin = f"{len(inputs)}\n" + " ".join(repr(float(v)) for v in inputs.ravel()) return subprocess.run(["./host"], cwd=tmp_path, input=stdin, capture_output=True, text=True, env=env) @@ -148,7 +151,8 @@ def test_device_pass_guard_needs_the_cuda_compiler_not_just_the_arch(tmp_path, g flags = ["-O2", "-Wall", "-Wextra", "-std=c11", "-D__CUDA_ARCH__=800"] for src in ("host.c", f"{name}.c"): r = subprocess.run([cc, *flags, "-c", src], cwd=tmp_path, capture_output=True, text=True) - assert r.returncode == 0 and r.stderr == "", (src, r.stderr) + assert r.returncode == 0, (src, r.stderr) + _assert_warning_free("gcc", r.stderr) pre = subprocess.run([cc, *flags, "-E", "host.c"], cwd=tmp_path, capture_output=True, text=True) assert pre.returncode == 0 assert f"{name}_devw" not in pre.stdout and f"{name}_w0[" in pre.stdout diff --git a/python/tests/test_device_fortran.py b/python/tests/test_device_fortran.py index c0662e1..b18584d 100644 --- a/python/tests/test_device_fortran.py +++ b/python/tests/test_device_fortran.py @@ -8,6 +8,7 @@ from rosenna.plan import build_plan from rosenna.weights import write_weights from rosenna.emit_fortran import emit_fortran, emit_fortran_recipe +from tests.conftest import _assert_warning_free from tests.test_emit_fortran import _live_reference @@ -67,10 +68,12 @@ def _build_and_run(tmp_path, name, embed, inputs, golden_model): fc = _omp_fc() flags = ["-O2", "-Wall", "-Wextra", "-std=f2008", "-fopenmp"] r = subprocess.run([fc, *flags, "-c", f"{name}_model.f90"], cwd=tmp_path, capture_output=True, text=True) - assert r.returncode == 0 and r.stderr == "", r.stderr + assert r.returncode == 0, r.stderr + _assert_warning_free("gfortran", r.stderr) subprocess.run(["ar", "rcs", f"lib{name}.a", f"{name}_model.o"], cwd=tmp_path, check=True) r = subprocess.run([fc, *flags, "host.f90", f"lib{name}.a", "-o", "host"], cwd=tmp_path, capture_output=True, text=True) - assert r.returncode == 0 and r.stderr == "", r.stderr + assert r.returncode == 0, r.stderr + _assert_warning_free("gfortran", r.stderr) stdin = f"{len(inputs)}\n" + "\n".join(" ".join(repr(float(v)) for v in row) for row in inputs) out = subprocess.run(["./host"], cwd=tmp_path, input=stdin, capture_output=True, text=True, check=True).stdout return np.array([[float(v) for v in line.split()] for line in out.strip().splitlines()]) @@ -172,7 +175,8 @@ def test_generated_fortran_is_warning_free_under_openacc(tmp_path, golden_model) [fc, "-O2", "-Wall", "-Wextra", "-std=f2008", "-fopenacc", "-c", src_path.name, "-o", f"{name}_{embed}_model.o"], cwd=tmp_path, capture_output=True, text=True) - assert r.returncode == 0 and r.stderr == "", (name, embed, r.stderr) + assert r.returncode == 0, (name, embed, r.stderr) + _assert_warning_free("gfortran", r.stderr) def test_bind_c_interface_targets_the_c_infer_batch_symbol(golden_model): diff --git a/python/tests/test_kernel.py b/python/tests/test_kernel.py index e5a2533..ba75574 100644 --- a/python/tests/test_kernel.py +++ b/python/tests/test_kernel.py @@ -10,7 +10,7 @@ from rosenna.emit_kernel import emit_kernel from rosenna.rt_header import rt_header from rosenna.emit_c import emit_c, emit_c_recipe, CONSTANT_MEMORY_LIMIT -from tests.conftest import save_model +from tests.conftest import _assert_warning_free, save_model def test_kernel_source_uses_only_the_rt_macros(golden_model): @@ -175,7 +175,8 @@ def test_generated_c_is_warning_free_under_openacc(tmp_path, golden_model): """) r = subprocess.run([cc, "-O2", "-Wall", "-Wextra", "-std=c11", "-fopenacc", f"{name}.c", "host.c", "-lm", "-o", "host"], cwd=d, capture_output=True, text=True) - assert r.returncode == 0 and r.stderr == "", r.stderr + assert r.returncode == 0, r.stderr + _assert_warning_free("gcc", r.stderr) if embed: assert subprocess.run(["./host"], cwd=d, capture_output=True).returncode == 0 # Status 10 is in the emitted legend, next to the routine that returns it. @@ -276,7 +277,8 @@ def _omp_build_and_run(tmp_path, name, plan, cc, init): assert "warning" not in r.stderr, r.stderr (tmp_path / "host.c").write_text(_omp_host(name, plan.input.shape[0], plan.output.shape[0], 16, init)) r = subprocess.run([cc, "-O2", "-Wall", "-Wextra", "-std=c11", "-fopenmp", "host.c", f"lib{name}.a", "-lm", "-o", "host"], cwd=tmp_path, capture_output=True, text=True) - assert r.returncode == 0 and r.stderr == "", r.stderr + assert r.returncode == 0, r.stderr + _assert_warning_free("gcc", r.stderr) return subprocess.run(["./host"], cwd=tmp_path, capture_output=True, text=True) @@ -335,7 +337,8 @@ def test_generated_c_is_warning_free_under_a_plain_compiler(tmp_path, golden_mod source, header = emit_c(plan) (d / f"{name}.c").write_text(source); (d / f"{name}.h").write_text(header) r = subprocess.run([cc, "-O2", "-Wall", "-Wextra", "-std=c11", "-c", f"{name}.c"], cwd=d, capture_output=True, text=True) - assert r.returncode == 0 and r.stderr == "", r.stderr + assert r.returncode == 0, r.stderr + _assert_warning_free("gcc", r.stderr) def test_generated_c_compiles_as_cpp_with_the_rt_header_stubbed(tmp_path, golden_model): diff --git a/python/tests/test_regressions.py b/python/tests/test_regressions.py index 3fe7fd9..4b8ee4f 100644 --- a/python/tests/test_regressions.py +++ b/python/tests/test_regressions.py @@ -21,7 +21,7 @@ from rosenna.frontend import load_graph from rosenna.plan import build_plan from rosenna.weights import write_weights -from tests.conftest import save_model +from tests.conftest import _assert_warning_free, _source_diagnostics, save_model from tests.test_emit_c import _build_and_run as _c_build_and_run from tests.test_emit_fortran import _build_and_run as _f_build_and_run from tests.test_emit_fortran import _live_reference @@ -264,31 +264,12 @@ def test_truncated_weights_file_returns_a_status(tmp_path, golden_model, lang): # --- item 10 / verification 3: generated code must compile warning-free ---- - -# A diagnostic about the generated source carries a :: location. -# A driver-level notice instead names the tool as its "location" -- for -# example Apple clang on the macOS CI runner prints, on every invocation and -# whatever the source, +# +# _DRIVER_NOTICE / _source_diagnostics / _assert_warning_free live in +# conftest.py so every test module (device/kernel tests included) can filter +# driver-level toolchain notices, e.g. Apple clang's on the macOS CI runner: # clang: warning: overriding deployment version from '16.0' to '26.0' [-Woverriding-deployment-version] -# which is about the SDK versus the deployment target and nothing to do with -# our C (ruling R21). gfortran's own multi-line diagnostics keep their -# `:::` header and a bare `Warning: ...` line, neither of -# which this pattern matches, so they survive. -_DRIVER_NOTICE = re.compile(r"^[^\s:]+: (warning|note): ") - - -def _source_diagnostics(stderr: str): - """Split compiler stderr into (about the source, driver-level noise).""" - kept, dropped = [], [] - for line in stderr.splitlines(): - (dropped if _DRIVER_NOTICE.match(line) else kept).append(line) - return "\n".join(kept).strip(), "\n".join(dropped).strip() - - -def _assert_warning_free(lang: str, stderr: str) -> None: - kept, dropped = _source_diagnostics(stderr) - assert kept == "", (f"{lang}: diagnostics about the generated source:\n{kept}\n" - f"(driver-level notices ignored: {dropped or 'none'})") +# (ruling R21, R30). def _compile_warnings(tmp_path, onnx_path, name, dtype="f64"): From 80b19f66c74b1cfdac32859524aaae10dc1ac8d6 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Mon, 14 Sep 2026 02:07:44 -0500 Subject: [PATCH 23/84] test: prove target regions by symbol, and skip rather than fail where libgomp ignores MANDATORY On the macOS CI runner (Homebrew GCC 13.4), test_fortran_target_regions_are_real ran a real !$omp target region to completion and returned 0 instead of being refused under OMP_TARGET_OFFLOAD=MANDATORY, while the identical C test passed on the same runner and compiler. MANDATORY enforcement is therefore not portable evidence on its own. Both tests now compile their host source to a standalone object (-c) and assert via nm that it references GOMP_target_ext -- a real target region cannot be compiled without a call to it -- as the primary, platform-independent evidence. The MANDATORY run is kept as corroborating evidence: it still passes the test where libgomp enforces it, and is downgraded to a pytest.skip (naming the platform and compiler) rather than a failure where it does not. --- python/tests/test_device_c.py | 39 +++++++++++++++++++--- python/tests/test_device_fortran.py | 50 +++++++++++++++++++++++------ 2 files changed, 75 insertions(+), 14 deletions(-) diff --git a/python/tests/test_device_c.py b/python/tests/test_device_c.py index 1789f3f..d9b54f3 100644 --- a/python/tests/test_device_c.py +++ b/python/tests/test_device_c.py @@ -1,4 +1,5 @@ import os +import platform import shutil import subprocess import numpy as np @@ -85,14 +86,44 @@ def test_host_region_calls_header_inline_and_matches(tmp_path, golden_model, nam def test_target_regions_are_real(tmp_path, golden_model): - # A host-only libgomp refuses a target region under OMP_TARGET_OFFLOAD=MANDATORY. If the pragmas - # were missing or ignored the program would succeed; this is the cheapest evidence without a GPU. + # Two-tier evidence (controller ruling R31). The only evidence here used to be that a + # host-only libgomp refuses a target region under OMP_TARGET_OFFLOAD=MANDATORY: if the + # pragmas were missing or ignored the program would succeed. That held locally, and this + # test itself passed on the macOS CI runner (Homebrew GCC 13.4) too -- but its Fortran twin + # (tests/test_device_fortran.py::test_fortran_target_regions_are_real) ran to completion and + # returned 0 -- not refused -- on that same runner and compiler. So MANDATORY enforcement is + # not portable evidence by itself, even here. The primary, platform-independent assertion is + # instead that the compiled host object references GOMP_target_ext: a real + # `#pragma omp target` region cannot be compiled without a call to it. The MANDATORY run is + # kept as corroborating evidence where libgomp does enforce it, and downgraded to a skip (not + # a failure) where it does not, naming the toolchain that let it through so the CI log + # records exactly which combination did this. name = "gemm_small" graph = load_graph(golden_model(name)); plan = build_plan(graph, dtype="f64") inputs = np.full((1, plan.input.shape[0]), 0.5) - r = _build_and_run(tmp_path, name, plan, graph, _omp_cc(), ["-O2", "-std=c11", "-fopenmp"], inputs, + cc = _omp_cc() + flags = ["-O2", "-std=c11", "-fopenmp"] + r = _build_and_run(tmp_path, name, plan, graph, cc, flags, inputs, env={**os.environ, "OMP_TARGET_OFFLOAD": "MANDATORY"}) - assert r.returncode != 0 and "MANDATORY" in r.stderr + + # host.c was written by _build_and_run; compile it standalone (-c) to inspect exactly + # what the host's own target region compiled to. + obj = subprocess.run([cc, *flags, "-c", "host.c", "-o", "host_check.o"], cwd=tmp_path, + capture_output=True, text=True) + assert obj.returncode == 0, obj.stderr + + nm = shutil.which("nm") + if not nm: + pytest.skip("no nm") + nm_out = subprocess.run([nm, "-u", "host_check.o"], cwd=tmp_path, capture_output=True, text=True).stdout + undefined = {line.split()[-1] for line in nm_out.splitlines() if line.strip()} + assert any("GOMP_target_ext" in sym for sym in undefined), nm_out + + if r.returncode == 0: + version = subprocess.run([cc, "--version"], capture_output=True, text=True).stdout.splitlines()[0] + pytest.skip(f"libgomp did not enforce OMP_TARGET_OFFLOAD=MANDATORY for a C target region " + f"on {platform.platform()} with {version}") + assert "MANDATORY" in r.stderr, r.stderr def test_plain_compiler_without_openmp_still_matches(tmp_path, golden_model): diff --git a/python/tests/test_device_fortran.py b/python/tests/test_device_fortran.py index b18584d..4297f91 100644 --- a/python/tests/test_device_fortran.py +++ b/python/tests/test_device_fortran.py @@ -1,4 +1,5 @@ import os +import platform import shutil import subprocess import numpy as np @@ -92,13 +93,21 @@ def test_host_region_calls_module_infer_and_matches(tmp_path, golden_model, name def test_fortran_target_regions_are_real(tmp_path, golden_model): - # A host-only libgomp refuses a target region under OMP_TARGET_OFFLOAD=MANDATORY. - # If the pragmas were missing or ignored the program would succeed; this is the - # cheapest evidence without a GPU (mirrors tests/test_device_c.py). This check only - # needs the program to run one point and be refused by libgomp -- it does not compare - # against onnxruntime -- so a constant input (not _live_reference) is enough, and - # cannot itself be a dead-model false pass/fail like test_host_region_calls_module_ - # infer_and_matches above needs to guard against. + # Two-tier evidence (controller ruling R31). The only evidence here used to be that a + # host-only libgomp refuses a target region under OMP_TARGET_OFFLOAD=MANDATORY: if the + # pragmas were missing or ignored the program would succeed. That held locally (gfortran + # 15) but on the macOS CI runner (Homebrew GCC 13.4, `gfortran` -> `gfortran-13`) this + # program ran to completion and returned 0 -- not refused -- while the identical C test + # (tests/test_device_c.py::test_target_regions_are_real) passed on that same runner and + # compiler. So MANDATORY enforcement is not portable evidence by itself. The primary, + # platform-independent assertion is instead that the compiled host object references + # GOMP_target_ext: a real `!$omp target` region cannot be compiled without a call to it. + # The MANDATORY run is kept as corroborating evidence where libgomp does enforce it, and + # downgraded to a skip (not a failure) where it does not, naming the toolchain that let it + # through so the CI log records exactly which combination did this. Neither tier compares + # against onnxruntime, so a constant input (not _live_reference) is enough, and this test + # cannot itself be a dead-model false pass/fail like test_host_region_calls_module_infer_ + # and_matches above needs to guard against. name = "gemm_small" graph = load_graph(golden_model(name)); plan = build_plan(graph, dtype="f64", embed=True) n_in, n_out = plan.input.shape[0], plan.output.shape[0] @@ -106,12 +115,33 @@ def test_fortran_target_regions_are_real(tmp_path, golden_model): (tmp_path / f"{name}_model.f90").write_text(emit_fortran(plan)) (tmp_path / "host.f90").write_text(HOST.format(name=name, n_in=n_in, n_out=n_out, init_lines="")) fc = _omp_fc() - subprocess.run([fc, "-O2", "-std=f2008", "-fopenmp", f"{name}_model.f90", "host.f90", "-o", "host"], - cwd=tmp_path, check=True, capture_output=True) + flags = ["-O2", "-std=f2008", "-fopenmp"] + + # Compile the module first (for its .mod) and the host as a standalone object, so the + # symbol check below inspects exactly what the host's own target region compiled to. + r = subprocess.run([fc, *flags, "-c", f"{name}_model.f90"], cwd=tmp_path, capture_output=True, text=True) + assert r.returncode == 0, r.stderr + r = subprocess.run([fc, *flags, "-c", "host.f90", "-o", "host.o"], cwd=tmp_path, capture_output=True, text=True) + assert r.returncode == 0, r.stderr + + nm = shutil.which("nm") + if not nm: + pytest.skip("no nm") + nm_out = subprocess.run([nm, "-u", "host.o"], cwd=tmp_path, capture_output=True, text=True).stdout + undefined = {line.split()[-1] for line in nm_out.splitlines() if line.strip()} + assert any("GOMP_target_ext" in sym for sym in undefined), nm_out + + r = subprocess.run([fc, *flags, "host.o", f"{name}_model.o", "-o", "host"], cwd=tmp_path, + capture_output=True, text=True) + assert r.returncode == 0, r.stderr stdin = "1\n" + " ".join(repr(float(v)) for v in inputs[0]) r = subprocess.run(["./host"], cwd=tmp_path, input=stdin, capture_output=True, text=True, env={**os.environ, "OMP_TARGET_OFFLOAD": "MANDATORY"}) - assert r.returncode != 0 and "MANDATORY" in r.stderr + if r.returncode == 0: + version = subprocess.run([fc, "--version"], capture_output=True, text=True).stdout.splitlines()[0] + pytest.skip(f"libgomp did not enforce OMP_TARGET_OFFLOAD=MANDATORY for a Fortran target " + f"region on {platform.platform()} with {version}") + assert "MANDATORY" in r.stderr, r.stderr def test_generated_fortran_still_fits_in_132_columns(golden_model): From 17efb7ad39ec28a9c7f8cfc54d985933cbca1fe4 Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Mon, 14 Sep 2026 18:12:31 -0400 Subject: [PATCH 24/84] Validate the GPU path on hardware, then cover every golden model and retire fLibrary Three pieces of work, all driven by running the code on a real A100 (NVIDIA HPC SDK 25.11, nvc/nvfortran/nvcc 13.0) rather than reasoning about it. GPU gate, first run on real hardware ------------------------------------ `rosenna gpu-gate --backend cuda` now PASSes, and the R5 device-residency claim is measured rather than asserted: zero cudaMemcpy inside the timed infer_batch call, embedded and file-loaded. Four defects it found: * nsys never opened its capture range (a plain nvtxRangePushA needs NSYS_NVTX_PROFILER_REGISTER_ONLY=0), resolved its -o path twice, and refused its own --stats export. All three exited 0 while producing nothing, so each is now checked explicitly. * nvfortran does not implement `has_device_addr` at all. The module is emitted as .F90 and the clause is chosen by the preprocessor, so gfortran keeps the standard OpenMP 5.1 spelling. Per-point offload was 30x slower than the batched kernel -------------------------------------------------------- Not IPO/LTO and not register pressure -- nvc's per-thread code is the better of the two. nvc maps one point to one *team*: 1,000,000 blocks of 32 threads with a single active lane each. `distribute parallel for`, which would use the whole block, made nvc emit a kernel that traps. The trigger is an accumulator seeded from a declare-target array, which is exactly how a dense layer is written, so both emitters now add the bias *after* the dot product. That reorders the sum by one term and changes results in the last ulp. Embedded weights also went to __constant__ up to 48 KB, chosen against the 64 KB bank -- a correctness bound, not a performance one. ncu showed 71% of warp-issue stalls on constant-cache misses. The cut is now 2 KB. Per point, every route through the library now costs the same: 1.5-1.8 ns. Operator coverage: 5 of 21 golden models to 21 of 21 ---------------------------------------------------- * rank 1-4 throughout; buffers stay flat and row-major, and a Spatial spec carries literal extents so no ONNX attribute reaches the emitters. * Conv, MaxPool, AveragePool, including auto_pad resolved at generation. * A folding pass evaluates every constant-only node away, which is also what removes the int64 shape tensors the emitters cannot carry. * Reshape/Squeeze/Unsqueeze/Flatten/Identity, and any Transpose that only moves size-1 axes, become buffer aliases: no code, no copy. Other Transposes gather. Add broadcasts a constant. * LSTM: forward, ONNX i/o/f/c gate order, optional initial states. Several graph inputs arrive concatenated in x, which keeps infer(x, y) and with it the batched entry point and the whole device contract. * Gemm gained a row count: an LSTM sequence feeding a Gemm applies it per timestep, which the single-row lowering got wrong by 3e-2. * verify allows a cancellation term in its tolerance (n*eps*scale). onnxruntime blocks and vectorises, so two correct implementations differ by more than rtol*|expected| when a sum cancels. mnist forced this: its f32 build is 4.6e-6 off ORT, and its f64 build matches an independent float64 reference to 2.4e-15. fLibrary retired ---------------- Every golden model now generates and verifies against onnxruntime on both backends, so the runtime library, its parser and the shell suite that drove it are removed. test_golden_suite.py replaces run.sh: 21 models, both backends, compared against onnxruntime rather than recorded output. Both READMEs and both doc/ files rewritten; the root examples/ now build generated code instead of libcorelib.a. 235 tests pass. CUDA gate and gcc/gfortran host-fallback gate both PASS. --- .github/workflows/CI.yml | 6 - .gitignore | 3 + README.md | 218 ++--- doc/methodology.md | 103 ++- doc/opensource.md | 296 +++---- examples/cAPI.c | 26 +- examples/capiTester.f90 | 30 +- examples/run_basic.sh | 29 + examples/run_basic_maclinux.sh | 16 - fLibrary/Makefile | 25 - fLibrary/activation_funcs.f90 | 66 -- fLibrary/derived_types.f90 | 53 -- fLibrary/layers.f90 | 297 ------- fLibrary/modelCreator.fpp | 177 ---- fLibrary/modelParserONNX.py | 475 ----------- fLibrary/objFiles/.gittouch | 1 - fLibrary/onnx_helpers.py | 192 ----- fLibrary/reader.f90 | 443 ---------- fLibrary/rosenna.f90 | 19 - gate-reports/gate-report-host-fallback.md | 358 +++++++++ gate-reports/gate-report.md | 757 ++++++++++++++++++ python/README.md | 74 +- python/examples/microfd_closure/patch.md | 6 +- python/examples/nvhpc_teams_mapping/README.md | 228 ++++++ .../nvhpc_teams_mapping/mapping_emulation.cu | 44 + .../nvhpc_teams_mapping/teams_mapping_repro.c | 46 ++ python/rosenna/cli.py | 7 +- python/rosenna/emit_c.py | 264 +++++- python/rosenna/emit_fortran.py | 265 +++++- python/rosenna/emit_kernel.py | 7 +- python/rosenna/errors.py | 9 + python/rosenna/fold.py | 150 ++++ python/rosenna/frontend.py | 33 +- python/rosenna/gate.py | 144 +++- python/rosenna/plan.py | 376 ++++++++- python/rosenna/validate.py | 205 ++++- python/rosenna/verify.py | 77 +- python/tests/conftest.py | 32 +- python/tests/test_cli.py | 42 +- python/tests/test_cli_smoke.py | 2 +- python/tests/test_device_fortran.py | 12 +- python/tests/test_emit_fortran.py | 12 +- python/tests/test_frontend.py | 11 +- python/tests/test_golden_suite.py | 65 ++ python/tests/test_kernel.py | 18 +- python/tests/test_plan.py | 2 +- python/tests/test_regressions.py | 32 +- python/tests/test_validate.py | 1 + test/Makefile | 75 -- test/c_default_paths.c | 9 - test/nnLSTM.py | 31 - test/run.sh | 30 - test/testChecker.py | 41 - test/test_parser.py | 220 ----- test/unit_tests.f90 | 106 --- test/userTesting.fpp | 53 -- test/weights_format_tests.f90 | 88 -- 57 files changed, 3412 insertions(+), 2995 deletions(-) create mode 100755 examples/run_basic.sh delete mode 100755 examples/run_basic_maclinux.sh delete mode 100644 fLibrary/Makefile delete mode 100644 fLibrary/activation_funcs.f90 delete mode 100644 fLibrary/derived_types.f90 delete mode 100644 fLibrary/layers.f90 delete mode 100644 fLibrary/modelCreator.fpp delete mode 100644 fLibrary/modelParserONNX.py delete mode 100644 fLibrary/objFiles/.gittouch delete mode 100644 fLibrary/onnx_helpers.py delete mode 100644 fLibrary/reader.f90 delete mode 100644 fLibrary/rosenna.f90 create mode 100644 gate-reports/gate-report-host-fallback.md create mode 100644 gate-reports/gate-report.md create mode 100644 python/examples/nvhpc_teams_mapping/README.md create mode 100644 python/examples/nvhpc_teams_mapping/mapping_emulation.cu create mode 100644 python/examples/nvhpc_teams_mapping/teams_mapping_repro.c create mode 100644 python/rosenna/errors.py create mode 100644 python/rosenna/fold.py create mode 100644 python/tests/test_golden_suite.py delete mode 100644 test/Makefile delete mode 100644 test/c_default_paths.c delete mode 100644 test/nnLSTM.py delete mode 100755 test/run.sh delete mode 100644 test/testChecker.py delete mode 100644 test/test_parser.py delete mode 100644 test/unit_tests.f90 delete mode 100644 test/userTesting.fpp delete mode 100644 test/weights_format_tests.f90 diff --git a/.github/workflows/CI.yml b/.github/workflows/CI.yml index 27766b7..16eb558 100644 --- a/.github/workflows/CI.yml +++ b/.github/workflows/CI.yml @@ -50,12 +50,6 @@ jobs: cd python && python3 -m pytest tests/test_device_c.py -v -rs 2>&1 | tee device.log ! grep -E -q "SKIPPED.*(no C compiler|-fopenmp|no gfortran)" device.log - - name: Run test cases - run: | - mkdir -p fLibrary/objFiles - chmod +x test/run.sh - cd test && ./run.sh - # Compile-only check of the generated CUDA sources. There is no GPU here and # no driver is installed: nvcc builds the kernel and the .c-as-C++ library for # sm_80 and the test asserts the archive exists. Running it is the GPU gate's diff --git a/.gitignore b/.gitignore index bf8d376..c0f75d5 100644 --- a/.gitignore +++ b/.gitignore @@ -15,7 +15,10 @@ !goldenFiles/mnist/mnist.onnx !instructions/* !python/examples/**/*.md +!python/examples/**/*.cu !python/README.md +!doc/*.md +!gate-reports/*.md reading.f90 userTesting.f90 linearV3copy.f90 diff --git a/README.md b/README.md index 9758905..33c286f 100644 --- a/README.md +++ b/README.md @@ -14,202 +14,96 @@

RoseNNa is a fast, portable, and minimally-intrusive library for neural network inference. -It can run inference on neural networks in [ONNX](https://onnx.ai/) format, which is universal and can be used with PyTorch, TensorFlow, Keras, and more. +It reads a neural network in [ONNX](https://onnx.ai/) format -- the format PyTorch, TensorFlow and Keras all export -- and **generates** a small, self-contained Fortran module and C library that computes it. __RoseNNa's intended use case is embedding neural networks in Fortran- and C-based HPC codebases.__ -One compiles RoseNNa and links it to an existing PDE (e.g., CFD) solver written in C or Fortran. -You can then evaluate your neural network from the PDE solver at Fortran/C speeds. +You link the generated code into an existing PDE (e.g. CFD) solver and call it per point, on the CPU or inside your own GPU offload loop. -RoseNNa currently supports RNNs, CNNs, and MLPs. -The library is optimized Fortran and outperforms PyTorch (by a factor between 2 and 5x) for the relatively small neural networks used in physics applications, like computational fluid dynamics. -RoseNNa is described in detail in A. Bati, S. H. Bryngelson (2024) Comp. Phys. Comm., 296, 109052.. +RoseNNa supports MLPs, CNNs and RNNs. +Because the generated code has literal loop bounds, no runtime shape logic, no allocation and no mutable global state, it inlines into a solver's own compute kernel -- including a device kernel. +RoseNNa is described in A. Bati, S. H. Bryngelson (2024) Comp. Phys. Comm., 296, 109052., which describes the earlier runtime-parsing library; the generator replaced it (see [History](#history)). ## Hello RoseNNa -``` fortran -program hello_roseNNa - - use rosenna - implicit none - - real, dimension(1,1,28,28) :: input ! model inputs - real, dimension(1,5) :: output ! model outputs - - call initialize() ! reads weights - call use_model(input, output) ! run inference - -end program +```sh +pip install -e python +rosenna generate model.onnx --lang both --out build/ ``` -This example program links to the roseNNa library, parses the model inputs, and runs inference on the loaded library. -Only a few lines are required to use the library: `use rosenna`, `call initialize()`, and `call use_model(args)`. +That writes `model_model.F90` and `model.c`/`model.h` (plus build recipes) into `build/`. Then, in Fortran: -With no arguments, `initialize` reads `onnxModel.txt` and `onnxWeights.bin` from the working directory. -If `onnxWeights.bin` does not exist, it reads a legacy `onnxWeights.txt` instead and prints a notice to standard error; it never does this when a weights path is passed explicitly. -To read the files from elsewhere, pass the paths. -`initialize` is a `bind(c)` procedure, so a Fortran caller must terminate each path with `c_null_char`: ``` fortran -use iso_c_binding -call initialize("path/onnxModel.txt"//c_null_char, "path/onnxWeights.bin"//c_null_char) -``` - -## Dependencies - -We have minimal dependencies. -For example, on MacOS you can get away with just -``` -brew install wget make cmake coreutils gcc -pip install torch onnx numpy fypp onnxruntime pandas -``` -## Basic Example -Here is a quick example of how **roseNNa** works. With just a few steps, you can see how to convert a basic feed-forward neural network originally built with PyTorch into usable, accurate code in Fortran. - -First, `cd` into the `fLibrary/` directory. - -Then, create PyTorch model and convert to ONNX: -``` bash -python ../goldenFiles/gemm_small/gemm_small.py -``` - -Read and interpret the corresponding output files from the last step via -``` bash -python modelParserONNX.py -f ../goldenFiles/gemm_small/gemm_small.onnx -``` -and compile the library -``` bash -make library -``` +program hello_roseNNa + use model_model + implicit none + real(real64) :: input(784), output(10) + integer :: status -Compile the "source files" (`capiTester.f90`) and link to the library file created: -``` bash -gfortran -c ../examples/capiTester.f90 -IobjFiles/ -gfortran -o flibrary capiTester.o libcorelib.a -./flibrary -``` -and finally check if the output from PyTorch model matches roseNNa's output -``` bash -python ../test/testChecker.py gemm_small + call model_init("model.rwt", status) ! only for a file-loaded model + call model_infer(input, output) ! run inference +end program ``` -## Compiling roseNNa - -1. **Save the neural network model that needs to be converted** +or in C: - Make sure to refer to the specific library's documentation about how to save the model. - -2. **Convert the saved model to an ONNX format** - - Details on converting a saved model to ONNX format can be found on their [website](https://onnx.ai/supported-tools.html#buildModel). - - - **Converting an LSTM?** - - ONNX's constant folding renames an LSTM's weight initializers and stores the - four gates in ONNX's `iofc` order, while roseNNa's `lstm_cell` consumes - PyTorch's `ifgo` order. The parser now remaps the gates internally and looks - every weight up by name, so a single `do_constant_folding=True` export is all - that is needed. Earlier versions required a second, unoptimized - (`do_constant_folding=False`) export passed via `-w`; that flag is now - accepted but ignored. +```c +#include "model.h" -```python -torch.onnx.export(model, # model being run - (inp, hidden), # model input (or a tuple for multiple inputs) - filePath+"lstm_gemm.onnx", # where to save the model (can be a file or file-like object) - export_params=True, # store the trained parameter weights inside the model file - opset_version=12, # the ONNX version to export the model to - do_constant_folding=True, # whether to execute constant folding for optimization - input_names = ['input', 'hidden_state','cell_state'], # the model's input names - output_names = ['output'], # the model's output names - ) +int main(void) { + double input[784], output[10]; + if (model_init("model.rwt") != 0) return 1; /* file-loaded models only */ + model_infer(input, output); +} ``` -3. **Preprocess the model** - -`fLibrary/` holds the library files that recreate and run inference on the model. Run `python modelParserONNX.py -f path/to/model.onnx` to reconstruct the model. - -4. **Compiling the library** - -Then, in the same `/fLibrary` directory, run `make library`. This compiles the library into `libcorelib.a`, which is required to link other `*.o` files with the library. This library file is now ready to be integrated into any Fortran/C workflow. +A model under a million parameters embeds its weights into the generated source by default, and then has no `init` to call at all. +`model_infer` is `pure` in Fortran, takes `restrict` pointers in C, does no I/O and allocates nothing, so it is safe to call from inside an OpenMP-target, OpenACC, CUDA or HIP loop. ## Supported ONNX operators and limits -roseNNa supports the following ONNX operators: `Gemm`, `MatMul`, `Conv`, `MaxPool`, `AveragePool`, `LSTM`, `Add`, -`Reshape`, `Transpose`, `Squeeze`, `Relu`, `Sigmoid`, `Tanh`. +roseNNa generates code for: `Gemm`, `MatMul`, `Conv`, `MaxPool`, `AveragePool`, `LSTM`, `Add`, +`Reshape`, `Transpose`, `Squeeze`, `Unsqueeze`, `Flatten`, `Identity`, `Relu`, `Sigmoid`, `Tanh`. -The parser rejects a model with `NotImplementedError` rather than silently producing a wrong answer when it -encounters an attribute it cannot honour. The limits it enforces: +Everything statically knowable is resolved at generation time: shapes, buffer sizes, padding (including `auto_pad`), and every node whose inputs are all constants -- so a `Reshape` of a weight, or an int64 shape tensor, never reaches the emitted code. -- `kernel_shape` is required for `MaxPool` and `AveragePool` (inferred from the weights for `Conv`) -- `dilations` must be 1 -- `ceil_mode` must be 0 -- kernels must be square -- pads must be symmetric per axis -- `Conv` `group` must be 1 (no grouped or depthwise convolution) -- `AveragePool` with nonzero pads requires `count_include_pad=1` -- `AveragePool` `auto_pad` must be `NOTSET` or `VALID` -- a `Pad` node must have all-zero pads -- `Gemm` `alpha` and `beta` must be 1, and `transA` must be 0 - -## Fortran use - -One can compile a Fortran example (like the `Hello RoseNNa` example above) by specifying the location of the module files and linking the library to other program files. -In practice, this looks like -``` shell -gfortran -c *.f90 -Ipath/to/objFiles -gfortran -o flibrary *.o path/to/libcorelib.a -./flibrary -``` +A model using something the generator cannot lower is **refused by name at generation time**, never silently mis-computed. `rosenna info model.onnx` reports what it found. The limits: -**Memory layout.** `use_model` expects inputs in Fortran (column-major) order. A C caller with a row-major array must transpose it first; a Fortran caller building an array from a row-major literal should use `RESHAPE(..., order=[2,1])`, as `examples/capiTester.f90` does. +- 2-D spatial ops only (rank-4 NCHW); `ceil_mode` must be 0 +- `Conv` `group` must be 1 (no grouped or depthwise convolution) +- `Gemm` `alpha` and `beta` must be 1, `transA` must be 0, and weights must be constant +- `LSTM` must be forward-direction with the default activations, no `clip`, `input_forget`, `sequence_lens` or peepholes +- one output; several inputs are fine and arrive concatenated (see below) +- every weight must be a constant initializer, not computed at runtime -## C use +## Verify it -One can readily call roseNNa from C. -Compile roseNNa, then use the following C program as an example: -```c -#include +```sh +rosenna verify model.onnx --cases 32 +``` -void use_model(double * i0, double * o0); -void initialize(const char * model_file, const char * weights_file); +compiles both backends and compares them against onnxruntime on random inputs. Every model in `goldenFiles/` is checked this way, on both backends, by `python/tests/test_golden_suite.py`. -int main(void) { +## Several inputs - /* roseNNa expects column-major (Fortran) ordering. */ - double a[2] = {1, 1}; - double b[3]; +A model with more than one graph input -- an LSTM's initial hidden and cell state, say -- takes them **concatenated in declaration order** in the single `x` buffer. That keeps one entry point, one input buffer, and so one device contract, for every model. - initialize("onnxModel.txt", "onnxWeights.bin"); - use_model(a, b); +## GPU use - for (int i = 0; i < 3; i++) { - printf("%f ", b[i]); - } - printf("\n"); - return 0; -} -``` -and compile it as -```shell -gcc -c *.c -gfortran -o capi *.o path/to/libcorelib.a -./capi -``` - -A weights path ending in `.txt` (in any letter case, trailing blanks ignored) is read as the legacy text format; -any other path is read as little-endian float64 binary, which must match the model exactly, or `initialize` -stops with an error. The `onnxWeights.txt` fallback described under Hello RoseNNa is read as text. +The generated code is callable from a device loop, and `rosenna gpu-gate` validates that end to end on real hardware. See [python/README.md](python/README.md) for the full story: the batched entry point, the CUDA/HIP kernel, the build recipes, and the measured per-point cost. ## Further documentation -Please see [this document](https://github.com/comp-physics/roseNNa/blob/master/doc/opensource.md) on how to extend roseNNa to new network models and [this document](https://github.com/comp-physics/roseNNa/blob/master/doc/methodology.md) on the details of the roseNNa pipeline. - -## Python code generator - -`python/` holds a second, newer way to use roseNNa: a generator that reads an ONNX model and emits a small, self-contained Fortran module and/or C library, callable per point from inside your own OpenMP-target, OpenACC, CUDA or HIP loop, with its weights device-resident. +- [python/README.md](python/README.md) -- install, generate, build, and call from C or Fortran +- [doc/methodology.md](doc/methodology.md) -- the roseNNa pipeline +- [doc/opensource.md](doc/opensource.md) -- extending roseNNa to new operators -Two paths currently coexist in this repository. The `fLibrary/` runtime library described in the rest of this README supports every op roseNNa implements (RNNs, CNNs, MLPs). The generator in `python/` supports dense (Gemm/MatMul + Relu/Tanh/Sigmoid) models only, but its output is GPU-callable. The generator is meant to replace the library once it covers everything the library does; until then, use `fLibrary/` for anything the generator does not yet support. +## History -See [python/README.md](python/README.md) for how to install, generate, build and call generated code from C or Fortran. +roseNNa began as `fLibrary/`: a Fortran library that parsed a model description at +startup and walked it at runtime. The generator in `python/` replaced it once it +covered every operator the library did and every model in `goldenFiles/`, which it +now verifies against onnxruntime on both backends rather than against recorded +output. The library, its `modelParserONNX.py`, and the shell suite that drove it +were removed at that point; they remain in the git history. ## Citation diff --git a/doc/methodology.md b/doc/methodology.md index 903d3cf..b5b89f3 100644 --- a/doc/methodology.md +++ b/doc/methodology.md @@ -1,13 +1,102 @@ # Pipeline -First, all the core files are compiled (`activation_funcs.f90`, `derived_types.f90`, `layers.f90`, `reader.f90`). `activation_funcs.f90` stores activation functions, `derived_types.f90` stores derived types for certain layer types, `layers.f90` stores the math behind certain layers (**currently we support GEMM, LSTM, Convolutional, and MaxPool layers**), and `reader.f90` loads in the weights that are stored in the system itself. -## Initialization and Preprocessing -Then, in each of the test case files in [`goldenFiles`](https://github.com/comp-physics/roseNNa/tree/master/goldenFiles), the **.py** file is run to create the model, randomly initialized with weights. It creates an intermediary file called inputs.fpp, which stores the exact inputs given to the model, which is later fed to the fortran built model. It also creates a "golden file" which represents the correct shape and output of the model. Lastly, the model that was run is stored in **.onnx** format. +roseNNa turns an ONNX model into Fortran and C source at *generation* time. The +generated code contains no parser, no shape logic, no allocation and no mutable +global state: every loop bound is a literal, so it inlines into a solver's own +compute kernel, including a device kernel. -[`modelParserONNX.py`](https://github.com/comp-physics/roseNNa/blob/master/fLibrary/modelParserONNX.py) is run to parse the onnx model and gathers information about the model and creates `onnxModel.txt` (layer names and weights dimensions) and `onnxWeights.bin` (the corresponding weights for each layer). It also creates a `variables.fpp` file that stores some key information about the model that fypp will process during model creation. +The pipeline is a chain of graph-to-graph passes in `python/rosenna/`, each of +which either resolves something or refuses the model by name. -## Running and Testing -Lastly, we have two **.fpp** files. [`modelCreator.fpp`](https://github.com/comp-physics/roseNNa/blob/master/fLibrary/modelCreator.fpp) is the module that builds the subroutine that stores the correct model architecture. It parses through `variables.fpp` and reconstructs the model with the subroutines in **layers.f90**. [`userTesting.fpp`](https://github.com/comp-physics/roseNNa/blob/master/test/userTesting.fpp) is used to create **userTesting.f90**, a sample file that calls "**initialize**" (which enables fortran to read in the weights and model structure from `onnxModel.txt` and `onnxWeights.bin`). Then it passes in the inputs from the intermediary file inputs.fpp, and runs the model. [`userTesting.fpp`](https://github.com/comp-physics/roseNNa/blob/master/test/userTesting.fpp) then stores the shape and output in a text file. +## 1. Load — `frontend.py` +`onnx.shape_inference` first, so every value has a literal shape. The result is +a `Graph` of `Node`s, `Tensor` values, and initializer arrays. A symbolic +dimension is refused here: roseNNa fixes every shape at generation. -[`testChecker.py`](https://github.com/comp-physics/roseNNa/blob/master/test/testChecker.py) compares the outputted text file to the test's "golden file". If the shapes match and the outputs are within reasonable range, the test case passes. Otherwise, the error is outputted as either a failure due to shape or mismatching values or to an external text file `output.txt` indicating there was a runtime failure somewhere (probably due to the model encoding, decoding, or running). +## 2. Fold — `fold.py` + +Two passes run before anything else looks at the graph. + +`fold_constants` evaluates every node whose inputs are all constants and turns +the result into an initializer. A real export is full of these: the shape +tensor of a `Reshape`, a `Constant` holding an LSTM's initial state, a weight +transposed once on the way in. Running them now is also what removes the int64 +tensors the generated code could never carry. + +`strip_shape_inputs` then drops the metadata operands of relabelling ops, and +any initializer nothing reads any more. + +## 3. Validate — `validate.py` + +Refuses, by node name, anything the emitters cannot lower: an unsupported op, a +rank the loop nests do not implement, a `Conv` with `group > 1`, an `LSTM` with +custom activations. Every rule here exists because the alternative is a model +that runs and returns confident nonsense — which is the failure mode this file +exists to prevent. + +## 4. Plan — `plan.py` + +Lowers the graph to an explicit `Plan`: a list of `Op`s, a set of flat rank-1 +buffers, and a weight layout. + +- **Shapes become arithmetic.** Buffers stay rank 1 and row-major whatever the + value's logical rank; a `Spatial` spec carries the literal extents a Conv or + pool loop nest needs, and `auto_pad` is resolved to begin-pads here, because + it depends on the input shape and the input shape is known. +- **Relabelling is free.** `Reshape`, `Squeeze`, `Unsqueeze`, `Flatten`, + `Identity`, and any `Transpose` that only moves size-1 axes move no bytes, so + they become buffer aliases: no code, no copy. Liveness is tracked on the root + of an alias chain, so a buffer is only reused after the last read of anything + sharing it. +- **Buffers are recycled.** Input and output get dedicated buffers; every + intermediate rotates through a free list. +- **Several inputs, one buffer.** A model with more than one graph input takes + them concatenated in `x` in declaration order, and each secondary input is + copied out of its slice. That is what keeps `infer(x, y)` — and with it + `infer_batch`, the native kernel and the device contract — unchanged. + +The plan carries a sha256 of itself, which the weights file records and the +generated reader checks. + +## 5. Emit — `emit_c.py`, `emit_fortran.py`, `emit_kernel.py` + +Both emitters render the *same* plan, so the two backends agree to 1e-12 and +emit identical buffer structure. `emit_kernel.py` writes the native CUDA/HIP +batched kernel, which calls the same header-inline body. + +One detail is not cosmetic: a dense layer adds its bias **after** the dot +product rather than seeding the accumulator with it. Seeding an accumulator +from a declare-target array is what makes nvc refuse to generate a +`distribute parallel for` body at all — it emits a kernel that traps — and the +reordering unlocks a ~30x faster per-point offload loop. See +[`python/examples/nvhpc_teams_mapping/`](../python/examples/nvhpc_teams_mapping/). + +## 6. Verify — `verify.py` + +`rosenna verify` generates, compiles and runs both backends and compares them +against onnxruntime on random inputs drawn from a fixed seed. + +Two things it deliberately does. It resamples until the reference is *alive*: a +model whose own weights compute all zeros would otherwise "pass" by reproducing +a dead network. And it allows a cancellation term in the tolerance — the +classical `n * eps * sum|terms|` bound for a summation — because onnxruntime +blocks and vectorises its convolutions and GEMMs, so two correct +implementations legitimately differ by more than `rtol * |expected|` when the +sum cancels. + +`python/tests/test_golden_suite.py` runs every model in `goldenFiles/` through +this, on both backends. That replaced the old shell suite, which compared +against recorded output — a recorded file pins whatever the library did the day +it was recorded, so a wrong-but-stable implementation records its own error as +the expectation. + +## 7. Gate — `gate.py` + +`rosenna gpu-gate` is the check that exercises the device path on real +hardware: it builds the model embedded and file-loaded, in both languages, and +runs three harnesses — a per-point C host, a per-point Fortran host, and a host +that hands device-resident data to `infer_batch` — each compared against +onnxruntime and timed. An `nsys` capture scoped to the timed call asserts zero +`cudaMemcpy` inside it. Every command and its output goes into +`gate-report.md`. diff --git a/doc/opensource.md b/doc/opensource.md index aaf1418..a664d57 100644 --- a/doc/opensource.md +++ b/doc/opensource.md @@ -1,209 +1,119 @@ -# Open Source Development -This project is ongoing and does not contain functionality of every layer available in ONNX. In order to embed new layers into roseNNa, certain steps must be followed: +# Adding an operator -## Parsing in modelParserONNX.py -This file reads in the ONNX interpretation of the model. At a higher level, it iterattes over all the layers in the ONNX model (called nodes in the graph), parses its contents by (1) sending some of its options to be parsed in f90 via fypp and (2) finding the weights that correspond to this layer and writing their dimensions to 'onnxModel.txt' and the weights to `onnxWeights.bin`. These two files will be read in by Fortran so it can store the weights and layers. Here is a pseudocode example from the "GEMM" layer in ONNX: +roseNNa does not implement every ONNX operator. Adding one means teaching four +places about it, in this order. The order matters: each step is refused loudly +by the one before it until you get there, so you are never debugging generated +code that should not have been generated. -```python -#an additional elif branch must be added so the parser knows to parse this layer -elif layer == "Gemm": - #the layer name tells reader.f90 which read routine to call - f.write(layer) - f.write("\n") - names = {n.name:n.i if n.type==2 else n.ints for n in node.attribute} - #(the full branch also rejects attributes roseNNa cannot honour: transA, alpha, beta, and a bias that is not rank 1) - - #modelArch stores the layer and options for layer (fypp input later on) - #ioMap is referenced to get the output name from the last layer (which is input to this layer) - modelArch.append(("Gemm", [ioMap[node.input[0]], names.get('transB', 0)], None)) - - #parsing the weight and bias inputs to the layer - #(when the bias is absent, the full branch writes a zero bias instead) - for inp in node.input[1:3]: - - #writing the dimensions to 'onnxModel.txt' - for dim in initializer[inp][0]: - f.write(str(dim)+ " ") - f.write("\n") - - #writing the weights to 'onnxWeights.bin' as little-endian float64 in column-major (Fortran) order - #findWeightsInitializer looks the tensor up by name among the initializers and Constant nodes - f2.write(np.asarray(findWeightsInitializer(inp), dtype='