From 448075a0ad91a7547829f46b792472f5f3e4a249 Mon Sep 17 00:00:00 2001 From: 37ng Date: Mon, 7 Sep 2026 19:07:09 -0700 Subject: [PATCH 1/5] add a CLI entry that dispatches to the existing scripts main.py was the uv init stub. It now lists every script with a main() and runs one by name: the script's directory goes on sys.path, sys.argv is rewritten so help reads "main.py ", the file is loaded as a module, and its main() runs. Flags are not repeated here; everything after the command name goes to that script's own parser, and its exit code is passed through. Co-Authored-By: Claude Opus 5 --- main.py | 80 +++++++++++++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 78 insertions(+), 2 deletions(-) diff --git a/main.py b/main.py index 260e80a..adb5778 100644 --- a/main.py +++ b/main.py @@ -1,6 +1,82 @@ +import importlib.util +import os +import sys + +ROOT = os.path.dirname(os.path.abspath(__file__)) +PROG = os.path.basename(__file__) + +GROUPS = [ + ("broker", [ + ("fetch-ms", "broker/fetch_ms.py", + "download acceleration history from mempool.space"), + ("run-ms", "broker/run_ms.py", + "build the out-of-band spend tables"), + ("export-ms", "broker/export_ms.py", + "publish the monthly out-of-band spend file"), + ]), + ("chain", [ + ("run-pipeline", "chain/run_pipeline.py", + "the block space pipeline, end to end"), + ("effective-fee", "chain/effective_fee.py", + "the union-find pass on its own"), + ("export-results", "chain/export_results.py", + "write the monthly result files"), + ("export-onchain-fee", "chain/export_onchain_monthly_fee.py", + "publish the monthly on-chain fee file"), + ("sanity-check", "chain/sanity_check.py", + "pool attribution quality and monthly shares"), + ("validate-mempool", "chain/validate_against_mempool.py", + "check low-fee blocks against mempool.space"), + ]), + ("utils", [ + ("refresh-pools", "utils/refresh_pools.py", + "update pools.json from upstream"), + ("delete-dataset", "utils/delete_dataset.py", + "drop the BigQuery working dataset"), + ]), +] + +COMMANDS = {name: path for _group, items in GROUPS for name, path, _what in items} + + +def usage(stream=sys.stdout): + print(f"usage: {PROG} [args]\n", file=stream) + for group, items in GROUPS: + print(group, file=stream) + for name, _path, what in items: + print(f" {name:20s} {what}", file=stream) + print(file=stream) + print(f"`{PROG} --help` lists the flags of one command", + file=stream) + + +def run(name, argv): + script = os.path.join(ROOT, COMMANDS[name]) + sys.path.insert(0, os.path.dirname(script)) + sys.argv = [f"{PROG} {name}"] + argv + + module_name = os.path.splitext(os.path.basename(script))[0] + spec = importlib.util.spec_from_file_location(module_name, script) + module = importlib.util.module_from_spec(spec) + sys.modules[module_name] = module + spec.loader.exec_module(module) + return module.main() + + def main(): - print("Hello from bitcoin-private-blockspace!") + argv = sys.argv[1:] + if not argv or argv[0] in ("-h", "--help", "help"): + usage() + return 0 + + name, rest = argv[0], argv[1:] + if name not in COMMANDS: + print(f"unknown command: {name}\n", file=sys.stderr) + usage(sys.stderr) + return 2 + + return run(name, rest) or 0 if __name__ == "__main__": - main() + sys.exit(main()) From f069692c3d5e6dc66164e4b46d6d74b39d6a4b26 Mon Sep 17 00:00:00 2001 From: 37ng Date: Mon, 7 Sep 2026 19:28:28 -0700 Subject: [PATCH 2/5] follow the broker changes in the CLI fetch_ms.py is now a library: no main(), and it writes month files into broker/data itself. The dispatcher gained a second kind of command, one the CLI parses on its own, and fetch-ms is the first: no flag fetches the missing months, --month fetches one, --from/--to fetch a range. export-ms is dropped; broker/export_ms.py no longer exists. Co-Authored-By: Claude Opus 5 --- main.py | 55 +++++++++++++++++++++++++++++++++++++++++++++++++------ 1 file changed, 49 insertions(+), 6 deletions(-) diff --git a/main.py b/main.py index adb5778..a5e8f4c 100644 --- a/main.py +++ b/main.py @@ -1,6 +1,8 @@ +import argparse import importlib.util import os import sys +from datetime import datetime ROOT = os.path.dirname(os.path.abspath(__file__)) PROG = os.path.basename(__file__) @@ -11,8 +13,6 @@ "download acceleration history from mempool.space"), ("run-ms", "broker/run_ms.py", "build the out-of-band spend tables"), - ("export-ms", "broker/export_ms.py", - "publish the monthly out-of-band spend file"), ]), ("chain", [ ("run-pipeline", "chain/run_pipeline.py", @@ -39,6 +39,42 @@ COMMANDS = {name: path for _group, items in GROUPS for name, path, _what in items} +def month(value): + try: + datetime.strptime(value, "%Y-%m") + except ValueError: + raise argparse.ArgumentTypeError(f"{value} is not a YYYY-MM month") + return value + + +def fetch_ms(module, argv): + parser = argparse.ArgumentParser( + description="with no flag, fetch every month after the newest file in " + "broker/data, up to the last complete month") + parser.add_argument("--month", type=month, metavar="YYYY-MM", + help="fetch this month alone") + parser.add_argument("--from", dest="from_month", type=month, + metavar="YYYY-MM", help="fetch this month and later") + parser.add_argument("--to", dest="to_month", type=month, metavar="YYYY-MM", + help="stop after this month") + args = parser.parse_args(argv) + + if args.month and (args.from_month or args.to_month): + parser.error("--month does not go with --from or --to") + if bool(args.from_month) != bool(args.to_month): + parser.error("--from and --to go together") + + if args.month: + module.fetch_ms(args.month) + elif args.from_month: + module.fetch_ms_range(args.from_month, args.to_month) + else: + module.fetch_missing_ms() + + +HANDLERS = {"fetch-ms": fetch_ms} + + def usage(stream=sys.stdout): print(f"usage: {PROG} [args]\n", file=stream) for group, items in GROUPS: @@ -50,16 +86,23 @@ def usage(stream=sys.stdout): file=stream) -def run(name, argv): - script = os.path.join(ROOT, COMMANDS[name]) +def load(script): sys.path.insert(0, os.path.dirname(script)) - sys.argv = [f"{PROG} {name}"] + argv - module_name = os.path.splitext(os.path.basename(script))[0] spec = importlib.util.spec_from_file_location(module_name, script) module = importlib.util.module_from_spec(spec) sys.modules[module_name] = module spec.loader.exec_module(module) + return module + + +def run(name, argv): + script = os.path.join(ROOT, COMMANDS[name]) + sys.argv = [f"{PROG} {name}"] + argv + module = load(script) + handler = HANDLERS.get(name) + if handler: + return handler(module, argv) return module.main() From 92bffad6b24dc94e85b619b12e739c6219e70bf2 Mon Sep 17 00:00:00 2001 From: 37ng Date: Mon, 7 Sep 2026 21:29:20 -0700 Subject: [PATCH 3/5] move the chain flags out of chain/ and into the CLI Every script in chain/ loses its argparse, its main(), and its __main__ block, the way fetch_ms.py already did. What each main() did is now a function that takes arguments: run_pipeline.run(month=, only=, from_step=, dry=, yes=, skip_checks=) export_onchain_monthly_fee.export(out_path) validate_against_mempool.validate(sample, sleep, sensitivity) select_steps takes only/from_step instead of an argparse namespace. sanity_check, export_results and effective_fee already had the functions the CLI needs, so they only lost code. main.py parses the flags for all six and keeps the old names, defaults and choices. Three lines that told a reader to run a script directly now name the command instead. Co-Authored-By: Claude Opus 5 --- chain/effective_fee.py | 19 ------ chain/export_onchain_monthly_fee.py | 15 +---- chain/export_results.py | 23 +------- chain/run_pipeline.py | 55 ++++++----------- chain/sanity_check.py | 18 ------ chain/validate_against_mempool.py | 23 ++------ main.py | 92 ++++++++++++++++++++++++++++- utils/config.py | 2 +- utils/refresh_pools.py | 2 +- 9 files changed, 121 insertions(+), 128 deletions(-) diff --git a/chain/effective_fee.py b/chain/effective_fee.py index b172e21..005b05a 100644 --- a/chain/effective_fee.py +++ b/chain/effective_fee.py @@ -1,4 +1,3 @@ -import argparse import itertools import json import os @@ -158,21 +157,3 @@ def run(source, writer, chunk_blocks=None, flush_rows=None, verbose=True): if verbose: print(f" wrote {writer.written} package rows", flush=True) return writer.written - - -def main(): - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--chunk-blocks", type=int, - default=config.UNIONFIND_CHUNK_BLOCKS) - parser.add_argument("--sqlite", help="run against a local database instead") - args = parser.parse_args() - - if args.sqlite: - source, writer = SqliteSource(args.sqlite), ListWriter() - else: - source, writer = BigQuerySource(), BigQueryWriter() - run(source, writer, chunk_blocks=args.chunk_blocks) - - -if __name__ == "__main__": - main() diff --git a/chain/export_onchain_monthly_fee.py b/chain/export_onchain_monthly_fee.py index c1d71e7..09cd489 100644 --- a/chain/export_onchain_monthly_fee.py +++ b/chain/export_onchain_monthly_fee.py @@ -1,4 +1,3 @@ -import argparse import json import os import sys @@ -87,21 +86,13 @@ def report(months): print(f"{'total':<9}{'':>9}{'':>12}{total:>14.4f}") -def main(): - parser = argparse.ArgumentParser() - parser.add_argument("--out", default=os.path.join(config.DATA_DIR, FILENAME)) - args = parser.parse_args() - +def export(out_path): fresh = build(monthly_rows()) if not fresh: print("the onchain_monthly_fee table is empty; nothing to publish") return 1 - months = merge(read(args.out), fresh) + months = merge(read(out_path), fresh) report(months) - write(payload(months), args.out) + write(payload(months), out_path) return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/chain/export_results.py b/chain/export_results.py index 96cdd23..4cb9fb2 100644 --- a/chain/export_results.py +++ b/chain/export_results.py @@ -1,4 +1,3 @@ -import argparse import datetime import decimal import json @@ -10,7 +9,6 @@ sys.path.insert(0, os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "utils")) import bqio -import config TABLES = { "monthly_summary": "SELECT * FROM `${dst}.monthly_summary` ORDER BY block_month", @@ -219,8 +217,8 @@ def write_summary(out_dir, monthly, sensitivity_grid, pools): "- A number that moves by an order of magnitude across the threshold " "grid is a statement about the cut-offs, not about the chain.", "- Per-pool rows depend on coinbase tag attribution. Run " - "`sanity_check.py` and compare against a public hashrate chart before " - "quoting any of them.", + "`main.py sanity-check` and compare against a public hashrate chart " + "before quoting any of them.", ] path = os.path.join(out_dir, "summary.md") @@ -272,20 +270,3 @@ def on_disk(name): def export_month(out_dir, replace=False): return merge_into(out_dir, fetch(), replace=replace) - - -def main(): - parser = argparse.ArgumentParser(description=__doc__, - formatter_class=argparse.RawDescriptionHelpFormatter) - parser.add_argument("--out", default=config.OUT_DIR) - parser.add_argument("--replace", action="store_true", - help="ignore what is already in --out and write only " - "the months the working dataset holds") - args = parser.parse_args() - - print(f"writing to {args.out}/") - export_month(args.out, replace=args.replace) - - -if __name__ == "__main__": - main() diff --git a/chain/run_pipeline.py b/chain/run_pipeline.py index ecf945c..9a6f4ba 100644 --- a/chain/run_pipeline.py +++ b/chain/run_pipeline.py @@ -1,4 +1,3 @@ -import argparse import os import sys @@ -65,19 +64,19 @@ def sql_name(step): return os.path.join(SQL_DIR, f"{step}.sql") -def select_steps(args): +def select_steps(only=None, from_step=None): names = [s[0] for s in STEPS] - if args.only: - wanted = set(args.only.split(",")) + if only: + wanted = set(only.split(",")) unknown = wanted - set(names) if unknown: sys.exit(f"unknown step(s): {', '.join(sorted(unknown))}") return [s for s in STEPS if s[0] in wanted] start = 0 - if getattr(args, "from_step", None): - if args.from_step not in names: - sys.exit(f"unknown step: {args.from_step}") - start = names.index(args.from_step) + if from_step: + if from_step not in names: + sys.exit(f"unknown step: {from_step}") + start = names.index(from_step) return STEPS[start:] @@ -139,30 +138,16 @@ def headline(): print(" read low_fee_sensitivity before quoting any of this") -def main(): - parser = argparse.ArgumentParser(description=__doc__, - formatter_class=argparse.RawDescriptionHelpFormatter) - parser.add_argument("--dry-run", action="store_true", - help="report bytes per step and stop") - parser.add_argument("--month", metavar="YYYY-MM", - help="run one month end to end, then merge it into out/") - parser.add_argument("--from", dest="from_step", metavar="STEP", - help="start at this step") - parser.add_argument("--only", metavar="STEP[,STEP]", - help="run only these steps") - parser.add_argument("--yes", action="store_true", - help="do not ask before spending") - parser.add_argument("--skip-checks", action="store_true") - args = parser.parse_args() - - if args.month: - config.set_month(args.month) +def run(month=None, only=None, from_step=None, dry=False, yes=False, + skip_checks=False): + if month: + config.set_month(month) print(f"month run: {config.START_DATE} .. {config.END_DATE}") - steps = select_steps(args) + steps = select_steps(only, from_step) bqio.ensure_dataset() - if args.dry_run: + if dry: dry_run(steps) return @@ -171,7 +156,7 @@ def main(): estimate = bqio.dry_run(bqio.render(sql_name("01_tx_base"))) print(f"step 01 will scan {bqio.human_bytes(estimate)} " f"(about ${bqio.usd(estimate):.2f})") - if not args.yes and not bqio.confirm("run it?"): + if not yes and not bqio.confirm("run it?"): sys.exit("stopped; nothing was run") for name, kind, what in steps: @@ -182,16 +167,12 @@ def main(): effective_fee.run(effective_fee.BigQuerySource(), effective_fee.BigQueryWriter()) - if not args.skip_checks: + if not skip_checks: run_checks() headline() - if args.month: - print(f"\nmerging {args.month} into {config.OUT_DIR}/") + if month: + print(f"\nmerging {month} into {config.OUT_DIR}/") export_results.export_month(config.OUT_DIR) print(f"\ndone. delete the BigQuery working dataset when ready:" - f"\n python ../utils/delete_dataset.py") - - -if __name__ == "__main__": - main() + f"\n .venv/bin/python main.py delete-dataset") diff --git a/chain/sanity_check.py b/chain/sanity_check.py index 9314cfb..734df84 100644 --- a/chain/sanity_check.py +++ b/chain/sanity_check.py @@ -1,4 +1,3 @@ -import argparse import os import sys @@ -72,20 +71,3 @@ def attribution_quality(): print("(a big count here is a missing tag in pools.py)") for r in rows: print(f" {r['blocks']:>5d} {r['coinbase_text']}") - - -def main(): - parser = argparse.ArgumentParser(description=__doc__, - formatter_class=argparse.RawDescriptionHelpFormatter) - parser.add_argument("--top", type=int, default=12, - help="pools listed per month") - parser.add_argument("--months", type=int, default=12) - parser.add_argument("--all", action="store_true", help="every month") - args = parser.parse_args() - - attribution_quality() - monthly_shares(None if args.all else args.months, args.top) - - -if __name__ == "__main__": - main() diff --git a/chain/validate_against_mempool.py b/chain/validate_against_mempool.py index d001baa..38397f6 100644 --- a/chain/validate_against_mempool.py +++ b/chain/validate_against_mempool.py @@ -1,4 +1,3 @@ -import argparse import json import os import random @@ -60,23 +59,15 @@ def sample_blocks(sample, sensitivity_column): return rows -def main(): - parser = argparse.ArgumentParser(description=__doc__, - formatter_class=argparse.RawDescriptionHelpFormatter) - parser.add_argument("--sample", type=int, default=50) - parser.add_argument("--sleep", type=float, default=1.5, - help="seconds between API calls") - parser.add_argument("--sensitivity", choices=["30", "50", "70"], default="50") - args = parser.parse_args() - - column = f"low_fee_{args.sensitivity}" - blocks = sample_blocks(args.sample, column) +def validate(sample, sleep, sensitivity): + column = f"low_fee_{sensitivity}" + blocks = sample_blocks(sample, column) if not blocks: print(f"no blocks with {column} above height {MIN_AUDITED_HEIGHT}") return print(f"checking {len(blocks)} blocks against mempool.space " - f"(sensitivity 0.{args.sensitivity})\n") + f"(sensitivity 0.{sensitivity})\n") audited = 0 total_low_fee = 0 @@ -85,7 +76,7 @@ def main(): blocks_with_any_overlap = 0 for block in blocks: - data = fetch_audit(block["block_hash"], args.sleep) + data = fetch_audit(block["block_hash"], sleep) if not data: print(f" {block['block_number']} no audit data") continue @@ -123,7 +114,3 @@ def main(): print("Transactions in acceleratedTxs were bought out of band through a " "public service: they confirm the mechanism, and they are the part " "of the count that is not invisible.") - - -if __name__ == "__main__": - main() diff --git a/main.py b/main.py index a5e8f4c..680aba3 100644 --- a/main.py +++ b/main.py @@ -7,6 +7,10 @@ ROOT = os.path.dirname(os.path.abspath(__file__)) PROG = os.path.basename(__file__) +sys.path.insert(0, os.path.join(ROOT, "utils")) + +import config + GROUPS = [ ("broker", [ ("fetch-ms", "broker/fetch_ms.py", @@ -72,7 +76,93 @@ def fetch_ms(module, argv): module.fetch_missing_ms() -HANDLERS = {"fetch-ms": fetch_ms} +def run_pipeline(module, argv): + parser = argparse.ArgumentParser() + parser.add_argument("--dry-run", action="store_true", + help="report bytes per step and stop") + parser.add_argument("--month", type=month, metavar="YYYY-MM", + help="run one month end to end, then merge it into out/") + parser.add_argument("--from", dest="from_step", metavar="STEP", + help="start at this step") + parser.add_argument("--only", metavar="STEP[,STEP]", + help="run only these steps") + parser.add_argument("--yes", action="store_true", + help="do not ask before spending") + parser.add_argument("--skip-checks", action="store_true") + args = parser.parse_args(argv) + + module.run(month=args.month, only=args.only, from_step=args.from_step, + dry=args.dry_run, yes=args.yes, skip_checks=args.skip_checks) + + +def effective_fee(module, argv): + parser = argparse.ArgumentParser() + parser.add_argument("--chunk-blocks", type=int, + default=config.UNIONFIND_CHUNK_BLOCKS) + parser.add_argument("--sqlite", help="run against a local database instead") + args = parser.parse_args(argv) + + if args.sqlite: + source, writer = module.SqliteSource(args.sqlite), module.ListWriter() + else: + source, writer = module.BigQuerySource(), module.BigQueryWriter() + module.run(source, writer, chunk_blocks=args.chunk_blocks) + + +def export_results(module, argv): + parser = argparse.ArgumentParser() + parser.add_argument("--out", default=config.OUT_DIR) + parser.add_argument("--replace", action="store_true", + help="ignore what is already in --out and write only " + "the months the working dataset holds") + args = parser.parse_args(argv) + + print(f"writing to {args.out}/") + module.export_month(args.out, replace=args.replace) + + +def export_onchain_fee(module, argv): + parser = argparse.ArgumentParser() + parser.add_argument("--out", + default=os.path.join(config.DATA_DIR, module.FILENAME)) + args = parser.parse_args(argv) + + return module.export(args.out) + + +def sanity_check(module, argv): + parser = argparse.ArgumentParser() + parser.add_argument("--top", type=int, default=12, + help="pools listed per month") + parser.add_argument("--months", type=int, default=12) + parser.add_argument("--all", action="store_true", help="every month") + args = parser.parse_args(argv) + + module.attribution_quality() + module.monthly_shares(None if args.all else args.months, args.top) + + +def validate_mempool(module, argv): + parser = argparse.ArgumentParser() + parser.add_argument("--sample", type=int, default=50) + parser.add_argument("--sleep", type=float, default=1.5, + help="seconds between API calls") + parser.add_argument("--sensitivity", choices=["30", "50", "70"], + default="50") + args = parser.parse_args(argv) + + module.validate(args.sample, args.sleep, args.sensitivity) + + +HANDLERS = { + "fetch-ms": fetch_ms, + "run-pipeline": run_pipeline, + "effective-fee": effective_fee, + "export-results": export_results, + "export-onchain-fee": export_onchain_fee, + "sanity-check": sanity_check, + "validate-mempool": validate_mempool, +} def usage(stream=sys.stdout): diff --git a/utils/config.py b/utils/config.py index d3f5e44..976bbab 100644 --- a/utils/config.py +++ b/utils/config.py @@ -23,7 +23,7 @@ # The source dataset is partitioned by month, and the pipeline aggregates by # month. So a run should cover exactly one month: set MONTH (env var) or pass -# `--month YYYY-MM` to `run_pipeline.py`. +# `--month YYYY-MM` to `main.py run-pipeline`. MONTH = os.environ.get("MONTH") diff --git a/utils/refresh_pools.py b/utils/refresh_pools.py index fdac573..ecc3b70 100644 --- a/utils/refresh_pools.py +++ b/utils/refresh_pools.py @@ -79,7 +79,7 @@ def main(): print(f"wrote {path}") print(f"{with_id} pool ids available to read the acceleration " f"`pools` array") - print("re-run sql/02_blocks.sql, then sanity_check.py") + print("re-run sql/02_blocks.sql, then main.py sanity-check") if __name__ == "__main__": From a1f99b684126c688d680e6018551e82f3a25345d Mon Sep 17 00:00:00 2001 From: 37ng Date: Mon, 7 Sep 2026 21:32:53 -0700 Subject: [PATCH 4/5] doc --- AGENTS.md | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/AGENTS.md b/AGENTS.md index dcbbe93..ccb6a23 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,5 +1,14 @@ # workflow +## commands + +run files in venv + +``` +.venv/bin/python main.py +``` + + ## git worktrees when asked to work on git worktrees From f604d5e0bf02c23fc8979844790ecf3c78226344 Mon Sep 17 00:00:00 2001 From: 37ng Date: Mon, 7 Sep 2026 22:28:25 -0700 Subject: [PATCH 5/5] feat: ci tests --- .github/workflows/tests.yml | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) create mode 100644 .github/workflows/tests.yml diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml new file mode 100644 index 0000000..22048c1 --- /dev/null +++ b/.github/workflows/tests.yml @@ -0,0 +1,17 @@ +name: tests + +on: + push: + branches: [main] + pull_request: + +jobs: + pytest: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: astral-sh/setup-uv@v5 + with: + enable-cache: true + - run: uv sync --group dev --frozen + - run: uv run --no-sync pytest -q