Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions QUICKSTART.md
Original file line number Diff line number Diff line change
Expand Up @@ -166,6 +166,11 @@ quantprobe run --gguf ./models/Qwen3-30B-A3B-Q2_K.gguf
quantprobe bench --gguf ./models/Qwen3-30B-A3B-Q2_K.gguf
```

`bench` refuses nonzero `llama-bench` exits, even when partial output contains a speed row.
It reports the failure with a bounded raw-output tail, without scoring or offering that run
as a contribution. A complete run that fails during teardown is refused too.


### Make your own compressed model

The one-command version — picks a requantizable source from the repo, fetches the eval corpus,
Expand Down
36 changes: 36 additions & 0 deletions quantprobe/runtime.py
Original file line number Diff line number Diff line change
Expand Up @@ -212,6 +212,32 @@ def run(a):
sys.exit(subprocess.call(cmd))


# Enough raw output to carry a backend's fatal message, bounded so a failing run cannot dump a
# multi-megabyte log into the terminal. The tail is the diagnostic; the whole log is noise.
BENCH_TAIL_LINES = 10
BENCH_TAIL_WIDTH = 140


def _bench_failed(rc, txt):
"""The refusal message for a llama-bench that exited non-zero, with a bounded raw tail."""
why = f"exit {rc}"
if rc < 0: # POSIX: a child killed by a signal reports the negative signal number
import signal

try:
why = f"{signal.Signals(-rc).name} (returncode {rc})"
except ValueError:
why = f"signal {-rc} (returncode {rc})"
lines = [ln[:BENCH_TAIL_WIDTH] for ln in txt.strip().splitlines()[-BENCH_TAIL_LINES:]]
tail = "\n".join(" " + ln for ln in lines) if lines else " (no output)"
return (
f"\n[quantprobe] llama-bench FAILED: {why}. No data point was taken.\n"
" Its output is not a result even when a tok/s row parses out of it - a run that died\n"
" partway prints rows on its way down, and half a sweep is not a benchmark.\n"
f" last output:\n{tail}"
)


def bench(a):
if getattr(a, "depth", None):
a.ctx = a.depth # prediction at the benched depth
Expand Down Expand Up @@ -269,6 +295,16 @@ def bench(a):
)
out = subprocess.run(cmd, capture_output=True, text=True, errors="replace", check=False)
txt = out.stdout + out.stderr
# A process that did not complete is not a measurement. llama-bench prints its result table
# row by row, so a run that dies partway - backend OOM, a crash at teardown, an external kill
# - can leave a perfectly parseable `tg32 | x +/- y` behind it. `check=False` was right (we
# want the output either way) but nothing ever read the status, so that leftover number was
# parsed, stamped with a machine state, printed as a result and offered to --contribute as a
# data point for the law. Everything below this line assumes a completed run, so the status
# is read FIRST - and the refusal leaves through a non-zero exit, not a printed note the
# shell cannot see.
if out.returncode != 0:
raise SystemExit(_bench_failed(out.returncode, txt))
mm = re.findall(r"tg\d+(?:\s*@\s*d\d+)?\s*\|\s*([0-9.]+)\s*(?:Â?±|\+/-)\s*([0-9.]+)", txt)
if not mm:
mm = re.findall(r"\|\s*([0-9.]+)\s*(?:Â?±)\s*([0-9.]+)\s*\|\s*$", txt, re.MULTILINE)
Expand Down
15 changes: 15 additions & 0 deletions tests/smoke.py
Original file line number Diff line number Diff line change
Expand Up @@ -3634,6 +3634,21 @@ def t_contribute_payload_carries_model_spec_not_none():
return None


def t_bench_never_scores_a_llama_bench_run_that_failed():
"""`bench` ran llama-bench with check=False and never read the returncode, so a process that
died AFTER printing one parseable `tg32 | x +/- y` row - backend OOM, a kill, a crash at
teardown - was parsed, stamped with a machine state, printed as `measured:` and offered to
--contribute as a data point for the law. A run that did not complete is not a measurement.

The case set lives in tests/test_bench_exit_status.py: rc 0 positive controls (still measured,
still contributes, --dry still runs nothing, argv unchanged), signal-style negative
returncodes, a bounded raw tail, a real failing executable through find_llama, and the exit
status the `quantprobe bench` subcommand hands back to the shell."""
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from test_bench_exit_status import run_smoke
return run_smoke()


def t_decon_screen_mutation_directions_pinned():
"""The Phase B decontamination screen is a kill rule (program law 2026-08-05): a verbatim
protected-bench text MUST flag, an 8-gram-sharing paraphrase MUST flag, a clean sample
Expand Down
Loading